Add CUDA Graph inference acceleration support

This commit is contained in:
RVC-Boss
2026-07-20 22:34:58 +08:00
parent f748ba4c48
commit b3f54df1be
26 changed files with 735 additions and 181 deletions

View File

@@ -10,6 +10,8 @@ from tools.file_io import read_text
import torch import torch
import logging import logging
from tools.cuda_graph import configure_cuda_graph
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -126,6 +128,12 @@ if infer_device.type != "cuda":
) )
# Run a real capture/replay probe on the selected inference device. Both
# application entry points import this module, so downstream inference code
# receives one consistent 0/1 switch without duplicating device checks.
CUDA_GRAPH_AVAILABLE = configure_cuda_graph(infer_device)
CONFIGS_DIR = Path(__file__).resolve().parent CONFIGS_DIR = Path(__file__).resolve().parent
MODEL_CONFIG_FILES = ( MODEL_CONFIG_FILES = (
"v1/32k.json", "v1/32k.json",
@@ -152,6 +160,7 @@ class Config:
self.device = str(infer_device) self.device = str(infer_device)
self.dtype = infer_dtype self.dtype = infer_dtype
self.is_half = infer_dtype == torch.float16 self.is_half = infer_dtype == torch.float16
self.cuda_graph = CUDA_GRAPH_AVAILABLE
self.n_cpu = 0 self.n_cpu = 0
self.gpu_name = None self.gpu_name = None
self.json_config = self.load_config_json() self.json_config = self.load_config_json()

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Vocal extraction aggressiveness", "人声提取激进程度": "Vocal extraction aggressiveness",
"%s → 成功": "%s → Success", "%s → 成功": "%s → Success",
"%s运行中请先停止该任务": "%s is running; stop it before starting another task", "%s运行中请先停止该任务": "%s is running; stop it before starting another task",
"%s进程已终止": "The %s process has been terminated", "%s进程已终止": "The %s process has been terminated",
@@ -49,8 +49,8 @@
"[索引训练][失败] 聚类失败,将使用原始特征继续\n%s": "[Index training][Failed] Clustering failed; continuing with the original features\n%s", "[索引训练][失败] 聚类失败,将使用原始特征继续\n%s": "[Index training][Failed] Clustering failed; continuing with the original features\n%s",
"[索引训练][失败] 请先进行特征提取": "[Index training][Failed] Extract features first", "[索引训练][失败] 请先进行特征提取": "[Index training][Failed] Extract features first",
"ckpt处理": "ckpt Processing", "ckpt处理": "ckpt Processing",
"index文件路径不可包含中文": "The index file path cannot contain Chinese characters", "index文件路径不可包含中文": "The index file path cannot contain Chinese characters",
"pth文件路径不可包含中文": "The .pth file path cannot contain Chinese characters", "pth文件路径不可包含中文": "The .pth file path cannot contain Chinese characters",
"rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Enter the GPU index(es) separated by '-', e.g., 0-0-1 to use 2 processes in GPU0 and 1 process in GPU1", "rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Enter the GPU index(es) separated by '-', e.g., 0-0-1 to use 2 processes in GPU0 and 1 process in GPU1",
"step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Step 1: Fill in the experimental configuration. Experimental data is stored in the 'logs' folder, with each experiment having a separate folder. Manually enter the experiment name path, which contains the experimental configuration, logs, and trained model files.", "step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Step 1: Fill in the experimental configuration. Experimental data is stored in the 'logs' folder, with each experiment having a separate folder. Manually enter the experiment name path, which contains the experimental configuration, logs, and trained model files.",
"step1:正在处理数据": "Step 1: Processing data", "step1:正在处理数据": "Step 1: Processing data",
@@ -71,8 +71,8 @@
"任务": "Task", "任务": "Task",
"伴奏人声分离&去混响&去回声": "Vocals/Accompaniment Separation & Reverberation Removal", "伴奏人声分离&去混响&去回声": "Vocals/Accompaniment Separation & Reverberation Removal",
"使用显卡:%s": "GPUs in use: %s", "使用显卡:%s": "GPUs in use: %s",
"使用模型采样率": "Use model sample rate", "使用模型采样率": "Use model sample rate",
"使用设备采样率": "Use device sample rate", "使用设备采样率": "Use device sample rate",
"保存名": "Save name:", "保存名": "Save name:",
"保存的文件名, 默认空为和源文件同名": "Save file name (default: same as the source file):", "保存的文件名, 默认空为和源文件同名": "Save file name (default: same as the source file):",
"保存的模型名不带后缀": "Saved model name (without extension):", "保存的模型名不带后缀": "Saved model name (without extension):",
@@ -177,7 +177,7 @@
"特征": "Features", "特征": "Features",
"特征提取": "Feature extraction", "特征提取": "Feature extraction",
"状态": "Status", "状态": "Status",
"独占 WASAPI 设备": "Exclusive WASAPI device", "独占 WASAPI 设备": "Exclusive WASAPI device",
"生成器预训练模型不存在将不使用assets/pretrained%s/%sG%s.pth": "Generator pretrained model not found; it will not be used: assets/pretrained%s/%sG%s.pth", "生成器预训练模型不存在将不使用assets/pretrained%s/%sG%s.pth": "Generator pretrained model not found; it will not be used: assets/pretrained%s/%sG%s.pth",
"男转女推荐+12key, 女转男推荐-12key, 如果音域爆炸导致音色失真也可以自己调整到合适音域. ": "Recommended +12 key for male to female conversion, and -12 key for female to male conversion. If the sound range goes too far and the voice is distorted, you can also adjust it to the appropriate range by yourself.", "男转女推荐+12key, 女转男推荐-12key, 如果音域爆炸导致音色失真也可以自己调整到合适音域. ": "Recommended +12 key for male to female conversion, and -12 key for female to male conversion. If the sound range goes too far and the voice is distorted, you can also adjust it to the appropriate range by yourself.",
"目标采样率": "Target sample rate:", "目标采样率": "Target sample rate:",
@@ -202,7 +202,7 @@
"训练结束, 您可查看控制台训练日志或实验文件夹下的train.log": "Training complete. You can check the training logs in the console or the 'train.log' file under the experiment folder.", "训练结束, 您可查看控制台训练日志或实验文件夹下的train.log": "Training complete. You can check the training logs in the console or the 'train.log' file under the experiment folder.",
"训练设备规则选择的精度:%s": "Training precision selected by device rules: %s", "训练设备规则选择的精度:%s": "Training precision selected by device rules: %s",
"训练轮次:{} [{:.0f}%]": "Training epoch: {} [{:.0f}%]", "训练轮次:{} [{:.0f}%]": "Training epoch: {} [{:.0f}%]",
"设备类型": "Device type", "设备类型": "Device type",
"请上传音频文件": "Upload an audio file", "请上传音频文件": "Upload an audio file",
"请填写输出文件夹路径": "Enter the output folder path", "请填写输出文件夹路径": "Enter the output folder path",
"请指定说话人id": "Please specify the speaker/singer ID:", "请指定说话人id": "Please specify the speaker/singer ID:",
@@ -233,7 +233,7 @@
"选择模型": "Select model", "选择模型": "Select model",
"选择索引": "Select index", "选择索引": "Select index",
"选择音高提取算法": "Select the pitch extraction algorithm", "选择音高提取算法": "Select the pitch extraction algorithm",
"采样率:": "Sample rate:", "采样率:": "Sample rate:",
"采样长度": "Sample length", "采样长度": "Sample length",
"重载设备列表": "Reload device list", "重载设备列表": "Reload device list",
"错误信息:%s": "Error details: %s", "错误信息:%s": "Error details: %s",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Index training] External index link already exists: %s", "[索引训练] 外部索引链接已存在:%s": "[Index training] External index link already exists: %s",
"[索引训练][跳过] added索引已存在%s": "[Index training][Skipped] added index already exists: %s", "[索引训练][跳过] added索引已存在%s": "[Index training][Skipped] added index already exists: %s",
"[索引训练][跳过] trained索引已存在%s": "[Index training][Skipped] trained index already exists: %s", "[索引训练][跳过] trained索引已存在%s": "[Index training][Skipped] trained index already exists: %s",
"当前设备:%s | 推理精度:%s": "Current device: %s | Inference precision: %s" "当前设备:%s | 推理精度:%s": "Current device: %s | Inference precision: %s",
"正在预热CUDA Graph": "Warming up CUDA Graph",
"CUDA Graph预热完成": "CUDA Graph warm-up complete"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Intensidad de extracción vocal", "人声提取激进程度": "Intensidad de extracción vocal",
"%s → 成功": "%s → Éxito", "%s → 成功": "%s → Éxito",
"%s运行中请先停止该任务": "%s está en ejecución; deténgalo antes de iniciar otra tarea", "%s运行中请先停止该任务": "%s está en ejecución; deténgalo antes de iniciar otra tarea",
"%s进程已终止": "El proceso %s ha finalizado", "%s进程已终止": "El proceso %s ha finalizado",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Entrenamiento de índice] El enlace externo ya existe: %s", "[索引训练] 外部索引链接已存在:%s": "[Entrenamiento de índice] El enlace externo ya existe: %s",
"[索引训练][跳过] added索引已存在%s": "[Entrenamiento de índice][Omitido] El índice added ya existe: %s", "[索引训练][跳过] added索引已存在%s": "[Entrenamiento de índice][Omitido] El índice added ya existe: %s",
"[索引训练][跳过] trained索引已存在%s": "[Entrenamiento de índice][Omitido] El índice trained ya existe: %s", "[索引训练][跳过] trained索引已存在%s": "[Entrenamiento de índice][Omitido] El índice trained ya existe: %s",
"当前设备:%s | 推理精度:%s": "Dispositivo actual: %s | Precisión de inferencia: %s" "当前设备:%s | 推理精度:%s": "Dispositivo actual: %s | Precisión de inferencia: %s",
"正在预热CUDA Graph": "Preparando CUDA Graph",
"CUDA Graph预热完成": "Preparación de CUDA Graph completada"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Intensité de lextraction vocale", "人声提取激进程度": "Intensité de lextraction vocale",
"%s → 成功": "%s → Réussite", "%s → 成功": "%s → Réussite",
"%s运行中请先停止该任务": "%s est en cours ; arrêtez cette tâche avant d'en démarrer une autre", "%s运行中请先停止该任务": "%s est en cours ; arrêtez cette tâche avant d'en démarrer une autre",
"%s进程已终止": "Le processus %s a été arrêté", "%s进程已终止": "Le processus %s a été arrêté",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Entraînement de lindex] Le lien externe existe déjà : %s", "[索引训练] 外部索引链接已存在:%s": "[Entraînement de lindex] Le lien externe existe déjà : %s",
"[索引训练][跳过] added索引已存在%s": "[Entraînement de lindex][Ignoré] Lindex added existe déjà : %s", "[索引训练][跳过] added索引已存在%s": "[Entraînement de lindex][Ignoré] Lindex added existe déjà : %s",
"[索引训练][跳过] trained索引已存在%s": "[Entraînement de lindex][Ignoré] Lindex trained existe déjà : %s", "[索引训练][跳过] trained索引已存在%s": "[Entraînement de lindex][Ignoré] Lindex trained existe déjà : %s",
"当前设备:%s | 推理精度:%s": "Périphérique actuel : %s | Précision dinférence : %s" "当前设备:%s | 推理精度:%s": "Périphérique actuel : %s | Précision dinférence : %s",
"正在预热CUDA Graph": "Préchauffage de CUDA Graph",
"CUDA Graph预热完成": "Préchauffage de CUDA Graph terminé"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Intensità dell'estrazione vocale", "人声提取激进程度": "Intensità dell'estrazione vocale",
"%s → 成功": "%s → Successo", "%s → 成功": "%s → Successo",
"%s运行中请先停止该任务": "%s è in esecuzione; interromperlo prima di avviare un'altra attività", "%s运行中请先停止该任务": "%s è in esecuzione; interromperlo prima di avviare un'altra attività",
"%s进程已终止": "Il processo %s è stato terminato", "%s进程已终止": "Il processo %s è stato terminato",
@@ -54,7 +54,7 @@
"rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Configurazione GPU RMVPE separata da trattini; ad esempio 0-0-1 avvia due processi sulla GPU 0 e uno sulla GPU 1", "rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Configurazione GPU RMVPE separata da trattini; ad esempio 0-0-1 avvia due processi sulla GPU 0 e uno sulla GPU 1",
"step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Passaggio 1: compilare la configurazione sperimentale. ", "step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Passaggio 1: compilare la configurazione sperimentale. ",
"step1:正在处理数据": "Passaggio 1: elaborazione dei dati", "step1:正在处理数据": "Passaggio 1: elaborazione dei dati",
"step2:正在提取音高&正在提取特征": "step2:Estrazione dell'intonazione e delle caratteristiche", "step2:正在提取音高&正在提取特征": "step2:Estrazione dell'intonazione e delle caratteristiche",
"step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Passaggio 2a: attraversa automaticamente tutti i file nella cartella di addestramento che possono essere decodificati in audio ed esegui la normalizzazione delle sezioni. ", "step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Passaggio 2a: attraversa automaticamente tutti i file nella cartella di addestramento che possono essere decodificati in audio ed esegui la normalizzazione delle sezioni. ",
"step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Passaggio 2b: utilizzare la CPU per estrarre il tono (se il modello ha il tono), utilizzare la GPU per estrarre le caratteristiche (selezionare l'indice GPU):", "step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Passaggio 2b: utilizzare la CPU per estrarre il tono (se il modello ha il tono), utilizzare la GPU per estrarre le caratteristiche (selezionare l'indice GPU):",
"step3: 填写训练设置, 开始训练模型和索引": "Passaggio 3: compilare le impostazioni di addestramento e avviare l'addestramento del modello e dell'indice", "step3: 填写训练设置, 开始训练模型和索引": "Passaggio 3: compilare le impostazioni di addestramento e avviare l'addestramento del modello e dell'indice",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Addestramento indice] Il collegamento esterno esiste già: %s", "[索引训练] 外部索引链接已存在:%s": "[Addestramento indice] Il collegamento esterno esiste già: %s",
"[索引训练][跳过] added索引已存在%s": "[Addestramento indice][Saltato] Lindice added esiste già: %s", "[索引训练][跳过] added索引已存在%s": "[Addestramento indice][Saltato] Lindice added esiste già: %s",
"[索引训练][跳过] trained索引已存在%s": "[Addestramento indice][Saltato] Lindice trained esiste già: %s", "[索引训练][跳过] trained索引已存在%s": "[Addestramento indice][Saltato] Lindice trained esiste già: %s",
"当前设备:%s | 推理精度:%s": "Dispositivo corrente: %s | Precisione di inferenza: %s" "当前设备:%s | 推理精度:%s": "Dispositivo corrente: %s | Precisione di inferenza: %s",
"正在预热CUDA Graph": "Riscaldamento di CUDA Graph",
"CUDA Graph预热完成": "Riscaldamento di CUDA Graph completato"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "ボーカル抽出の強度", "人声提取激进程度": "ボーカル抽出の強度",
"%s → 成功": "%s → 成功", "%s → 成功": "%s → 成功",
"%s运行中请先停止该任务": "%sを実行中です。先に停止してください", "%s运行中请先停止该任务": "%sを実行中です。先に停止してください",
"%s进程已终止": "%sプロセスを終了しました", "%s进程已终止": "%sプロセスを終了しました",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[インデックス学習] 外部インデックスリンクは既に存在します:%s", "[索引训练] 外部索引链接已存在:%s": "[インデックス学習] 外部インデックスリンクは既に存在します:%s",
"[索引训练][跳过] added索引已存在%s": "[インデックス学習][スキップ] addedインデックスは既に存在します%s", "[索引训练][跳过] added索引已存在%s": "[インデックス学習][スキップ] addedインデックスは既に存在します%s",
"[索引训练][跳过] trained索引已存在%s": "[インデックス学習][スキップ] trainedインデックスは既に存在します%s", "[索引训练][跳过] trained索引已存在%s": "[インデックス学習][スキップ] trainedインデックスは既に存在します%s",
"当前设备:%s | 推理精度:%s": "現在のデバイス:%s | 推論精度:%s" "当前设备:%s | 推理精度:%s": "現在のデバイス:%s | 推論精度:%s",
"正在预热CUDA Graph": "CUDA Graphをウォームアップしています",
"CUDA Graph预热完成": "CUDA Graphのウォームアップが完了しました"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "보컬 추출 강도", "人声提取激进程度": "보컬 추출 강도",
"%s → 成功": "%s → 성공", "%s → 成功": "%s → 성공",
"%s运行中请先停止该任务": "%s이(가) 실행 중입니다. 먼저 중지하세요", "%s运行中请先停止该任务": "%s이(가) 실행 중입니다. 먼저 중지하세요",
"%s进程已终止": "%s 프로세스가 종료되었습니다", "%s进程已终止": "%s 프로세스가 종료되었습니다",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[인덱스 학습] 외부 인덱스 링크가 이미 있습니다: %s", "[索引训练] 外部索引链接已存在:%s": "[인덱스 학습] 외부 인덱스 링크가 이미 있습니다: %s",
"[索引训练][跳过] added索引已存在%s": "[인덱스 학습][건너뜀] added 인덱스가 이미 있습니다: %s", "[索引训练][跳过] added索引已存在%s": "[인덱스 학습][건너뜀] added 인덱스가 이미 있습니다: %s",
"[索引训练][跳过] trained索引已存在%s": "[인덱스 학습][건너뜀] trained 인덱스가 이미 있습니다: %s", "[索引训练][跳过] trained索引已存在%s": "[인덱스 학습][건너뜀] trained 인덱스가 이미 있습니다: %s",
"当前设备:%s | 推理精度:%s": "현재 장치: %s | 추론 정밀도: %s" "当前设备:%s | 推理精度:%s": "현재 장치: %s | 추론 정밀도: %s",
"正在预热CUDA Graph": "CUDA Graph 워밍업 중",
"CUDA Graph预热完成": "CUDA Graph 워밍업 완료"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Intensidade da extração vocal", "人声提取激进程度": "Intensidade da extração vocal",
"%s → 成功": "%s → Sucesso", "%s → 成功": "%s → Sucesso",
"%s运行中请先停止该任务": "%s está em execução; interrompa antes de iniciar outra tarefa", "%s运行中请先停止该任务": "%s está em execução; interrompa antes de iniciar outra tarefa",
"%s进程已终止": "O processo %s foi encerrado", "%s进程已终止": "O processo %s foi encerrado",
@@ -54,7 +54,7 @@
"rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Configuração do número do cartão rmvpe: Use - para separar os números dos cartões de entrada de diferentes processos. Por exemplo, 0-0-1 é usado para executar 2 processos no cartão 0 e 1 processo no cartão 1.", "rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Configuração do número do cartão rmvpe: Use - para separar os números dos cartões de entrada de diferentes processos. Por exemplo, 0-0-1 é usado para executar 2 processos no cartão 0 e 1 processo no cartão 1.",
"step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Etapa 1: Preencha a configuração experimental. Os dados experimentais são armazenados na pasta 'logs', com cada experimento tendo uma pasta separada. Digite manualmente o caminho do nome do experimento, que contém a configuração experimental, os logs e os arquivos de modelo treinados.", "step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Etapa 1: Preencha a configuração experimental. Os dados experimentais são armazenados na pasta 'logs', com cada experimento tendo uma pasta separada. Digite manualmente o caminho do nome do experimento, que contém a configuração experimental, os logs e os arquivos de modelo treinados.",
"step1:正在处理数据": "Etapa 1: Processamento de dados", "step1:正在处理数据": "Etapa 1: Processamento de dados",
"step2:正在提取音高&正在提取特征": "step2:Extração de tom e características", "step2:正在提取音高&正在提取特征": "step2:Extração de tom e características",
"step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Etapa 2a: Percorra automaticamente todos os arquivos na pasta de treinamento que podem ser decodificados em áudio e execute a normalização da fatia. Gera 2 pastas wav no diretório do experimento. Atualmente, apenas o treinamento de um único cantor/palestrante é suportado.", "step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Etapa 2a: Percorra automaticamente todos os arquivos na pasta de treinamento que podem ser decodificados em áudio e execute a normalização da fatia. Gera 2 pastas wav no diretório do experimento. Atualmente, apenas o treinamento de um único cantor/palestrante é suportado.",
"step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Etapa 2b: Use a CPU para extrair o tom (se o modelo tiver tom), use a GPU para extrair recursos (selecione o índice da GPU):", "step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Etapa 2b: Use a CPU para extrair o tom (se o modelo tiver tom), use a GPU para extrair recursos (selecione o índice da GPU):",
"step3: 填写训练设置, 开始训练模型和索引": "Etapa 3: Preencha as configurações de treinamento e comece a treinar o modelo e o Index", "step3: 填写训练设置, 开始训练模型和索引": "Etapa 3: Preencha as configurações de treinamento e comece a treinar o modelo e o Index",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Treinamento de índice] O link externo já existe: %s", "[索引训练] 外部索引链接已存在:%s": "[Treinamento de índice] O link externo já existe: %s",
"[索引训练][跳过] added索引已存在%s": "[Treinamento de índice][Ignorado] O índice added já existe: %s", "[索引训练][跳过] added索引已存在%s": "[Treinamento de índice][Ignorado] O índice added já existe: %s",
"[索引训练][跳过] trained索引已存在%s": "[Treinamento de índice][Ignorado] O índice trained já existe: %s", "[索引训练][跳过] trained索引已存在%s": "[Treinamento de índice][Ignorado] O índice trained já existe: %s",
"当前设备:%s | 推理精度:%s": "Dispositivo atual: %s | Precisão de inferência: %s" "当前设备:%s | 推理精度:%s": "Dispositivo atual: %s | Precisão de inferência: %s",
"正在预热CUDA Graph": "Aquecendo o CUDA Graph",
"CUDA Graph预热完成": "Aquecimento do CUDA Graph concluído"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Интенсивность выделения вокала", "人声提取激进程度": "Интенсивность выделения вокала",
"%s → 成功": "%s → Успешно", "%s → 成功": "%s → Успешно",
"%s运行中请先停止该任务": "%s уже выполняется; сначала остановите эту задачу", "%s运行中请先停止该任务": "%s уже выполняется; сначала остановите эту задачу",
"%s进程已终止": "Процесс %s завершён", "%s进程已终止": "Процесс %s завершён",
@@ -54,7 +54,7 @@
"rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Введите номера графических процессоров, разделенные символом «-», например, 0-0-1, чтобы запустить два процесса на GPU 0 и один процесс на GPU 1:", "rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "Введите номера графических процессоров, разделенные символом «-», например, 0-0-1, чтобы запустить два процесса на GPU 0 и один процесс на GPU 1:",
"step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Шаг 1. Конфигурирование модели. Данные обучения модели сохраняются в папку 'logs', и для каждой модели создаётся отдельная папка. Введите вручную путь к настройкам для модели, в которой находятся логи и тренировочные файлы.", "step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Шаг 1. Конфигурирование модели. Данные обучения модели сохраняются в папку 'logs', и для каждой модели создаётся отдельная папка. Введите вручную путь к настройкам для модели, в которой находятся логи и тренировочные файлы.",
"step1:正在处理数据": "Шаг 1. Переработка данных", "step1:正在处理数据": "Шаг 1. Переработка данных",
"step2:正在提取音高&正在提取特征": "step2:Извлечение высоты тона и признаков", "step2:正在提取音高&正在提取特征": "step2:Извлечение высоты тона и признаков",
"step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Шаг 2А. Автоматическая обработка исходных аудиозаписей для обучения и выполнение нормализации среза. Создаст 2 папки wav в папке модели. В данный момент поддерживается обучение только на одноголосных записях.", "step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Шаг 2А. Автоматическая обработка исходных аудиозаписей для обучения и выполнение нормализации среза. Создаст 2 папки wav в папке модели. В данный момент поддерживается обучение только на одноголосных записях.",
"step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Шаг 2Б. Оценка и извлечение тональности в аудиофайлах с помощью процессора (если включена поддержка изменения высоты звука), извлечение черт с помощью GPU (выберите номер GPU):", "step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Шаг 2Б. Оценка и извлечение тональности в аудиофайлах с помощью процессора (если включена поддержка изменения высоты звука), извлечение черт с помощью GPU (выберите номер GPU):",
"step3: 填写训练设置, 开始训练模型和索引": "Шаг 3. Заполнение дополнительных настроек обучения и запуск обучения модели и индекса", "step3: 填写训练设置, 开始训练模型和索引": "Шаг 3. Заполнение дополнительных настроек обучения и запуск обучения модели и индекса",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Обучение индекса] Внешняя ссылка уже существует: %s", "[索引训练] 外部索引链接已存在:%s": "[Обучение индекса] Внешняя ссылка уже существует: %s",
"[索引训练][跳过] added索引已存在%s": "[Обучение индекса][Пропущено] Индекс added уже существует: %s", "[索引训练][跳过] added索引已存在%s": "[Обучение индекса][Пропущено] Индекс added уже существует: %s",
"[索引训练][跳过] trained索引已存在%s": "[Обучение индекса][Пропущено] Индекс trained уже существует: %s", "[索引训练][跳过] trained索引已存在%s": "[Обучение индекса][Пропущено] Индекс trained уже существует: %s",
"当前设备:%s | 推理精度:%s": "Текущее устройство: %s | Точность вывода: %s" "当前设备:%s | 推理精度:%s": "Текущее устройство: %s | Точность вывода: %s",
"正在预热CUDA Graph": "Прогрев CUDA Graph",
"CUDA Graph预热完成": "Прогрев CUDA Graph завершён"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "Vokal ayırma yoğunluğu", "人声提取激进程度": "Vokal ayırma yoğunluğu",
"%s → 成功": "%s → Başarılı", "%s → 成功": "%s → Başarılı",
"%s运行中请先停止该任务": "%s çalışıyor; başka bir görev başlatmadan önce durdurun", "%s运行中请先停止该任务": "%s çalışıyor; başka bir görev başlatmadan önce durdurun",
"%s进程已终止": "%s işlemi sonlandırıldı", "%s进程已终止": "%s işlemi sonlandırıldı",
@@ -54,7 +54,7 @@
"rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "RMVPE GPU yapılandırmasını tirelerle ayırın; örneğin 0-0-1, GPU 0'da iki ve GPU 1'de bir işlem çalıştırır", "rmvpe卡号配置以-分隔输入使用的不同进程卡号,例如0-0-1使用在卡0上跑2个进程并在卡1上跑1个进程": "RMVPE GPU yapılandırmasını tirelerle ayırın; örneğin 0-0-1, GPU 0'da iki ve GPU 1'de bir işlem çalıştırır",
"step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Adım 1: Deneysel yapılandırmayı doldurun. Deneysel veriler 'logs' klasöründe saklanır ve her bir deney için ayrı bir klasör vardır. Deneysel adı yolu manuel olarak girin; bu yol, deneysel yapılandırmayı, günlükleri ve eğitilmiş model dosyalarını içerir.", "step1: 填写实验配置. 实验数据放在logs下, 每个实验一个文件夹, 需手工输入实验名路径, 内含实验配置, 日志, 训练得到的模型文件. ": "Adım 1: Deneysel yapılandırmayı doldurun. Deneysel veriler 'logs' klasöründe saklanır ve her bir deney için ayrı bir klasör vardır. Deneysel adı yolu manuel olarak girin; bu yol, deneysel yapılandırmayı, günlükleri ve eğitilmiş model dosyalarını içerir.",
"step1:正在处理数据": "Adım 1: Veri işleme", "step1:正在处理数据": "Adım 1: Veri işleme",
"step2:正在提取音高&正在提取特征": "step2:Perde ve özellik çıkarımı", "step2:正在提取音高&正在提取特征": "step2:Perde ve özellik çıkarımı",
"step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Adım 2a: Eğitim klasöründe ses dosyalarını otomatik olarak gezinerek dilimleme normalizasyonu yapın. Deney dizini içinde 2 wav klasörü oluşturur. Şu anda sadece tek kişilik eğitim desteklenmektedir.", "step2a: 自动遍历训练文件夹下所有可解码成音频的文件并进行切片归一化, 在实验目录下生成2个wav文件夹; 暂时只支持单人训练. ": "Adım 2a: Eğitim klasöründe ses dosyalarını otomatik olarak gezinerek dilimleme normalizasyonu yapın. Deney dizini içinde 2 wav klasörü oluşturur. Şu anda sadece tek kişilik eğitim desteklenmektedir.",
"step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Adım 2b: Ses yüksekliği (Pitch) çıkartmak için CPU kullanın (eğer model ses yüksekliği içeriyorsa), özellikleri çıkartmak için GPU kullanın (GPU indeksini seçin):", "step2b: 使用CPU提取音高(如果模型带音高), 使用GPU提取特征(选择卡号)": "Adım 2b: Ses yüksekliği (Pitch) çıkartmak için CPU kullanın (eğer model ses yüksekliği içeriyorsa), özellikleri çıkartmak için GPU kullanın (GPU indeksini seçin):",
"step3: 填写训练设置, 开始训练模型和索引": "Adım 3: Eğitim ayarlarını doldurun ve modeli ve dizini eğitmeye başlayın", "step3: 填写训练设置, 开始训练模型和索引": "Adım 3: Eğitim ayarlarını doldurun ve modeli ve dizini eğitmeye başlayın",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[Dizin eğitimi] Harici dizin bağlantısı zaten var: %s", "[索引训练] 外部索引链接已存在:%s": "[Dizin eğitimi] Harici dizin bağlantısı zaten var: %s",
"[索引训练][跳过] added索引已存在%s": "[Dizin eğitimi][Atlandı] added dizini zaten var: %s", "[索引训练][跳过] added索引已存在%s": "[Dizin eğitimi][Atlandı] added dizini zaten var: %s",
"[索引训练][跳过] trained索引已存在%s": "[Dizin eğitimi][Atlandı] trained dizini zaten var: %s", "[索引训练][跳过] trained索引已存在%s": "[Dizin eğitimi][Atlandı] trained dizini zaten var: %s",
"当前设备:%s | 推理精度:%s": "Geçerli cihaz: %s | Çıkarım hassasiyeti: %s" "当前设备:%s | 推理精度:%s": "Geçerli cihaz: %s | Çıkarım hassasiyeti: %s",
"正在预热CUDA Graph": "CUDA Graph ısınıyor",
"CUDA Graph预热完成": "CUDA Graph ısınması tamamlandı"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "人声提取激进程度", "人声提取激进程度": "人声提取激进程度",
"%s → 成功": "%s → 成功", "%s → 成功": "%s → 成功",
"%s运行中请先停止该任务": "%s运行中请先停止该任务", "%s运行中请先停止该任务": "%s运行中请先停止该任务",
"%s进程已终止": "%s进程已终止", "%s进程已终止": "%s进程已终止",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[索引训练] 外部索引链接已存在:%s", "[索引训练] 外部索引链接已存在:%s": "[索引训练] 外部索引链接已存在:%s",
"[索引训练][跳过] added索引已存在%s": "[索引训练][跳过] added索引已存在%s", "[索引训练][跳过] added索引已存在%s": "[索引训练][跳过] added索引已存在%s",
"[索引训练][跳过] trained索引已存在%s": "[索引训练][跳过] trained索引已存在%s", "[索引训练][跳过] trained索引已存在%s": "[索引训练][跳过] trained索引已存在%s",
"当前设备:%s | 推理精度:%s": "当前设备:%s | 推理精度:%s" "当前设备:%s | 推理精度:%s": "当前设备:%s | 推理精度:%s",
"正在预热CUDA Graph": "正在预热CUDA Graph",
"CUDA Graph预热完成": "CUDA Graph预热完成"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "人聲提取激進程度", "人声提取激进程度": "人聲提取激進程度",
"%s → 成功": "%s → 成功", "%s → 成功": "%s → 成功",
"%s运行中请先停止该任务": "%s运行中请先停止该任务", "%s运行中请先停止该任务": "%s运行中请先停止该任务",
"%s进程已终止": "%s進程已终止", "%s进程已终止": "%s進程已终止",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[索引訓練] 外部索引連結已存在:%s", "[索引训练] 外部索引链接已存在:%s": "[索引訓練] 外部索引連結已存在:%s",
"[索引训练][跳过] added索引已存在%s": "[索引訓練][跳過] added索引已存在%s", "[索引训练][跳过] added索引已存在%s": "[索引訓練][跳過] added索引已存在%s",
"[索引训练][跳过] trained索引已存在%s": "[索引訓練][跳過] trained索引已存在%s", "[索引训练][跳过] trained索引已存在%s": "[索引訓練][跳過] trained索引已存在%s",
"当前设备:%s | 推理精度:%s": "目前裝置:%s | 推理精度:%s" "当前设备:%s | 推理精度:%s": "目前裝置:%s | 推理精度:%s",
"正在预热CUDA Graph": "正在預熱 CUDA Graph",
"CUDA Graph预热完成": "CUDA Graph 預熱完成"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "人声提取激进程度", "人声提取激进程度": "人声提取激进程度",
"%s → 成功": "%s → 成功", "%s → 成功": "%s → 成功",
"%s运行中请先停止该任务": "%s运行中请先停止该任务", "%s运行中请先停止该任务": "%s运行中请先停止该任务",
"%s进程已终止": "%s进程已终止", "%s进程已终止": "%s进程已终止",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[索引训练] 外部索引链接已存在:%s", "[索引训练] 外部索引链接已存在:%s": "[索引训练] 外部索引链接已存在:%s",
"[索引训练][跳过] added索引已存在%s": "[索引训练][跳过] added索引已存在%s", "[索引训练][跳过] added索引已存在%s": "[索引训练][跳过] added索引已存在%s",
"[索引训练][跳过] trained索引已存在%s": "[索引训练][跳过] trained索引已存在%s", "[索引训练][跳过] trained索引已存在%s": "[索引训练][跳过] trained索引已存在%s",
"当前设备:%s | 推理精度:%s": "当前设备:%s | 推理精度:%s" "当前设备:%s | 推理精度:%s": "当前设备:%s | 推理精度:%s",
"正在预热CUDA Graph": "正在预热CUDA Graph",
"CUDA Graph预热完成": "CUDA Graph预热完成"
} }

View File

@@ -1,5 +1,5 @@
{ {
"人声提取激进程度": "人聲提取激進程度", "人声提取激进程度": "人聲提取激進程度",
"%s → 成功": "%s → 成功", "%s → 成功": "%s → 成功",
"%s运行中请先停止该任务": "%s运行中请先停止该任务", "%s运行中请先停止该任务": "%s运行中请先停止该任务",
"%s进程已终止": "%s進程已终止", "%s进程已终止": "%s進程已终止",
@@ -257,5 +257,7 @@
"[索引训练] 外部索引链接已存在:%s": "[索引訓練] 外部索引連結已存在:%s", "[索引训练] 外部索引链接已存在:%s": "[索引訓練] 外部索引連結已存在:%s",
"[索引训练][跳过] added索引已存在%s": "[索引訓練][跳過] added索引已存在%s", "[索引训练][跳过] added索引已存在%s": "[索引訓練][跳過] added索引已存在%s",
"[索引训练][跳过] trained索引已存在%s": "[索引訓練][跳過] trained索引已存在%s", "[索引训练][跳过] trained索引已存在%s": "[索引訓練][跳過] trained索引已存在%s",
"当前设备:%s | 推理精度:%s": "目前裝置:%s | 推論精度:%s" "当前设备:%s | 推理精度:%s": "目前裝置:%s | 推論精度:%s",
"正在预热CUDA Graph": "正在預熱 CUDA Graph",
"CUDA Graph预热完成": "CUDA Graph 預熱完成"
} }

View File

@@ -1,5 +1,7 @@
import torch import torch
from tools.cuda_graph import cuda_graph_enabled, run_cuda_graph
def _is_directml_device(device): def _is_directml_device(device):
"""Return whether *device* is the PrivateUse1 device registered by DirectML.""" """Return whether *device* is the PrivateUse1 device registered by DirectML."""
@@ -35,6 +37,46 @@ class FCPEInfer:
self.infer_model.model.to(device).eval() self.infer_model.model.to(device).eval()
else: else:
self.infer_model = spawn_bundled_infer_model(device) self.infer_model = spawn_bundled_infer_model(device)
if getattr(device, "type", None) == "cuda" or str(device).startswith(
"cuda"
):
# torchfcpe creates this tensor on CPU and copies it to CUDA in
# every local-argmax decode. Host-to-device copies are forbidden
# during CUDA Graph capture, so keep the immutable offsets on the
# target device and use the graph-safe equivalent decoder below.
self.local_offsets = torch.arange(
9, device=device, dtype=torch.long
).view(1, 1, 9)
def _graphable_model_infer(self, mel, decoder_mode, threshold):
"""Run FCPE network and an exactly equivalent capture-safe decoder."""
model = self.infer_model.model
latent = model(mel)
batch, frames, _ = latent.shape
cents = model.cent_table[None, None, :].expand(batch, frames, -1)
if decoder_mode == "argmax":
confidence = torch.max(latent, dim=-1, keepdim=True).values
decoded = torch.sum(cents * latent, dim=-1, keepdim=True) / torch.sum(
latent, dim=-1, keepdim=True
)
elif decoder_mode == "local_argmax":
confidence, max_index = torch.max(latent, dim=-1, keepdim=True)
local_index = self.local_offsets + (max_index - 4)
local_index = local_index.clamp(0, model.out_dims - 1)
local_cents = torch.gather(cents, -1, local_index)
local_latent = torch.gather(latent, -1, local_index)
decoded = torch.sum(
local_cents * local_latent, dim=-1, keepdim=True
) / torch.sum(local_latent, dim=-1, keepdim=True)
else:
raise ValueError("Unknown FCPE decoder mode: %s" % decoder_mode)
# Match torchfcpe's masking and cent-to-Hz formulas operation-for-operation.
confidence_mask = torch.ones_like(confidence)
confidence_mask.masked_fill_(confidence <= threshold, float("-inf"))
decoded = decoded * confidence_mask
return 10.0 * torch.pow(2.0, decoded / 1200.0)
def _decode_on_cpu(self, latent, decoder_mode, threshold): def _decode_on_cpu(self, latent, decoder_mode, threshold):
"""Decode DML network logits on CPU with torchfcpe's exact formulas. """Decode DML network logits on CPU with torchfcpe's exact formulas.
@@ -79,11 +121,33 @@ class FCPEInfer:
threshold=0.006, threshold=0.006,
): ):
if not self.is_directml: if not self.is_directml:
return self.infer_model.infer( wav = wav.to(self.device)
if cuda_graph_enabled(wav.device):
# Wav2MelModule contains tensor-dependent Python conditionals and
# cannot be captured. Running it eagerly also creates/caches its
# STFT window before the graph. The much larger FCPE neural net
# and decoder form the stable-shape CUDA Graph boundary.
mel = self.infer_model.wav2mel(wav, sr)
return run_cuda_graph(
self.infer_model.model,
"fcpe-core-%s-%s" % (decoder_mode, threshold),
lambda input_mel: self._graphable_model_infer(
input_mel,
decoder_mode,
threshold,
),
mel,
)
return run_cuda_graph(
self.infer_model,
"fcpe-%s-%s-%s" % (sr, decoder_mode, threshold),
lambda input_wav: self.infer_model.infer(
input_wav,
sr=sr,
decoder_mode=decoder_mode,
threshold=threshold,
),
wav, wav,
sr=sr,
decoder_mode=decoder_mode,
threshold=threshold,
) )
wav_cpu = wav.detach().to(device="cpu", dtype=torch.float32) wav_cpu = wav.detach().to(device="cpu", dtype=torch.float32)

View File

@@ -4,7 +4,9 @@ from pathlib import Path
import torch import torch
from torch import nn from torch import nn
from transformers import AutoFeatureExtractor, HubertModel from transformers import AutoFeatureExtractor, HubertModel
from tools.cuda_graph import run_cuda_graph
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -63,7 +65,7 @@ def hubert_audio_requires_normalization():
return bool(feature_extractor.do_normalize) return bool(feature_extractor.do_normalize)
def extract_hubert_features(model, source, version, padding_mask=None): def extract_hubert_features(model, source, version, padding_mask=None):
"""Return the RVC v1 (256-D) or v2 (768-D) HuBERT representation. """Return the RVC v1 (256-D) or v2 (768-D) HuBERT representation.
Transformers hidden_states[N] is numerically equivalent to the source checkpoint's Transformers hidden_states[N] is numerically equivalent to the source checkpoint's
@@ -77,20 +79,49 @@ def extract_hubert_features(model, source, version, padding_mask=None):
if padding_mask is not None and bool(torch.any(padding_mask).item()): if padding_mask is not None and bool(torch.any(padding_mask).item()):
attention_mask = (~padding_mask.bool()).long() attention_mask = (~padding_mask.bool()).long()
if version == "v1": if version == "v1":
outputs = model( if attention_mask is None:
input_values=source, def forward(input_values):
attention_mask=attention_mask, outputs = model(
output_hidden_states=True, input_values=input_values,
return_dict=True, attention_mask=None,
) output_hidden_states=True,
features = outputs.hidden_states[9] return_dict=True,
return model.final_proj(features) )
return model.final_proj(outputs.hidden_states[9])
outputs = model(
input_values=source, return run_cuda_graph(model, "hubert-v1-no-mask", forward, source)
attention_mask=attention_mask,
output_hidden_states=False, def forward(input_values, mask):
return_dict=True, outputs = model(
) input_values=input_values,
return outputs.last_hidden_state attention_mask=mask,
output_hidden_states=True,
return_dict=True,
)
return model.final_proj(outputs.hidden_states[9])
return run_cuda_graph(
model, "hubert-v1-mask", forward, source, attention_mask
)
if attention_mask is None:
def forward(input_values):
return model(
input_values=input_values,
attention_mask=None,
output_hidden_states=False,
return_dict=True,
).last_hidden_state
return run_cuda_graph(model, "hubert-v2-no-mask", forward, source)
def forward(input_values, mask):
return model(
input_values=input_values,
attention_mask=mask,
output_hidden_states=False,
return_dict=True,
).last_hidden_state
return run_cuda_graph(model, "hubert-v2-mask", forward, source, attention_mask)

View File

@@ -67,8 +67,11 @@ class TextEncoder(nn.Module):
) )
x = self.encoder(x * x_mask, x_mask) x = self.encoder(x * x_mask, x_mask)
if skip_head is not None: if skip_head is not None:
assert isinstance(skip_head, torch.Tensor) head = (
head = int(skip_head.item()) int(skip_head.item())
if isinstance(skip_head, torch.Tensor)
else int(skip_head)
)
x = x[:, :, head:] x = x[:, :, head:]
x_mask = x_mask[:, :, head:] x_mask = x_mask[:, :, head:]
stats = self.proj(x) * x_mask stats = self.proj(x) * x_mask
@@ -231,8 +234,7 @@ class Generator(torch.nn.Module):
n_res = None, n_res = None,
): ):
if n_res is not None: if n_res is not None:
assert isinstance(n_res, torch.Tensor) n = int(n_res.item()) if isinstance(n_res, torch.Tensor) else int(n_res)
n = int(n_res.item())
if n != x.shape[-1]: if n != x.shape[-1]:
x = F.interpolate(x, size=n, mode="linear") x = F.interpolate(x, size=n, mode="linear")
x = self.conv_pre(x) x = self.conv_pre(x)
@@ -501,8 +503,7 @@ class GeneratorNSF(torch.nn.Module):
har_source, noi_source, uv = self.m_source(f0, self.upp) har_source, noi_source, uv = self.m_source(f0, self.upp)
har_source = har_source.transpose(1, 2) har_source = har_source.transpose(1, 2)
if n_res is not None: if n_res is not None:
assert isinstance(n_res, torch.Tensor) n = int(n_res.item()) if isinstance(n_res, torch.Tensor) else int(n_res)
n = int(n_res.item())
if n * self.upp != har_source.shape[-1]: if n * self.upp != har_source.shape[-1]:
har_source = F.interpolate(har_source, size=n * self.upp, mode="linear") har_source = F.interpolate(har_source, size=n * self.upp, mode="linear")
if n != x.shape[-1]: if n != x.shape[-1]:
@@ -674,12 +675,14 @@ class SynthesizerTrnMs256NSFsid(nn.Module):
): ):
g = self.emb_g(sid).unsqueeze(-1) g = self.emb_g(sid).unsqueeze(-1)
if skip_head is not None and return_length is not None: if skip_head is not None and return_length is not None:
assert isinstance(skip_head, torch.Tensor) head = int(skip_head.item()) if isinstance(skip_head, torch.Tensor) else int(skip_head)
assert isinstance(return_length, torch.Tensor) length = (
head = int(skip_head.item()) int(return_length.item())
length = int(return_length.item()) if isinstance(return_length, torch.Tensor)
flow_head = torch.clamp(skip_head - 24, min=0) else int(return_length)
dec_head = head - int(flow_head.item()) )
flow_head = max(head - 24, 0)
dec_head = head - flow_head
m_p, logs_p, x_mask = self.enc_p(phone, pitch, phone_lengths, flow_head) m_p, logs_p, x_mask = self.enc_p(phone, pitch, phone_lengths, flow_head)
z_p = (m_p + torch.exp(logs_p) * torch.randn_like(m_p) * 0.66666) * x_mask z_p = (m_p + torch.exp(logs_p) * torch.randn_like(m_p) * 0.66666) * x_mask
z = self.flow(z_p, x_mask, g=g, reverse=True) z = self.flow(z_p, x_mask, g=g, reverse=True)
@@ -862,12 +865,14 @@ class SynthesizerTrnMs256NSFsid_nono(nn.Module):
): ):
g = self.emb_g(sid).unsqueeze(-1) g = self.emb_g(sid).unsqueeze(-1)
if skip_head is not None and return_length is not None: if skip_head is not None and return_length is not None:
assert isinstance(skip_head, torch.Tensor) head = int(skip_head.item()) if isinstance(skip_head, torch.Tensor) else int(skip_head)
assert isinstance(return_length, torch.Tensor) length = (
head = int(skip_head.item()) int(return_length.item())
length = int(return_length.item()) if isinstance(return_length, torch.Tensor)
flow_head = torch.clamp(skip_head - 24, min=0) else int(return_length)
dec_head = head - int(flow_head.item()) )
flow_head = max(head - 24, 0)
dec_head = head - flow_head
m_p, logs_p, x_mask = self.enc_p(phone, None, phone_lengths, flow_head) m_p, logs_p, x_mask = self.enc_p(phone, None, phone_lengths, flow_head)
z_p = (m_p + torch.exp(logs_p) * torch.randn_like(m_p) * 0.66666) * x_mask z_p = (m_p + torch.exp(logs_p) * torch.randn_like(m_p) * 0.66666) * x_mask
z = self.flow(z_p, x_mask, g=g, reverse=True) z = self.flow(z_p, x_mask, g=g, reverse=True)

View File

@@ -6,7 +6,9 @@ import torch
import torch.nn as nn import torch.nn as nn
import torch.nn.functional as F import torch.nn.functional as F
from librosa.util import normalize, pad_center, tiny from librosa.util import normalize, pad_center, tiny
from scipy.signal import get_window from scipy.signal import get_window
from tools.cuda_graph import run_cuda_graph
import logging import logging
@@ -536,7 +538,7 @@ class RMVPE:
cents_mapping = 20 * np.arange(360) + 1997.3794084376191 cents_mapping = 20 * np.arange(360) + 1997.3794084376191
self.cents_mapping = np.pad(cents_mapping, (4, 4)) # 368 self.cents_mapping = np.pad(cents_mapping, (4, 4)) # 368
def mel2hidden(self, mel): def mel2hidden(self, mel):
with torch.no_grad(): with torch.no_grad():
n_frames = mel.shape[-1] n_frames = mel.shape[-1]
n_pad = 32 * ((n_frames - 1) // 32 + 1) - n_frames n_pad = 32 * ((n_frames - 1) // 32 + 1) - n_frames
@@ -549,10 +551,30 @@ class RMVPE:
[onnx_outputs_names], [onnx_outputs_names],
input_feed={onnx_input_name: mel.cpu().numpy()}, input_feed={onnx_input_name: mel.cpu().numpy()},
)[0] )[0]
else: else:
mel = mel.half() if self.is_half else mel.float() mel = mel.half() if self.is_half else mel.float()
hidden = self.model(mel) hidden = run_cuda_graph(
return hidden[:, :n_frames] self.model,
"rmvpe-network",
lambda input_mel: self.model(input_mel),
mel,
)
return hidden[:, :n_frames]
def extract_mel(self, audio, center=True):
if not torch.is_tensor(audio):
audio = torch.from_numpy(audio)
audio = audio.float().to(self.device)
if audio.dim() == 1:
audio = audio.unsqueeze(0)
if "privateuseone" in str(self.device):
return self.mel_extractor(audio, center=center)
return run_cuda_graph(
self.mel_extractor,
"rmvpe-mel-center-%s" % int(bool(center)),
lambda input_audio: self.mel_extractor(input_audio, center=center),
audio,
)
def decode(self, hidden, thred=0.03): def decode(self, hidden, thred=0.03):
cents_pred = self.to_local_average_cents(hidden, thred=thred) cents_pred = self.to_local_average_cents(hidden, thred=thred)
@@ -564,11 +586,7 @@ class RMVPE:
def infer_from_audio(self, audio, thred=0.03): def infer_from_audio(self, audio, thred=0.03):
# torch.cuda.synchronize() # torch.cuda.synchronize()
# t0 = ttime() # t0 = ttime()
if not torch.is_tensor(audio): mel = self.extract_mel(audio, center=True)
audio = torch.from_numpy(audio)
mel = self.mel_extractor(
audio.float().to(self.device).unsqueeze(0), center=True
)
# print(123123123,mel.device.type) # print(123123123,mel.device.type)
# torch.cuda.synchronize() # torch.cuda.synchronize()
# t1 = ttime() # t1 = ttime()

View File

@@ -10,6 +10,7 @@ from torchaudio.transforms import Resample
from infer.hubert import extract_hubert_features, load_hubert_model from infer.hubert import extract_hubert_features, load_hubert_model
from i18n.i18n import I18nAuto from i18n.i18n import I18nAuto
from tools.cuda_graph import run_cuda_graph
i18n = I18nAuto() i18n = I18nAuto()
@@ -93,6 +94,7 @@ class RVC:
self.cache_pitchf = torch.zeros( self.cache_pitchf = torch.zeros(
1024, device=self.device, dtype=torch.float32 1024, device=self.device, dtype=torch.float32
) )
self.infer_count = 0
self.resample_kernel = {} self.resample_kernel = {}
@@ -179,11 +181,9 @@ class RVC:
if len(f0) < p_len: if len(f0) < p_len:
f0 = np.pad(f0, (0, p_len - len(f0))) f0 = np.pad(f0, (0, p_len - len(f0)))
f0 = f0[:p_len] f0 = f0[:p_len]
try: uv = f0 == 0
uv = f0 == 0 if np.any(~uv):
f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv]) f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv])
except Exception:
traceback.print_exc()
f0 *= pow(2, f0_up_key / 12) f0 *= pow(2, f0_up_key / 12)
return self.get_f0_post(f0) return self.get_f0_post(f0)
@@ -198,11 +198,9 @@ class RVC:
device=self.device, device=self.device,
) )
f0 = self.model_rmvpe.infer_from_audio(x, thred=0.03) f0 = self.model_rmvpe.infer_from_audio(x, thred=0.03)
try: uv = f0 == 0
uv = f0 == 0 if np.any(~uv):
f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv]) f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv])
except Exception:
traceback.print_exc()
f0 *= pow(2, f0_up_key / 12) f0 *= pow(2, f0_up_key / 12)
return self.get_f0_post(f0) return self.get_f0_post(f0)
@@ -218,11 +216,9 @@ class RVC:
decoder_mode="local_argmax", decoder_mode="local_argmax",
threshold=0.006, threshold=0.006,
).squeeze().detach().cpu().numpy() ).squeeze().detach().cpu().numpy()
try: uv = f0 == 0
uv = f0 == 0 if np.any(~uv):
f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv]) f0[uv] = np.interp(np.where(uv)[0], np.where(~uv)[0], f0[~uv])
except Exception:
traceback.print_exc()
f0 *= pow(2, f0_up_key / 12) f0 *= pow(2, f0_up_key / 12)
return self.get_f0_post(f0) return self.get_f0_post(f0)
@@ -234,6 +230,8 @@ class RVC:
return_length, return_length,
f0method, f0method,
) : ) :
report_status = self.infer_count < 3 or self.infer_count % 100 == 0
self.infer_count += 1
t1 = ttime() t1 = ttime()
with torch.no_grad(): with torch.no_grad():
if self.config.is_half: if self.config.is_half:
@@ -271,7 +269,8 @@ class RVC:
i18n("索引无效必须使用added_xxxx.index不能使用trained_xxxx.index") i18n("索引无效必须使用added_xxxx.index不能使用trained_xxxx.index")
) )
else: else:
printt(i18n("索引检索失败或未启用")) if report_status:
printt(i18n("索引检索失败或未启用"))
except Exception: except Exception:
traceback.print_exc() traceback.print_exc()
printt(i18n("索引检索失败")) printt(i18n("索引检索失败"))
@@ -298,26 +297,49 @@ class RVC:
t4 = ttime() t4 = ttime()
feats = F.interpolate(feats.permute(0, 2, 1), scale_factor=2).permute(0, 2, 1) feats = F.interpolate(feats.permute(0, 2, 1), scale_factor=2).permute(0, 2, 1)
feats = feats[:, :p_len, :] feats = feats[:, :p_len, :]
p_len = torch.LongTensor([p_len]).to(self.device) p_len_tensor = torch.LongTensor([p_len]).to(self.device)
sid = torch.LongTensor([0]).to(self.device) sid = torch.LongTensor([0]).to(self.device)
skip_head = torch.LongTensor([skip_head]) skip_head_value = int(skip_head)
return_length2 = torch.LongTensor([return_length2]) return_length_value = int(return_length)
return_length = torch.LongTensor([return_length]) return_length2_value = int(return_length2)
with torch.no_grad(): with torch.no_grad():
if self.if_f0 == 1: if self.if_f0 == 1:
infered_audio, _, _ = self.net_g.infer( infered_audio = run_cuda_graph(
self.net_g,
"rvc-realtime-f0-%s-%s-%s"
% (skip_head_value, return_length_value, return_length2_value),
lambda phone, lengths, coarse, continuous, speaker: self.net_g.infer(
phone,
lengths,
coarse,
continuous,
speaker,
skip_head_value,
return_length_value,
return_length2_value,
)[0],
feats, feats,
p_len, p_len_tensor,
cache_pitch, cache_pitch,
cache_pitchf, cache_pitchf,
sid, sid,
skip_head,
return_length,
return_length2,
) )
else: else:
infered_audio, _, _ = self.net_g.infer( infered_audio = run_cuda_graph(
feats, p_len, sid, skip_head, return_length, return_length2 self.net_g,
"rvc-realtime-no-f0-%s-%s-%s"
% (skip_head_value, return_length_value, return_length2_value),
lambda phone, lengths, speaker: self.net_g.infer(
phone,
lengths,
speaker,
skip_head_value,
return_length_value,
return_length2_value,
)[0],
feats,
p_len_tensor,
sid,
) )
infered_audio = infered_audio.squeeze(1).float() infered_audio = infered_audio.squeeze(1).float()
upp_res = int(np.floor(factor * self.tgt_sr // 100)) upp_res = int(np.floor(factor * self.tgt_sr // 100))
@@ -332,11 +354,12 @@ class RVC:
infered_audio[:, : return_length * upp_res] infered_audio[:, : return_length * upp_res]
) )
t5 = ttime() t5 = ttime()
printt( if report_status:
i18n("耗时:特征=%.3f秒,索引=%.3f秒,音高=%.3f秒,模型=%.3f"), printt(
t2 - t1, i18n("耗时:特征=%.3f秒,索引=%.3f秒,音高=%.3f秒,模型=%.3f"),
t3 - t2, t2 - t1,
t4 - t3, t3 - t2,
t5 - t4, t4 - t3,
) t5 - t4,
)
return infered_audio.squeeze() return infered_audio.squeeze()

View File

@@ -19,6 +19,7 @@ from infer.vc.pipeline import Pipeline
from infer.vc.utils import * from infer.vc.utils import *
from i18n.i18n import I18nAuto from i18n.i18n import I18nAuto
from tools.progress import batch_status, should_report from tools.progress import batch_status, should_report
from tools.cuda_graph import clear_cuda_graph_cache
i18n = I18nAuto() i18n = I18nAuto()
@@ -68,6 +69,8 @@ class VC:
self.hubert_model is not None self.hubert_model is not None
): # 考虑到轮询, 需要加个判断看是否 sid 是由有模型切换到无模型的 ): # 考虑到轮询, 需要加个判断看是否 sid 是由有模型切换到无模型的
logger.info(i18n("清理模型缓存")) logger.info(i18n("清理模型缓存"))
clear_cuda_graph_cache(self.net_g)
clear_cuda_graph_cache(self.hubert_model)
del (self.net_g, self.n_spk, self.hubert_model, self.tgt_sr) # ,cpt del (self.net_g, self.n_spk, self.hubert_model, self.tgt_sr) # ,cpt
self.hubert_model = self.net_g = self.n_spk = self.hubert_model = ( self.hubert_model = self.net_g = self.n_spk = self.hubert_model = (
self.tgt_sr self.tgt_sr
@@ -112,6 +115,9 @@ class VC:
person = f'{os.getenv("weight_root")}/{sid}' person = f'{os.getenv("weight_root")}/{sid}'
logger.info("%s: %s", i18n("正在加载模型"), person) logger.info("%s: %s", i18n("正在加载模型"), person)
if self.net_g is not None:
clear_cuda_graph_cache(self.net_g)
self.cpt = torch.load(person, map_location="cpu") self.cpt = torch.load(person, map_location="cpu")
self.tgt_sr = self.cpt["config"][-1] self.tgt_sr = self.cpt["config"][-1]
self.cpt["config"][-3] = self.cpt["weight"]["emb_g.weight"].shape[0] # n_spk self.cpt["config"][-3] = self.cpt["weight"]["emb_g.weight"].shape[0] # n_spk

View File

@@ -15,6 +15,7 @@ import torch.nn.functional as F
from scipy import signal from scipy import signal
from infer.hubert import extract_hubert_features from infer.hubert import extract_hubert_features
from tools.cuda_graph import cuda_graph_enabled, run_cuda_graph
bh, ah = signal.butter(N=5, Wn=48, btype="high", fs=16000) bh, ah = signal.butter(N=5, Wn=48, btype="high", fs=16000)
@@ -217,11 +218,34 @@ class Pipeline(object):
p_len = torch.tensor([p_len], device=self.device).long() p_len = torch.tensor([p_len], device=self.device).long()
with torch.no_grad(): with torch.no_grad():
hasp = pitch is not None and pitchf is not None hasp = pitch is not None and pitchf is not None
arg = (feats, p_len, pitch, pitchf, sid) if hasp else (feats, p_len, sid) if hasp:
audio1 = (net_g.infer(*arg)[0][0, 0]).data.cpu().float().numpy() synthesized = run_cuda_graph(
del hasp, arg net_g,
"rvc-synth-f0",
lambda phone, lengths, coarse, continuous, speaker: net_g.infer(
phone, lengths, coarse, continuous, speaker
)[0],
feats,
p_len,
pitch,
pitchf,
sid,
)
else:
synthesized = run_cuda_graph(
net_g,
"rvc-synth-no-f0",
lambda phone, lengths, speaker: net_g.infer(
phone, lengths, speaker
)[0],
feats,
p_len,
sid,
)
audio1 = synthesized[0, 0].data.cpu().float().numpy()
del hasp, synthesized
del feats, p_len, padding_mask del feats, p_len, padding_mask
if torch.cuda.is_available(): if torch.cuda.is_available() and not cuda_graph_enabled(self.device):
torch.cuda.empty_cache() torch.cuda.empty_cache()
t2 = ttime() t2 = ttime()
times[0] += t1 - t0 times[0] += t1 - t0
@@ -381,6 +405,6 @@ class Pipeline(object):
max_int16 /= audio_max max_int16 /= audio_max
audio_opt = (audio_opt * max_int16).astype(np.int16) audio_opt = (audio_opt * max_int16).astype(np.int16)
del pitch, pitchf, sid del pitch, pitchf, sid
if torch.cuda.is_available(): if torch.cuda.is_available() and not cuda_graph_enabled(self.device):
torch.cuda.empty_cache() torch.cuda.empty_cache()
return audio_opt return audio_opt

View File

@@ -36,9 +36,10 @@ if __name__ == "__main__":
import torch.nn.functional as F import torch.nn.functional as F
import torchaudio.transforms as tat import torchaudio.transforms as tat
from configs.config import Config
from infer import rtrvc as rvc_for_realtime from infer import rtrvc as rvc_for_realtime
from i18n.i18n import I18nAuto from i18n.i18n import I18nAuto
from configs.config import Config from tools.cuda_graph import cuda_graph_enabled, run_cuda_graph
i18n = I18nAuto() i18n = I18nAuto()
@@ -64,9 +65,10 @@ if __name__ == "__main__":
self.sg_output_device = "" self.sg_output_device = ""
class GUI: class GUI:
def __init__(self) : def __init__(self) :
self.gui_config = GUIConfig() self.gui_config = GUIConfig()
self.config = Config() self.config = Config()
printt("RVC_CUDA_GRAPH=%s", os.environ.get("RVC_CUDA_GRAPH", "0"))
self.function = "vc" self.function = "vc"
self.delay_time = 0 self.delay_time = 0
self.hostapis = None self.hostapis = None
@@ -629,9 +631,16 @@ if __name__ == "__main__":
dtype=torch.float32, dtype=torch.float32,
) )
self.rms_buffer = np.zeros(4 * self.zc, dtype="float32") self.rms_buffer = np.zeros(4 * self.zc, dtype="float32")
self.sola_buffer = torch.zeros( self.sola_buffer = torch.zeros(
self.sola_buffer_frame, device=self.config.device, dtype=torch.float32 self.sola_buffer_frame, device=self.config.device, dtype=torch.float32
) )
self.sola_den_kernel = torch.ones(
1,
1,
self.sola_buffer_frame,
device=self.config.device,
dtype=torch.float32,
)
self.nr_buffer = self.sola_buffer.clone() self.nr_buffer = self.sola_buffer.clone()
self.output_buffer = self.input_wav.clone() self.output_buffer = self.input_wav.clone()
self.skip_head = self.extra_frame // self.zc self.skip_head = self.extra_frame // self.zc
@@ -666,12 +675,85 @@ if __name__ == "__main__":
).to(self.config.device) ).to(self.config.device)
else: else:
self.resampler2 = None self.resampler2 = None
self.tg = TorchGate( self.tg = TorchGate(
sr=self.gui_config.samplerate, n_fft=4 * self.zc, prop_decrease=0.9 sr=self.gui_config.samplerate, n_fft=4 * self.zc, prop_decrease=0.9
).to(self.config.device) ).to(self.config.device)
self.start_stream() self.prewarm_cuda_graph()
self.start_stream()
def start_stream(self):
def prewarm_cuda_graph(self):
if not cuda_graph_enabled(self.config.device):
return
try:
printt(i18n("正在预热CUDA Graph"))
samples = self.input_wav_res.shape[0]
phase = torch.arange(
samples, device=self.config.device, dtype=torch.float32
)
probe = 0.05 * torch.sin(2 * np.pi * 220.0 * phase / 16000.0)
self.input_wav_res.copy_(probe)
if self.gui_config.I_noise_reduce:
short = self.input_wav[
-self.sola_buffer_frame - self.block_frame :
].unsqueeze(0)
run_cuda_graph(
self.tg,
"realtime-input-noise-reduction",
lambda short_audio, full_audio: self.tg(
short_audio, full_audio
),
short,
self.input_wav.unsqueeze(0),
)
resample_input = self.input_wav[-self.block_frame - 2 * self.zc :]
run_cuda_graph(
self.resampler,
"realtime-input-resample",
lambda audio: self.resampler(audio),
resample_input,
)
inferred = self.rvc.infer(
self.input_wav_res,
self.block_frame_16k,
self.skip_head,
self.return_length,
self.gui_config.f0method,
)
if self.resampler2 is not None:
inferred = run_cuda_graph(
self.resampler2,
"realtime-output-resample",
lambda audio: self.resampler2(audio),
inferred,
)
if self.gui_config.O_noise_reduce:
run_cuda_graph(
self.tg,
"realtime-output-noise-reduction",
lambda short_audio, full_audio: self.tg(
short_audio, full_audio
),
inferred.unsqueeze(0),
self.output_buffer.unsqueeze(0),
)
torch.cuda.synchronize(self.config.device)
printt(i18n("CUDA Graph预热完成"))
except Exception:
printt(traceback.format_exc())
finally:
self.input_wav.zero_()
self.input_wav_denoise.zero_()
self.input_wav_res.zero_()
self.output_buffer.zero_()
self.sola_buffer.zero_()
self.nr_buffer.zero_()
self.rvc.cache_pitch.zero_()
self.rvc.cache_pitchf.zero_()
def start_stream(self):
global flag_vc global flag_vc
if not flag_vc: if not flag_vc:
flag_vc = True flag_vc = True
@@ -739,9 +821,13 @@ if __name__ == "__main__":
self.block_frame : self.block_frame :
].clone() ].clone()
input_wav = self.input_wav[-self.sola_buffer_frame - self.block_frame :] input_wav = self.input_wav[-self.sola_buffer_frame - self.block_frame :]
input_wav = self.tg( input_wav = run_cuda_graph(
input_wav.unsqueeze(0), self.input_wav.unsqueeze(0) self.tg,
).squeeze(0) "realtime-input-noise-reduction",
lambda short, full: self.tg(short, full),
input_wav.unsqueeze(0),
self.input_wav.unsqueeze(0),
).squeeze(0)
input_wav[: self.sola_buffer_frame] *= self.fade_in_window input_wav[: self.sola_buffer_frame] *= self.fade_in_window
input_wav[: self.sola_buffer_frame] += ( input_wav[: self.sola_buffer_frame] += (
self.nr_buffer * self.fade_out_window self.nr_buffer * self.fade_out_window
@@ -750,15 +836,23 @@ if __name__ == "__main__":
: self.block_frame : self.block_frame
] ]
self.nr_buffer[:] = input_wav[self.block_frame :] self.nr_buffer[:] = input_wav[self.block_frame :]
self.input_wav_res[-self.block_frame_16k - 160 :] = self.resampler( resample_input = self.input_wav_denoise[
self.input_wav_denoise[-self.block_frame - 2 * self.zc :] -self.block_frame - 2 * self.zc :
)[160:] ]
else: self.input_wav_res[-self.block_frame_16k - 160 :] = run_cuda_graph(
self.input_wav_res[-160 * (indata.shape[0] // self.zc + 1) :] = ( self.resampler,
self.resampler(self.input_wav[-indata.shape[0] - 2 * self.zc :])[ "realtime-input-resample",
160: lambda audio: self.resampler(audio),
] resample_input,
) )[160:]
else:
resample_input = self.input_wav[-indata.shape[0] - 2 * self.zc :]
self.input_wav_res[-160 * (indata.shape[0] // self.zc + 1) :] = run_cuda_graph(
self.resampler,
"realtime-input-resample",
lambda audio: self.resampler(audio),
resample_input,
)[160:]
# infer # infer
if self.function == "vc": if self.function == "vc":
infer_wav = self.rvc.infer( infer_wav = self.rvc.infer(
@@ -767,9 +861,14 @@ if __name__ == "__main__":
self.skip_head, self.skip_head,
self.return_length, self.return_length,
self.gui_config.f0method, self.gui_config.f0method,
) )
if self.resampler2 is not None: if self.resampler2 is not None:
infer_wav = self.resampler2(infer_wav) infer_wav = run_cuda_graph(
self.resampler2,
"realtime-output-resample",
lambda audio: self.resampler2(audio),
infer_wav,
)
elif self.gui_config.I_noise_reduce: elif self.gui_config.I_noise_reduce:
infer_wav = self.input_wav_denoise[self.extra_frame :].clone() infer_wav = self.input_wav_denoise[self.extra_frame :].clone()
else: else:
@@ -780,9 +879,13 @@ if __name__ == "__main__":
self.block_frame : self.block_frame :
].clone() ].clone()
self.output_buffer[-self.block_frame :] = infer_wav[-self.block_frame :] self.output_buffer[-self.block_frame :] = infer_wav[-self.block_frame :]
infer_wav = self.tg( infer_wav = run_cuda_graph(
infer_wav.unsqueeze(0), self.output_buffer.unsqueeze(0) self.tg,
).squeeze(0) "realtime-output-noise-reduction",
lambda short, full: self.tg(short, full),
infer_wav.unsqueeze(0),
self.output_buffer.unsqueeze(0),
).squeeze(0)
# volume envelop mixing # volume envelop mixing
if self.gui_config.rms_mix_rate < 1 and self.function == "vc": if self.gui_config.rms_mix_rate < 1 and self.function == "vc":
if self.gui_config.I_noise_reduce: if self.gui_config.I_noise_reduce:
@@ -813,20 +916,20 @@ if __name__ == "__main__":
mode="linear", mode="linear",
align_corners=True, align_corners=True,
)[0, 0, :-1] )[0, 0, :-1]
rms2 = torch.max(rms2, torch.zeros_like(rms2) + 1e-3) rms2 = torch.max(rms2, torch.zeros_like(rms2) + 1e-3)
infer_wav *= torch.pow( infer_wav *= torch.pow(
rms1 / rms2, torch.tensor(1 - self.gui_config.rms_mix_rate) rms1 / rms2, 1.0 - self.gui_config.rms_mix_rate
) )
# SOLA algorithm from https://github.com/yxlllc/DDSP-SVC # SOLA algorithm from https://github.com/yxlllc/DDSP-SVC
conv_input = infer_wav[ conv_input = infer_wav[
None, None, : self.sola_buffer_frame + self.sola_search_frame None, None, : self.sola_buffer_frame + self.sola_search_frame
] ]
cor_nom = F.conv1d(conv_input, self.sola_buffer[None, None, :]) cor_nom = F.conv1d(conv_input, self.sola_buffer[None, None, :])
cor_den = torch.sqrt( cor_den = torch.sqrt(
F.conv1d( F.conv1d(
conv_input**2, conv_input**2,
torch.ones(1, 1, self.sola_buffer_frame, device=self.config.device), self.sola_den_kernel,
) )
+ 1e-8 + 1e-8
) )
if sys.platform == "darwin": if sys.platform == "darwin":

227
tools/cuda_graph.py Normal file
View File

@@ -0,0 +1,227 @@
import logging
import os
import threading
import time
from collections import OrderedDict
import torch
logger = logging.getLogger(__name__)
ENV_NAME = "RVC_CUDA_GRAPH"
MAX_CACHE_ENV = "RVC_CUDA_GRAPH_MAX_CACHE"
_probe_lock = threading.Lock()
_probe_result = None
def _device_type(device):
if isinstance(device, torch.device):
return device.type
return str(device).split(":", 1)[0].lower()
def _cuda_device(device):
parsed = device if isinstance(device, torch.device) else torch.device(device)
if parsed.index is None:
parsed = torch.device("cuda", torch.cuda.current_device())
return parsed
def _clone_output(value):
if torch.is_tensor(value):
return value.clone()
if isinstance(value, tuple):
return tuple(_clone_output(item) for item in value)
if isinstance(value, list):
return [_clone_output(item) for item in value]
if isinstance(value, dict):
return {key: _clone_output(item) for key, item in value.items()}
return value
def detect_cuda_graph_support(device):
if _device_type(device) != "cuda" or not torch.cuda.is_available():
return False
if not hasattr(torch.cuda, "CUDAGraph") or not hasattr(torch.cuda, "graph"):
return False
cuda_device = _cuda_device(device)
try:
with torch.cuda.device(cuda_device):
current = torch.cuda.current_stream(cuda_device)
warmup = torch.cuda.Stream(device=cuda_device)
warmup.wait_stream(current)
with torch.cuda.stream(warmup):
probe = torch.arange(32, device=cuda_device, dtype=torch.float32)
for _ in range(3):
expected = probe.square().add_(1)
current.wait_stream(warmup)
torch.cuda.synchronize(cuda_device)
graph = torch.cuda.CUDAGraph()
with torch.cuda.graph(graph):
captured = probe.square() + 1
probe.copy_(torch.arange(32, device=cuda_device, dtype=torch.float32))
graph.replay()
torch.cuda.synchronize(cuda_device)
valid = torch.equal(
captured.cpu(), torch.arange(32, dtype=torch.float32).square() + 1
)
del captured, expected, graph, probe
return bool(valid)
except Exception:
logger.exception("CUDA Graph support probe failed on %s", cuda_device)
return False
def configure_cuda_graph(device):
global _probe_result
explicit = os.environ.get(ENV_NAME)
if explicit in {"0", "1"}:
if explicit == "0":
return False
if _device_type(device) != "cuda":
os.environ[ENV_NAME] = "0"
return False
with _probe_lock:
if _probe_result is None:
_probe_result = detect_cuda_graph_support(device)
os.environ[ENV_NAME] = "1" if _probe_result else "0"
return bool(_probe_result)
def cuda_graph_enabled(device):
return (
os.environ.get(ENV_NAME) == "1"
and _device_type(device) == "cuda"
and torch.cuda.is_available()
)
def _tensor_signature(tensor):
return (
tuple(tensor.shape),
tuple(tensor.stride()),
str(tensor.dtype),
str(tensor.device),
bool(tensor.requires_grad),
)
class _CapturedCall:
def __init__(self, function, inputs):
started = time.perf_counter()
self.lock = threading.RLock()
self.inputs = tuple(torch.empty_like(value) for value in inputs)
for static, value in zip(self.inputs, inputs):
static.copy_(value)
device = self.inputs[0].device
current = torch.cuda.current_stream(device)
warmup = torch.cuda.Stream(device=device)
warmup.wait_stream(current)
with torch.cuda.stream(warmup), torch.no_grad():
for _ in range(3):
output = function(*self.inputs)
current.wait_stream(warmup)
torch.cuda.synchronize(device)
self.graph = torch.cuda.CUDAGraph()
with torch.cuda.graph(self.graph), torch.no_grad():
self.output = function(*self.inputs)
self.capture_ms = (time.perf_counter() - started) * 1000.0
self.done_event = None
del output
def replay(self, inputs):
with self.lock:
stream = torch.cuda.current_stream(self.inputs[0].device)
if self.done_event is not None:
stream.wait_event(self.done_event)
for static, value in zip(self.inputs, inputs):
static.copy_(value, non_blocking=True)
self.graph.replay()
output = _clone_output(self.output)
self.done_event = torch.cuda.Event(blocking=False)
self.done_event.record(stream)
return output
class _GraphCache:
def __init__(self):
self.entries = OrderedDict()
self.failures = set()
self.lock = threading.RLock()
self.capture_count = 0
self.replay_count = 0
self.fallback_count = 0
self.eviction_count = 0
self.capture_ms = 0.0
def run(self, key, function, inputs):
signature = key + tuple(_tensor_signature(value) for value in inputs)
with self.lock:
if signature in self.failures:
self.fallback_count += 1
return function(*inputs)
entry = self.entries.get(signature)
if entry is None:
try:
entry = _CapturedCall(function, inputs)
self.entries[signature] = entry
self.capture_count += 1
self.capture_ms += entry.capture_ms
max_entries = max(1, int(os.environ.get(MAX_CACHE_ENV, "8")))
while len(self.entries) > max_entries:
self.entries.popitem(last=False)
self.eviction_count += 1
except Exception:
self.failures.add(signature)
self.fallback_count += 1
logger.exception("CUDA Graph capture failed for %s; using eager", key)
return function(*inputs)
else:
self.entries.move_to_end(signature)
output = entry.replay(inputs)
with self.lock:
self.replay_count += 1
return output
def run_cuda_graph(owner, namespace, function, *inputs):
if not inputs or not cuda_graph_enabled(inputs[0].device):
return function(*inputs)
cache = getattr(owner, "_rvc_cuda_graph_cache", None)
if cache is None:
cache = _GraphCache()
setattr(owner, "_rvc_cuda_graph_cache", cache)
return cache.run((str(namespace),), function, tuple(inputs))
def clear_cuda_graph_cache(owner):
cache = getattr(owner, "_rvc_cuda_graph_cache", None)
if cache is not None:
cache.entries.clear()
cache.failures.clear()
delattr(owner, "_rvc_cuda_graph_cache")
def get_cuda_graph_stats(owner):
cache = getattr(owner, "_rvc_cuda_graph_cache", None)
if cache is None:
return {
"entries": 0,
"failures": 0,
"captures": 0,
"replays": 0,
"fallbacks": 0,
"evictions": 0,
"capture_ms": 0.0,
}
with cache.lock:
return {
"entries": len(cache.entries),
"failures": len(cache.failures),
"captures": cache.capture_count,
"replays": cache.replay_count,
"fallbacks": cache.fallback_count,
"evictions": cache.eviction_count,
"capture_ms": cache.capture_ms,
}

View File

@@ -9,6 +9,8 @@ import soundfile as sf
import torch import torch
import torch.nn as nn import torch.nn as nn
import yaml import yaml
from tools.cuda_graph import run_cuda_graph
from tqdm import tqdm from tqdm import tqdm
from tools.file_io import read_text from tools.file_io import read_text
from i18n.i18n import I18nAuto from i18n.i18n import I18nAuto
@@ -178,7 +180,12 @@ class Roformer_Loader:
if len(batch_data) >= batch_size or (i >= mix.shape[1]): if len(batch_data) >= batch_size or (i >= mix.shape[1]):
arr = torch.stack(batch_data, dim=0) arr = torch.stack(batch_data, dim=0)
# print(23333333,arr.dtype) # print(23333333,arr.dtype)
x = model(arr) x = run_cuda_graph(
model,
"uvr-bsroformer",
lambda audio: model(audio),
arr,
)
window = window_middle window = window_middle
if i - step == 0: # First audio chunk, no fadein if i - step == 0: # First audio chunk, no fadein

View File

@@ -1,5 +1,6 @@
import numpy as np import numpy as np
import torch import torch
from tools.cuda_graph import run_cuda_graph
from tqdm import tqdm from tqdm import tqdm
@@ -34,7 +35,12 @@ def inference(X_spec, device, model, aggressiveness, data):
X_mag_window = X_mag_window.half() X_mag_window = X_mag_window.half()
X_mag_window = X_mag_window.to(device) X_mag_window = X_mag_window.to(device)
pred = model.predict(X_mag_window, aggressiveness) pred = run_cuda_graph(
model,
"uvr-vr-%s" % repr(aggressiveness),
lambda window: model.predict(window, aggressiveness),
X_mag_window,
)
pred = pred.detach().cpu().numpy() pred = pred.detach().cpu().numpy()
preds.append(pred[0]) preds.append(pred[0])

View File

@@ -25,6 +25,7 @@ for name in os.listdir(tmp):
except Exception as error: except Exception as error:
print(str(error)) print(str(error))
from configs.config import Config, GPU_INDEX, GPU_INFOS, GPU_MEMORY, IS_GPU
from infer.vc.modules import VC from infer.vc.modules import VC
from tools.uvr5.webui import uvr from tools.uvr5.webui import uvr
from tools.file_io import read_text from tools.file_io import read_text
@@ -35,7 +36,6 @@ from train.process_ckpt import (
show_info, show_info,
) )
from i18n.i18n import I18nAuto from i18n.i18n import I18nAuto
from configs.config import Config, GPU_INDEX, GPU_INFOS, GPU_MEMORY, IS_GPU
import torch, platform import torch, platform
import numpy as np import numpy as np
import gradio as gr import gradio as gr
@@ -143,6 +143,7 @@ print(
i18n("当前设备:%s | 推理精度:%s") % (config.device, config.dtype), i18n("当前设备:%s | 推理精度:%s") % (config.device, config.dtype),
flush=True, flush=True,
) )
logger.info("RVC_CUDA_GRAPH=%s", os.environ.get("RVC_CUDA_GRAPH", "0"))
# GPU filtering and precision rules are shared with inference/extraction/training. # GPU filtering and precision rules are shared with inference/extraction/training.
gpu_infos = list(GPU_INFOS) gpu_infos = list(GPU_INFOS)
gpu_indices = sorted(GPU_INDEX) gpu_indices = sorted(GPU_INDEX)
@@ -394,6 +395,10 @@ def train_task_stopped(state):
def start_train_process(state, cmd): def start_train_process(state, cmd):
kwargs = {"shell": True, "cwd": now_dir} kwargs = {"shell": True, "cwd": now_dir}
if "train/train.py" in cmd.replace("\\", "/"):
training_env = os.environ.copy()
training_env["RVC_CUDA_GRAPH"] = "0"
kwargs["env"] = training_env
if platform.system() == "Windows": if platform.system() == "Windows":
kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
else: else: