Retrieval-based-Voice-Conve.../infer/lib/audio.py

import librosa
import ffmpeg
import numpy as np


def load_audio(file, sr):
    try:
        # https://github.com/openai/whisper/blob/main/whisper/audio.py#L26
        # This launches a subprocess to decode audio while down-mixing and resampling as necessary.
        # Requires the ffmpeg CLI and `ffmpeg-python` package to be installed.
        file = (
            file.strip(" ").strip('"').strip("\n").strip('"').strip(" ")
        )  # 防止小白拷路径头尾带了空格和"和回车
        out, _ = (
            ffmpeg.input(file, threads=0)
            .output("-", format="f32le", acodec="pcm_f32le", ac=1, ar=sr)
            .run(cmd=["ffmpeg", "-nostdin"], capture_stdout=True, capture_stderr=True)
        )
        return np.frombuffer(out, np.float32).flatten()

    except AttributeError:
        audio = file[1] / 32768.0
        if len(audio.shape) == 2:
            audio = np.mean(audio, -1)
        return librosa.resample(audio, orig_sr=file[0], target_sr=16000)

    except Exception as e:
        raise RuntimeError(f"Failed to load audio: {e}")
load audio with gradio-file 2023-08-27 12:14:01 +02:00			`import librosa`
replace lib 2023-08-19 13:00:56 +02:00			`import ffmpeg`
			`import numpy as np`


			`def load_audio(file, sr):`
			`try:`
			`# https://github.com/openai/whisper/blob/main/whisper/audio.py#L26`
			`# This launches a subprocess to decode audio while down-mixing and resampling as necessary.`
			# Requires the ffmpeg CLI and `ffmpeg-python` package to be installed.
			`file = (`
			`file.strip(" ").strip('"').strip("\n").strip('"').strip(" ")`
			`) # 防止小白拷路径头尾带了空格和"和回车`
			`out, _ = (`
			`ffmpeg.input(file, threads=0)`
			`.output("-", format="f32le", acodec="pcm_f32le", ac=1, ar=sr)`
			`.run(cmd=["ffmpeg", "-nostdin"], capture_stdout=True, capture_stderr=True)`
			`)`
load audio with gradio-file 2023-08-27 12:14:01 +02:00			`return np.frombuffer(out, np.float32).flatten()`

			`except AttributeError:`
			`audio = file[1] / 32768.0`
			`if len(audio.shape) == 2:`
			`audio = np.mean(audio, -1)`
			`return librosa.resample(audio, orig_sr=file[0], target_sr=16000)`

replace lib 2023-08-19 13:00:56 +02:00			`except Exception as e:`
			`raise RuntimeError(f"Failed to load audio: {e}")`