Retrieval-based-Voice-Conve.../infer/lib/audio.py

import platform, os
import ffmpeg
import numpy as np
import av
from io import BytesIO
import traceback
import re


def wav2(i, o, format):
    inp = av.open(i, "rb")
    if format == "m4a":
        format = "mp4"
    out = av.open(o, "wb", format=format)
    if format == "ogg":
        format = "libvorbis"
    if format == "mp4":
        format = "aac"

    ostream = out.add_stream(format)

    for frame in inp.decode(audio=0):
        for p in ostream.encode(frame):
            out.mux(p)

    for p in ostream.encode(None):
        out.mux(p)

    out.close()
    inp.close()


def load_audio(file, sr):
    try:
        # https://github.com/openai/whisper/blob/main/whisper/audio.py#L26
        # This launches a subprocess to decode audio while down-mixing and resampling as necessary.
        # Requires the ffmpeg CLI and `ffmpeg-python` package to be installed.
        file = clean_path(file)  # 防止小白拷路径头尾带了空格和"和回车
        if os.path.exists(file) == False:
            raise RuntimeError(
                "You input a wrong audio path that does not exists, please fix it!"
            )
        out, _ = (
            ffmpeg.input(file, threads=0)
            .output("-", format="f32le", acodec="pcm_f32le", ac=1, ar=sr)
            .run(cmd=["ffmpeg", "-nostdin"], capture_stdout=True, capture_stderr=True)
        )
    except Exception as e:
        traceback.print_exc()
        raise RuntimeError(f"Failed to load audio: {e}")

    return np.frombuffer(out, np.float32).flatten()


def clean_path(path_str):
    if platform.system() == "Windows":
        path_str = path_str.replace("/", "\\")
    path_str = re.sub(r'[\u202a\u202b\u202c\u202d\u202e]', '', path_str)  # 移除 Unicode 控制字符
    return path_str.strip(" ").strip('"').strip("\n").strip('"').strip(" ")
chore(format): run black on main 2024-01-26 16:10:04 +08:00			`import platform, os`
Update audio.py 2024-01-23 15:15:04 +08:00			`import ffmpeg`
replace lib 2023-08-19 19:00:56 +08:00			`import numpy as np`
optimize(ffmpeg): replace cmdline ffmpeg to pyav except uvr5 2023-09-03 15:19:19 +08:00			`import av`
			`from io import BytesIO`
The import for traceback is missing. 2024-06-26 21:15:01 +08:00			`import traceback`
移除音频文件路径 Unicode 控制字符 (#2334) 2024-11-23 20:46:42 +08:00			`import re`
replace lib 2023-08-19 19:00:56 +08:00
Format code (#1193) Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com> 2023-09-14 08:34:30 +08:00
optimize(ffmpeg): replace cmdline ffmpeg to pyav except uvr5 2023-09-03 15:19:19 +08:00			`def wav2(i, o, format):`
Format code (#1193) Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com> 2023-09-14 08:34:30 +08:00			`inp = av.open(i, "rb")`
			`if format == "m4a":`
			`format = "mp4"`
			`out = av.open(o, "wb", format=format)`
			`if format == "ogg":`
			`format = "libvorbis"`
			`if format == "mp4":`
			`format = "aac"`
optimize(ffmpeg): replace cmdline ffmpeg to pyav except uvr5 2023-09-03 15:19:19 +08:00
			`ostream = out.add_stream(format)`

			`for frame in inp.decode(audio=0):`
Format code (#1193) Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com> 2023-09-14 08:34:30 +08:00			`for p in ostream.encode(frame):`
			`out.mux(p)`
optimize(ffmpeg): replace cmdline ffmpeg to pyav except uvr5 2023-09-03 15:19:19 +08:00
Format code (#1193) Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com> 2023-09-14 08:34:30 +08:00			`for p in ostream.encode(None):`
			`out.mux(p)`
optimize(ffmpeg): replace cmdline ffmpeg to pyav except uvr5 2023-09-03 15:19:19 +08:00
			`out.close()`
			`inp.close()`

Format code (#1193) Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com> 2023-09-14 08:34:30 +08:00
replace lib 2023-08-19 19:00:56 +08:00			`def load_audio(file, sr):`
			`try:`
Update audio.py 2024-01-23 15:15:04 +08:00			`# https://github.com/openai/whisper/blob/main/whisper/audio.py#L26`
			`# This launches a subprocess to decode audio while down-mixing and resampling as necessary.`
			# Requires the ffmpeg CLI and `ffmpeg-python` package to be installed.
			`file = clean_path(file) # 防止小白拷路径头尾带了空格和"和回车`
Update audio.py 2024-06-14 19:56:10 +08:00			`if os.path.exists(file) == False:`
			`raise RuntimeError(`
			`"You input a wrong audio path that does not exists, please fix it!"`
			`)`
Update audio.py 2024-01-23 15:15:04 +08:00			`out, _ = (`
			`ffmpeg.input(file, threads=0)`
			`.output("-", format="f32le", acodec="pcm_f32le", ac=1, ar=sr)`
			`.run(cmd=["ffmpeg", "-nostdin"], capture_stdout=True, capture_stderr=True)`
			`)`
			`except Exception as e:`
Update audio.py 2024-06-14 19:56:10 +08:00			`traceback.print_exc()`
Update audio.py 2024-01-23 15:15:04 +08:00			`raise RuntimeError(f"Failed to load audio: {e}")`

			`return np.frombuffer(out, np.float32).flatten()`
load audio with gradio-file 2023-08-27 18:14:01 +08:00

Update audio.py 2024-06-14 19:56:10 +08:00
Update audio.py 2024-01-23 15:15:04 +08:00			`def clean_path(path_str):`
chore(format): run black on main 2024-01-26 16:10:04 +08:00			`if platform.system() == "Windows":`
			`path_str = path_str.replace("/", "\\")`
移除音频文件路径 Unicode 控制字符 (#2334) 2024-11-23 20:46:42 +08:00			`path_str = re.sub(r'[\u202a\u202b\u202c\u202d\u202e]', '', path_str) # 移除 Unicode 控制字符`
Update audio.py 2024-01-23 15:15:04 +08:00			`return path_str.strip(" ").strip('"').strip("\n").strip('"').strip(" ")`