202601 MP3转语音代码

写于

MP3转语音转换器使用说明

功能介绍

这个Python程序可以将MP3音频文件转换为语音,主要功能包括:

  1. 语音识别:从MP3文件中提取文本内容
  2. 文本转语音:将提取的文本转换为语音
  3. 音频播放:播放转换后的语音文件
  4. 批量处理:支持批量转换多个MP3文件

安装依赖

首先安装所需的Python包:

pip install -r requirements_audio.txt

使用方法

1. 运行程序

python mp3_to_speech.py

2. 选择操作模式

程序运行后会显示菜单: - 1. 转换单个MP3文件 - 2. 批量转换MP3文件
- 3. 播放MP3文件 - 4. 文本转语音

3. 功能详解

单个文件转换

输入MP3文件路径,程序会: 1. 提取音频中的文本内容 2. 将文本转换为语音 3. 播放转换后的语音

批量转换

输入包含MP3文件的文件夹路径,程序会自动处理该文件夹下的所有MP3文件。

直接播放

直接播放指定的MP3文件。

文本转语音

输入任意文本,选择语言(默认为中文),程序会将文本转换为语音并播放。

注意事项

  1. 语音识别限制:语音识别功能依赖于网络服务,需要联网使用
  2. 语言支持:支持多种语言,中文请使用 zh-CN,英文请使用 en-US
  3. 音频格式:程序主要处理MP3格式,其他格式可能需要转换
  4. 网络要求:文本转语音和语音识别功能需要网络连接

示例

from mp3_to_speech import MP3ToSpeechConverter

# 创建转换器对象
converter = MP3ToSpeechConverter()

# 转换单个文件
converter.convert_mp3_to_speech('input.mp3')

# 批量转换
converter.batch_convert('./mp3_files')

# 文本转语音
converter.text_to_speech('你好,这是一个测试', lang='zh-CN')

故障排除

  1. 音频播放失败:检查是否安装了pygame库
  2. 语音识别失败:检查网络连接和音频文件质量
  3. 文本转语音失败:检查网络连接和文本内容

代码如下

import os
import pygame
import tempfile
from gtts import gTTS
from pydub import AudioSegment
import speech_recognition as sr


class MP3ToSpeechConverter:
    """MP3文件转语音转换器"""

    def __init__(self):
        """初始化转换器"""
        pygame.mixer.init()
        self.recognizer = sr.Recognizer()

    def extract_text_from_mp3(self, mp3_file_path):
        """
        从MP3文件中提取文本(使用语音识别)

        参数:
            mp3_file_path (str): MP3文件路径

        返回:
            str: 提取的文本内容
        """
        try:
            # 使用语音识别API
            with sr.AudioFile(mp3_file_path) as source:
                audio_data = self.recognizer.record(source)

            # 使用Google语音识别
            text = self.recognizer.recognize_google(audio_data, language='zh-CN')
            print(f"提取到的文本: {text}")
            return text

        except sr.UnknownValueError:
            print("无法识别音频内容")
            return ""
        except sr.RequestError as e:
            print(f"语音识别服务错误: {e}")
            return ""
        except Exception as e:
            print(f"处理音频文件时出错: {e}")
            return ""

    def text_to_speech(self, text, output_file=None, lang='zh-CN'):
        """
        将文本转换为语音

        参数:
            text (str): 要转换的文本
            output_file (str): 输出文件路径,如果为None则自动生成
            lang (str): 语言代码,默认为中文

        返回:
            str: 输出文件路径
        """
        if not text:
            print("文本为空,无法转换")
            return None

        try:
            # 创建gTTS对象
            tts = gTTS(text=text, lang=lang, slow=False)

            # 如果没有指定输出文件,创建临时文件
            if output_file is None:
                temp_file = tempfile.NamedTemporaryFile(delete=False, suffix='.mp3')
                output_file = temp_file.name
                temp_file.close()

            # 保存音频文件
            tts.save(output_file)
            print(f"语音文件已生成: {output_file}")
            return output_file

        except Exception as e:
            print(f"文本转语音时出错: {e}")
            return None

    def play_audio(self, audio_file_path):
        """
        播放音频文件

        参数:
            audio_file_path (str): 音频文件路径
        """
        try:
            if not os.path.exists(audio_file_path):
                print(f"音频文件不存在: {audio_file_path}")
                return

            pygame.mixer.music.load(audio_file_path)
            pygame.mixer.music.play()

            print("正在播放音频...")
            while pygame.mixer.music.get_busy():
                pygame.time.Clock().tick(10)

            print("音频播放完成")

        except Exception as e:
            print(f"播放音频时出错: {e}")

    def convert_mp3_to_speech(self, mp3_file_path, output_lang='zh-CN'):
        """
        完整的MP3转语音流程

        参数:
            mp3_file_path (str): 输入MP3文件路径
            output_lang (str): 输出语音的语言

        返回:
            str: 生成的语音文件路径
        """
        print(f"开始处理MP3文件: {mp3_file_path}")

        # 步骤1: 从MP3中提取文本
        print("步骤1: 提取音频中的文本...")
        text = self.extract_text_from_mp3(mp3_file_path)

        if not text:
            print("无法从MP3中提取文本,尝试直接播放原文件...")
            self.play_audio(mp3_file_path)
            return mp3_file_path

        # 步骤2: 将文本转换为语音
        print("步骤2: 将文本转换为语音...")
        output_file = self.text_to_speech(text, lang=output_lang)

        if output_file:
            # 步骤3: 播放生成的语音
            print("步骤3: 播放生成的语音...")
            self.play_audio(output_file)
            return output_file
        else:
            print("转换失败,尝试直接播放原文件...")
            self.play_audio(mp3_file_path)
            return mp3_file_path

    def batch_convert(self, mp3_folder, output_folder='./speech_output'):
        """
        批量转换MP3文件

        参数:
            mp3_folder (str): 包含MP3文件的文件夹路径
            output_folder (str): 输出文件夹路径
        """
        if not os.path.exists(output_folder):
            os.makedirs(output_folder)

        mp3_files = [f for f in os.listdir(mp3_folder) if f.endswith('.mp3')]

        if not mp3_files:
            print("未找到MP3文件")
            return

        print(f"找到 {len(mp3_files)} 个MP3文件")

        for i, mp3_file in enumerate(mp3_files):
            print(f"\n处理文件 {i+1}/{len(mp3_files)}: {mp3_file}")
            mp3_path = os.path.join(mp3_folder, mp3_file)

            # 生成输出文件名
            base_name = os.path.splitext(mp3_file)[0]
            output_file = os.path.join(output_folder, f"{base_name}_speech.mp3")

            try:
                self.convert_mp3_to_speech(mp3_path)
                print(f"完成: {mp3_file}")
            except Exception as e:
                print(f"处理 {mp3_file} 时出错: {e}")


def main():
    """主函数"""
    converter = MP3ToSpeechConverter()

    print("=== MP3转语音转换器 ===")
    print("1. 转换单个MP3文件")
    print("2. 批量转换MP3文件")
    print("3. 播放MP3文件")
    print("4. 文本转语音")

    choice = input("\n请选择操作 (1-4): ").strip()

    if choice == '1':
        mp3_path = input("请输入MP3文件路径: ").strip()
        if os.path.exists(mp3_path):
            converter.convert_mp3_to_speech(mp3_path)
        else:
            print("文件不存在")

    elif choice == '2':
        folder_path = input("请输入MP3文件夹路径: ").strip()
        if os.path.exists(folder_path):
            converter.batch_convert(folder_path)
        else:
            print("文件夹不存在")

    elif choice == '3':
        mp3_path = input("请输入MP3文件路径: ").strip()
        if os.path.exists(mp3_path):
            converter.play_audio(mp3_path)
        else:
            print("文件不存在")

    elif choice == '4':
        text = input("请输入要转换的文本: ").strip()
        lang = input("请输入语言代码 (如zh-CN, en-US): ").strip() or 'zh-CN'
        output_file = converter.text_to_speech(text, lang=lang)
        if output_file:
            converter.play_audio(output_file)

    else:
        print("无效的选择")


if __name__ == "__main__":
    main()

对应的库

# MP3转语音转换器依赖包

# 音频处理
pydub>=0.25.1
pygame>=2.5.0

# 文本转语音
gtts>=2.4.0

# 语音识别
SpeechRecognition>=3.10.0

# 其他依赖
requests>=2.31.0