python语音转文字

百度英语中文语音识别为文字服务 python demo【使用requests重写官方demo】

python • 李魔佛发表了文章 • 0 个评论 • 3071 次浏览 • 2021-04-25 11:49 • 来自相关话题

官方使用的稍微底层的urllib写的，用过了requests库的人看着不习惯。故重写之，并做了封装。
官方demo：
https://github.com/Baidu-AIP/speech-demo
# -*- coding: utf-8 -*-
# @Time : 2021/4/24 20:50
# @File : baidu_voice_service.py
# @Author : Rocky C@www.30daydo.com

import os
import requests
import sys
import pickle
sys.path.append('..')
from config import API_KEY,SECRET_KEY
from base64 import b64encode
from pathlib import PurePath

BASE = PurePath(__file__).parent

# 需要识别的文件
AUDIO_FILE = r'C:\OtherGit\speech-demo\rest-api-asr\python\audio\2.m4a' # 只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式
# 文件格式
FORMAT = AUDIO_FILE[-3:] # 文件后缀只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式

CUID = '24057753'
# 采样率
RATE = 16000 # 固定值

ASR_URL = 'http://vop.baidu.com/server_api'

#测试自训练平台需要打开以下信息，自训练平台模型上线后，您会看见第二步：“”获取专属模型参数pid:8001，modelid:1234”，按照这个信息获取 dev_pid=8001，lm_id=1234
'''
http://vop.baidu.com/server_api
1537 普通话(纯中文识别) 输入法模型有标点支持自定义词库
1737 英语英语模型无标点不支持自定义词库
1637 粤语粤语模型有标点不支持自定义词库
1837 四川话四川话模型有标点不支持自定义词库
1936 普通话远场
'''
DEV_PID = 1737

SCOPE = 'brain_enhanced_asr' # 有此scope表示有asr能力，没有请在网页里开通极速版

class DemoError(Exception):
pass

TOKEN_URL = 'http://openapi.baidu.com/oauth/2.0/token'

def fetch_token():

params = {'grant_type': 'client_credentials',
'client_id': API_KEY,
'client_secret': SECRET_KEY}
r = requests.post(
url=TOKEN_URL,
data=params
)

result = r.json()
if ('access_token' in result.keys() and 'scope' in result.keys()):
if SCOPE and (not SCOPE in result['scope'].split(' ')): # SCOPE = False 忽略检查
raise DemoError('scope is not correct')

return result['access_token']

else:
raise DemoError('MAYBE API_KEY or SECRET_KEY not correct: access_token or scope not found in token response')

""" TOKEN end """

def dump_token(token):
with open(os.path.join(BASE,'token.pkl'),'wb') as fp:
pickle.dump({'token':token},fp)

def load_token(filename):

if not os.path.exists(filename):
token=fetch_token()
dump_token(token)
return token
else:
with open(filename,'rb') as fp:
token = pickle.load(fp)
return token['token']

def recognize_service(token,filename):

with open(filename, 'rb') as speech_file:
speech_data = speech_file.read()

length = len(speech_data)
if length == 0:
raise DemoError('file %s length read 0 bytes' % AUDIO_FILE)

b64_data = b64encode(speech_data)
params = {'cuid': CUID, 'token': token, 'dev_pid': DEV_PID,'speech':b64_data,'len':length,'format':FORMAT,'rate':RATE,'channel':1}

headers = {
'Content-Type':'application/json',
}
r = requests.post(url=ASR_URL,json=params,headers=headers)
return r.json()

def main():
filename = 'token.pkl'
token = load_token(filename)
result = recognize_service(token,AUDIO_FILE)
print(result['result'])

if __name__ == '__main__':
main()

只需要替换自己的key就可以使用。
自己录了几段英文测试了下，还是蛮准的。查看全部

官方使用的稍微底层的urllib写的，用过了requests库的人看着不习惯。故重写之，并做了封装。
官方demo：
https://github.com/Baidu-AIP/speech-demo

# -*- coding: utf-8 -*-

# @Time : 2021/4/24 20:50

# @File : baidu_voice_service.py

# @Author : Rocky C@www.30daydo.com



import os

import requests

import sys

import pickle

sys.path.append('..')

from config import API_KEY,SECRET_KEY

from base64 import b64encode

from pathlib import PurePath





BASE = PurePath(__file__).parent



# 需要识别的文件

AUDIO_FILE = r'C:\OtherGit\speech-demo\rest-api-asr\python\audio\2.m4a'  # 只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式

# 文件格式

FORMAT = AUDIO_FILE[-3:]  # 文件后缀只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式



CUID = '24057753'

# 采样率

RATE = 16000  # 固定值





ASR_URL = 'http://vop.baidu.com/server_api'



#测试自训练平台需要打开以下信息， 自训练平台模型上线后，您会看见 第二步：“”获取专属模型参数pid:8001，modelid:1234”，按照这个信息获取 dev_pid=8001，lm_id=1234

'''

http://vop.baidu.com/server_api

1537	普通话(纯中文识别)	输入法模型	有标点	支持自定义词库

1737	英语	英语模型	无标点	不支持自定义词库

1637	粤语	粤语模型	有标点	不支持自定义词库

1837	四川话	四川话模型	有标点	不支持自定义词库

1936	普通话远场

'''

DEV_PID = 1737



SCOPE = 'brain_enhanced_asr'  # 有此scope表示有asr能力，没有请在网页里开通极速版





class DemoError(Exception):

    pass





TOKEN_URL = 'http://openapi.baidu.com/oauth/2.0/token'



def fetch_token():



    params = {'grant_type': 'client_credentials',

              'client_id': API_KEY,

              'client_secret': SECRET_KEY}

    r = requests.post(

        url=TOKEN_URL,

        data=params

    )



    result = r.json()

    if ('access_token' in result.keys() and 'scope' in result.keys()):

        if SCOPE and (not SCOPE in result['scope'].split(' ')):  # SCOPE = False 忽略检查

            raise DemoError('scope is not correct')



        return result['access_token']



    else:

        raise DemoError('MAYBE API_KEY or SECRET_KEY not correct: access_token or scope not found in token response')





"""  TOKEN end """



def dump_token(token):

    with open(os.path.join(BASE,'token.pkl'),'wb') as fp:

        pickle.dump({'token':token},fp)



def load_token(filename):



    if not os.path.exists(filename):

        token=fetch_token()

        dump_token(token)

        return token

    else:

        with open(filename,'rb') as fp:

            token = pickle.load(fp)

            return token['token']



def recognize_service(token,filename):



    with open(filename, 'rb') as speech_file:

        speech_data = speech_file.read()



    length = len(speech_data)

    if length == 0:

        raise DemoError('file %s length read 0 bytes' % AUDIO_FILE)



    b64_data = b64encode(speech_data)

    params = {'cuid': CUID, 'token': token, 'dev_pid': DEV_PID,'speech':b64_data,'len':length,'format':FORMAT,'rate':RATE,'channel':1}



    headers = {

        'Content-Type':'application/json',

    }

    r = requests.post(url=ASR_URL,json=params,headers=headers)

    return r.json()





def main():

    filename = 'token.pkl'

    token = load_token(filename)

    result = recognize_service(token,AUDIO_FILE)

    print(result['result'])



if __name__ == '__main__':

    main()

只需要替换自己的key就可以使用。
自己录了几段英文测试了下，还是蛮准的。

# -*- coding: utf-8 -*-
# @Time : 2021/4/24 20:50
# @File : baidu_voice_service.py
# @Author : Rocky C@www.30daydo.com

import os
import requests
import sys
import pickle
sys.path.append('..')
from config import API_KEY,SECRET_KEY
from base64 import b64encode
from pathlib import PurePath

BASE = PurePath(__file__).parent

# 需要识别的文件
AUDIO_FILE = r'C:\OtherGit\speech-demo\rest-api-asr\python\audio\2.m4a' # 只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式
# 文件格式
FORMAT = AUDIO_FILE[-3:] # 文件后缀只支持 pcm/wav/amr 格式，极速版额外支持m4a 格式

CUID = '24057753'
# 采样率
RATE = 16000 # 固定值

ASR_URL = 'http://vop.baidu.com/server_api'

#测试自训练平台需要打开以下信息，自训练平台模型上线后，您会看见第二步：“”获取专属模型参数pid:8001，modelid:1234”，按照这个信息获取 dev_pid=8001，lm_id=1234
'''
http://vop.baidu.com/server_api
1537 普通话(纯中文识别) 输入法模型有标点支持自定义词库
1737 英语英语模型无标点不支持自定义词库
1637 粤语粤语模型有标点不支持自定义词库
1837 四川话四川话模型有标点不支持自定义词库
1936 普通话远场
'''
DEV_PID = 1737

SCOPE = 'brain_enhanced_asr' # 有此scope表示有asr能力，没有请在网页里开通极速版

class DemoError(Exception):
pass

TOKEN_URL = 'http://openapi.baidu.com/oauth/2.0/token'

def fetch_token():

params = {'grant_type': 'client_credentials',
'client_id': API_KEY,
'client_secret': SECRET_KEY}
r = requests.post(
url=TOKEN_URL,
data=params
)

result = r.json()
if ('access_token' in result.keys() and 'scope' in result.keys()):
if SCOPE and (not SCOPE in result['scope'].split(' ')): # SCOPE = False 忽略检查
raise DemoError('scope is not correct')

return result['access_token']

else:
raise DemoError('MAYBE API_KEY or SECRET_KEY not correct: access_token or scope not found in token response')

""" TOKEN end """

def dump_token(token):
with open(os.path.join(BASE,'token.pkl'),'wb') as fp:
pickle.dump({'token':token},fp)

def load_token(filename):

if not os.path.exists(filename):
token=fetch_token()
dump_token(token)
return token
else:
with open(filename,'rb') as fp:
token = pickle.load(fp)
return token['token']

def recognize_service(token,filename):

with open(filename, 'rb') as speech_file:
speech_data = speech_file.read()

length = len(speech_data)
if length == 0:
raise DemoError('file %s length read 0 bytes' % AUDIO_FILE)

b64_data = b64encode(speech_data)
params = {'cuid': CUID, 'token': token, 'dev_pid': DEV_PID,'speech':b64_data,'len':length,'format':FORMAT,'rate':RATE,'channel':1}

headers = {
'Content-Type':'application/json',
}
r = requests.post(url=ASR_URL,json=params,headers=headers)
return r.json()

def main():
filename = 'token.pkl'
token = load_token(filename)
result = recognize_service(token,AUDIO_FILE)
print(result['result'])

if __name__ == '__main__':
main()

百度英语中文语音识别为文字服务 python demo【使用requests重写官方demo】

百度英语中文语音识别为文字服务 python demo【使用requests重写官方demo】

话题描述

相关话题

最佳回复者

1 人关注该话题