百度語音合成api/sdk及demo

1.流程web

  1)換取tokenjson

    用Api Key 和 SecretKey。訪問https://openapi.baidu.com/oauth/2.0/token 換取 token   api

// appKey = Va5yQRHl********LT0vuXV4
// appSecret = 0rDSjzQ20XUj5i********PQSzr5pVw2

https://openapi.baidu.com/oauth/2.0/token?grant_type=client_credentials&client_id=Va5yQRHl********LT0vuXV4&client_secret=0rDSjzQ20XUj5i********PQSzr5pVw2

  返回:服務器

{
    "access_token": "1.a6b7dbd428f731035f771b8d********.86400.1292922000-2346678-124328",
    "expires_in": 2592000,
    "refresh_token": "2.385d55f8615fdfd9edb7c4b********.604800.1293440400-2346678-124328",
    "scope": "public audio_tts_post ...",
    "session_key": "ANXxSNjwQDugf8615Onqeik********CdlLxn",
    "session_secret": "248APxvxjCZ0VEC********aK4oZExMB",
}
scope中含有audio_tts_post 表示有語音合成能力,沒有該audio_tts_post 的token調用接口會返回502錯誤。 在結果中能夠看見 token = 1.a6b7dbd428f731035f771b8d********.86400.1292922000-2346678-124328,在2592000秒(30天)後過時

  2)訪問合成接口session

#從上文中咱們獲取的token是 1.a6b7dbd428f731035f771b8d********.86400.1292922000-2346678-124328
http://tsn.baidu.com/text2audio?lan=zh&ctp=1&cuid=abcdxxx&tok=1.a6b7dbd428f731035f771b8d****.86400.1292922000-2346678-124328&tex=%e7%99%be%e5%ba%a6%e4%bd%a0%e5%a5%bd&vol=9&per=0&spd=5&pit=5&aue=3
// 這是一個正常MP3的下載url
// tex在實際開發過程當中請urlencode2次
"""
tex 必填 合成的文本,使用UTF-8編碼。小於2048箇中文字或者英文數字。(文本在百度服務器內轉換爲GBK後,長度必須小於4096字節)
tok 必填 開放平臺獲取到的開發者access_token(見上面的「鑑權認證機制」段落)
cuid 必填 用戶惟一標識,用來計算UV值。建議填寫能區分用戶的機器 MAC 地址或 IMEI 碼,長度爲60字符之內
ctp 必填 客戶端類型選擇,web端填寫固定值1
lan 必填 固定值zh。語言選擇,目前只有中英文混合模式,填寫固定值zh
spd 選填 語速,取值0-15,默認爲5中語速
pit 選填 音調,取值0-15,默認爲5中語調
vol 選填 音量,取值0-15,默認爲5中音量
per 選填 發音人選擇, 0爲普通女聲,1爲普通男生,3爲情感合成-度逍遙,4爲情感合成-度丫丫,默認爲普通女聲
aue 選填 3爲mp3格式(默認); 4爲pcm-16k;5爲pcm-8k;6爲wav(內容同pcm-16k); 注意aue=4或者6是語音識別要求的格式,可是音頻內容不是語音識別要求的天然人發音,因此識別效果會受影
  • aue =3 ,返回爲二進制mp3文件,具體header信息 Content-Type: audio/mp3;
  • aue =4 ,返回爲二進制pcm文件,具體header信息 Content-Type:audio/basic;codec=pcm;rate=16000;channel=1
  • aue =5 ,返回爲二進制pcm文件,具體header信息 Content-Type:audio/basic;codec=pcm;rate=8000;channel=1
  • aue =6 ,返回爲二進制wav文件,具體header信息 Content-Type: audio/wav;

 

 """app

2.api 代碼demopost

# coding=utf-8
import sys
import json

IS_PY3 = sys.version_info.major == 3
if IS_PY3:
    from urllib.request import urlopen
    from urllib.request import Request
    from urllib.error import URLError
    from urllib.parse import urlencode
    from urllib.parse import quote_plus
else:
    import urllib2
    from urllib import quote_plus
    from urllib2 import urlopen
    from urllib2 import Request
    from urllib2 import URLError
    from urllib import urlencode

API_KEY = '4E1BG9lTnlSeIf1NQFlrSq6h'
SECRET_KEY = '544ca4657ba8002e3dea3ac2f5fdd241'

TEXT = "歡迎使用百度語音合成。"

# 發音人選擇, 0爲普通女聲,1爲普通男生,3爲情感合成-度逍遙,4爲情感合成-度丫丫,默認爲普通女聲
PER = 4
# 語速,取值0-15,默認爲5中語速
SPD = 5
# 音調,取值0-15,默認爲5中語調
PIT = 5
# 音量,取值0-9,默認爲5中音量
VOL = 5
# 下載的文件格式, 3:mp3(default) 4: pcm-16k 5: pcm-8k 6. wav
AUE = 3

FORMATS = {3: "mp3", 4: "pcm", 5: "pcm", 6: "wav"}
FORMAT = FORMATS[AUE]

CUID = "123456PYTHON"

TTS_URL = 'http://tsn.baidu.com/text2audio'


class DemoError(Exception):
    pass


"""  TOKEN start """

TOKEN_URL = 'http://openapi.baidu.com/oauth/2.0/token'
SCOPE = 'audio_tts_post'  # 有此scope表示有tts能力,沒有請在網頁裏勾選


def fetch_token():
    print("fetch token begin")
    params = {'grant_type': 'client_credentials',
              'client_id': API_KEY,
              'client_secret': SECRET_KEY}
    post_data = urlencode(params)
    if (IS_PY3):
        post_data = post_data.encode('utf-8')
    req = Request(TOKEN_URL, post_data)
    try:
        f = urlopen(req, timeout=5)
        result_str = f.read()
    except URLError as err:
        print('token http response http code : ' + str(err.code))
        result_str = err.read()
    if (IS_PY3):
        result_str = result_str.decode()

    print(result_str)
    result = json.loads(result_str)
    print(result)
    if ('access_token' in result.keys() and 'scope' in result.keys()):
        if not SCOPE in result['scope'].split(' '):
            raise DemoError('scope is not correct')
        print('SUCCESS WITH TOKEN: %s ; EXPIRES IN SECONDS: %s' % (result['access_token'], result['expires_in']))
        return result['access_token']
    else:
        raise DemoError('MAYBE API_KEY or SECRET_KEY not correct: access_token or scope not found in token response')


"""  TOKEN end """

if __name__ == '__main__':
    token = fetch_token()
    tex = quote_plus(TEXT)  # 此處TEXT須要兩次urlencode
    print(tex)
    params = {'tok': token, 'tex': tex, 'per': PER, 'spd': SPD, 'pit': PIT, 'vol': VOL, 'aue': AUE, 'cuid': CUID,
              'lan': 'zh', 'ctp': 1}  # lan ctp 固定參數

    data = urlencode(params)
    print('test on Web Browser' + TTS_URL + '?' + data)

    req = Request(TTS_URL, data.encode('utf-8'))
    has_error = False
    try:
        f = urlopen(req)
        result_str = f.read()

        headers = dict((name.lower(), value) for name, value in f.headers.items())

        has_error = ('content-type' not in headers.keys() or headers['content-type'].find('audio/') < 0)
    except  URLError as err:
        print('asr http response http code : ' + str(err.code))
        result_str = err.read()
        has_error = True

    save_file = "error.txt" if has_error else 'result.' + FORMAT
    with open(save_file, 'wb') as of:
        of.write(result_str)

    if has_error:
        if (IS_PY3):
            result_str = str(result_str, 'utf-8')
        print("tts api  error:" + result_str)

    print("result saved as :" + save_file)

3.sdk下載fetch

pip install baidu-aip

4.sdk demo代碼ui

# coding=utf-8
from aip import AipSpeech

API_KEY = 'DHEl3FqU9oxxx' # 替換成你的api_key 在建立的應用中查找
SECRET_KEY = 'ZWGxEZjxxxxIMx7BeGnoW83u' # 替換成你的secret_key 在建立的應用中查找
APP_ID = '1xx3x3' # 替換成你的app_id
client = AipSpeech(APP_ID, API_KEY, SECRET_KEY)
result = client.synthesis(u'baidu是一家科技公司', 'zh', 1, {
    'vol': 5, 'per': 5, 'spd': 4
})
# vol 控制音量 per 控制男女聲 spd 控制速度
if not isinstance(result, dict):
    with open('auido.mp3', 'wb') as f:
        f.write(result)
相關文章
相關標籤/搜索