mirror of
https://github.com/omnivore-app/omnivore.git
synced 2026-03-11 08:54:26 +00:00
Read SSML file to stream
This commit is contained in:
parent
7baafc96bb
commit
0aa5447b9b
5 changed files with 84 additions and 64 deletions
8
packages/text-to-speech/data/ssml.xml
Normal file
8
packages/text-to-speech/data/ssml.xml
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="en-US">
|
||||
<voice name="en-US-JennyNeural">
|
||||
Good morning!
|
||||
</voice>
|
||||
<voice name="en-US-ChristopherNeural">
|
||||
Good morning to you too Jenny!
|
||||
</voice>
|
||||
</speak>
|
||||
|
|
@ -14,7 +14,8 @@
|
|||
"lint": "eslint src --ext ts,js,tsx,jsx",
|
||||
"compile": "tsc",
|
||||
"build": "tsc",
|
||||
"start": "functions-framework --source=build/src/ --target=textToSpeechHandler",
|
||||
"start": "functions-framework --target=textToSpeechHandler",
|
||||
"start_streaming": "functions-framework --target=textToSpeechStreamingHandler",
|
||||
"dev": "concurrently \"tsc -w\" \"nodemon --watch ./build/ --exec npm run start\"",
|
||||
"gcloud-deploy": "gcloud functions deploy text-to-speech --gen2 --entry-point=textToSpeechHandler --trigger-http --allow-unauthenticated --region=us-west2 --runtime nodejs14",
|
||||
"deploy": "yarn build && yarn gcloud-deploy"
|
||||
|
|
|
|||
|
|
@ -9,9 +9,14 @@ import * as jwt from 'jsonwebtoken'
|
|||
import * as dotenv from 'dotenv' // see https://github.com/motdotla/dotenv#how-do-i-use-dotenv-with-import
|
||||
import { synthesizeTextToSpeech, TextToSpeechInput } from './textToSpeech'
|
||||
import { File, Storage } from '@google-cloud/storage'
|
||||
import { createWriteStream } from 'fs'
|
||||
import { PassThrough } from 'stream'
|
||||
import * as fs from 'fs'
|
||||
|
||||
dotenv.config()
|
||||
Sentry.GCPFunction.init({
|
||||
dsn: process.env.SENTRY_DSN,
|
||||
tracesSampleRate: 0,
|
||||
})
|
||||
|
||||
const storage = new Storage()
|
||||
|
||||
|
|
@ -51,18 +56,13 @@ const updateSpeech = async (
|
|||
return response.status === 200
|
||||
}
|
||||
|
||||
Sentry.GCPFunction.init({
|
||||
dsn: process.env.SENTRY_DSN,
|
||||
tracesSampleRate: 0,
|
||||
})
|
||||
|
||||
export const textToSpeechHandler = Sentry.GCPFunction.wrapHttpFunction(
|
||||
async (req, res) => {
|
||||
console.debug('New text to speech request', req)
|
||||
const token = req.query.token as string
|
||||
if (!process.env.JWT_SECRET) {
|
||||
console.error('JWT_SECRET not exists')
|
||||
return res.status(500).send('JWT_SECRET not exists')
|
||||
return res.status(500).send({ errorCodes: 'JWT_SECRET_NOT_EXISTS' })
|
||||
}
|
||||
try {
|
||||
jwt.verify(token, process.env.JWT_SECRET)
|
||||
|
|
@ -71,25 +71,31 @@ export const textToSpeechHandler = Sentry.GCPFunction.wrapHttpFunction(
|
|||
return res.status(200).send('UNAUTHENTICATED')
|
||||
}
|
||||
const input = req.body as TextToSpeechInput
|
||||
const id = input.id
|
||||
const bucket = input.bucket
|
||||
if (!id || !bucket) {
|
||||
return res.status(200).send('Invalid data')
|
||||
}
|
||||
try {
|
||||
const audioFileName = `speech/${input.id}.mp3`
|
||||
const audioFile = createGCSFile(input.bucket, audioFileName)
|
||||
const audioFileName = `speech/${id}.mp3`
|
||||
const audioFile = createGCSFile(bucket, audioFileName)
|
||||
const writeStream = audioFile.createWriteStream({
|
||||
resumable: true,
|
||||
})
|
||||
const { speechMarks } = await synthesizeTextToSpeech({
|
||||
...input,
|
||||
textType: 'html',
|
||||
writeStream,
|
||||
})
|
||||
// upload Speech Marks file to GCS
|
||||
const speechMarksFileName = `speech/${input.id}.json`
|
||||
const speechMarksFileName = `speech/${id}.json`
|
||||
await uploadToBucket(
|
||||
speechMarksFileName,
|
||||
Buffer.from(JSON.stringify(speechMarks)),
|
||||
input.bucket
|
||||
bucket
|
||||
)
|
||||
const updated = await updateSpeech(
|
||||
input.id,
|
||||
id,
|
||||
token,
|
||||
'COMPLETED',
|
||||
audioFileName,
|
||||
|
|
@ -97,12 +103,12 @@ export const textToSpeechHandler = Sentry.GCPFunction.wrapHttpFunction(
|
|||
)
|
||||
|
||||
if (!updated) {
|
||||
return res.status(500).send('Failed to update speech')
|
||||
return res.status(500).send({ errorCodes: 'DB_ERROR' })
|
||||
}
|
||||
} catch (e) {
|
||||
console.error(e)
|
||||
await updateSpeech(input.id, token, 'FAILED')
|
||||
return res.status(500).send('Failed to synthesize')
|
||||
console.error('Text to speech cloud function error', e)
|
||||
await updateSpeech(id, token, 'FAILED')
|
||||
return res.status(500).send({ errorCodes: 'SYNTHESIZER_ERROR' })
|
||||
}
|
||||
|
||||
res.send('OK')
|
||||
|
|
@ -125,12 +131,10 @@ export const textToSpeechStreamingHandler = Sentry.GCPFunction.wrapHttpFunction(
|
|||
}
|
||||
|
||||
try {
|
||||
const audioFileName = `./tmp/speech-${Date.now()}.mp3`
|
||||
const writeStream = createWriteStream(audioFileName)
|
||||
const ssml = fs.readFileSync('./data/ssml.xml', 'utf8')
|
||||
const writeStream = new PassThrough()
|
||||
const input: TextToSpeechInput = {
|
||||
id: req.query.id as string,
|
||||
text: 'text',
|
||||
bucket: req.query.bucket as string,
|
||||
text: ssml,
|
||||
textType: 'ssml',
|
||||
writeStream,
|
||||
}
|
||||
|
|
@ -142,8 +146,8 @@ export const textToSpeechStreamingHandler = Sentry.GCPFunction.wrapHttpFunction(
|
|||
})
|
||||
writeStream.pipe(res)
|
||||
} catch (e) {
|
||||
console.error(e)
|
||||
return res.status(500).send('Failed to synthesize')
|
||||
console.error('Text to speech streaming error', e)
|
||||
return res.status(500).send({ errorCodes: 'SYNTHESIZER_ERROR' })
|
||||
}
|
||||
}
|
||||
)
|
||||
|
|
|
|||
|
|
@ -10,15 +10,15 @@ import {
|
|||
import { htmlToSsml, ssmlItemText } from './htmlToSsml'
|
||||
|
||||
export interface TextToSpeechInput {
|
||||
id: string
|
||||
id?: string
|
||||
text: string
|
||||
voice?: string
|
||||
languageCode?: string
|
||||
textType?: 'text' | 'ssml'
|
||||
textType?: 'html' | 'ssml'
|
||||
rate?: number
|
||||
volume?: number
|
||||
complimentaryVoice?: string
|
||||
bucket: string
|
||||
bucket?: string
|
||||
writeStream: NodeJS.WritableStream
|
||||
}
|
||||
|
||||
|
|
@ -45,7 +45,7 @@ export const synthesizeTextToSpeech = async (
|
|||
process.env.AZURE_SPEECH_KEY,
|
||||
process.env.AZURE_SPEECH_REGION
|
||||
)
|
||||
const textType = input.textType || 'text'
|
||||
const textType = input.textType || 'html'
|
||||
speechConfig.speechSynthesisOutputFormat =
|
||||
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3
|
||||
|
||||
|
|
@ -53,12 +53,14 @@ export const synthesizeTextToSpeech = async (
|
|||
const synthesizer = new SpeechSynthesizer(speechConfig)
|
||||
const speechMarks: SpeechMark[] = []
|
||||
let timeOffset = 0
|
||||
const characterOffset = 0
|
||||
|
||||
synthesizer.synthesizing = function (s, e) {
|
||||
// convert arrayBuffer to stream and write to gcs file
|
||||
writeStream.write(Buffer.from(e.result.audioData))
|
||||
}
|
||||
// synthesizer.synthesizing = function (s, e) {
|
||||
// // convert arrayBuffer to stream and write to stream
|
||||
// console.debug(
|
||||
// `(synthesizing): Audio length: ${e.result.audioData.byteLength}`
|
||||
// )
|
||||
// writeStream.write(Buffer.from(e.result.audioData))
|
||||
// }
|
||||
|
||||
// The event synthesis completed signals that the synthesis is completed.
|
||||
synthesizer.synthesisCompleted = (s, e) => {
|
||||
|
|
@ -83,15 +85,20 @@ export const synthesizeTextToSpeech = async (
|
|||
if (cancellationDetails.reason === CancellationReason.Error) {
|
||||
str += ': ' + e.result.errorDetails
|
||||
}
|
||||
console.info(str)
|
||||
console.error(str)
|
||||
}
|
||||
|
||||
// The unit of e.audioOffset is tick (1 tick = 100 nanoseconds), divide by 10,000 to convert to milliseconds.
|
||||
synthesizer.wordBoundary = (s, e) => {
|
||||
console.debug(
|
||||
`(word boundary) Audio offset: ${e.audioOffset / 10000}ms, text: ${
|
||||
e.text
|
||||
}`
|
||||
)
|
||||
speechMarks.push({
|
||||
word: e.text,
|
||||
time: (timeOffset + e.audioOffset) / 10000,
|
||||
start: characterOffset + e.textOffset,
|
||||
start: e.textOffset,
|
||||
length: e.wordLength,
|
||||
type: 'word',
|
||||
})
|
||||
|
|
@ -99,7 +106,7 @@ export const synthesizeTextToSpeech = async (
|
|||
|
||||
synthesizer.bookmarkReached = (s, e) => {
|
||||
console.debug(
|
||||
`(Bookmark reached), Audio offset: ${
|
||||
`(bookmark reached) Audio offset: ${
|
||||
e.audioOffset / 10000
|
||||
}ms, bookmark text: ${e.text}`
|
||||
)
|
||||
|
|
@ -111,12 +118,14 @@ export const synthesizeTextToSpeech = async (
|
|||
}
|
||||
|
||||
const speakSsmlAsyncPromise = (
|
||||
text: string
|
||||
ssml: string,
|
||||
writeStream: NodeJS.WritableStream
|
||||
): Promise<SpeechSynthesisResult> => {
|
||||
return new Promise((resolve, reject) => {
|
||||
synthesizer.speakSsmlAsync(
|
||||
text,
|
||||
ssml,
|
||||
(result) => {
|
||||
writeStream.write(Buffer.from(result.audioData))
|
||||
resolve(result)
|
||||
},
|
||||
(error) => {
|
||||
|
|
@ -126,36 +135,34 @@ export const synthesizeTextToSpeech = async (
|
|||
})
|
||||
}
|
||||
|
||||
if (textType === 'text') {
|
||||
const ssmlItems = htmlToSsml(input.text, {
|
||||
primaryVoice: input.voice || 'en-US-JennyNeural',
|
||||
secondaryVoice: 'en-US-GuyNeural',
|
||||
language: input.languageCode || 'en-US',
|
||||
rate: '1',
|
||||
})
|
||||
try {
|
||||
if (textType === 'html') {
|
||||
const ssmlItems = htmlToSsml(input.text, {
|
||||
primaryVoice: input.voice || 'en-US-JennyNeural',
|
||||
secondaryVoice: input.complimentaryVoice || 'en-US-GuyNeural',
|
||||
language: input.languageCode || 'en-US',
|
||||
rate: '1',
|
||||
})
|
||||
|
||||
for (const ssmlItem of Array.from(ssmlItems)) {
|
||||
const ssml = ssmlItemText(ssmlItem)
|
||||
console.debug(`synthesizing ${ssml}`)
|
||||
const result = await speakSsmlAsyncPromise(ssml)
|
||||
if (result.reason === ResultReason.Canceled) {
|
||||
writeStream.end()
|
||||
synthesizer.close()
|
||||
throw new Error(result.errorDetails)
|
||||
for (const ssmlItem of Array.from(ssmlItems)) {
|
||||
const ssml = ssmlItemText(ssmlItem)
|
||||
console.debug('start synthesizing', ssml)
|
||||
const result = await speakSsmlAsyncPromise(ssml, writeStream)
|
||||
timeOffset = timeOffset + result.audioDuration
|
||||
}
|
||||
timeOffset = timeOffset + result.audioDuration
|
||||
// characterOffset = characterOffset + htmlElement.innerText.length
|
||||
}
|
||||
} else {
|
||||
const result = await speakSsmlAsyncPromise(input.text)
|
||||
if (result.reason === ResultReason.Canceled) {
|
||||
writeStream.end()
|
||||
synthesizer.close()
|
||||
throw new Error(result.errorDetails)
|
||||
} else {
|
||||
console.debug('start synthesizing', input.text)
|
||||
await speakSsmlAsyncPromise(input.text, writeStream)
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('synthesis error', error)
|
||||
throw error
|
||||
} finally {
|
||||
console.debug('closing synthesizer')
|
||||
writeStream.end()
|
||||
synthesizer.close()
|
||||
console.debug('synthesizer closed')
|
||||
}
|
||||
writeStream.end()
|
||||
synthesizer.close()
|
||||
|
||||
return {
|
||||
speechMarks,
|
||||
|
|
|
|||
Loading…
Reference in a new issue