diff --git a/packages/api/src/routers/article_router.ts b/packages/api/src/routers/article_router.ts index 027cf91cd..59fa773da 100644 --- a/packages/api/src/routers/article_router.ts +++ b/packages/api/src/routers/article_router.ts @@ -22,28 +22,9 @@ import { getPageById, updatePage } from '../elastic/pages' import { generateDownloadSignedUrl } from '../utils/uploads' import { enqueueTextToSpeech } from '../utils/createTask' import { createPubSubClient } from '../datalayer/pubsub' -import { htmlToSsmlItems, SSMLItem } from '@omnivore/text-to-speech-handler' -import { WordPunctTokenizer } from 'natural' -import { htmlToText } from 'html-to-text' - -interface Utterance { - wordOffset: number - wordCount: number - voice?: string - text: string - idx: number -} - -interface SSMLOutput { - wordCount: number - averageWPM: number - language: string - defaultVoice: string - utterances: Utterance[] -} +import { htmlToSpeechFile } from '@omnivore/text-to-speech-handler' const logger = buildLogger('app.dispatch') -const WORDS_PER_MINUTE = 200 export function articleRouter() { const router = express.Router() @@ -93,16 +74,16 @@ export function articleRouter() { }) router.get( - '/:id/:outputFormat/:priority/:voice?', + '/:id/:outputFormat/:priority?/:voice?/:secondaryVoice?', cors(corsConfig), async (req, res) => { const articleId = req.params.id const outputFormat = req.params.outputFormat - const voice = req.params.voice || 'en-US-JennyNeural' - const priority = req.params.priority + const voice = req.params.voice + const priority = req.params.priority || 'high' if ( !articleId || - !['mp3', 'speech-marks', 'ssml'].includes(outputFormat) || + !['mp3', 'speech-marks', 'speech-file'].includes(outputFormat) || !['low', 'high'].includes(priority) ) { return res.status(400).send('Invalid data') @@ -120,26 +101,17 @@ export function articleRouter() { }, }) - if (outputFormat === 'ssml') { + if (outputFormat === 'speech-file') { const page = await getPageById(articleId) if (!page) { return res.status(404).send('Page not found') } - const ssmlItems = htmlToSsmlItems(page.content, { + const speechFile = htmlToSpeechFile(page.content, { primaryVoice: voice, - secondaryVoice: 'en-US-GuyNeural', - rate: '1', - language: page.language || 'en-US', + secondaryVoice: req.params.secondaryVoice, + language: page.language, }) - const [utterances, wordCount] = ssmlItemsToUtterances(ssmlItems) - const ssmlOutput: SSMLOutput = { - wordCount, - averageWPM: WORDS_PER_MINUTE, - language: page.language || 'en-US', - defaultVoice: voice, - utterances, - } - return res.send(ssmlOutput) + return res.send(speechFile) } const existingSpeech = await getRepository(Speech).findOne({ @@ -219,24 +191,3 @@ const redirectUrl = async (speech: Speech, outputFormat: string) => { return generateDownloadSignedUrl(speech.audioFileName) } } - -const ssmlItemsToUtterances = (items: SSMLItem[]): [Utterance[], number] => { - const tokenizer = new WordPunctTokenizer() - let wordOffset = 0 - return [ - items.map((item) => { - const text = htmlToText(item.textItems.join(''), { wordwrap: false }) - const wordCount = tokenizer.tokenize(text).length - const utterance: Utterance = { - wordOffset, - wordCount, - text, - voice: item.voice, - idx: item.idx, - } - wordOffset += wordCount - return utterance - }), - wordOffset, - ] -} diff --git a/packages/api/src/textToSpeech.d.ts b/packages/api/src/textToSpeech.d.ts index ab955ae1b..051b9be8f 100644 --- a/packages/api/src/textToSpeech.d.ts +++ b/packages/api/src/textToSpeech.d.ts @@ -1,21 +1,29 @@ declare module '@omnivore/text-to-speech-handler' { - export function htmlToSsmlItems( + export function htmlToSpeechFile( html: string, options: SSMLOptions - ): SSMLItem[] + ): SpeechFile export interface SSMLOptions { - primaryVoice: string - secondaryVoice: string - rate: string - language: string + primaryVoice?: string + secondaryVoice?: string + rate?: number + language?: string } - export interface SSMLItem { - open: string - close: string - textItems: string[] + interface Utterance { idx: number + wordOffset: number + wordCount: number voice?: string + text: string + } + + export interface SpeechFile { + wordCount: number + averageWPM: number + language: string + defaultVoice: string + utterances: Utterance[] } } diff --git a/packages/text-to-speech/src/index.ts b/packages/text-to-speech/src/index.ts index a9b42a308..5ca4d54f7 100644 --- a/packages/text-to-speech/src/index.ts +++ b/packages/text-to-speech/src/index.ts @@ -9,11 +9,7 @@ import * as jwt from 'jsonwebtoken' import * as dotenv from 'dotenv' // see https://github.com/motdotla/dotenv#how-do-i-use-dotenv-with-import import { synthesizeTextToSpeech, TextToSpeechInput } from './textToSpeech' import { File, Storage } from '@google-cloud/storage' -import { htmlToSsmlItems } from './htmlToSsml' - -interface SSMLInput { - text: string -} +import { htmlToSpeechFile } from './htmlToSsml' interface UtteranceInput { voice?: string @@ -176,9 +172,9 @@ export const textToSpeechStreamingHandler = Sentry.GCPFunction.wrapHttpFunction( return res.status(500).send({ errorCode: 'SYNTHESIZER_ERROR' }) } res.send({ + idx: utteranceInput.idx, audioData: audioData.toString('hex'), speechMarks, - idx: utteranceInput.idx, }) } catch (e) { console.error('Text to speech streaming error', e) @@ -188,7 +184,7 @@ export const textToSpeechStreamingHandler = Sentry.GCPFunction.wrapHttpFunction( ) module.exports = { - htmlToSsmlItems, + htmlToSpeechFile, textToSpeechStreamingHandler, textToSpeechHandler, }