From a8617f260595bf9950943872bd96855ef9a8df18 Mon Sep 17 00:00:00 2001 From: Hongbo Wu Date: Fri, 26 Aug 2022 14:54:12 +0800 Subject: [PATCH] Synthesize and upload in cloud function --- packages/text-to-speech/package.json | 5 +- packages/text-to-speech/src/index.ts | 385 ++++++++++++++++++++++++++- yarn.lock | 63 ++++- 3 files changed, 447 insertions(+), 6 deletions(-) diff --git a/packages/text-to-speech/package.json b/packages/text-to-speech/package.json index d55567b35..14ccdd9e3 100644 --- a/packages/text-to-speech/package.json +++ b/packages/text-to-speech/package.json @@ -26,8 +26,11 @@ "dependencies": { "@google-cloud/functions-framework": "3.1.2", "@google-cloud/pubsub": "^2.18.4", + "@google-cloud/storage": "^6.4.1", "@sentry/serverless": "^6.16.1", "axios": "^0.27.2", - "jsonwebtoken": "^8.5.1" + "jsonwebtoken": "^8.5.1", + "linkedom": "^0.14.12", + "microsoft-cognitiveservices-speech-sdk": "^1.22.0" } } diff --git a/packages/text-to-speech/src/index.ts b/packages/text-to-speech/src/index.ts index 06317208e..b10172f92 100644 --- a/packages/text-to-speech/src/index.ts +++ b/packages/text-to-speech/src/index.ts @@ -4,11 +4,390 @@ /* eslint-disable @typescript-eslint/no-unused-vars */ import * as Sentry from '@sentry/serverless' -import parseHeaders from 'parse-headers' +import { parseHTML } from 'linkedom' +import { File, Storage } from '@google-cloud/storage' +import { + CancellationDetails, + CancellationReason, + ResultReason, + SpeechConfig, + SpeechSynthesisOutputFormat, + SpeechSynthesisResult, + SpeechSynthesizer, +} from 'microsoft-cognitiveservices-speech-sdk' +import { PubSub } from '@google-cloud/pubsub' + +interface TextToSpeechInput { + id: string + text: string + voice?: string + languageCode?: string + textType?: 'text' | 'ssml' + rate?: number + volume?: number + complimentaryVoice?: string + bucket: string +} + +interface TextToSpeechOutput { + audioFileName: string + speechMarksFileName: string +} + +interface SpeechMark { + time: number + start?: number + length?: number + word: string + type: 'word' | 'bookmark' +} + +const storage = new Storage() +const pubsub = new PubSub() +const SPEECH_UPDATE_TOPIC = 'speech-update' + +const uploadToBucket = async ( + filePath: string, + data: Buffer, + bucket: string, + options?: { contentType?: string; public?: boolean } +): Promise => { + await storage.bucket(bucket).file(filePath).save(data, options) +} + +const createGCSFile = (bucket: string, filename: string): File => { + return storage.bucket(bucket).file(filename) +} + +const updateSpeech = ( + speechId: string, + audioFileName: string, + speechMarksFileName: string +): Promise => { + return pubsub + .topic(SPEECH_UPDATE_TOPIC) + .publishMessage({ json: { speechId, audioFileName, speechMarksFileName } }) + .catch((err) => { + console.error('error publishing speech update:', err) + return undefined + }) +} + +const synthesizeTextToSpeech = async ( + input: TextToSpeechInput +): Promise => { + if (!process.env.azureSpeechKey || !process.env.azureSpeechRegion) { + throw new Error('Azure Speech Key or Region not set') + } + const audioFileName = `speech/${input.id}.mp3` + const audioFile = createGCSFile(input.bucket, audioFileName) + const writeStream = audioFile.createWriteStream({ + resumable: true, + }) + const speechConfig = SpeechConfig.fromSubscription( + process.env.azureSpeechKey, + process.env.azureSpeechRegion + ) + const textType = input.textType || 'text' + if (textType === 'text') { + speechConfig.speechSynthesisLanguage = input.languageCode || 'en-US' + speechConfig.speechSynthesisVoiceName = input.voice || 'en-US-JennyNeural' + } + speechConfig.speechSynthesisOutputFormat = + SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3 + + // Create the speech synthesizer. + const synthesizer = new SpeechSynthesizer(speechConfig) + const speechMarks: SpeechMark[] = [] + let timeOffset = 0 + let characterOffset = 0 + + synthesizer.synthesizing = function (s, e) { + // convert arrayBuffer to stream and write to gcs file + writeStream.write(Buffer.from(e.result.audioData)) + } + + // The event synthesis completed signals that the synthesis is completed. + synthesizer.synthesisCompleted = (s, e) => { + console.info( + `(synthesized) Reason: ${ResultReason[e.result.reason]} Audio length: ${ + e.result.audioData.byteLength + }` + ) + } + + // The synthesis started event signals that the synthesis is started. + synthesizer.synthesisStarted = (s, e) => { + console.info('(synthesis started)') + } + + // The event signals that the service has stopped processing speech. + // This can happen when an error is encountered. + synthesizer.SynthesisCanceled = (s, e) => { + const cancellationDetails = CancellationDetails.fromResult(e.result) + let str = + '(cancel) Reason: ' + CancellationReason[cancellationDetails.reason] + if (cancellationDetails.reason === CancellationReason.Error) { + str += ': ' + e.result.errorDetails + } + console.info(str) + } + + // The unit of e.audioOffset is tick (1 tick = 100 nanoseconds), divide by 10,000 to convert to milliseconds. + synthesizer.wordBoundary = (s, e) => { + speechMarks.push({ + word: e.text, + time: (timeOffset + e.audioOffset) / 10000, + start: characterOffset + e.textOffset, + length: e.wordLength, + type: 'word', + }) + } + + synthesizer.bookmarkReached = (s, e) => { + console.debug( + `(Bookmark reached), Audio offset: ${ + e.audioOffset / 10000 + }ms, bookmark text: ${e.text}` + ) + speechMarks.push({ + word: e.text, + time: (timeOffset + e.audioOffset) / 10000, + type: 'bookmark', + }) + } + + const speakTextAsyncPromise = ( + text: string + ): Promise => { + return new Promise((resolve, reject) => { + synthesizer.speakTextAsync( + text, + (result) => { + resolve(result) + }, + (error) => { + reject(error) + } + ) + }) + } + + const speakSsmlAsyncPromise = ( + text: string + ): Promise => { + return new Promise((resolve, reject) => { + synthesizer.speakSsmlAsync( + text, + (result) => { + resolve(result) + }, + (error) => { + reject(error) + } + ) + }) + } + + if (textType === 'text') { + // slice the text into chunks of 5,000 characters + let currentTextChunk = '' + const textChunks = input.text.split('\n') + for (let i = 0; i < textChunks.length; i++) { + currentTextChunk += textChunks[i] + '\n' + if (currentTextChunk.length < 5000 && i < textChunks.length - 1) { + continue + } + console.debug(`synthesizing ${currentTextChunk}`) + const result = await speakTextAsyncPromise(currentTextChunk) + timeOffset = timeOffset + result.audioDuration + characterOffset = characterOffset + currentTextChunk.length + currentTextChunk = '' + } + } else { + const document = parseHTML(input.text).document + const elements = document.querySelectorAll( + 'h1, h2, h3, p, ul, ol, blockquote' + ) + // convert html elements to the ssml document + for (const e of Array.from(elements)) { + const htmlElement = e as HTMLElement + if (htmlElement.innerText) { + // use complimentary voice for blockquote, hardcoded for now + const voice = + htmlElement.tagName.toLowerCase() === 'blockquote' + ? input.complimentaryVoice || 'en-US-AriaNeural' + : input.voice + const ssml = htmlElementToSsml({ + htmlElement: e, + language: input.languageCode, + rate: input.rate, + volume: input.volume, + voice, + }) + console.debug(`synthesizing ${ssml}`) + const result = await speakSsmlAsyncPromise(ssml) + // if (result.reason === ResultReason.Canceled) { + // synthesizer.close() + // throw new Error(result.errorDetails) + // } + timeOffset = timeOffset + result.audioDuration + // characterOffset = characterOffset + htmlElement.innerText.length + } + } + } + writeStream.end() + synthesizer.close() + + console.debug(`audio file: ${audioFileName}`) + + // upload Speech Marks file to GCS + const speechMarksFileName = `speech/${input.id}.json` + await uploadToBucket( + speechMarksFileName, + Buffer.from(JSON.stringify(speechMarks)), + input.bucket + ) + + return { + audioFileName, + speechMarksFileName, + } +} + +const htmlElementToSsml = ({ + htmlElement, + language = 'en-US', + voice = 'en-US-JennyNeural', + rate = 1, + volume = 100, +}: { + htmlElement: Element + language?: string + voice?: string + rate?: number + volume?: number +}): string => { + const replaceElement = (newElement: Element, oldElement: Element) => { + const id = oldElement.getAttribute('data-omnivore-anchor-idx') + if (id) { + const e = htmlElement.querySelector(`[data-omnivore-anchor-idx="${id}"]`) + e?.parentNode?.replaceChild(newElement, e) + } + } + + const appendBookmarkElement = (parent: Element, element: Element) => { + const id = element.getAttribute('data-omnivore-anchor-idx') + if (id) { + const bookMark = ssml.createElement('bookmark') + bookMark.setAttribute('mark', `data-omnivore-anchor-idx-${id}`) + parent.appendChild(bookMark) + } + } + + const replaceWithEmphasis = (element: Element, level: string) => { + const parent = ssml.createDocumentFragment() as unknown as Element + appendBookmarkElement(parent, element) + const emphasisElement = ssml.createElement('emphasis') + emphasisElement.setAttribute('level', level) + emphasisElement.innerHTML = element.innerHTML.trim() + parent.appendChild(emphasisElement) + replaceElement(parent, element) + } + + const replaceWithSentence = (element: Element) => { + const parent = ssml.createDocumentFragment() as unknown as Element + appendBookmarkElement(parent, element) + const sentenceElement = ssml.createElement('s') + sentenceElement.innerHTML = element.innerHTML.trim() + parent.appendChild(sentenceElement) + replaceElement(parent, element) + } + + // create new ssml document + const ssml = parseHTML('').document + const speakElement = ssml.createElement('speak') + speakElement.setAttribute('version', '1.0') + speakElement.setAttribute('xmlns', 'http://www.w3.org/2001/10/synthesis') + speakElement.setAttribute('xml:lang', language) + const voiceElement = ssml.createElement('voice') + voiceElement.setAttribute('name', voice) + speakElement.appendChild(voiceElement) + const prosodyElement = ssml.createElement('prosody') + prosodyElement.setAttribute('rate', `${rate}`) + prosodyElement.setAttribute('volume', volume.toString()) + voiceElement.appendChild(prosodyElement) + // add each paragraph to the ssml document + appendBookmarkElement(prosodyElement, htmlElement) + // replace emphasis elements with ssml + htmlElement.querySelectorAll('*').forEach((e) => { + switch (e.tagName.toLowerCase()) { + case 's': + replaceWithEmphasis(e, 'moderate') + break + case 'sub': + if (e.getAttribute('alias') === null) { + replaceWithEmphasis(e, 'moderate') + } + break + case 'i': + case 'em': + case 'q': + case 'blockquote': + case 'cite': + case 'del': + case 'strike': + case 'sup': + case 'summary': + case 'caption': + case 'figcaption': + replaceWithEmphasis(e, 'moderate') + break + case 'b': + case 'strong': + case 'dt': + case 'dfn': + case 'u': + case 'mark': + case 'th': + case 'title': + case 'var': + replaceWithEmphasis(e, 'moderate') + break + case 'li': + replaceWithSentence(e) + break + default: { + const parent = ssml.createDocumentFragment() as unknown as Element + appendBookmarkElement(parent, e) + const text = (e as HTMLElement).innerText.trim() + const textElement = ssml.createTextNode(text) + parent.appendChild(textElement) + replaceElement(parent, e) + } + } + }) + prosodyElement.appendChild(htmlElement) + + return speakElement.outerHTML.replace(/ |\n/g, '') +} export const textToSpeechHandler = Sentry.GCPFunction.wrapHttpFunction( async (req, res) => { - console.log('new request', req) - res.send('OK') + console.debug('New text to speech request', req) + const input = req.body as TextToSpeechInput + const { audioFileName, speechMarksFileName } = await synthesizeTextToSpeech( + input + ) + const result = await updateSpeech( + input.id, + audioFileName, + speechMarksFileName + ) + + if (!result) { + return res.status(500).send('Failed to update speech') + } + res.send('OK') } ) diff --git a/yarn.lock b/yarn.lock index e03b59cf2..b8a67aeb9 100644 --- a/yarn.lock +++ b/yarn.lock @@ -2458,7 +2458,7 @@ resolved "https://registry.yarnpkg.com/@google-cloud/opentelemetry-resource-util/-/opentelemetry-resource-util-1.1.0.tgz#0bd1fe708ba27288f6efc9712fbd3705fd325540" integrity sha512-AXfQiqIxeespEYcRNaotC05ddiy2Vgk2yqY73b7Hl1UoJ75Gt4kSRcswrVn18eoDI0YQkSTBh7Ye9ugfFLN5HA== -"@google-cloud/paginator@^3.0.0", "@google-cloud/paginator@^3.0.6": +"@google-cloud/paginator@^3.0.0", "@google-cloud/paginator@^3.0.6", "@google-cloud/paginator@^3.0.7": version "3.0.7" resolved "https://registry.yarnpkg.com/@google-cloud/paginator/-/paginator-3.0.7.tgz#fb6f8e24ec841f99defaebf62c75c2e744dd419b" integrity sha512-jJNutk0arIQhmpUUQJPJErsojqo834KcyB6X7a1mxuic8i1tKXxde8E69IZxNZawRIlZdIK2QY4WALvlK5MzYQ== @@ -2548,6 +2548,30 @@ stream-events "^1.0.4" xdg-basedir "^4.0.0" +"@google-cloud/storage@^6.4.1": + version "6.4.1" + resolved "https://registry.yarnpkg.com/@google-cloud/storage/-/storage-6.4.1.tgz#83334150d4e224cb48691de4d7f9c38e143a0970" + integrity sha512-lAddmRJ8tvxPykUqJfONBQA5XGwGk0vut1POXublc64+nCdB5aQMxwuBMf7J1zubx19QGpYPQwW6wR7YTWrvLw== + dependencies: + "@google-cloud/paginator" "^3.0.7" + "@google-cloud/projectify" "^3.0.0" + "@google-cloud/promisify" "^3.0.0" + abort-controller "^3.0.0" + arrify "^2.0.0" + async-retry "^1.3.3" + compressible "^2.0.12" + duplexify "^4.0.0" + ent "^2.2.0" + extend "^3.0.2" + gaxios "^5.0.0" + google-auth-library "^8.0.1" + mime "^3.0.0" + mime-types "^2.0.8" + p-limit "^3.0.1" + retry-request "^5.0.0" + teeny-request "^8.0.0" + uuid "^8.0.0" + "@google-cloud/tasks@^2.3.0": version "2.5.0" resolved "https://registry.yarnpkg.com/@google-cloud/tasks/-/tasks-2.5.0.tgz#e6c2598038001550c408845e91570d176c18a25a" @@ -14122,7 +14146,7 @@ gaxios@^4.0.0: is-stream "^2.0.0" node-fetch "^2.3.0" -gaxios@^5.0.0: +gaxios@^5.0.0, gaxios@^5.0.1: version "5.0.1" resolved "https://registry.yarnpkg.com/gaxios/-/gaxios-5.0.1.tgz#50fc76a2d04bc1700ed8c3ff1561e52255dfc6e0" integrity sha512-keK47BGKHyyOVQxgcUaSaFvr3ehZYAlvhvpHXy0YB2itzZef+GqZR8TBsfVRWghdwlKrYsn+8L8i3eblF7Oviw== @@ -14512,6 +14536,21 @@ google-auth-library@^7.0.0, google-auth-library@^7.6.1, google-auth-library@^7.9 jws "^4.0.0" lru-cache "^6.0.0" +google-auth-library@^8.0.1: + version "8.4.0" + resolved "https://registry.yarnpkg.com/google-auth-library/-/google-auth-library-8.4.0.tgz#3a5414344bb313ee64ceeef1f7e5162cc1fdf04b" + integrity sha512-cg/usxyQEmq4PPDBQRt+kGIrfL3k+mOrAoS9Xv1hitQL66AoY7iWvRBcYo3Rb0w4V1t9e/GqW2/D4honlAtMDg== + dependencies: + arrify "^2.0.0" + base64-js "^1.3.0" + ecdsa-sig-formatter "^1.0.11" + fast-text-encoding "^1.0.0" + gaxios "^5.0.0" + gcp-metadata "^5.0.0" + gtoken "^6.1.0" + jws "^4.0.0" + lru-cache "^6.0.0" + google-auth-library@^8.0.2: version "8.1.0" resolved "https://registry.yarnpkg.com/google-auth-library/-/google-auth-library-8.1.0.tgz#879e8d2e90a9d47e6eab32fd1d5fd9ed52d7d441" @@ -14744,6 +14783,15 @@ gtoken@^6.0.0: google-p12-pem "^4.0.0" jws "^4.0.0" +gtoken@^6.1.0: + version "6.1.1" + resolved "https://registry.yarnpkg.com/gtoken/-/gtoken-6.1.1.tgz#29ebf3e6893719176d180f5694f1cad525ce3c04" + integrity sha512-HPM4VzzPEGxjQ7T2xLrdSYBs+h1c0yHAUiN+8RHPDoiZbndlpg9Sx3SjWcrTt9+N3FHsSABEpjvdQVan5AAuZQ== + dependencies: + gaxios "^5.0.1" + google-p12-pem "^4.0.0" + jws "^4.0.0" + gzip-size@^6.0.0: version "6.0.0" resolved "https://registry.yarnpkg.com/gzip-size/-/gzip-size-6.0.0.tgz#065367fd50c239c0671cbcbad5be3e2eeb10e462" @@ -17490,6 +17538,17 @@ lines-and-columns@^1.1.6: resolved "https://registry.yarnpkg.com/lines-and-columns/-/lines-and-columns-1.1.6.tgz#1c00c743b433cd0a4e80758f7b64a57440d9ff00" integrity sha1-HADHQ7QzzQpOgHWPe2SldEDZ/wA= +linkedom@^0.14.12: + version "0.14.12" + resolved "https://registry.yarnpkg.com/linkedom/-/linkedom-0.14.12.tgz#3b19442e41de33a9ef9b035ccdd97bf5b66c77e1" + integrity sha512-8uw8LZifCwyWeVWr80T79sQTMmNXt4Da7oN5yH5gTXRqQM+TuZWJyBqRMcIp32zx/f8anHNHyil9Avw9y76ziQ== + dependencies: + css-select "^5.1.0" + cssom "^0.5.0" + html-escaper "^3.0.3" + htmlparser2 "^8.0.1" + uhyphen "^0.1.0" + linkedom@^0.14.9: version "0.14.9" resolved "https://registry.yarnpkg.com/linkedom/-/linkedom-0.14.9.tgz#34c6f15eddc809406f42d8ee48cd30b0222eccb0"