diff --git a/packages/text-to-speech/src/htmlToSsml.ts b/packages/text-to-speech/src/htmlToSsml.ts
index c74a463ea..e8e379a8b 100644
--- a/packages/text-to-speech/src/htmlToSsml.ts
+++ b/packages/text-to-speech/src/htmlToSsml.ts
@@ -5,6 +5,7 @@ import {
SentenceTokenizerNew,
WordPunctTokenizer,
} from 'natural'
+import { isElement } from 'underscore'
// this code needs to be kept in sync with the
// frontend code in: useReadingProgressAnchor
@@ -49,6 +50,7 @@ const DEFAULT_LANGUAGE = 'en-US'
const DEFAULT_VOICE = 'en-US-JennyNeural'
const DEFAULT_SECONDARY_VOICE = 'en-US-GuyNeural'
const DEFAULT_RATE = '1.1'
+const ELEMENT_INDEX_ATTRIBUTE = 'data-omnivore-anchor-idx'
const ANCHOR_ELEMENTS_BLOCKED_ATTRIBUTES = [
'omnivore-highlight-id',
@@ -84,6 +86,9 @@ const SKIP_TAGS = [
'CODE',
]
+const getAnchorIndex = (element: Element): number =>
+ Number(element.getAttribute(ELEMENT_INDEX_ATTRIBUTE))
+
function parseDomTree(pageNode: Element) {
if (!pageNode || pageNode.childNodes.length == 0) {
console.log('no child nodes found')
@@ -118,7 +123,7 @@ function parseDomTree(pageNode: Element) {
visitedNodeList.forEach((node, index) => {
// We start at index 3, because the frontend starts two nodes above us
// on the #readability-page-1 element that wraps the entire content.
- node.setAttribute('data-omnivore-anchor-idx', (index + 3).toString())
+ node.setAttribute(ELEMENT_INDEX_ATTRIBUTE, (index + 3).toString())
})
return visitedNodeList
}
@@ -157,7 +162,7 @@ function emitElement(
isTopLevel: boolean
) {
const topLevelTags = ssmlTagsForTopLevelElement()
- const idx = element.getAttribute('data-omnivore-anchor-idx')
+ const idx = getAnchorIndex(element)
let maxVisitedIdx = Number(idx)
if (isTopLevel) {
@@ -166,6 +171,11 @@ function emitElement(
for (const child of Array.from(element.childNodes)) {
if (SKIP_TAGS.indexOf(child.nodeName) >= 0) {
+ // Skip unwanted tags and update the index
+ if (isElement(child)) {
+ const childIdx = getAnchorIndex(child)
+ maxVisitedIdx = Math.max(maxVisitedIdx, childIdx)
+ }
continue
}
@@ -180,8 +190,8 @@ function emitElement(
}
emitTextNode(textItems, cleanedText, child)
}
- if (child.nodeType == 1 /* Node.ELEMENT_NODE */) {
- maxVisitedIdx = emitElement(textItems, child as HTMLElement, false)
+ if (isElement(child)) {
+ maxVisitedIdx = emitElement(textItems, child, false)
if (child.nodeName === 'LI') {
// add a new line after each list item
emit(textItems, '\n')
@@ -265,6 +275,11 @@ export const stripEmojis = (text: string): string => {
return text.replace(emojiRegex, '').replace(/\s+/g, ' ')
}
+const filterUtterances = (utterances: Utterance[]): Utterance[] => {
+ const punctuationOrSpaceOnly = /^[\s.,;:!?]+$/
+ return utterances.filter((u) => !punctuationOrSpaceOnly.test(u.text))
+}
+
const textToUtterances = ({
wordTokenizer,
idx,
@@ -309,7 +324,10 @@ const textToUtterances = ({
const sentenceTokenizer = new SentenceTokenizerNew()
sentences = sentenceTokenizer.tokenize(text)
} catch (err) {
- console.debug('Unable to tokenize sentences')
+ console.log(
+ 'Unable to tokenize sentences, falling back to old tokenizer',
+ text
+ )
// fallback to old sentence tokenizer
const sentenceTokenizer = new SentenceTokenizer()
sentences = sentenceTokenizer.tokenize(text)
@@ -371,15 +389,13 @@ const replaceSmartQuotes = (text: string): string => {
// get the max idx of the element and its children
const getMaxVisitedIdx = (element: Element): number => {
- let maxVisitedIdx = Number(element.getAttribute('data-omnivore-anchor-idx'))
+ let maxVisitedIdx = getAnchorIndex(element)
for (const child of Array.from(element.childNodes)) {
- if (child.nodeType === 1) {
- maxVisitedIdx = Math.max(
- maxVisitedIdx,
- getMaxVisitedIdx(child as Element)
- )
+ if (isElement(child)) {
+ maxVisitedIdx = Math.max(maxVisitedIdx, getMaxVisitedIdx(child))
}
}
+
return maxVisitedIdx
}
@@ -461,10 +477,12 @@ export const htmlToSpeechFile = (htmlInput: HtmlInput): SpeechFile => {
}
}
+ const filteredUtterances = filterUtterances(utterances)
+
return {
wordCount: wordOffset,
language,
defaultVoice,
- utterances,
+ utterances: filteredUtterances,
}
}