mirror of
https://github.com/bewcloud/bewcloud.git
synced 2026-03-11 08:54:49 +00:00
fix: properly strip HTML tags and resolve entities in feed article summaries (#149)
* fix: properly strip HTML tags and resolve entities in feed article summaries Fixes #146 The parseTextFromHtml function was using document.textContent directly on the parsed HTML document, which could leave raw HTML tags and unresolved entities in feed article summaries. Changes: - Extract text from body element to avoid document wrapper artifacts - Collapse multiple whitespace/newlines into single spaces for cleaner output - Add early return for empty/whitespace-only input - Use optional chaining for safer null handling * fix: preserve single line breaks, only collapse 2+ consecutive whitespace Address review feedback: the previous \s+ regex was too aggressive and broke text-only summaries with legitimate line breaks. Now: - Collapse runs of 2+ non-newline whitespace into a single space - Collapse 3+ consecutive newlines into double newline (paragraph break) - Single line breaks are preserved --------- Co-authored-by: User <user@example.com>
This commit is contained in:
parent
6d5ee7b53c
commit
1aca444b22
1 changed files with 9 additions and 2 deletions
11
lib/feed.ts
11
lib/feed.ts
|
|
@ -222,13 +222,20 @@ export async function getUrlInfo(url: string): Promise<{ title: string; htmlBody
|
|||
}
|
||||
|
||||
export async function parseTextFromHtml(html: string): Promise<string> {
|
||||
let text = '';
|
||||
if (!html || !html.trim()) {
|
||||
return '';
|
||||
}
|
||||
|
||||
await initParser();
|
||||
|
||||
const document = new DOMParser().parseFromString(html, 'text/html');
|
||||
|
||||
text = document!.textContent;
|
||||
// Extract text from body to avoid any artifacts from the document wrapper
|
||||
const text = (document?.querySelector('body')?.textContent || document?.textContent || '')
|
||||
// Collapse runs of 2+ whitespace/newline characters, preserving single line breaks
|
||||
.replace(/[^\S\n]{2,}/g, ' ')
|
||||
.replace(/\n{3,}/g, '\n\n')
|
||||
.trim();
|
||||
|
||||
return text;
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in a new issue