diff --git a/app/utils/string.test.ts b/app/utils/string.test.ts index 34c57ce94..61c97ac2c 100644 --- a/app/utils/string.test.ts +++ b/app/utils/string.test.ts @@ -1,5 +1,9 @@ import { describe, expect, test } from "vitest"; -import { pathnameFromPotentialURL, truncateBySentence } from "./strings"; +import { + pathnameFromPotentialURL, + removeMarkdown, + truncateBySentence, +} from "./strings"; describe("pathnameFromPotentialURL()", () => { test("Resolves path name from valid URL", () => { @@ -44,3 +48,30 @@ describe("truncateBySentence()", () => { expect(truncateBySentence(text, 20)).toBe("First line"); }); }); + +describe("removeMarkdown()", () => { + test("Decodes   entities and collapses runs", () => { + const text = "     Global Gauntlet is an event"; + expect(removeMarkdown(text)).toBe("Global Gauntlet is an event"); + }); + + test("Decodes common named HTML entities", () => { + expect(removeMarkdown("Tom & Jerry <3 "hi"")).toBe( + 'Tom & Jerry <3 "hi"', + ); + }); + + test("Decodes numeric HTML entities", () => { + expect(removeMarkdown("café & tea")).toBe("café & tea"); + }); + + test("Leaves unknown named entities untouched", () => { + expect(removeMarkdown("AT&T &fakeentity; rules")).toBe( + "AT&T &fakeentity; rules", + ); + }); + + test("Strips HTML tags and markdown emphasis", () => { + expect(removeMarkdown("

Hello **world**!

")).toBe("Hello world!"); + }); +}); diff --git a/app/utils/strings.ts b/app/utils/strings.ts index 60aac7e53..c38d5e409 100644 --- a/app/utils/strings.ts +++ b/app/utils/strings.ts @@ -76,12 +76,35 @@ export function truncateBySentence(value: string, max: number) { } // based on https://github.com/zuchka/remove-markdown +const NAMED_HTML_ENTITIES: Record = { + nbsp: " ", + amp: "&", + lt: "<", + gt: ">", + quot: '"', + apos: "'", +}; + export function removeMarkdown(value: string) { const htmlReplaceRegex = /<[^>]*>/g; return ( value // Remove HTML tags .replace(htmlReplaceRegex, "") + // Decode named HTML entities (e.g.  , &) + .replace(/&([a-zA-Z]+);/g, (match, name: string) => { + const replacement = NAMED_HTML_ENTITIES[name.toLowerCase()]; + return replacement ?? match; + }) + // Decode numeric HTML entities (e.g.   or  ) + .replace(/&#(x?[0-9a-fA-F]+);/g, (_, code: string) => { + const codePoint = code.startsWith("x") + ? Number.parseInt(code.slice(1), 16) + : Number.parseInt(code, 10); + return Number.isFinite(codePoint) + ? String.fromCodePoint(codePoint) + : ""; + }) // Remove setext-style headers .replace(/^[=-]{2,}\s*$/g, "") // Remove footnotes? @@ -113,5 +136,8 @@ export function removeMarkdown(value: string) { // .replace(/(\S+)\n\s*(\S+)/g, '$1 $2') // Replace strike through .replace(/~(.*?)~/g, "$1") + // Collapse runs of whitespace (e.g. from decoded   or stripped tags) + .replace(/[ \t ]{2,}/g, " ") + .trim() ); }