From b75a0d86da88822428725731fa722d78eac27261 Mon Sep 17 00:00:00 2001 From: dj013 Date: Tue, 11 Aug 2026 17:28:53 +0200 Subject: [PATCH] [events] Strip inline markdown from extracted descriptions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Event descriptions are excerpted from the first body paragraph but consumed as plain text everywhere — cards, the event page, descriptions, OpenGraph, JSON-LD — so markdown survived verbatim: "**C++ Serbia Beer Wednesday**" rendered with literal asterisks on the events list, and seven events showed raw "[text](url)" link syntax. Truncation made it worse by cutting links in half, leaving "([subroutines]( { expect(url).toMatch(/^https?:\/\//); }); }); + +// The extracted description renders as plain text everywhere it is consumed — event +// cards, descriptions, OpenGraph, JSON-LD — so leftover markdown would show +// literally ("**Beer Wednesday**" on a card). Guards the stripInlineMarkdown pass in +// lib/events-server.ts against new markdown constructs appearing in event bodies. +describe("Event descriptions are plain text", () => { + const events = getAllEventsServer(); + + it("extracts a description for every event", () => { + expect(events.length).toBeGreaterThan(0); + }); + + it.each(events.map((e) => ({ slug: e.slug, description: e.description })))( + "$slug — description carries no markdown markup", + ({ slug, description }) => { + expect( + description, + `${slug}: description still contains markdown markup.\n Value: ${description}\n Hint: extend stripInlineMarkdown in lib/strip-markdown.ts` + ).not.toMatch(/\*\*|__|\[[^\]]+\]\(|!\[|~~|, OpenGraph, JSON-LD), so + // inline markdown comes off here — and before truncation, which could otherwise + // cut a link in half. + let description = stripInlineMarkdown(eventHeader.description || ""); if (!description && content) { // Extract first meaningful paragraph from content const lines = content.split("\n"); @@ -159,7 +163,8 @@ function parseEventFile(fileName: string): Event | null { !trimmed.startsWith("-") && trimmed.length > 50 ) { - description = trimmed.substring(0, 200) + (trimmed.length > 200 ? "..." : ""); + const plain = stripInlineMarkdown(trimmed); + description = plain.substring(0, 200) + (plain.length > 200 ? "..." : ""); break; } } diff --git a/lib/strip-markdown.test.ts b/lib/strip-markdown.test.ts new file mode 100644 index 0000000..9e5eaef --- /dev/null +++ b/lib/strip-markdown.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it } from "vitest"; + +import { stripInlineMarkdown } from "./strip-markdown"; + +describe("stripInlineMarkdown", () => { + it("leaves plain text untouched", () => { + expect(stripInlineMarkdown("Pozivamo vas na okupljanje.")).toBe("Pozivamo vas na okupljanje."); + }); + + it("unwraps bold", () => { + expect(stripInlineMarkdown("Pozivamo vas na prvi **C++ Serbia Beer Wednesday**!")).toBe( + "Pozivamo vas na prvi C++ Serbia Beer Wednesday!" + ); + expect(stripInlineMarkdown("a __bold__ word")).toBe("a bold word"); + }); + + it("unwraps italics with either marker", () => { + expect(stripInlineMarkdown("an *italic* and an _italic_ word")).toBe( + "an italic and an italic word" + ); + }); + + it("unwraps bold italics — the visible *** case", () => { + expect(stripInlineMarkdown("this is ***important*** here")).toBe("this is important here"); + }); + + it("keeps intraword underscores — identifiers are not emphasis", () => { + expect(stripInlineMarkdown("std::lower_bound and upper_bound differ")).toBe( + "std::lower_bound and upper_bound differ" + ); + }); + + it("reduces links to their text", () => { + expect( + stripInlineMarkdown( + "susreli sa [templejt metaprogramiranjem](https://en.wikipedia.org/wiki/Template_metaprogramming)" + ) + ).toBe("susreli sa templejt metaprogramiranjem"); + }); + + it("handles angle-bracketed link destinations with parentheses in the URL", () => { + expect( + stripInlineMarkdown( + "generalizacija podrutina ([subroutines]())" + ) + ).toBe("generalizacija podrutina (subroutines)"); + }); + + it("collapses images to their alt text", () => { + expect(stripInlineMarkdown("before ![a banner](https://example.org/x.png) after")).toBe( + "before a banner after" + ); + }); + + it("unwraps inline code", () => { + expect(stripInlineMarkdown("call `co_await` here")).toBe("call co_await here"); + }); + + it("bares autolinks", () => { + expect(stripInlineMarkdown("see for more")).toBe( + "see https://cppserbia.org for more" + ); + }); + + it("unwraps strikethrough", () => { + expect(stripInlineMarkdown("it is ~~cancelled~~ rescheduled")).toBe( + "it is cancelled rescheduled" + ); + }); + + it("strips markup nested inside a link", () => { + expect(stripInlineMarkdown("[**bold link**](https://example.org)")).toBe("bold link"); + }); +}); diff --git a/lib/strip-markdown.ts b/lib/strip-markdown.ts new file mode 100644 index 0000000..93fc178 --- /dev/null +++ b/lib/strip-markdown.ts @@ -0,0 +1,31 @@ +/** + * Reduces inline Markdown to plain text: `[text](url)` becomes `text`, emphasis and + * code markers are removed, images collapse to their alt text. + * + * Event descriptions are excerpted from Markdown bodies but consumed as plain text + * everywhere — cards, `` descriptions, OpenGraph, JSON-LD — so the markup has + * to come off at extraction time, before the excerpt is truncated. Stripping first + * also keeps truncation from cutting a link in half and leaving `[text](https://…` + * in the output. + * + * Inline constructs only: block-level Markdown (headings, list markers, tables) + * never reaches the extractor, which walks paragraph lines. + */ +export function stripInlineMarkdown(text: string): string { + return ( + text + // Images first so the link pass doesn't eat `![alt](url)` as a link + .replace(/!\[([^\]]*)\]\((?:<[^>]*>|[^)]*)\)/g, "$1") + // `` destinations exist to protect parentheses inside the URL + .replace(/\[([^\]]+)\]\((?:<[^>]*>|[^)]*)\)/g, "$1") + .replace(/<(https?:\/\/[^>\s]+)>/g, "$1") + // `***text***` sheds the bold pair here and the italic pair below + .replace(/(\*\*|__)(?=\S)([^*_](?:.*?\S)?)\1/g, "$2") + .replace(/~~(?=\S)((?:[^~]*\S)?)~~/g, "$1") + .replace(/\*(?=\S)((?:[^*]*\S)?)\*/g, "$1") + // Word-boundary guards keep identifiers like `lower_bound` intact + .replace(/(?