diff --git a/.cspell/project-terms.txt b/.cspell/project-terms.txt index 783d1e8..bcaaec1 100644 --- a/.cspell/project-terms.txt +++ b/.cspell/project-terms.txt @@ -3,6 +3,7 @@ # Keep entries lowercase unless the cased form is the one that appears # in source (proper nouns, acronyms). apikey +autolinks citty # Employers named in event frontmatter / tests KDAB @@ -25,8 +26,10 @@ opengraph Schniz urlname vaul +walkthrough VCALENDAR VEVENT webcal winget youtu +intraword diff --git a/.cspell/serbian-terms.txt b/.cspell/serbian-terms.txt index 856c136..62ad55a 100644 --- a/.cspell/serbian-terms.txt +++ b/.cspell/serbian-terms.txt @@ -32,10 +32,13 @@ trifunovic # Common Serbian words (diacritic-less Latin; the config's diacritic # regex catches anything with č/ć/ž/š/đ so those don't need to be here) duboko +generalizacija grupe koda +metaprogramiranjem modernizaciju okupljanje +podrutina poslednji Prebaci Propustili @@ -45,3 +48,7 @@ srpski Srpski uranjanje Vidimo +Pozivamo +prvi +susreli +templejt diff --git a/lib/event-validation.test.ts b/lib/event-validation.test.ts index 3d3e8fb..eaa41f0 100644 --- a/lib/event-validation.test.ts +++ b/lib/event-validation.test.ts @@ -1,9 +1,13 @@ +// @vitest-environment node +// (events-server imports bare `fs`, which the default jsdom environment can't resolve) + import fs from "node:fs"; import path from "node:path"; import matter from "gray-matter"; import { describe, expect, it } from "vitest"; +import { getAllEventsServer } from "./events-server"; import { getSpeaker, type Speaker, SPEAKERS } from "./speakers"; const EVENTS_DIR = path.join(process.cwd(), "events"); @@ -160,3 +164,25 @@ describe("Speaker registry", () => { expect(url).toMatch(/^https?:\/\//); }); }); + +// The extracted description renders as plain text everywhere it is consumed — event +// cards, descriptions, OpenGraph, JSON-LD — so leftover markdown would show +// literally ("**Beer Wednesday**" on a card). Guards the stripInlineMarkdown pass in +// lib/events-server.ts against new markdown constructs appearing in event bodies. +describe("Event descriptions are plain text", () => { + const events = getAllEventsServer(); + + it("extracts a description for every event", () => { + expect(events.length).toBeGreaterThan(0); + }); + + it.each(events.map((e) => ({ slug: e.slug, description: e.description })))( + "$slug — description carries no markdown markup", + ({ slug, description }) => { + expect( + description, + `${slug}: description still contains markdown markup.\n Value: ${description}\n Hint: extend stripInlineMarkdown in lib/strip-markdown.ts` + ).not.toMatch(/\*\*|__|\[[^\]]+\]\(|!\[|~~|, OpenGraph, JSON-LD), so + // inline markdown comes off here — and before truncation, which could otherwise + // cut a link in half. + let description = stripInlineMarkdown(eventHeader.description || ""); if (!description && content) { // Extract first meaningful paragraph from content const lines = content.split("\n"); @@ -159,7 +163,8 @@ function parseEventFile(fileName: string): Event | null { !trimmed.startsWith("-") && trimmed.length > 50 ) { - description = trimmed.substring(0, 200) + (trimmed.length > 200 ? "..." : ""); + const plain = stripInlineMarkdown(trimmed); + description = plain.substring(0, 200) + (plain.length > 200 ? "..." : ""); break; } } diff --git a/lib/strip-markdown.test.ts b/lib/strip-markdown.test.ts new file mode 100644 index 0000000..9e5eaef --- /dev/null +++ b/lib/strip-markdown.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it } from "vitest"; + +import { stripInlineMarkdown } from "./strip-markdown"; + +describe("stripInlineMarkdown", () => { + it("leaves plain text untouched", () => { + expect(stripInlineMarkdown("Pozivamo vas na okupljanje.")).toBe("Pozivamo vas na okupljanje."); + }); + + it("unwraps bold", () => { + expect(stripInlineMarkdown("Pozivamo vas na prvi **C++ Serbia Beer Wednesday**!")).toBe( + "Pozivamo vas na prvi C++ Serbia Beer Wednesday!" + ); + expect(stripInlineMarkdown("a __bold__ word")).toBe("a bold word"); + }); + + it("unwraps italics with either marker", () => { + expect(stripInlineMarkdown("an *italic* and an _italic_ word")).toBe( + "an italic and an italic word" + ); + }); + + it("unwraps bold italics — the visible *** case", () => { + expect(stripInlineMarkdown("this is ***important*** here")).toBe("this is important here"); + }); + + it("keeps intraword underscores — identifiers are not emphasis", () => { + expect(stripInlineMarkdown("std::lower_bound and upper_bound differ")).toBe( + "std::lower_bound and upper_bound differ" + ); + }); + + it("reduces links to their text", () => { + expect( + stripInlineMarkdown( + "susreli sa [templejt metaprogramiranjem](https://en.wikipedia.org/wiki/Template_metaprogramming)" + ) + ).toBe("susreli sa templejt metaprogramiranjem"); + }); + + it("handles angle-bracketed link destinations with parentheses in the URL", () => { + expect( + stripInlineMarkdown( + "generalizacija podrutina ([subroutines]())" + ) + ).toBe("generalizacija podrutina (subroutines)"); + }); + + it("collapses images to their alt text", () => { + expect(stripInlineMarkdown("before ![a banner](https://example.org/x.png) after")).toBe( + "before a banner after" + ); + }); + + it("unwraps inline code", () => { + expect(stripInlineMarkdown("call `co_await` here")).toBe("call co_await here"); + }); + + it("bares autolinks", () => { + expect(stripInlineMarkdown("see for more")).toBe( + "see https://cppserbia.org for more" + ); + }); + + it("unwraps strikethrough", () => { + expect(stripInlineMarkdown("it is ~~cancelled~~ rescheduled")).toBe( + "it is cancelled rescheduled" + ); + }); + + it("strips markup nested inside a link", () => { + expect(stripInlineMarkdown("[**bold link**](https://example.org)")).toBe("bold link"); + }); +}); diff --git a/lib/strip-markdown.ts b/lib/strip-markdown.ts new file mode 100644 index 0000000..93fc178 --- /dev/null +++ b/lib/strip-markdown.ts @@ -0,0 +1,31 @@ +/** + * Reduces inline Markdown to plain text: `[text](url)` becomes `text`, emphasis and + * code markers are removed, images collapse to their alt text. + * + * Event descriptions are excerpted from Markdown bodies but consumed as plain text + * everywhere — cards, `` descriptions, OpenGraph, JSON-LD — so the markup has + * to come off at extraction time, before the excerpt is truncated. Stripping first + * also keeps truncation from cutting a link in half and leaving `[text](https://…` + * in the output. + * + * Inline constructs only: block-level Markdown (headings, list markers, tables) + * never reaches the extractor, which walks paragraph lines. + */ +export function stripInlineMarkdown(text: string): string { + return ( + text + // Images first so the link pass doesn't eat `![alt](url)` as a link + .replace(/!\[([^\]]*)\]\((?:<[^>]*>|[^)]*)\)/g, "$1") + // `` destinations exist to protect parentheses inside the URL + .replace(/\[([^\]]+)\]\((?:<[^>]*>|[^)]*)\)/g, "$1") + .replace(/<(https?:\/\/[^>\s]+)>/g, "$1") + // `***text***` sheds the bold pair here and the italic pair below + .replace(/(\*\*|__)(?=\S)([^*_](?:.*?\S)?)\1/g, "$2") + .replace(/~~(?=\S)((?:[^~]*\S)?)~~/g, "$1") + .replace(/\*(?=\S)((?:[^*]*\S)?)\*/g, "$1") + // Word-boundary guards keep identifiers like `lower_bound` intact + .replace(/(?