From 3dbebfbc9345d2603908f32c0dabebc0ff21feb3 Mon Sep 17 00:00:00 2001 From: srdusr <99972264+srdusr@users.noreply.github.com> Date: Sun, 14 Dec 2025 09:24:00 +0200 Subject: Add community text submissions, and make custom text usable for study Submissions - POST /api/texts proposes a passage; nothing reaches players until a moderator approves it. GET /api/texts serves the approved set, which the client merges on top of its bundled packs at startup. - Validation the server enforces rather than trusts: category from a fixed list, 40 to 600 characters, no control characters (a newline makes a passage untypeable in a single-line input), attribution length, and a unique index on md5(lower(btrim(content))) so the same passage cannot be submitted twice under different whitespace or casing. - Moderation is a flag on users. The queue and the review endpoint both refuse a non-moderator, and reviewing an already-reviewed submission is a 404 rather than a silent second write. - Submissions are rate limited per user: enough for a real contributor, not enough to fill the queue from a script. - A Contribute screen carries the form, your own submissions with their status, and - for moderators only - the review queue. This is the half of TypeRacer's model the packs could not reach by authoring: their corpus is roughly twelve thousand passages, grown by submission. Custom text as a study tool - Imported documents are kept between visits, with how far through each one you are. Custom text lived only in memory, so importing a set of notes and reloading the page lost them - fine for pasting a paragraph to race, useless for working through a file over several sittings. - Position is recorded when a segment is finished, not when the next is started, so closing the tab after a segment does not lose it. - Markdown is chunked as markdown: fenced code blocks are kept whole and typed line by line, and the decoration - hashes, asterisks, backticks, link brackets, table pipes - is stripped so what you retype is the material rather than the punctuation around it. - Everything stays on the device. Notes are not uploaded anywhere. Fixed while doing it: a chunk could contain a newline, which cannot be typed in a single-line input at all. Any paragraph with a line break inside it -- ordinary in notes and in wrapped prose - produced an unfinishable segment. Whitespace inside a chunk is now flattened. --- web/src/customText.js | 85 +++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 83 insertions(+), 2 deletions(-) (limited to 'web/src/customText.js') diff --git a/web/src/customText.js b/web/src/customText.js index dc74541..ca8c98b 100644 --- a/web/src/customText.js +++ b/web/src/customText.js @@ -65,6 +65,12 @@ function splitLong(str, maxLen = 400) { return parts; } +/// Collapses the whitespace inside a chunk. The typing input is a single +/// line, so a chunk containing a newline can never be finished. +function flattenWhitespace(text) { + return text.replace(/\s+/g, ' ').trim(); +} + export function chunkPlainText(raw) { const normalized = raw.replace(/\r\n/g, '\n').trim(); if (!normalized) return []; @@ -73,7 +79,7 @@ export function chunkPlainText(raw) { const chunks = []; for (const block of source) { for (const piece of splitLong(block)) { - const trimmed = piece.trim(); + const trimmed = flattenWhitespace(piece); if (trimmed) chunks.push({ content: trimmed, time: null }); } } @@ -145,11 +151,86 @@ function parseLrc(raw) { return chunks; } -export function parseCustomContent(raw, filename) { + +// Markdown notes are the most common thing someone brings to a typing app to +// study: lecture notes, a cheatsheet, a page of documentation. Typed +// verbatim, most of what you retype is punctuation - hashes, asterisks, +// backticks and link brackets - rather than the material itself. +// +// `strip` removes the decoration and keeps the prose, which is the mode for +// studying what the notes say. Left off, the file is typed exactly as +// written, which is the mode for learning the syntax. +export function chunkMarkdown(raw, { strip = true } = {}) { + const normalized = raw.replace(/\r\n/g, '\n').trim(); + if (!normalized) return []; + + // Fenced code blocks are extracted whole and never stripped: their + // punctuation is the point, and paragraph splitting would cut them apart. + const segments = []; + const fence = /```[^\n]*\n([\s\S]*?)```/g; + let last = 0; + let m; + while ((m = fence.exec(normalized)) !== null) { + if (m.index > last) segments.push({ text: normalized.slice(last, m.index), code: false }); + segments.push({ text: m[1], code: true }); + last = m.index + m[0].length; + } + if (last < normalized.length) segments.push({ text: normalized.slice(last), code: false }); + + const chunks = []; + for (const seg of segments) { + if (seg.code) { + // One line at a time: a code block is typed the way it is written. + for (const line of seg.text.split('\n')) { + const t = line.trim(); + if (t) chunks.push({ content: t, time: null, code: true }); + } + continue; + } + for (const block of seg.text.split(/\n\s*\n/)) { + let text = block.trim(); + if (!text) continue; + if (strip) { + text = text + .replace(/^\s{0,3}#{1,6}\s+/gm, '') // heading markers + .replace(/^\s{0,3}>\s?/gm, '') // block quotes + .replace(/^\s*[-*+]\s+/gm, '') // bullet markers + .replace(/^\s*\d+[.)]\s+/gm, '') // ordered list markers + .replace(/!\[([^\]]*)\]\([^)]*\)/g, '$1') // images -> alt text + .replace(/\[([^\]]+)\]\([^)]*\)/g, '$1') // links -> label + .replace(/`([^`]+)`/g, '$1') // inline code + .replace(/(\*\*|__)(.*?)\1/g, '$2') // bold + .replace(/(\*|_)(.*?)\1/g, '$2') // italics + .replace(/^\s*([-*_]\s*){3,}$/gm, '') // horizontal rules + .replace(/\|/g, ' ') // table pipes + .replace(/[ \t]+/g, ' ') + .trim(); + } + // A heading on its own becomes a one-word chunk that is not worth + // typing; fold it into nothing and let the paragraph follow. + if (!text || text.length < 3) continue; + for (const piece of splitLong(text)) { + const t = flattenWhitespace(piece); + if (t) chunks.push({ content: t, time: null }); + } + } + } + return chunks; +} + +export function parseCustomContent(raw, filename, options = {}) { const ext = (filename || '').split('.').pop().toLowerCase(); if (ext === 'srt') return { chunks: parseSrt(raw), language: null, timed: true }; if (ext === 'vtt') return { chunks: parseVtt(raw), language: null, timed: true }; if (ext === 'lrc') return { chunks: parseLrc(raw), language: null, timed: true }; const language = languageForFilename(filename); + if (ext === 'md' || ext === 'markdown') { + return { + chunks: chunkMarkdown(raw, { strip: options.stripMarkdown !== false }), + language: null, + timed: false, + markdown: true, + }; + } return { chunks: chunkPlainText(raw), language, timed: false }; } -- cgit v1.2.3