srdusr
aboutsummaryrefslogtreecommitdiffstats
path: root/scripts/extract_texts.js
diff options
context:
space:
mode:
authorsrdusr <[email protected]>2025-09-11 21:53:00 +0200
committersrdusr <[email protected]>2025-09-11 21:53:00 +0200
commita726d9f5fb56e1fd7983c5ac806d408ad78daa86 (patch)
tree4129a058e2a9e8fa4e0a51e3ed943b746750c3cf /scripts/extract_texts.js
parent8c95e90fa54db565919ec323818f77e7256812c8 (diff)
downloadtyperpunk-a726d9f5fb56e1fd7983c5ac806d408ad78daa86.tar.gz
typerpunk-a726d9f5fb56e1fd7983c5ac806d408ad78daa86.zip
Add multiplayer bots, typing languages, and rework the UI layout
Multiplayer - Quick match: POST /api/multiplayer/quickmatch returns whichever room is still filling, or opens one. Players never see a room code; joining by code stays for racing specific people. - Bots fill quick-match rooms after a short wait so a new game is never an empty lobby. They only ever join quick-match rooms, never a room opened by code. One or two per room, drawn from separate ~40 and ~80 WPM tiers so two bots are never near each other's pace, and they stall to correct mistakes rather than typing a clean straight line. - Live player count via GET /api/multiplayer/online, shown on the Multiplayer control and under the main menu's Multiplayer button. - Per-racer colours: you are the theme accent, opponents take distinct hues that stay the same from lobby to race. - The countdown no longer holds the room lock for its full three seconds, which is what reset clients mid-countdown. Typing languages - 16 languages for the generated-word modes, each with its own high-frequency vocabulary rather than a translation of the English list. - Picker in the top-right rail; non-English uses its own list at every difficulty tier instead of falling back to English words. Fix UTF-8 accuracy in the game core - update_game_state mixed byte and character counts: total_characters_typed accumulated byte-length deltas while total_correct_characters compared a char index against that byte count. Equal on ASCII, so it went unnoticed; a correctly typed Spanish passage scored 6%. The old byte slicing would also have panicked if an index landed inside a multi-byte character. Rewritten char-based, with regression tests. Programming mode - Replaced prose about programming with real code: 26 syntax-highlighted snippets across JavaScript, Python, Rust, C/Go/Java and shell. Single-line by necessity, since the typing input is a single-line field. Layout and readability - One icon rail arrangement on every screen: Settings/Store under the wordmark, Language/Theme/Friends/Account top-right, Stats/Leaderboard/ Multiplayer bottom-right. - Main menu: mode picker moved out of the Single Player button, which it was notching a divider through and pushing the label off-centre. - Escape returns to the menu, closing any open popover first, and confirms before abandoning a live race. - Split --text-color and --sub-color per theme; they shared one value that measured 3.65:1 against the background, below the 4.5:1 body-text floor. - Semantic colours used in exactly one place each: gold for a personal best, amber for the race countdown and the mobile-result badge. - Passage now sits in the same place on the typing and end screens, and its column is a whole number of characters wide so wrapping cannot leave a permanent gap on the right. - End screen: keystrokes and a correct/wrong/extra/missed split, attribution carried over from the typing screen, and a graph with a separate error axis, axis titles including seconds, and smoothed lines.
Diffstat (limited to 'scripts/extract_texts.js')
-rw-r--r--scripts/extract_texts.js148
1 files changed, 148 insertions, 0 deletions
diff --git a/scripts/extract_texts.js b/scripts/extract_texts.js
new file mode 100644
index 0000000..8e8bdd3
--- /dev/null
+++ b/scripts/extract_texts.js
@@ -0,0 +1,148 @@
+#!/usr/bin/env node
+/*
+ Extract paragraphs from mirrored sites under similar/ to build a large texts.json.
+ - Scans HTML files in similar/play.typeracer.com, similar/monkeytype.com, etc.
+ - Extracts visible text from common content tags, splits into paragraphs, filters by length.
+ - Deduplicates and shuffles, attaches category from source directory, and attribution as the source path.
+ - Writes to repo-root texts.json for both CLI and Web to use.
+*/
+const fs = require('fs');
+const path = require('path');
+
+const ROOT = path.resolve(__dirname, '..');
+const SIMILAR_DIR = path.join(ROOT, 'similar');
+const OUTPUT = path.join(ROOT, 'texts.json');
+
+const CONTENT_TAGS = ['p', 'article', 'main', 'section'];
+
+function findHtmlFiles(dir) {
+ const results = [];
+ let entries;
+ try {
+ entries = fs.readdirSync(dir, { withFileTypes: true });
+ } catch {
+ return results;
+ }
+ for (const entry of entries) {
+ const full = path.join(dir, entry.name);
+ if (entry.isDirectory()) {
+ results.push(...findHtmlFiles(full));
+ } else if (entry.isFile() && entry.name.toLowerCase().endsWith('.html')) {
+ results.push(full);
+ }
+ }
+ return results;
+}
+
+function stripBoilerplateTags(html) {
+ return html.replace(/<(script|style|nav|footer|header|noscript)[^>]*>[\s\S]*?<\/\1>/gi, ' ');
+}
+
+function decodeEntities(text) {
+ return text
+ .replace(/&nbsp;/g, ' ')
+ .replace(/&amp;/g, '&')
+ .replace(/&lt;/g, '<')
+ .replace(/&gt;/g, '>')
+ .replace(/&quot;/g, '"')
+ .replace(/&#39;/g, "'");
+}
+
+function extractTagText(html, tag) {
+ const regex = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, 'gi');
+ const out = [];
+ let match;
+ while ((match = regex.exec(html)) !== null) {
+ const text = decodeEntities(match[1].replace(/<[^>]+>/g, ' '));
+ out.push(text);
+ }
+ return out;
+}
+
+function isLikelyVisibleText(text) {
+ const t = text.replace(/\s+/g, ' ').trim();
+ if (!t) return false;
+ if (t.length < 60) return false; // avoid too-short snippets
+ // avoid nav/footer boilerplate
+ if (/©|copyright|cookie|privacy|terms|policy|subscribe|sign in|login|menu|footer|header/i.test(t)) return false;
+ return true;
+}
+
+function splitIntoParagraphs(text) {
+ const blocks = text
+ .split(/\n\s*\n|\r\n\r\n/)
+ .map(s => s.replace(/\s+/g, ' ').trim())
+ .filter(Boolean);
+ const paras = [];
+ for (const b of blocks) {
+ if (b.length <= 400) {
+ paras.push(b);
+ } else {
+ let start = 0;
+ while (start < b.length) {
+ const end = Math.min(start + 350, b.length);
+ const slice = b.slice(start, end);
+ const lastPeriod = slice.lastIndexOf('. ');
+ const lastComma = slice.lastIndexOf(', ');
+ const cut = lastPeriod > 150 ? lastPeriod + 1 : (lastComma > 150 ? lastComma + 1 : slice.length);
+ paras.push(slice.slice(0, cut).trim());
+ start += cut;
+ }
+ }
+ }
+ return paras;
+}
+
+try {
+ const htmlFiles = findHtmlFiles(SIMILAR_DIR);
+ const items = [];
+ const seen = new Set();
+
+ for (const file of htmlFiles) {
+ const rel = path.relative(SIMILAR_DIR, file);
+ const parts = rel.split(path.sep);
+ const category = parts[0]?.replace(/\W+/g, '').toLowerCase() || 'general';
+ const attribution = `similar/${rel}`;
+
+ const html = stripBoilerplateTags(fs.readFileSync(file, 'utf8'));
+ const textBits = [];
+ for (const tag of CONTENT_TAGS) {
+ for (const text of extractTagText(html, tag)) {
+ if (isLikelyVisibleText(text)) textBits.push(text);
+ }
+ }
+
+ const combined = textBits.join('\n\n');
+ if (!combined.trim()) continue;
+
+ const paras = splitIntoParagraphs(combined)
+ .map(s => s.replace(/\s+/g, ' ').trim())
+ .filter(s => s.length >= 80 && s.length <= 400);
+
+ for (const content of paras) {
+ const key = content.toLowerCase();
+ if (seen.has(key)) continue;
+ seen.add(key);
+ items.push({ category, content, attribution });
+ }
+ }
+
+ // Shuffle
+ for (let i = items.length - 1; i > 0; i--) {
+ const j = Math.floor(Math.random() * (i + 1));
+ [items[i], items[j]] = [items[j], items[i]];
+ }
+
+ // If not enough, keep existing texts.json and merge
+ let existing = [];
+ if (fs.existsSync(OUTPUT)) {
+ try { existing = JSON.parse(fs.readFileSync(OUTPUT, 'utf8')); } catch {}
+ }
+ const merged = [...items, ...existing].slice(0, 5000); // cap to 5k entries
+
+ fs.writeFileSync(OUTPUT, JSON.stringify(merged, null, 2));
+ console.log(`Wrote ${merged.length} texts to ${OUTPUT}`);
+} catch (err) {
+ console.error('extract_texts failed:', err);
+ process.exit(1);
+}