Skip to content

Commit ca85de6

Browse files
committed
translation and stats skipped words review
1 parent 6cff667 commit ca85de6

27 files changed

Lines changed: 5040 additions & 452 deletions

File tree

‎browserbible/js/lib/stopwords.js‎

Lines changed: 88 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,88 @@
1+
// Stop words for word-frequency statistics.
2+
//
3+
// Word lists live in content/stopwords/<iso639-3>.json (one flat array per
4+
// language, served from public/ so they stay out of the JS bundle and new
5+
// languages can be added without code changes). The English list covers
6+
// function words, light verbs, contractions, and KJV-era archaic forms;
7+
// distinctive frequent words (God, Lord, verily, behold, amen) are
8+
// deliberately left out so they stay visible in the stats.
9+
10+
// textInfo.lang is usually a bare ISO 639-3 code but may be suffixed
11+
// ("eng-Latn-US") or, from some providers, a 2-letter ISO 639-1 code.
12+
const LANG_ALIASES = {
13+
en: 'eng', es: 'spa', pt: 'por', fr: 'fra', hi: 'hin',
14+
ar: 'ara', arb: 'ara', ja: 'jpn', ko: 'kor', zh: 'zho', cmn: 'zho'
15+
};
16+
17+
export function normalizeLang(lang) {
18+
if (!lang) return undefined;
19+
const primary = String(lang).toLowerCase().split('-')[0];
20+
if (!/^[a-z]{2,3}$/.test(primary)) return undefined;
21+
return LANG_ALIASES[primary] ?? primary;
22+
}
23+
24+
const cache = new Map();
25+
26+
/**
27+
* Resolve the stop-word Set for a language, or undefined when no list
28+
* exists (missing file, network failure, malformed JSON). Results are
29+
* cached per language; concurrent callers share one fetch.
30+
*/
31+
export function loadStopwords(lang) {
32+
const code = normalizeLang(lang);
33+
if (!code) return Promise.resolve(undefined);
34+
35+
if (!cache.has(code)) {
36+
cache.set(code, fetch(`content/stopwords/${code}.json`)
37+
.then((response) => (response.ok ? response.json() : undefined))
38+
.then((words) => (Array.isArray(words) ? new Set(words) : undefined))
39+
.catch(() => undefined));
40+
}
41+
return cache.get(code);
42+
}
43+
44+
// Unicode letters plus combining marks (Devanagari matras, Arabic diacritics)
45+
// with word-internal straight or curly apostrophes; trailing possessive 's is
46+
// stripped so "God's" counts as "God".
47+
const WORD_PATTERN = /[\p{L}\p{M}]+(?:['’][\p{L}\p{M}]+)*/gu;
48+
49+
// Languages written without spaces, segmented via Intl.Segmenter (which needs
50+
// a BCP-47 locale). Everything else splits on the regex above.
51+
const SEGMENTER_LOCALES = {
52+
zho: 'zh', cmn: 'zh', yue: 'zh', jpn: 'ja', tha: 'th', khm: 'km', lao: 'lo', mya: 'my'
53+
};
54+
const segmenters = new Map();
55+
56+
// Languages where an apostrophe marks elision (l'Éternel, qu'il), so tokens
57+
// split there; in English it marks contractions (don't) and stays internal.
58+
const ELISION_LANGS = new Set(['fra']);
59+
60+
export function tokenizeWords(text, lang) {
61+
if (!text) return [];
62+
const code = normalizeLang(lang);
63+
64+
const locale = SEGMENTER_LOCALES[code];
65+
if (locale && typeof Intl !== 'undefined' && Intl.Segmenter) {
66+
if (!segmenters.has(locale)) {
67+
segmenters.set(locale, new Intl.Segmenter(locale, { granularity: 'word' }));
68+
}
69+
const words = [];
70+
for (const seg of segmenters.get(locale).segment(text)) {
71+
if (seg.isWordLike && /[\p{L}\p{M}]/u.test(seg.segment)) words.push(seg.segment);
72+
}
73+
return words;
74+
}
75+
76+
const matches = text.match(WORD_PATTERN) ?? [];
77+
const tokens = matches.map((token) => token.replace(/['’]s$/i, ''));
78+
if (ELISION_LANGS.has(code)) {
79+
return tokens.flatMap((token) => token.split(/['’]/).filter(Boolean));
80+
}
81+
return tokens;
82+
}
83+
84+
// Canonical form for counting and stop-word lookup: lowercase with curly
85+
// apostrophes normalized, so "Don’t" matches the "don't" list entry.
86+
export function wordKey(token) {
87+
return token.toLowerCase().replace(/’/g, "'");
88+
}

‎browserbible/js/resources/index.js‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -12,5 +12,6 @@ export const AVAILABLE_LANGUAGES = [
1212
'ur', // Urdu
1313
'id', // Indonesian
1414
'de', // German
15-
'ja' // Japanese
15+
'ja', // Japanese
16+
'ko' // Korean
1617
];

‎browserbible/js/windows/StatisticsWindow.js‎

Lines changed: 29 additions & 38 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,7 @@ import { getApp } from '../core/registry.js';
55
import { getText, loadSection } from '../texts/TextLoader.js';
66
import { renderWordCloud } from '../lib/SimpleWordCloud.js';
77
import { escapeRegExp, highlightTextMatches } from '../lib/textHighlighter.js';
8+
import { loadStopwords, tokenizeWords, wordKey } from '../lib/stopwords.js';
89

910
const INIT_DELAY_MS = 1500;
1011
const FONT_SIZE_MIN = 9;
@@ -16,26 +17,6 @@ const GREEK_STOPWORDS = ['G2532', 'G3588', 'G846', 'G1722', 'G1519', 'G1537', 'G
1617
const getTextAsync = (textId) => AsyncHelpers.promisifyWithError(getText, textId);
1718
const loadSectionAsync = (textInfo, sectionId) => AsyncHelpers.promisifyWithError(loadSection, textInfo, sectionId);
1819

19-
const exclusions = {
20-
"es": ["de"],
21-
"chs": ["-", ":", ",", "。", "(", ")", "!", ";", "一", "?"],
22-
"eng": [
23-
"a", "abaft", "aboard", "about", "above", "absent", "across", "afore", "after",
24-
"against", "along", "alongside", "amid", "amidst", "among", "amongst", "an",
25-
"anenst", "apud", "around", "as", "aside", "astride", "at", "athwart", "atop",
26-
"barring", "before", "behind", "below", "beneath", "beside", "besides", "between",
27-
"beyond", "but", "by", "circa", "concerning", "despite", "down", "during", "except",
28-
"excluding", "failing", "following", "for", "forenenst", "from", "given", "in",
29-
"including", "inside", "into", "lest", "like", "minus", "modulo", "near", "next",
30-
"notwithstanding", "of", "off", "on", "onto", "opposite", "out", "outside", "over",
31-
"pace", "past", "per", "plus", "pro", "qua", "regarding", "round", "sans", "save",
32-
"since", "than", "through", "throughout", "till", "to", "toward", "towards", "under",
33-
"underneath", "unlike", "until", "unto", "up", "upon", "versus", "via", "with",
34-
"within", "without", "worth", "the", "him", "his", "he", "she", "it", "her", "hers",
35-
"and", "yet", "that", "was", "were", "be", "being", "been", "had", "its", "i"
36-
]
37-
};
38-
3920
const byCountDescending = (a, b) => b.count - a.count;
4021

4122
function lerp(start, end, min, max, value) {
@@ -56,6 +37,8 @@ class StatisticsWindowComponent extends BaseWindow {
5637
lemmaData: [],
5738
hasLemma: false
5839
};
40+
41+
this._wordIndex = new Map();
5942
}
6043

6144
async render() {
@@ -93,7 +76,7 @@ class StatisticsWindowComponent extends BaseWindow {
9376
this.startProcess(bibleSettings.data.textid, bibleSettings.data.sectionid);
9477
} else {
9578
this.refs.statsMainNode.innerHTML =
96-
'<div class="statistics-empty">Open a Bible window to see statistics for its current chapter.</div>';
79+
`<div class="statistics-empty">${i18n.t('windows.stats.intro')}</div>`;
9780
}
9881
}, INIT_DELAY_MS);
9982
}
@@ -119,6 +102,7 @@ class StatisticsWindowComponent extends BaseWindow {
119102

120103
this.removeHighlights();
121104
this._statsEpoch = (this._statsEpoch ?? 0) + 1;
105+
this._wordIndex = new Map();
122106

123107
Object.assign(this.state, {
124108
textid: tid,
@@ -169,9 +153,19 @@ class StatisticsWindowComponent extends BaseWindow {
169153
}
170154

171155
processLemmaVerse(verse) {
156+
const stopwords = this._stopwords;
157+
172158
verse.querySelectorAll('l[s]').forEach((lemma) => {
173159
const word = lemma.innerHTML;
174160

161+
// Lemma-tagged translations (e.g. ENGWEB) carry surface text in the
162+
// target language; skip occurrences that are entirely stop words.
163+
const tokens = tokenizeWords(lemma.textContent, this.state.textInfo.lang);
164+
if (stopwords && tokens.length > 0 &&
165+
tokens.every((t) => stopwords.has(wordKey(t)))) {
166+
return;
167+
}
168+
175169
for (const strongs of lemma.getAttribute('s').split(' ')) {
176170
if (GREEK_STOPWORDS.includes(strongs)) continue;
177171

@@ -187,26 +181,19 @@ class StatisticsWindowComponent extends BaseWindow {
187181
}
188182

189183
processTextVerse(verse) {
190-
const { lang } = this.state.textInfo;
191-
let verseText = verse.innerHTML.replace(/<.*?>/gi, '');
192-
193-
if (lang.startsWith('en')) {
194-
verseText = verseText.replace(/[^A-Za-z\s]/g, '');
195-
}
196-
197-
const langExclusions = exclusions[lang];
198-
199-
for (const word of verseText.split(' ')) {
200-
if (word === '' || langExclusions?.includes(word.toLowerCase())) continue;
184+
const stopwords = this._stopwords;
201185

202-
const entry = this.state.wordStats.find(
203-
(wi) => wi.word.toLowerCase() === word.toLowerCase()
204-
);
186+
for (const word of tokenizeWords(verse.textContent, this.state.textInfo.lang)) {
187+
const key = wordKey(word);
188+
if (stopwords?.has(key)) continue;
205189

190+
const entry = this._wordIndex.get(key);
206191
if (entry) {
207192
entry.count++;
208193
} else {
209-
this.state.wordStats.push({ word, count: 1 });
194+
const newEntry = { word, count: 1 };
195+
this._wordIndex.set(key, newEntry);
196+
this.state.wordStats.push(newEntry);
210197
}
211198
}
212199
}
@@ -223,11 +210,15 @@ class StatisticsWindowComponent extends BaseWindow {
223210
const wordCloudNode = resultsNode.querySelector('.statistics-wordcloud');
224211

225212
try {
226-
const content = await loadSectionAsync(this.state.textInfo, this.state.sectionid);
213+
const [content, stopwords] = await Promise.all([
214+
loadSectionAsync(this.state.textInfo, this.state.sectionid),
215+
loadStopwords(this.state.textInfo.lang)
216+
]);
227217
if (epoch !== this._statsEpoch) {
228218
resultsNode.remove();
229219
return;
230220
}
221+
this._stopwords = stopwords;
231222

232223
let contentEl = content;
233224
if (typeof content === 'string') {
@@ -329,7 +320,7 @@ class StatisticsWindowComponent extends BaseWindow {
329320

330321
async loadLemmaInfo(epoch) {
331322
const lemmaNodeWrapper = this.createElement(`<div class="statistics-section statistics-rare-words">
332-
<h3>Rare Words</h3>
323+
<h3>${i18n.t('windows.stats.rarewords')}</h3>
333324
<div class="statistics-results loading-indicator"></div>
334325
</div>`);
335326
this.refs.statsMainNode.appendChild(lemmaNodeWrapper);

0 commit comments

Comments
 (0)