1//! Spelling as OneNote 2010 checks it as you type. Its Proofing defaults decide what is a
2//! word: words in UPPERCASE, words holding numbers, and Internet and file addresses are
3//! left alone, and a word repeating the one before it is marked as repeated. A platform
4//! dictionary checks each paragraph's words on a thread of its own, and results are kept
5//! per paragraph text. Nothing about spelling is stored in the page.
6
7use onestore::page::text::Paragraph;
8use std::collections::{HashMap, HashSet};
9use std::hash::{DefaultHasher, Hash, Hasher};
10use std::ops::Range;
11use std::sync::atomic::{AtomicU32, Ordering};
12use std::sync::{Arc, Mutex, mpsc};
13
14/// Checked paragraphs kept before the cache starts over.
15const KEPT: usize = 8192;
16/// Paragraphs waiting to be checked that one call to the dictionary takes.
17#[cfg(not(target_arch = "wasm32"))]
18const BATCH: usize = 64;
19
20/// A platform spell checker, called from the spelling thread and the host's. Languages are
21/// Windows LCIDs, as runs store them.
22pub trait Dictionary: Send + Sync {
23 /// Whether each word is misspelled in its language; a word in a language no dictionary
24 /// serves is not.
25 fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool>;
26 /// Corrections for `word`, best first.
27 fn suggest(&self, word: &str, language: u32) -> Vec<String>;
28 /// Add to Dictionary: accepts `word` from now on.
29 fn learn(&self, word: &str);
30}
31
32/// Which of a platform's dictionaries, named `en_GB`, `en-GB` or `en`, serves `language`, an
33/// LCID: its locale's, else its language's.
34pub fn pick(language: u32, available: &[String]) -> Option<&str> {
35 let tag = crate::language::tag(language)?.replace('-', "_");
36 let (bare, region) = tag.split_once('_').unwrap_or((&tag, ""));
37 // A bare tag is the locale Windows picks, most often the language's own country.
38 let home = match bare {
39 "en" => "US".to_owned(),
40 "pt" => "BR".to_owned(),
41 _ => bare.to_uppercase(),
42 };
43 let wanted = if region.is_empty() {
44 vec![format!("{bare}_{home}"), bare.to_owned()]
45 } else {
46 vec![tag.clone(), bare.to_owned()]
47 };
48 let name = |dictionary: &String| dictionary.replace('-', "_");
49 wanted
50 .iter()
51 .find_map(|wanted| {
52 available
53 .iter()
54 .find(|dictionary| name(dictionary) == *wanted)
55 })
56 .or_else(|| {
57 available
58 .iter()
59 .find(|dictionary| name(dictionary).starts_with(&format!("{bare}_")))
60 })
61 .map(String::as_str)
62}
63
64/// A marked word: misspelled, or repeating the word before it.
65#[derive(Clone, Debug, PartialEq, Eq)]
66pub(crate) struct Mark {
67 /// Bytes of the paragraph's text.
68 pub(crate) range: Range<usize>,
69 pub(crate) repeated: bool,
70}
71
72/// A word of a paragraph: its bytes, its language, and whether its spelling is checked.
73struct Word {
74 range: Range<usize>,
75 /// The run's language, an LCID; `None` where untagged.
76 language: Option<u32>,
77 checked: bool,
78}
79
80/// Letters, digits, underscores and combining marks, as against spaces, punctuation and
81/// symbols.
82fn word_character(character: char) -> bool {
83 !character.is_whitespace()
84 && !character.is_ascii_punctuation()
85 && !matches!(u32::from(character),
86 0x80..=0xbf | 0xd7 | 0xf7 | 0x2000..=0x2bff | 0x3000..=0x303f | 0xfe30..=0xfe4f
87 | 0xff00..=0xff0f | 0x1f000..=0x1faff)
88 || character == '_'
89}
90
91fn apostrophe(character: char) -> bool {
92 matches!(character, '\'' | '\u{2019}' | '\u{02bc}')
93}
94
95/// An Internet or file address, which OneNote leaves unchecked whole.
96fn address(chunk: &str) -> bool {
97 let lower = chunk.to_lowercase();
98 lower.contains("://")
99 || lower.starts_with("www.")
100 || chunk.contains('\\')
101 || chunk
102 .split_once('@')
103 .is_some_and(|(user, host)| !user.is_empty() && host.contains('.'))
104}
105
106/// The words of `paragraph`'s shown text, in order. Hidden field codes and equations are
107/// not text to check, and a word split by one is two.
108fn words(paragraph: &Paragraph) -> Vec<Word> {
109 let text = paragraph.text();
110 let spans = paragraph.spans();
111 let format = |byte: usize| &spans[spans.partition_point(|span| span.end <= byte)].format;
112 // Runs of shown text between spaces, which formatting changes don't split.
113 let mut chunks = Vec::new();
114 let mut start = None;
115 for (byte, character) in text.char_indices() {
116 let format = format(byte);
117 let apart =
118 character.is_whitespace() || format.hidden == Some(true) || format.math == Some(true);
119 match (apart, start) {
120 (true, Some(from)) => {
121 chunks.push(from..byte);
122 start = None;
123 }
124 (false, None) => start = Some(byte),
125 _ => {}
126 }
127 }
128 chunks.extend(start.map(|from| from..text.len()));
129 let mut words = Vec::new();
130 for chunk in chunks {
131 let at = chunk.start;
132 let chunk = &text[chunk];
133 if address(chunk) {
134 continue;
135 }
136 let mut characters = chunk.char_indices().peekable();
137 while let Some((offset, character)) = characters.next() {
138 if !word_character(character) {
139 continue;
140 }
141 let mut end = offset + character.len_utf8();
142 while let Some(&(next, character)) = characters.peek() {
143 let joined = apostrophe(character)
144 && chunk[next + character.len_utf8()..]
145 .chars()
146 .next()
147 .is_some_and(char::is_alphabetic);
148 if !word_character(character) && !joined {
149 break;
150 }
151 end = next + character.len_utf8();
152 characters.next();
153 }
154 let word = &chunk[offset..end];
155 if word
156 .chars()
157 .any(|character| character.is_numeric() || crate::search::ideographic(character))
158 {
159 continue;
160 }
161 words.push(Word {
162 range: at + offset..at + end,
163 language: format(at + offset).language,
164 checked: word.chars().any(char::is_lowercase),
165 });
166 }
167 }
168 words
169}
170
171/// Whether two words are the same word, as a repeated one is.
172fn same(a: &str, b: &str) -> bool {
173 let fold = |word: &str| {
174 word.chars()
175 .flat_map(char::to_lowercase)
176 .map(|character| {
177 if apostrophe(character) {
178 '\''
179 } else {
180 character
181 }
182 })
183 .collect::<String>()
184 };
185 fold(a) == fold(b)
186}
187
188/// Which of `words` repeat the word before them, with only spaces between.
189fn repeated(text: &str, words: &[Word]) -> Vec<bool> {
190 words
191 .iter()
192 .enumerate()
193 .map(|(index, word)| {
194 index.checked_sub(1).is_some_and(|previous| {
195 let previous = &words[previous].range;
196 text[previous.end..word.range.start]
197 .chars()
198 .all(char::is_whitespace)
199 && same(&text[previous.clone()], &text[word.range.clone()])
200 })
201 })
202 .collect()
203}
204
205/// The marks `dictionary` gives each of `paragraphs`, asking it once: each word repeating
206/// the one before it, and each other checked word it finds misspelled.
207/// Checks `paragraphs`, reading text no run tags as language `untagged`.
208fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary, untagged: u32) -> Vec<Vec<Mark>> {
209 let words: Vec<(Vec<Word>, Vec<bool>)> = paragraphs
210 .iter()
211 .map(|paragraph| {
212 let words = words(paragraph);
213 let repeated = repeated(paragraph.text(), &words);
214 (words, repeated)
215 })
216 .collect();
217 let asked: Vec<(&str, u32)> = paragraphs
218 .iter()
219 .zip(&words)
220 .flat_map(|(paragraph, (words, repeated))| {
221 words
222 .iter()
223 .zip(repeated)
224 .filter(|(word, repeated)| word.checked && !**repeated)
225 .map(|(word, _)| {
226 let language = word.language.unwrap_or(untagged);
227 (&paragraph.text()[word.range.clone()], language)
228 })
229 })
230 .collect();
231 let mut misspelled = dictionary.misspelled(&asked).into_iter();
232 words
233 .into_iter()
234 .map(|(words, repeated)| {
235 words
236 .into_iter()
237 .zip(repeated)
238 .filter_map(|(word, repeated)| {
239 let marked = repeated || word.checked && misspelled.next().unwrap_or(false);
240 marked.then_some(Mark {
241 range: word.range,
242 repeated,
243 })
244 })
245 .collect()
246 })
247 .collect()
248}
249
250/// What a paragraph's marks are kept under: its text and the formatting that decides them.
251fn key(paragraph: &Paragraph) -> u64 {
252 let mut hasher = DefaultHasher::new();
253 paragraph.text().hash(&mut hasher);
254 for span in paragraph.spans() {
255 (
256 span.end,
257 span.format.language,
258 span.format.hidden,
259 span.format.math,
260 )
261 .hash(&mut hasher);
262 }
263 hasher.finish()
264}
265
266struct Checked {
267 text: String,
268 marks: Vec<Mark>,
269}
270
271#[derive(Default)]
272struct State {
273 checked: HashMap<u64, Checked>,
274 pending: HashSet<u64>,
275 /// Words Ignore or Add to Dictionary accepted, misspelled or repeated, left unmarked
276 /// everywhere.
277 accepted: HashSet<String>,
278}
279
280impl State {
281 /// The marks kept for `paragraph` under `key`, less accepted words.
282 fn marks(&self, key: u64, paragraph: &Paragraph) -> Option<Vec<Mark>> {
283 let checked = self
284 .checked
285 .get(&key)
286 .filter(|checked| checked.text == paragraph.text())?;
287 Some(
288 checked
289 .marks
290 .iter()
291 .filter(|mark| {
292 !self
293 .accepted
294 .contains(&paragraph.text()[mark.range.clone()])
295 })
296 .cloned()
297 .collect(),
298 )
299 }
300
301 fn keep(&mut self, key: u64, paragraph: &Paragraph, marks: Vec<Mark>) {
302 if self.checked.len() >= KEPT {
303 self.checked.clear();
304 }
305 self.checked.insert(
306 key,
307 Checked {
308 text: paragraph.text().to_owned(),
309 marks,
310 },
311 );
312 }
313}
314
315struct Shared {
316 dictionary: Box<dyn Dictionary>,
317 /// The language text no run tags is checked in, an LCID: US English, as OneNote's.
318 untagged: AtomicU32,
319 state: Mutex<State>,
320}
321
322/// Spell checking shared by the pages of one app. Paragraphs are checked when first asked
323/// for, on a thread of its own, which wakes `redraw` when marks are ready.
324#[derive(Clone)]
325pub struct Spelling {
326 shared: Arc<Shared>,
327 jobs: mpsc::Sender<(u64, Paragraph)>,
328}
329
330impl Spelling {
331 pub fn new(dictionary: Box<dyn Dictionary>, redraw: std::task::Waker) -> Self {
332 let shared = Arc::new(Shared {
333 dictionary,
334 untagged: AtomicU32::new(crate::language::EN_US),
335 state: Mutex::default(),
336 });
337 let (jobs, receiver) = mpsc::channel::<(u64, Paragraph)>();
338 let worker = Arc::clone(&shared);
339 // The browser gives a page one thread: paragraphs are checked as they are drawn.
340 #[cfg(target_arch = "wasm32")]
341 drop((receiver, worker, redraw));
342 #[cfg(not(target_arch = "wasm32"))]
343 std::thread::Builder::new()
344 .name("spelling".into())
345 .spawn(move || {
346 while let Ok(first) = receiver.recv() {
347 let jobs: Vec<(u64, Paragraph)> = std::iter::once(first)
348 .chain(receiver.try_iter().take(BATCH - 1))
349 .collect();
350 let paragraphs: Vec<&Paragraph> =
351 jobs.iter().map(|(_, paragraph)| paragraph).collect();
352 let untagged = worker.untagged.load(Ordering::Relaxed);
353 let marks = check(&paragraphs, &*worker.dictionary, untagged);
354 let mut state = worker.state.lock().unwrap();
355 for ((key, paragraph), marks) in jobs.iter().zip(marks) {
356 state.pending.remove(key);
357 state.keep(*key, paragraph, marks);
358 }
359 drop(state);
360 redraw.wake_by_ref();
361 }
362 })
363 .expect("the spelling thread starts");
364 Self { shared, jobs }
365 }
366
367 /// The marks on `paragraph` once it is checked, less accepted words; until then none,
368 /// and it is checked.
369 pub(crate) fn marks(&self, paragraph: &Paragraph) -> Vec<Mark> {
370 if cfg!(target_arch = "wasm32") {
371 return self.marks_now(paragraph);
372 }
373 let key = key(paragraph);
374 let mut state = self.shared.state.lock().unwrap();
375 if let Some(marks) = state.marks(key, paragraph) {
376 return marks;
377 }
378 if state.pending.insert(key) {
379 let _ = self.jobs.send((key, paragraph.clone()));
380 }
381 Vec::new()
382 }
383
384 /// The marks on `paragraph`, checked on this thread where not yet checked, as the
385 /// Spelling pane walks the page.
386 pub(crate) fn marks_now(&self, paragraph: &Paragraph) -> Vec<Mark> {
387 let key = key(paragraph);
388 if let Some(marks) = self.shared.state.lock().unwrap().marks(key, paragraph) {
389 return marks;
390 }
391 let untagged = self.shared.untagged.load(Ordering::Relaxed);
392 let marks = check(&[paragraph], &*self.shared.dictionary, untagged).remove(0);
393 let mut state = self.shared.state.lock().unwrap();
394 state.keep(key, paragraph, marks);
395 state.marks(key, paragraph).unwrap_or_default()
396 }
397
398 /// Checks every paragraph again when next asked, as after a dictionary arrived.
399 pub fn recheck(&self) {
400 self.shared.state.lock().unwrap().checked.clear();
401 }
402
403 /// Checks text no run tags in `language`, an LCID, as a browser's text is in its own.
404 pub fn untagged(&self, language: u32) {
405 self.shared.untagged.store(language, Ordering::Relaxed);
406 self.recheck();
407 }
408
409 /// Corrections for `word` in `language`, best first.
410 pub(crate) fn suggest(&self, word: &str, language: Option<u32>) -> Vec<String> {
411 self.shared.dictionary.suggest(
412 word,
413 language.unwrap_or(self.shared.untagged.load(Ordering::Relaxed)),
414 )
415 }
416
417 /// Ignore: leaves `word` unmarked everywhere until the app quits.
418 pub fn ignore(&self, word: &str) {
419 self.shared
420 .state
421 .lock()
422 .unwrap()
423 .accepted
424 .insert(word.to_owned());
425 }
426
427 /// Add to Dictionary: the platform's dictionary accepts `word` from now on.
428 pub fn learn(&self, word: &str) {
429 self.shared.dictionary.learn(word);
430 self.ignore(word);
431 }
432}
433
434#[cfg(test)]
435pub(crate) mod tests {
436 use super::*;
437 use onestore::document::Format;
438
439 /// Knows a few English and French words; English only by default.
440 pub(crate) struct Fake;
441
442 const ENGLISH: &[&str] = &[
443 "this", "sentence", "has", "the", "typos", "word", "don't", "i", "a", "bar", "is", "here",
444 "well", "known", "hello",
445 ];
446 const FRENCH: &[&str] = &["bonjour", "le", "monde", "une", "phrase"];
447
448 impl Dictionary for Fake {
449 fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> {
450 words
451 .iter()
452 .map(|(word, language)| {
453 let known = match language {
454 1033 => ENGLISH,
455 1036 => FRENCH,
456 _ => return false,
457 };
458 !known.contains(&word.to_lowercase().replace('\u{2019}', "'").as_str())
459 })
460 .collect()
461 }
462
463 fn suggest(&self, word: &str, _: u32) -> Vec<String> {
464 match word {
465 "Ths" => vec!["This".into(), "Thus".into()],
466 "sentense" => vec!["sentence".into()],
467 _ => Vec::new(),
468 }
469 }
470
471 fn learn(&self, _: &str) {}
472 }
473
474 #[test]
475 fn languages_pick_their_locale_then_their_language() {
476 let available: Vec<String> = ["en", "en_GB", "fr", "de-AT", "de_DE", "pt_BR"]
477 .map(String::from)
478 .into();
479 for (lcid, picked) in [
480 (1033, Some("en")),
481 (2057, Some("en_GB")),
482 (3081, Some("en")),
483 (3084, Some("fr")),
484 (1031, Some("de_DE")),
485 (3079, Some("de-AT")),
486 (2070, Some("pt_BR")),
487 (1049, None),
488 (0x1007f, None),
489 ] {
490 assert_eq!(pick(lcid, &available), picked, "{lcid}");
491 }
492 let hunspell: Vec<String> = ["en_AU", "en_US", "fr_FR"].map(String::from).into();
493 assert_eq!(pick(1033, &hunspell), Some("en_US"));
494 assert_eq!(pick(1036, &hunspell), Some("fr_FR"));
495 let portuguese: Vec<String> = ["pt_PT", "pt_BR"].map(String::from).into();
496 assert_eq!(pick(1046, &portuguese), Some("pt_BR"));
497 assert_eq!(pick(2070, &portuguese), Some("pt_PT"));
498 }
499
500 fn marked(paragraph: &Paragraph) -> Vec<(&str, bool)> {
501 check(&[paragraph], &Fake, crate::language::EN_US)
502 .remove(0)
503 .into_iter()
504 .map(|mark| (&paragraph.text()[mark.range], mark.repeated))
505 .collect()
506 }
507
508 /// OneNote 2010 under its default Proofing options (`/tmp/snowbound-spelling/lab`).
509 #[test]
510 fn words_are_marked_as_onenote_marks_them() {
511 let paragraph = Paragraph::new(
512 "Ths sentense has the typos. HELLO WRLD NASA abc123 x86 3rd foo_bar \
513 www.exmple.com http://exmple.com/qwrt a@exmple.com C:\\Users\\qwrt \
514 don't dont word word the The don't don\u{2019}t well-knwn (sentense) \
515 \u{65e5}\u{672c}\u{8a9e}"
516 .into(),
517 Format::default(),
518 );
519 assert_eq!(
520 marked(&paragraph),
521 [
522 ("Ths", false),
523 ("sentense", false),
524 ("foo_bar", false),
525 ("dont", false),
526 ("word", true),
527 ("The", true),
528 ("don\u{2019}t", true),
529 ("knwn", false),
530 ("sentense", false),
531 ]
532 );
533 }
534
535 #[test]
536 fn each_run_is_checked_in_its_language_and_hidden_codes_are_not_text() {
537 let french = Format {
538 language: Some(1036),
539 ..Format::default()
540 };
541 let paragraph = Paragraph::from_runs([
542 ("bonjour le monde ".to_owned(), french.clone()),
543 ("bonjjour ".to_owned(), french),
544 (
545 "hello ".to_owned(),
546 Format {
547 language: Some(1061),
548 ..Format::default()
549 },
550 ),
551 (
552 "wo".to_owned(),
553 Format {
554 bold: Some(true),
555 ..Format::default()
556 },
557 ),
558 ("rd ".to_owned(), Format::default()),
559 (
560 "HYPERLINK \"qwrt\"".to_owned(),
561 Format {
562 hidden: Some(true),
563 ..Format::default()
564 },
565 ),
566 ("heere".to_owned(), Format::default()),
567 ]);
568 assert_eq!(marked(&paragraph), [("bonjjour", false), ("heere", false)]);
569 }
570
571 #[test]
572 fn marks_arrive_from_the_thread_and_accepted_words_drop_out() {
573 struct Wake(mpsc::Sender<()>);
574 impl std::task::Wake for Wake {
575 fn wake(self: Arc<Self>) {
576 let _ = self.0.send(());
577 }
578 }
579 let (woken, wakes) = mpsc::channel();
580 let spelling = Spelling::new(Box::new(Fake), Arc::new(Wake(woken)).into());
581 let paragraph = Paragraph::new("Ths sentense here".into(), Format::default());
582 assert!(spelling.marks(&paragraph).is_empty());
583 wakes
584 .recv_timeout(std::time::Duration::from_secs(10))
585 .expect("the spelling thread wakes the host");
586 let words = |spelling: &Spelling| {
587 spelling
588 .marks(&paragraph)
589 .into_iter()
590 .map(|mark| paragraph.text()[mark.range].to_owned())
591 .collect::<Vec<_>>()
592 };
593 assert_eq!(words(&spelling), ["Ths", "sentense"]);
594 spelling.ignore("sentense");
595 assert_eq!(words(&spelling), ["Ths"]);
596 assert_eq!(spelling.suggest("Ths", None), ["This", "Thus"]);
597 // The same text in another language is checked again.
598 let french = Paragraph::new(
599 "Ths sentense here".into(),
600 Format {
601 language: Some(1036),
602 ..Format::default()
603 },
604 );
605 assert!(spelling.marks(&french).is_empty());
606 wakes
607 .recv_timeout(std::time::Duration::from_secs(10))
608 .expect("the spelling thread wakes the host");
609 assert_eq!(spelling.marks(&french).len(), 2);
610 }
611
612 #[test]
613 fn arriving_dictionaries_and_untagged_languages_check_paragraphs_again() {
614 struct Late(Arc<std::sync::atomic::AtomicBool>);
615 impl Dictionary for Late {
616 fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> {
617 match self.0.load(std::sync::atomic::Ordering::Relaxed) {
618 true => Fake.misspelled(words),
619 false => vec![false; words.len()],
620 }
621 }
622 fn suggest(&self, _: &str, _: u32) -> Vec<String> {
623 Vec::new()
624 }
625 fn learn(&self, _: &str) {}
626 }
627 let arrived = Arc::new(std::sync::atomic::AtomicBool::new(false));
628 let spelling = Spelling::new(
629 Box::new(Late(Arc::clone(&arrived))),
630 std::task::Waker::noop().clone(),
631 );
632 let paragraph = Paragraph::new("Ths sentense here".into(), Format::default());
633 assert!(spelling.marks_now(&paragraph).is_empty());
634 arrived.store(true, std::sync::atomic::Ordering::Relaxed);
635 assert!(
636 spelling.marks_now(&paragraph).is_empty(),
637 "kept until asked again"
638 );
639 spelling.recheck();
640 assert_eq!(spelling.marks_now(&paragraph).len(), 2);
641 // Untagged text is US English until the host says otherwise.
642 spelling.untagged(1036);
643 assert_eq!(spelling.marks_now(&paragraph).len(), 3);
644 }
645}