| 1 | //! Spelling as OneNote 2010 checks it as you type. Its Proofing defaults decide what is a |
| 2 | //! word: words in UPPERCASE, words holding numbers, and Internet and file addresses are |
| 3 | //! left alone, and a word repeating the one before it is marked as repeated. A platform |
| 4 | //! dictionary checks each paragraph's words on a thread of its own, and results are kept |
| 5 | //! per paragraph text. Nothing about spelling is stored in the page. |
| 6 | |
| 7 | use onestore::page::text::Paragraph; |
| 8 | use std::collections::{HashMap, HashSet}; |
| 9 | use std::hash::{DefaultHasher, Hash, Hasher}; |
| 10 | use std::ops::Range; |
| 11 | use std::sync::atomic::{AtomicU32, Ordering}; |
| 12 | use std::sync::{Arc, Mutex, mpsc}; |
| 13 | |
| 14 | /// Checked paragraphs kept before the cache starts over. |
| 15 | const KEPT: usize = 8192; |
| 16 | /// Paragraphs waiting to be checked that one call to the dictionary takes. |
| 17 | #[cfg(not(target_arch = "wasm32"))] |
| 18 | const BATCH: usize = 64; |
| 19 | |
| 20 | /// A platform spell checker, called from the spelling thread and the host's. Languages are |
| 21 | /// Windows LCIDs, as runs store them. |
| 22 | pub trait Dictionary: Send + Sync { |
| 23 | /// Whether each word is misspelled in its language; a word in a language no dictionary |
| 24 | /// serves is not. |
| 25 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool>; |
| 26 | /// Corrections for `word`, best first. |
| 27 | fn suggest(&self, word: &str, language: u32) -> Vec<String>; |
| 28 | /// Add to Dictionary: accepts `word` from now on. |
| 29 | fn learn(&self, word: &str); |
| 30 | } |
| 31 | |
| 32 | /// Which of a platform's dictionaries, named `en_GB`, `en-GB` or `en`, serves `language`, an |
| 33 | /// LCID: its locale's, else its language's. |
| 34 | pub fn pick(language: u32, available: &[String]) -> Option<&str> { |
| 35 | let tag = crate::language::tag(language)?.replace('-', "_"); |
| 36 | let (bare, region) = tag.split_once('_').unwrap_or((&tag, "")); |
| 37 | // A bare tag is the locale Windows picks, most often the language's own country. |
| 38 | let home = match bare { |
| 39 | "en" => "US".to_owned(), |
| 40 | "pt" => "BR".to_owned(), |
| 41 | _ => bare.to_uppercase(), |
| 42 | }; |
| 43 | let wanted = if region.is_empty() { |
| 44 | vec![format!("{bare}_{home}"), bare.to_owned()] |
| 45 | } else { |
| 46 | vec![tag.clone(), bare.to_owned()] |
| 47 | }; |
| 48 | let name = |dictionary: &String| dictionary.replace('-', "_"); |
| 49 | wanted |
| 50 | .iter() |
| 51 | .find_map(|wanted| { |
| 52 | available |
| 53 | .iter() |
| 54 | .find(|dictionary| name(dictionary) == *wanted) |
| 55 | }) |
| 56 | .or_else(|| { |
| 57 | available |
| 58 | .iter() |
| 59 | .find(|dictionary| name(dictionary).starts_with(&format!("{bare}_"))) |
| 60 | }) |
| 61 | .map(String::as_str) |
| 62 | } |
| 63 | |
| 64 | /// A marked word: misspelled, or repeating the word before it. |
| 65 | #[derive(Clone, Debug, PartialEq, Eq)] |
| 66 | pub(crate) struct Mark { |
| 67 | /// Bytes of the paragraph's text. |
| 68 | pub(crate) range: Range<usize>, |
| 69 | pub(crate) repeated: bool, |
| 70 | } |
| 71 | |
| 72 | /// A word of a paragraph: its bytes, its language, and whether its spelling is checked. |
| 73 | struct Word { |
| 74 | range: Range<usize>, |
| 75 | /// The run's language, an LCID; `None` where untagged. |
| 76 | language: Option<u32>, |
| 77 | checked: bool, |
| 78 | } |
| 79 | |
| 80 | /// Letters, digits, underscores and combining marks, as against spaces, punctuation and |
| 81 | /// symbols. |
| 82 | fn word_character(character: char) -> bool { |
| 83 | !character.is_whitespace() |
| 84 | && !character.is_ascii_punctuation() |
| 85 | && !matches!(u32::from(character), |
| 86 | 0x80..=0xbf | 0xd7 | 0xf7 | 0x2000..=0x2bff | 0x3000..=0x303f | 0xfe30..=0xfe4f |
| 87 | | 0xff00..=0xff0f | 0x1f000..=0x1faff) |
| 88 | || character == '_' |
| 89 | } |
| 90 | |
| 91 | fn apostrophe(character: char) -> bool { |
| 92 | matches!(character, '\'' | '\u{2019}' | '\u{02bc}') |
| 93 | } |
| 94 | |
| 95 | /// An Internet or file address, which OneNote leaves unchecked whole. |
| 96 | fn address(chunk: &str) -> bool { |
| 97 | let lower = chunk.to_lowercase(); |
| 98 | lower.contains("://") |
| 99 | || lower.starts_with("www.") |
| 100 | || chunk.contains('\\') |
| 101 | || chunk |
| 102 | .split_once('@') |
| 103 | .is_some_and(|(user, host)| !user.is_empty() && host.contains('.')) |
| 104 | } |
| 105 | |
| 106 | /// The words of `paragraph`'s shown text, in order. Hidden field codes and equations are |
| 107 | /// not text to check, and a word split by one is two. |
| 108 | fn words(paragraph: &Paragraph) -> Vec<Word> { |
| 109 | let text = paragraph.text(); |
| 110 | let spans = paragraph.spans(); |
| 111 | let format = |byte: usize| &spans[spans.partition_point(|span| span.end <= byte)].format; |
| 112 | // Runs of shown text between spaces, which formatting changes don't split. |
| 113 | let mut chunks = Vec::new(); |
| 114 | let mut start = None; |
| 115 | for (byte, character) in text.char_indices() { |
| 116 | let format = format(byte); |
| 117 | let apart = |
| 118 | character.is_whitespace() || format.hidden == Some(true) || format.math == Some(true); |
| 119 | match (apart, start) { |
| 120 | (true, Some(from)) => { |
| 121 | chunks.push(from..byte); |
| 122 | start = None; |
| 123 | } |
| 124 | (false, None) => start = Some(byte), |
| 125 | _ => {} |
| 126 | } |
| 127 | } |
| 128 | chunks.extend(start.map(|from| from..text.len())); |
| 129 | let mut words = Vec::new(); |
| 130 | for chunk in chunks { |
| 131 | let at = chunk.start; |
| 132 | let chunk = &text[chunk]; |
| 133 | if address(chunk) { |
| 134 | continue; |
| 135 | } |
| 136 | let mut characters = chunk.char_indices().peekable(); |
| 137 | while let Some((offset, character)) = characters.next() { |
| 138 | if !word_character(character) { |
| 139 | continue; |
| 140 | } |
| 141 | let mut end = offset + character.len_utf8(); |
| 142 | while let Some(&(next, character)) = characters.peek() { |
| 143 | let joined = apostrophe(character) |
| 144 | && chunk[next + character.len_utf8()..] |
| 145 | .chars() |
| 146 | .next() |
| 147 | .is_some_and(char::is_alphabetic); |
| 148 | if !word_character(character) && !joined { |
| 149 | break; |
| 150 | } |
| 151 | end = next + character.len_utf8(); |
| 152 | characters.next(); |
| 153 | } |
| 154 | let word = &chunk[offset..end]; |
| 155 | if word |
| 156 | .chars() |
| 157 | .any(|character| character.is_numeric() || crate::search::ideographic(character)) |
| 158 | { |
| 159 | continue; |
| 160 | } |
| 161 | words.push(Word { |
| 162 | range: at + offset..at + end, |
| 163 | language: format(at + offset).language, |
| 164 | checked: word.chars().any(char::is_lowercase), |
| 165 | }); |
| 166 | } |
| 167 | } |
| 168 | words |
| 169 | } |
| 170 | |
| 171 | /// Whether two words are the same word, as a repeated one is. |
| 172 | fn same(a: &str, b: &str) -> bool { |
| 173 | let fold = |word: &str| { |
| 174 | word.chars() |
| 175 | .flat_map(char::to_lowercase) |
| 176 | .map(|character| { |
| 177 | if apostrophe(character) { |
| 178 | '\'' |
| 179 | } else { |
| 180 | character |
| 181 | } |
| 182 | }) |
| 183 | .collect::<String>() |
| 184 | }; |
| 185 | fold(a) == fold(b) |
| 186 | } |
| 187 | |
| 188 | /// Which of `words` repeat the word before them, with only spaces between. |
| 189 | fn repeated(text: &str, words: &[Word]) -> Vec<bool> { |
| 190 | words |
| 191 | .iter() |
| 192 | .enumerate() |
| 193 | .map(|(index, word)| { |
| 194 | index.checked_sub(1).is_some_and(|previous| { |
| 195 | let previous = &words[previous].range; |
| 196 | text[previous.end..word.range.start] |
| 197 | .chars() |
| 198 | .all(char::is_whitespace) |
| 199 | && same(&text[previous.clone()], &text[word.range.clone()]) |
| 200 | }) |
| 201 | }) |
| 202 | .collect() |
| 203 | } |
| 204 | |
| 205 | /// The marks `dictionary` gives each of `paragraphs`, asking it once: each word repeating |
| 206 | /// the one before it, and each other checked word it finds misspelled. |
| 207 | /// Checks `paragraphs`, reading text no run tags as language `untagged`. |
| 208 | fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary, untagged: u32) -> Vec<Vec<Mark>> { |
| 209 | let words: Vec<(Vec<Word>, Vec<bool>)> = paragraphs |
| 210 | .iter() |
| 211 | .map(|paragraph| { |
| 212 | let words = words(paragraph); |
| 213 | let repeated = repeated(paragraph.text(), &words); |
| 214 | (words, repeated) |
| 215 | }) |
| 216 | .collect(); |
| 217 | let asked: Vec<(&str, u32)> = paragraphs |
| 218 | .iter() |
| 219 | .zip(&words) |
| 220 | .flat_map(|(paragraph, (words, repeated))| { |
| 221 | words |
| 222 | .iter() |
| 223 | .zip(repeated) |
| 224 | .filter(|(word, repeated)| word.checked && !**repeated) |
| 225 | .map(|(word, _)| { |
| 226 | let language = word.language.unwrap_or(untagged); |
| 227 | (&paragraph.text()[word.range.clone()], language) |
| 228 | }) |
| 229 | }) |
| 230 | .collect(); |
| 231 | let mut misspelled = dictionary.misspelled(&asked).into_iter(); |
| 232 | words |
| 233 | .into_iter() |
| 234 | .map(|(words, repeated)| { |
| 235 | words |
| 236 | .into_iter() |
| 237 | .zip(repeated) |
| 238 | .filter_map(|(word, repeated)| { |
| 239 | let marked = repeated || word.checked && misspelled.next().unwrap_or(false); |
| 240 | marked.then_some(Mark { |
| 241 | range: word.range, |
| 242 | repeated, |
| 243 | }) |
| 244 | }) |
| 245 | .collect() |
| 246 | }) |
| 247 | .collect() |
| 248 | } |
| 249 | |
| 250 | /// What a paragraph's marks are kept under: its text and the formatting that decides them. |
| 251 | fn key(paragraph: &Paragraph) -> u64 { |
| 252 | let mut hasher = DefaultHasher::new(); |
| 253 | paragraph.text().hash(&mut hasher); |
| 254 | for span in paragraph.spans() { |
| 255 | ( |
| 256 | span.end, |
| 257 | span.format.language, |
| 258 | span.format.hidden, |
| 259 | span.format.math, |
| 260 | ) |
| 261 | .hash(&mut hasher); |
| 262 | } |
| 263 | hasher.finish() |
| 264 | } |
| 265 | |
| 266 | struct Checked { |
| 267 | text: String, |
| 268 | marks: Vec<Mark>, |
| 269 | } |
| 270 | |
| 271 | #[derive(Default)] |
| 272 | struct State { |
| 273 | checked: HashMap<u64, Checked>, |
| 274 | pending: HashSet<u64>, |
| 275 | /// Words Ignore or Add to Dictionary accepted, misspelled or repeated, left unmarked |
| 276 | /// everywhere. |
| 277 | accepted: HashSet<String>, |
| 278 | } |
| 279 | |
| 280 | impl State { |
| 281 | /// The marks kept for `paragraph` under `key`, less accepted words. |
| 282 | fn marks(&self, key: u64, paragraph: &Paragraph) -> Option<Vec<Mark>> { |
| 283 | let checked = self |
| 284 | .checked |
| 285 | .get(&key) |
| 286 | .filter(|checked| checked.text == paragraph.text())?; |
| 287 | Some( |
| 288 | checked |
| 289 | .marks |
| 290 | .iter() |
| 291 | .filter(|mark| { |
| 292 | !self |
| 293 | .accepted |
| 294 | .contains(&paragraph.text()[mark.range.clone()]) |
| 295 | }) |
| 296 | .cloned() |
| 297 | .collect(), |
| 298 | ) |
| 299 | } |
| 300 | |
| 301 | fn keep(&mut self, key: u64, paragraph: &Paragraph, marks: Vec<Mark>) { |
| 302 | if self.checked.len() >= KEPT { |
| 303 | self.checked.clear(); |
| 304 | } |
| 305 | self.checked.insert( |
| 306 | key, |
| 307 | Checked { |
| 308 | text: paragraph.text().to_owned(), |
| 309 | marks, |
| 310 | }, |
| 311 | ); |
| 312 | } |
| 313 | } |
| 314 | |
| 315 | struct Shared { |
| 316 | dictionary: Box<dyn Dictionary>, |
| 317 | /// The language text no run tags is checked in, an LCID: US English, as OneNote's. |
| 318 | untagged: AtomicU32, |
| 319 | state: Mutex<State>, |
| 320 | } |
| 321 | |
| 322 | /// Spell checking shared by the pages of one app. Paragraphs are checked when first asked |
| 323 | /// for, on a thread of its own, which wakes `redraw` when marks are ready. |
| 324 | #[derive(Clone)] |
| 325 | pub struct Spelling { |
| 326 | shared: Arc<Shared>, |
| 327 | jobs: mpsc::Sender<(u64, Paragraph)>, |
| 328 | } |
| 329 | |
| 330 | impl Spelling { |
| 331 | pub fn new(dictionary: Box<dyn Dictionary>, redraw: std::task::Waker) -> Self { |
| 332 | let shared = Arc::new(Shared { |
| 333 | dictionary, |
| 334 | untagged: AtomicU32::new(crate::language::EN_US), |
| 335 | state: Mutex::default(), |
| 336 | }); |
| 337 | let (jobs, receiver) = mpsc::channel::<(u64, Paragraph)>(); |
| 338 | let worker = Arc::clone(&shared); |
| 339 | // The browser gives a page one thread: paragraphs are checked as they are drawn. |
| 340 | #[cfg(target_arch = "wasm32")] |
| 341 | drop((receiver, worker, redraw)); |
| 342 | #[cfg(not(target_arch = "wasm32"))] |
| 343 | std::thread::Builder::new() |
| 344 | .name("spelling".into()) |
| 345 | .spawn(move || { |
| 346 | while let Ok(first) = receiver.recv() { |
| 347 | let jobs: Vec<(u64, Paragraph)> = std::iter::once(first) |
| 348 | .chain(receiver.try_iter().take(BATCH - 1)) |
| 349 | .collect(); |
| 350 | let paragraphs: Vec<&Paragraph> = |
| 351 | jobs.iter().map(|(_, paragraph)| paragraph).collect(); |
| 352 | let untagged = worker.untagged.load(Ordering::Relaxed); |
| 353 | let marks = check(&paragraphs, &*worker.dictionary, untagged); |
| 354 | let mut state = worker.state.lock().unwrap(); |
| 355 | for ((key, paragraph), marks) in jobs.iter().zip(marks) { |
| 356 | state.pending.remove(key); |
| 357 | state.keep(*key, paragraph, marks); |
| 358 | } |
| 359 | drop(state); |
| 360 | redraw.wake_by_ref(); |
| 361 | } |
| 362 | }) |
| 363 | .expect("the spelling thread starts"); |
| 364 | Self { shared, jobs } |
| 365 | } |
| 366 | |
| 367 | /// The marks on `paragraph` once it is checked, less accepted words; until then none, |
| 368 | /// and it is checked. |
| 369 | pub(crate) fn marks(&self, paragraph: &Paragraph) -> Vec<Mark> { |
| 370 | if cfg!(target_arch = "wasm32") { |
| 371 | return self.marks_now(paragraph); |
| 372 | } |
| 373 | let key = key(paragraph); |
| 374 | let mut state = self.shared.state.lock().unwrap(); |
| 375 | if let Some(marks) = state.marks(key, paragraph) { |
| 376 | return marks; |
| 377 | } |
| 378 | if state.pending.insert(key) { |
| 379 | let _ = self.jobs.send((key, paragraph.clone())); |
| 380 | } |
| 381 | Vec::new() |
| 382 | } |
| 383 | |
| 384 | /// The marks on `paragraph`, checked on this thread where not yet checked, as the |
| 385 | /// Spelling pane walks the page. |
| 386 | pub(crate) fn marks_now(&self, paragraph: &Paragraph) -> Vec<Mark> { |
| 387 | let key = key(paragraph); |
| 388 | if let Some(marks) = self.shared.state.lock().unwrap().marks(key, paragraph) { |
| 389 | return marks; |
| 390 | } |
| 391 | let untagged = self.shared.untagged.load(Ordering::Relaxed); |
| 392 | let marks = check(&[paragraph], &*self.shared.dictionary, untagged).remove(0); |
| 393 | let mut state = self.shared.state.lock().unwrap(); |
| 394 | state.keep(key, paragraph, marks); |
| 395 | state.marks(key, paragraph).unwrap_or_default() |
| 396 | } |
| 397 | |
| 398 | /// Checks every paragraph again when next asked, as after a dictionary arrived. |
| 399 | pub fn recheck(&self) { |
| 400 | self.shared.state.lock().unwrap().checked.clear(); |
| 401 | } |
| 402 | |
| 403 | /// Checks text no run tags in `language`, an LCID, as a browser's text is in its own. |
| 404 | pub fn untagged(&self, language: u32) { |
| 405 | self.shared.untagged.store(language, Ordering::Relaxed); |
| 406 | self.recheck(); |
| 407 | } |
| 408 | |
| 409 | /// Corrections for `word` in `language`, best first. |
| 410 | pub(crate) fn suggest(&self, word: &str, language: Option<u32>) -> Vec<String> { |
| 411 | self.shared.dictionary.suggest( |
| 412 | word, |
| 413 | language.unwrap_or(self.shared.untagged.load(Ordering::Relaxed)), |
| 414 | ) |
| 415 | } |
| 416 | |
| 417 | /// Ignore: leaves `word` unmarked everywhere until the app quits. |
| 418 | pub fn ignore(&self, word: &str) { |
| 419 | self.shared |
| 420 | .state |
| 421 | .lock() |
| 422 | .unwrap() |
| 423 | .accepted |
| 424 | .insert(word.to_owned()); |
| 425 | } |
| 426 | |
| 427 | /// Add to Dictionary: the platform's dictionary accepts `word` from now on. |
| 428 | pub fn learn(&self, word: &str) { |
| 429 | self.shared.dictionary.learn(word); |
| 430 | self.ignore(word); |
| 431 | } |
| 432 | } |
| 433 | |
| 434 | #[cfg(test)] |
| 435 | pub(crate) mod tests { |
| 436 | use super::*; |
| 437 | use onestore::document::Format; |
| 438 | |
| 439 | /// Knows a few English and French words; English only by default. |
| 440 | pub(crate) struct Fake; |
| 441 | |
| 442 | const ENGLISH: &[&str] = &[ |
| 443 | "this", "sentence", "has", "the", "typos", "word", "don't", "i", "a", "bar", "is", "here", |
| 444 | "well", "known", "hello", |
| 445 | ]; |
| 446 | const FRENCH: &[&str] = &["bonjour", "le", "monde", "une", "phrase"]; |
| 447 | |
| 448 | impl Dictionary for Fake { |
| 449 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { |
| 450 | words |
| 451 | .iter() |
| 452 | .map(|(word, language)| { |
| 453 | let known = match language { |
| 454 | 1033 => ENGLISH, |
| 455 | 1036 => FRENCH, |
| 456 | _ => return false, |
| 457 | }; |
| 458 | !known.contains(&word.to_lowercase().replace('\u{2019}', "'").as_str()) |
| 459 | }) |
| 460 | .collect() |
| 461 | } |
| 462 | |
| 463 | fn suggest(&self, word: &str, _: u32) -> Vec<String> { |
| 464 | match word { |
| 465 | "Ths" => vec!["This".into(), "Thus".into()], |
| 466 | "sentense" => vec!["sentence".into()], |
| 467 | _ => Vec::new(), |
| 468 | } |
| 469 | } |
| 470 | |
| 471 | fn learn(&self, _: &str) {} |
| 472 | } |
| 473 | |
| 474 | #[test] |
| 475 | fn languages_pick_their_locale_then_their_language() { |
| 476 | let available: Vec<String> = ["en", "en_GB", "fr", "de-AT", "de_DE", "pt_BR"] |
| 477 | .map(String::from) |
| 478 | .into(); |
| 479 | for (lcid, picked) in [ |
| 480 | (1033, Some("en")), |
| 481 | (2057, Some("en_GB")), |
| 482 | (3081, Some("en")), |
| 483 | (3084, Some("fr")), |
| 484 | (1031, Some("de_DE")), |
| 485 | (3079, Some("de-AT")), |
| 486 | (2070, Some("pt_BR")), |
| 487 | (1049, None), |
| 488 | (0x1007f, None), |
| 489 | ] { |
| 490 | assert_eq!(pick(lcid, &available), picked, "{lcid}"); |
| 491 | } |
| 492 | let hunspell: Vec<String> = ["en_AU", "en_US", "fr_FR"].map(String::from).into(); |
| 493 | assert_eq!(pick(1033, &hunspell), Some("en_US")); |
| 494 | assert_eq!(pick(1036, &hunspell), Some("fr_FR")); |
| 495 | let portuguese: Vec<String> = ["pt_PT", "pt_BR"].map(String::from).into(); |
| 496 | assert_eq!(pick(1046, &portuguese), Some("pt_BR")); |
| 497 | assert_eq!(pick(2070, &portuguese), Some("pt_PT")); |
| 498 | } |
| 499 | |
| 500 | fn marked(paragraph: &Paragraph) -> Vec<(&str, bool)> { |
| 501 | check(&[paragraph], &Fake, crate::language::EN_US) |
| 502 | .remove(0) |
| 503 | .into_iter() |
| 504 | .map(|mark| (&paragraph.text()[mark.range], mark.repeated)) |
| 505 | .collect() |
| 506 | } |
| 507 | |
| 508 | /// OneNote 2010 under its default Proofing options (`/tmp/snowbound-spelling/lab`). |
| 509 | #[test] |
| 510 | fn words_are_marked_as_onenote_marks_them() { |
| 511 | let paragraph = Paragraph::new( |
| 512 | "Ths sentense has the typos. HELLO WRLD NASA abc123 x86 3rd foo_bar \ |
| 513 | www.exmple.com http://exmple.com/qwrt a@exmple.com C:\\Users\\qwrt \ |
| 514 | don't dont word word the The don't don\u{2019}t well-knwn (sentense) \ |
| 515 | \u{65e5}\u{672c}\u{8a9e}" |
| 516 | .into(), |
| 517 | Format::default(), |
| 518 | ); |
| 519 | assert_eq!( |
| 520 | marked(&paragraph), |
| 521 | [ |
| 522 | ("Ths", false), |
| 523 | ("sentense", false), |
| 524 | ("foo_bar", false), |
| 525 | ("dont", false), |
| 526 | ("word", true), |
| 527 | ("The", true), |
| 528 | ("don\u{2019}t", true), |
| 529 | ("knwn", false), |
| 530 | ("sentense", false), |
| 531 | ] |
| 532 | ); |
| 533 | } |
| 534 | |
| 535 | #[test] |
| 536 | fn each_run_is_checked_in_its_language_and_hidden_codes_are_not_text() { |
| 537 | let french = Format { |
| 538 | language: Some(1036), |
| 539 | ..Format::default() |
| 540 | }; |
| 541 | let paragraph = Paragraph::from_runs([ |
| 542 | ("bonjour le monde ".to_owned(), french.clone()), |
| 543 | ("bonjjour ".to_owned(), french), |
| 544 | ( |
| 545 | "hello ".to_owned(), |
| 546 | Format { |
| 547 | language: Some(1061), |
| 548 | ..Format::default() |
| 549 | }, |
| 550 | ), |
| 551 | ( |
| 552 | "wo".to_owned(), |
| 553 | Format { |
| 554 | bold: Some(true), |
| 555 | ..Format::default() |
| 556 | }, |
| 557 | ), |
| 558 | ("rd ".to_owned(), Format::default()), |
| 559 | ( |
| 560 | "HYPERLINK \"qwrt\"".to_owned(), |
| 561 | Format { |
| 562 | hidden: Some(true), |
| 563 | ..Format::default() |
| 564 | }, |
| 565 | ), |
| 566 | ("heere".to_owned(), Format::default()), |
| 567 | ]); |
| 568 | assert_eq!(marked(&paragraph), [("bonjjour", false), ("heere", false)]); |
| 569 | } |
| 570 | |
| 571 | #[test] |
| 572 | fn marks_arrive_from_the_thread_and_accepted_words_drop_out() { |
| 573 | struct Wake(mpsc::Sender<()>); |
| 574 | impl std::task::Wake for Wake { |
| 575 | fn wake(self: Arc<Self>) { |
| 576 | let _ = self.0.send(()); |
| 577 | } |
| 578 | } |
| 579 | let (woken, wakes) = mpsc::channel(); |
| 580 | let spelling = Spelling::new(Box::new(Fake), Arc::new(Wake(woken)).into()); |
| 581 | let paragraph = Paragraph::new("Ths sentense here".into(), Format::default()); |
| 582 | assert!(spelling.marks(&paragraph).is_empty()); |
| 583 | wakes |
| 584 | .recv_timeout(std::time::Duration::from_secs(10)) |
| 585 | .expect("the spelling thread wakes the host"); |
| 586 | let words = |spelling: &Spelling| { |
| 587 | spelling |
| 588 | .marks(&paragraph) |
| 589 | .into_iter() |
| 590 | .map(|mark| paragraph.text()[mark.range].to_owned()) |
| 591 | .collect::<Vec<_>>() |
| 592 | }; |
| 593 | assert_eq!(words(&spelling), ["Ths", "sentense"]); |
| 594 | spelling.ignore("sentense"); |
| 595 | assert_eq!(words(&spelling), ["Ths"]); |
| 596 | assert_eq!(spelling.suggest("Ths", None), ["This", "Thus"]); |
| 597 | // The same text in another language is checked again. |
| 598 | let french = Paragraph::new( |
| 599 | "Ths sentense here".into(), |
| 600 | Format { |
| 601 | language: Some(1036), |
| 602 | ..Format::default() |
| 603 | }, |
| 604 | ); |
| 605 | assert!(spelling.marks(&french).is_empty()); |
| 606 | wakes |
| 607 | .recv_timeout(std::time::Duration::from_secs(10)) |
| 608 | .expect("the spelling thread wakes the host"); |
| 609 | assert_eq!(spelling.marks(&french).len(), 2); |
| 610 | } |
| 611 | |
| 612 | #[test] |
| 613 | fn arriving_dictionaries_and_untagged_languages_check_paragraphs_again() { |
| 614 | struct Late(Arc<std::sync::atomic::AtomicBool>); |
| 615 | impl Dictionary for Late { |
| 616 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { |
| 617 | match self.0.load(std::sync::atomic::Ordering::Relaxed) { |
| 618 | true => Fake.misspelled(words), |
| 619 | false => vec![false; words.len()], |
| 620 | } |
| 621 | } |
| 622 | fn suggest(&self, _: &str, _: u32) -> Vec<String> { |
| 623 | Vec::new() |
| 624 | } |
| 625 | fn learn(&self, _: &str) {} |
| 626 | } |
| 627 | let arrived = Arc::new(std::sync::atomic::AtomicBool::new(false)); |
| 628 | let spelling = Spelling::new( |
| 629 | Box::new(Late(Arc::clone(&arrived))), |
| 630 | std::task::Waker::noop().clone(), |
| 631 | ); |
| 632 | let paragraph = Paragraph::new("Ths sentense here".into(), Format::default()); |
| 633 | assert!(spelling.marks_now(&paragraph).is_empty()); |
| 634 | arrived.store(true, std::sync::atomic::Ordering::Relaxed); |
| 635 | assert!( |
| 636 | spelling.marks_now(&paragraph).is_empty(), |
| 637 | "kept until asked again" |
| 638 | ); |
| 639 | spelling.recheck(); |
| 640 | assert_eq!(spelling.marks_now(&paragraph).len(), 2); |
| 641 | // Untagged text is US English until the host says otherwise. |
| 642 | spelling.untagged(1036); |
| 643 | assert_eq!(spelling.marks_now(&paragraph).len(), 3); |
| 644 | } |
| 645 | } |