| author | |
| committer | |
| log | f837389ba02268ce5ee180e09bba96a337be406b |
| tree | 44c9676ca81879e4006f58c1b30dadbe14ef43a4 |
| parent | 21adbb5dfae140e40626cefd52ed377a44977c7f |
| signature | Signed by SSH key SHA256:52mNGHRsVFBDED9IAX5pe+LRWUefqTbxEReunq21QvU |
Dictionaries are fetched the first time a word in their language is
checked, and text no run tags counts as the browser's language. The site
offers en_US, en_GB, de_DE, fr_FR, es_ES, pt_BR, pt_PT, it_IT, nl_NL,
sv_SE and ru_RU from nixpkgs' Hunspell packages, converted to UTF-8 and
gzipped, each with its readme and its licences' texts; a row in
release_web.py's DICTIONARIES adds another. A bare pt picks pt_BR, as
Windows does.
Assisted-by: claude-opus-5.57 files changed, 233 insertions(+), 73 deletions(-)
arc/platforms.md+3-2| ... | @@ -177,8 +177,9 @@ keyboard, the toolbar and the macOS menu bar all run commands from it. | ... | @@ -177,8 +177,9 @@ keyboard, the toolbar and the macOS menu bar all run commands from it. |
| 177 | browser keeps its own window and tab chords. Dialogs are the browser's; Insert, and Open | 177 | browser keeps its own window and tab chords. Dialogs are the browser's; Insert, and Open |
| 178 | without a folder to give, ask for files to copy in; printing downloads the PDF. Servers and | 178 | without a folder to give, ask for files to copy in; printing downloads the PDF. Servers and |
| 179 | recording are still to come. | 179 | recording are still to come. |
| 180 | - Spelling is Hunspell's American English dictionary through `spellbook`, fetched beside the | 180 | - Spelling is Hunspell's dictionaries through `spellbook`, each fetched beside the module the |
| 181 | module; Add to Dictionary keeps its words in the browser's files. Text the bundled faces | 181 | first time a word in its language is checked, text no run tags counting as the browser's |
| 182 | language; Add to Dictionary keeps its words in the browser's files. Text the bundled faces | ||
| 182 | lack takes Noto's, fetched by script the first time a page holds it (a CJK face cut to | 183 | lack takes Noto's, fetched by script the first time a page holds it (a CJK face cut to |
| 183 | the national standards' characters by `tools/web/subset_cjk.py`; emoji in Noto Color | 184 | the national standards' characters by `tools/web/subset_cjk.py`; emoji in Noto Color |
| 184 | Emoji's COLRv1 outlines, which `draw` paints itself), and the page is laid out again. | 185 | Emoji's COLRv1 outlines, which `draw` paints itself), and the page is laid out again. |
crates/canvas/src/spelling.rs+71-12| ... | @@ -8,6 +8,7 @@ use onestore::page::text::Paragraph; | ... | @@ -8,6 +8,7 @@ use onestore::page::text::Paragraph; |
| 8 | use std::collections::{HashMap, HashSet}; | 8 | use std::collections::{HashMap, HashSet}; |
| 9 | use std::hash::{DefaultHasher, Hash, Hasher}; | 9 | use std::hash::{DefaultHasher, Hash, Hasher}; |
| 10 | use std::ops::Range; | 10 | use std::ops::Range; |
| 11 | use std::sync::atomic::{AtomicU32, Ordering}; | ||
| 11 | use std::sync::{Arc, Mutex, mpsc}; | 12 | use std::sync::{Arc, Mutex, mpsc}; |
| 12 | 13 | ||
| 13 | /// Checked paragraphs kept before the cache starts over. | 14 | /// Checked paragraphs kept before the cache starts over. |
| ... | @@ -36,6 +37,7 @@ pub fn pick(language: u32, available: &[String]) -> Option<&str> { | ... | @@ -36,6 +37,7 @@ pub fn pick(language: u32, available: &[String]) -> Option<&str> { |
| 36 | // A bare tag is the locale Windows picks, most often the language's own country. | 37 | // A bare tag is the locale Windows picks, most often the language's own country. |
| 37 | let home = match bare { | 38 | let home = match bare { |
| 38 | "en" => "US".to_owned(), | 39 | "en" => "US".to_owned(), |
| 40 | "pt" => "BR".to_owned(), | ||
| 39 | _ => bare.to_uppercase(), | 41 | _ => bare.to_uppercase(), |
| 40 | }; | 42 | }; |
| 41 | let wanted = if region.is_empty() { | 43 | let wanted = if region.is_empty() { |
| ... | @@ -70,7 +72,8 @@ pub(crate) struct Mark { | ... | @@ -70,7 +72,8 @@ pub(crate) struct Mark { |
| 70 | /// A word of a paragraph: its bytes, its language, and whether its spelling is checked. | 72 | /// A word of a paragraph: its bytes, its language, and whether its spelling is checked. |
| 71 | struct Word { | 73 | struct Word { |
| 72 | range: Range<usize>, | 74 | range: Range<usize>, |
| 73 | language: u32, | 75 | /// The run's language, an LCID; `None` where untagged. |
| 76 | language: Option<u32>, | ||
| 74 | checked: bool, | 77 | checked: bool, |
| 75 | } | 78 | } |
| 76 | 79 | ||
| ... | @@ -157,9 +160,7 @@ fn words(paragraph: &Paragraph) -> Vec<Word> { | ... | @@ -157,9 +160,7 @@ fn words(paragraph: &Paragraph) -> Vec<Word> { |
| 157 | } | 160 | } |
| 158 | words.push(Word { | 161 | words.push(Word { |
| 159 | range: at + offset..at + end, | 162 | range: at + offset..at + end, |
| 160 | language: format(at + offset) | 163 | language: format(at + offset).language, |
| 161 | .language | ||
| 162 | .unwrap_or(crate::language::EN_US), | ||
| 163 | checked: word.chars().any(char::is_lowercase), | 164 | checked: word.chars().any(char::is_lowercase), |
| 164 | }); | 165 | }); |
| 165 | } | 166 | } |
| ... | @@ -203,7 +204,8 @@ fn repeated(text: &str, words: &[Word]) -> Vec<bool> { | ... | @@ -203,7 +204,8 @@ fn repeated(text: &str, words: &[Word]) -> Vec<bool> { |
| 203 | 204 | ||
| 204 | /// The marks `dictionary` gives each of `paragraphs`, asking it once: each word repeating | 205 | /// The marks `dictionary` gives each of `paragraphs`, asking it once: each word repeating |
| 205 | /// the one before it, and each other checked word it finds misspelled. | 206 | /// the one before it, and each other checked word it finds misspelled. |
| 206 | fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary) -> Vec<Vec<Mark>> { | 207 | /// Checks `paragraphs`, reading text no run tags as language `untagged`. |
| 208 | fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary, untagged: u32) -> Vec<Vec<Mark>> { | ||
| 207 | let words: Vec<(Vec<Word>, Vec<bool>)> = paragraphs | 209 | let words: Vec<(Vec<Word>, Vec<bool>)> = paragraphs |
| 208 | .iter() | 210 | .iter() |
| 209 | .map(|paragraph| { | 211 | .map(|paragraph| { |
| ... | @@ -220,7 +222,10 @@ fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary) -> Vec<Vec<Mark | ... | @@ -220,7 +222,10 @@ fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary) -> Vec<Vec<Mark |
| 220 | .iter() | 222 | .iter() |
| 221 | .zip(repeated) | 223 | .zip(repeated) |
| 222 | .filter(|(word, repeated)| word.checked && !**repeated) | 224 | .filter(|(word, repeated)| word.checked && !**repeated) |
| 223 | .map(|(word, _)| (&paragraph.text()[word.range.clone()], word.language)) | 225 | .map(|(word, _)| { |
| 226 | let language = word.language.unwrap_or(untagged); | ||
| 227 | (&paragraph.text()[word.range.clone()], language) | ||
| 228 | }) | ||
| 224 | }) | 229 | }) |
| 225 | .collect(); | 230 | .collect(); |
| 226 | let mut misspelled = dictionary.misspelled(&asked).into_iter(); | 231 | let mut misspelled = dictionary.misspelled(&asked).into_iter(); |
| ... | @@ -309,6 +314,8 @@ impl State { | ... | @@ -309,6 +314,8 @@ impl State { |
| 309 | 314 | ||
| 310 | struct Shared { | 315 | struct Shared { |
| 311 | dictionary: Box<dyn Dictionary>, | 316 | dictionary: Box<dyn Dictionary>, |
| 317 | /// The language text no run tags is checked in, an LCID: US English, as OneNote's. | ||
| 318 | untagged: AtomicU32, | ||
| 312 | state: Mutex<State>, | 319 | state: Mutex<State>, |
| 313 | } | 320 | } |
| 314 | 321 | ||
| ... | @@ -324,6 +331,7 @@ impl Spelling { | ... | @@ -324,6 +331,7 @@ impl Spelling { |
| 324 | pub fn new(dictionary: Box<dyn Dictionary>, redraw: std::task::Waker) -> Self { | 331 | pub fn new(dictionary: Box<dyn Dictionary>, redraw: std::task::Waker) -> Self { |
| 325 | let shared = Arc::new(Shared { | 332 | let shared = Arc::new(Shared { |
| 326 | dictionary, | 333 | dictionary, |
| 334 | untagged: AtomicU32::new(crate::language::EN_US), | ||
| 327 | state: Mutex::default(), | 335 | state: Mutex::default(), |
| 328 | }); | 336 | }); |
| 329 | let (jobs, receiver) = mpsc::channel::<(u64, Paragraph)>(); | 337 | let (jobs, receiver) = mpsc::channel::<(u64, Paragraph)>(); |
| ... | @@ -341,7 +349,8 @@ impl Spelling { | ... | @@ -341,7 +349,8 @@ impl Spelling { |
| 341 | .collect(); | 349 | .collect(); |
| 342 | let paragraphs: Vec<&Paragraph> = | 350 | let paragraphs: Vec<&Paragraph> = |
| 343 | jobs.iter().map(|(_, paragraph)| paragraph).collect(); | 351 | jobs.iter().map(|(_, paragraph)| paragraph).collect(); |
| 344 | let marks = check(&paragraphs, &*worker.dictionary); | 352 | let untagged = worker.untagged.load(Ordering::Relaxed); |
| 353 | let marks = check(&paragraphs, &*worker.dictionary, untagged); | ||
| 345 | let mut state = worker.state.lock().unwrap(); | 354 | let mut state = worker.state.lock().unwrap(); |
| 346 | for ((key, paragraph), marks) in jobs.iter().zip(marks) { | 355 | for ((key, paragraph), marks) in jobs.iter().zip(marks) { |
| 347 | state.pending.remove(key); | 356 | state.pending.remove(key); |
| ... | @@ -379,17 +388,30 @@ impl Spelling { | ... | @@ -379,17 +388,30 @@ impl Spelling { |
| 379 | if let Some(marks) = self.shared.state.lock().unwrap().marks(key, paragraph) { | 388 | if let Some(marks) = self.shared.state.lock().unwrap().marks(key, paragraph) { |
| 380 | return marks; | 389 | return marks; |
| 381 | } | 390 | } |
| 382 | let marks = check(&[paragraph], &*self.shared.dictionary).remove(0); | 391 | let untagged = self.shared.untagged.load(Ordering::Relaxed); |
| 392 | let marks = check(&[paragraph], &*self.shared.dictionary, untagged).remove(0); | ||
| 383 | let mut state = self.shared.state.lock().unwrap(); | 393 | let mut state = self.shared.state.lock().unwrap(); |
| 384 | state.keep(key, paragraph, marks); | 394 | state.keep(key, paragraph, marks); |
| 385 | state.marks(key, paragraph).unwrap_or_default() | 395 | state.marks(key, paragraph).unwrap_or_default() |
| 386 | } | 396 | } |
| 387 | 397 | ||
| 398 | /// Checks every paragraph again when next asked, as after a dictionary arrived. | ||
| 399 | pub fn recheck(&self) { | ||
| 400 | self.shared.state.lock().unwrap().checked.clear(); | ||
| 401 | } | ||
| 402 | |||
| 403 | /// Checks text no run tags in `language`, an LCID, as a browser's text is in its own. | ||
| 404 | pub fn untagged(&self, language: u32) { | ||
| 405 | self.shared.untagged.store(language, Ordering::Relaxed); | ||
| 406 | self.recheck(); | ||
| 407 | } | ||
| 408 | |||
| 388 | /// Corrections for `word` in `language`, best first. | 409 | /// Corrections for `word` in `language`, best first. |
| 389 | pub(crate) fn suggest(&self, word: &str, language: Option<u32>) -> Vec<String> { | 410 | pub(crate) fn suggest(&self, word: &str, language: Option<u32>) -> Vec<String> { |
| 390 | self.shared | 411 | self.shared.dictionary.suggest( |
| 391 | .dictionary | 412 | word, |
| 392 | .suggest(word, language.unwrap_or(crate::language::EN_US)) | 413 | language.unwrap_or(self.shared.untagged.load(Ordering::Relaxed)), |
| 414 | ) | ||
| 393 | } | 415 | } |
| 394 | 416 | ||
| 395 | /// Ignore: leaves `word` unmarked everywhere until the app quits. | 417 | /// Ignore: leaves `word` unmarked everywhere until the app quits. |
| ... | @@ -470,10 +492,13 @@ pub(crate) mod tests { | ... | @@ -470,10 +492,13 @@ pub(crate) mod tests { |
| 470 | let hunspell: Vec<String> = ["en_AU", "en_US", "fr_FR"].map(String::from).into(); | 492 | let hunspell: Vec<String> = ["en_AU", "en_US", "fr_FR"].map(String::from).into(); |
| 471 | assert_eq!(pick(1033, &hunspell), Some("en_US")); | 493 | assert_eq!(pick(1033, &hunspell), Some("en_US")); |
| 472 | assert_eq!(pick(1036, &hunspell), Some("fr_FR")); | 494 | assert_eq!(pick(1036, &hunspell), Some("fr_FR")); |
| 495 | let portuguese: Vec<String> = ["pt_PT", "pt_BR"].map(String::from).into(); | ||
| 496 | assert_eq!(pick(1046, &portuguese), Some("pt_BR")); | ||
| 497 | assert_eq!(pick(2070, &portuguese), Some("pt_PT")); | ||
| 473 | } | 498 | } |
| 474 | 499 | ||
| 475 | fn marked(paragraph: &Paragraph) -> Vec<(&str, bool)> { | 500 | fn marked(paragraph: &Paragraph) -> Vec<(&str, bool)> { |
| 476 | check(&[paragraph], &Fake) | 501 | check(&[paragraph], &Fake, crate::language::EN_US) |
| 477 | .remove(0) | 502 | .remove(0) |
| 478 | .into_iter() | 503 | .into_iter() |
| 479 | .map(|mark| (&paragraph.text()[mark.range], mark.repeated)) | 504 | .map(|mark| (&paragraph.text()[mark.range], mark.repeated)) |
| ... | @@ -583,4 +608,38 @@ pub(crate) mod tests { | ... | @@ -583,4 +608,38 @@ pub(crate) mod tests { |
| 583 | .expect("the spelling thread wakes the host"); | 608 | .expect("the spelling thread wakes the host"); |
| 584 | assert_eq!(spelling.marks(&french).len(), 2); | 609 | assert_eq!(spelling.marks(&french).len(), 2); |
| 585 | } | 610 | } |
| 611 | |||
| 612 | #[test] | ||
| 613 | fn arriving_dictionaries_and_untagged_languages_check_paragraphs_again() { | ||
| 614 | struct Late(Arc<std::sync::atomic::AtomicBool>); | ||
| 615 | impl Dictionary for Late { | ||
| 616 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { | ||
| 617 | match self.0.load(std::sync::atomic::Ordering::Relaxed) { | ||
| 618 | true => Fake.misspelled(words), | ||
| 619 | false => vec![false; words.len()], | ||
| 620 | } | ||
| 621 | } | ||
| 622 | fn suggest(&self, _: &str, _: u32) -> Vec<String> { | ||
| 623 | Vec::new() | ||
| 624 | } | ||
| 625 | fn learn(&self, _: &str) {} | ||
| 626 | } | ||
| 627 | let arrived = Arc::new(std::sync::atomic::AtomicBool::new(false)); | ||
| 628 | let spelling = Spelling::new( | ||
| 629 | Box::new(Late(Arc::clone(&arrived))), | ||
| 630 | std::task::Waker::noop().clone(), | ||
| 631 | ); | ||
| 632 | let paragraph = Paragraph::new("Ths sentense here".into(), Format::default()); | ||
| 633 | assert!(spelling.marks_now(&paragraph).is_empty()); | ||
| 634 | arrived.store(true, std::sync::atomic::Ordering::Relaxed); | ||
| 635 | assert!( | ||
| 636 | spelling.marks_now(&paragraph).is_empty(), | ||
| 637 | "kept until asked again" | ||
| 638 | ); | ||
| 639 | spelling.recheck(); | ||
| 640 | assert_eq!(spelling.marks_now(&paragraph).len(), 2); | ||
| 641 | // Untagged text is US English until the host says otherwise. | ||
| 642 | spelling.untagged(1036); | ||
| 643 | assert_eq!(spelling.marks_now(&paragraph).len(), 3); | ||
| 644 | } | ||
| 586 | } | 645 | } |
crates/snowbound/src/spell_web.rs+62-32| ... | @@ -1,59 +1,89 @@ | ... | @@ -1,59 +1,89 @@ |
| 1 | //! Spelling in the browser, which gives pages no checker to ask: Hunspell's American English | 1 | //! Spelling in the browser, which gives pages no checker to ask: Hunspell dictionaries through |
| 2 | //! dictionary through `spellbook`, which `index.html` fetches beside the module, with the | 2 | //! `spellbook`, each fetched beside the module the first time a word in its language is |
| 3 | //! words Add to Dictionary learns kept in the browser's files. | 3 | //! checked, with the words Add to Dictionary learns kept in the browser's files. |
| 4 | 4 | ||
| 5 | use canvas::spelling::{Dictionary, pick}; | 5 | use canvas::spelling::{Dictionary, pick}; |
| 6 | use std::{io::Write, sync::Mutex}; | 6 | use std::{collections::BTreeMap, io::Write, sync::Mutex}; |
| 7 | 7 | ||
| 8 | /// The dictionary's files, where `web::start` puts them, and the words learned. | ||
| 9 | pub const AFFIX: &str = "/Dictionaries/en_US.aff"; | ||
| 10 | pub const WORDS: &str = "/Dictionaries/en_US.dic"; | ||
| 11 | const LEARNED: &str = "/Settings/dictionary.txt"; | 8 | const LEARNED: &str = "/Settings/dictionary.txt"; |
| 12 | 9 | ||
| 13 | struct Checker(Mutex<spellbook::Dictionary>); | 10 | /// The dictionaries the site has, as `en_US`, which `platform::start` lists. |
| 11 | static AVAILABLE: Mutex<Vec<String>> = Mutex::new(Vec::new()); | ||
| 12 | /// The dictionaries asked for, by name: `None` until one arrives, or where it failed to. | ||
| 13 | static LOADED: Mutex<BTreeMap<String, Option<spellbook::Dictionary>>> = Mutex::new(BTreeMap::new()); | ||
| 14 | 14 | ||
| 15 | struct Checker; | ||
| 16 | |||
| 17 | /// The checker, where the site has dictionaries. | ||
| 15 | pub fn dictionary() -> Option<Box<dyn Dictionary>> { | 18 | pub fn dictionary() -> Option<Box<dyn Dictionary>> { |
| 16 | let affix = notebook::fs::read_to_string(AFFIX).ok()?; | 19 | let available = AVAILABLE.lock().ok()?; |
| 17 | let words = notebook::fs::read_to_string(WORDS).ok()?; | 20 | (!available.is_empty()).then(|| Box::new(Checker) as Box<dyn Dictionary>) |
| 18 | let mut dictionary = spellbook::Dictionary::new(&affix, &words).ok()?; | 21 | } |
| 19 | for word in notebook::fs::read_to_string(LEARNED) | 22 | |
| 20 | .unwrap_or_default() | 23 | /// Lists the dictionaries the site has. |
| 21 | .lines() | 24 | pub fn offer(names: Vec<String>) { |
| 22 | { | 25 | if let Ok(mut available) = AVAILABLE.lock() { |
| 23 | let _ = dictionary.add(word); | 26 | *available = names; |
| 24 | } | 27 | } |
| 25 | Some(Box::new(Checker(Mutex::new(dictionary)))) | ||
| 26 | } | 28 | } |
| 27 | 29 | ||
| 28 | /// Whether the dictionary serves `language`, an LCID. | 30 | /// Dictionary `name` arrived as its affix and word files. |
| 29 | fn serves(language: u32) -> bool { | 31 | pub fn arrived(name: &str, affix: &str, words: &str) { |
| 30 | pick(language, &["en_US".to_owned()]).is_some() | 32 | let dictionary = spellbook::Dictionary::new(affix, words) |
| 33 | .ok() | ||
| 34 | .map(|mut dictionary| { | ||
| 35 | for word in notebook::fs::read_to_string(LEARNED) | ||
| 36 | .unwrap_or_default() | ||
| 37 | .lines() | ||
| 38 | { | ||
| 39 | let _ = dictionary.add(word); | ||
| 40 | } | ||
| 41 | dictionary | ||
| 42 | }); | ||
| 43 | if let Ok(mut loaded) = LOADED.lock() { | ||
| 44 | loaded.insert(name.to_owned(), dictionary); | ||
| 45 | } | ||
| 46 | } | ||
| 47 | |||
| 48 | /// Runs `check` with the dictionary serving `language`, an LCID, asking for it the first | ||
| 49 | /// time; `None` until it arrives, or where none serves the language. | ||
| 50 | fn with<T>(language: u32, check: impl FnOnce(&mut spellbook::Dictionary) -> T) -> Option<T> { | ||
| 51 | let name = pick(language, &AVAILABLE.lock().ok()?)?.to_owned(); | ||
| 52 | let mut loaded = LOADED.lock().ok()?; | ||
| 53 | match loaded.get_mut(&name) { | ||
| 54 | Some(dictionary) => dictionary.as_mut().map(check), | ||
| 55 | None => { | ||
| 56 | loaded.insert(name.clone(), None); | ||
| 57 | crate::platform::fetch_dictionary(&name); | ||
| 58 | None | ||
| 59 | } | ||
| 60 | } | ||
| 31 | } | 61 | } |
| 32 | 62 | ||
| 33 | impl Dictionary for Checker { | 63 | impl Dictionary for Checker { |
| 34 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { | 64 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { |
| 35 | let Ok(dictionary) = self.0.lock() else { | ||
| 36 | return vec![false; words.len()]; | ||
| 37 | }; | ||
| 38 | words | 65 | words |
| 39 | .iter() | 66 | .iter() |
| 40 | .map(|(word, language)| serves(*language) && !dictionary.check(word)) | 67 | .map(|(word, language)| { |
| 68 | with(*language, |dictionary| !dictionary.check(word)).unwrap_or(false) | ||
| 69 | }) | ||
| 41 | .collect() | 70 | .collect() |
| 42 | } | 71 | } |
| 43 | 72 | ||
| 44 | fn suggest(&self, word: &str, language: u32) -> Vec<String> { | 73 | fn suggest(&self, word: &str, language: u32) -> Vec<String> { |
| 45 | let mut suggestions = Vec::new(); | 74 | with(language, |dictionary| { |
| 46 | if serves(language) | 75 | let mut suggestions = Vec::new(); |
| 47 | && let Ok(dictionary) = self.0.lock() | ||
| 48 | { | ||
| 49 | dictionary.suggest(word, &mut suggestions); | 76 | dictionary.suggest(word, &mut suggestions); |
| 50 | } | 77 | suggestions |
| 51 | suggestions | 78 | }) |
| 79 | .unwrap_or_default() | ||
| 52 | } | 80 | } |
| 53 | 81 | ||
| 54 | fn learn(&self, word: &str) { | 82 | fn learn(&self, word: &str) { |
| 55 | if let Ok(mut dictionary) = self.0.lock() { | 83 | if let Ok(mut loaded) = LOADED.lock() { |
| 56 | let _ = dictionary.add(word); | 84 | for dictionary in loaded.values_mut().flatten() { |
| 85 | let _ = dictionary.add(word); | ||
| 86 | } | ||
| 57 | } | 87 | } |
| 58 | let _ = notebook::fs::OpenOptions::new() | 88 | let _ = notebook::fs::OpenOptions::new() |
| 59 | .create(true) | 89 | .create(true) |
crates/snowbound/src/web.rs+26-10| ... | @@ -80,6 +80,9 @@ extern "C" { | ... | @@ -80,6 +80,9 @@ extern "C" { |
| 80 | /// Fetches `fonts/{name}` for `font_arrived`. | 80 | /// Fetches `fonts/{name}` for `font_arrived`. |
| 81 | #[wasm_bindgen(js_name = fetchFont)] | 81 | #[wasm_bindgen(js_name = fetchFont)] |
| 82 | fn fetch_font(name: &str); | 82 | fn fetch_font(name: &str); |
| 83 | /// Fetches dictionary `name`'s files for `dictionary_arrived`. | ||
| 84 | #[wasm_bindgen(js_name = fetchDictionary)] | ||
| 85 | pub fn fetch_dictionary(name: &str); | ||
| 83 | #[wasm_bindgen(js_name = storeFiles)] | 86 | #[wasm_bindgen(js_name = storeFiles)] |
| 84 | fn store_files(changes: js_sys::Array); | 87 | fn store_files(changes: js_sys::Array); |
| 85 | } | 88 | } |
| ... | @@ -865,14 +868,14 @@ pub mod smb { | ... | @@ -865,14 +868,14 @@ pub mod smb { |
| 865 | } | 868 | } |
| 866 | } | 869 | } |
| 867 | 870 | ||
| 868 | /// Starts Snowbound on the page's canvas, `index.html` having fetched `fonts` and, where it | 871 | /// Starts Snowbound on the page's canvas, `index.html` having fetched `fonts` and the names |
| 869 | /// could, the spelling dictionary's affix and word files. `module` is the module's own | 872 | /// of the spelling dictionaries the site has. `module` is the module's own exports, which |
| 870 | /// exports, which the glue calls. | 873 | /// the glue calls. |
| 871 | #[wasm_bindgen] | 874 | #[wasm_bindgen] |
| 872 | pub async fn start( | 875 | pub async fn start( |
| 873 | module: JsValue, | 876 | module: JsValue, |
| 874 | fonts: Vec<js_sys::Uint8Array>, | 877 | fonts: Vec<js_sys::Uint8Array>, |
| 875 | dictionary: Vec<js_sys::Uint8Array>, | 878 | dictionaries: Vec<String>, |
| 876 | ) -> Result<(), JsValue> { | 879 | ) -> Result<(), JsValue> { |
| 877 | std::panic::set_hook(Box::new(|info| report(info))); | 880 | std::panic::set_hook(Box::new(|info| report(info))); |
| 878 | let window = web_sys::window().ok_or("No window")?; | 881 | let window = web_sys::window().ok_or("No window")?; |
| ... | @@ -901,12 +904,7 @@ pub async fn start( | ... | @@ -901,12 +904,7 @@ pub async fn start( |
| 901 | canvas.set_width(host(|host| host.size.width)); | 904 | canvas.set_width(host(|host| host.size.width)); |
| 902 | canvas.set_height(host(|host| host.size.height)); | 905 | canvas.set_height(host(|host| host.size.height)); |
| 903 | restore(load_files().await?); | 906 | restore(load_files().await?); |
| 904 | if let [affix, words] = dictionary.as_slice() { | 907 | crate::spell::offer(dictionaries); |
| 905 | notebook::fs::restore("/Dictionaries", notebook::fs::Saved::Directory); | ||
| 906 | for (path, file) in [(crate::spell::AFFIX, affix), (crate::spell::WORDS, words)] { | ||
| 907 | notebook::fs::restore(path, notebook::fs::Saved::File(file.to_vec(), 0.0)); | ||
| 908 | } | ||
| 909 | } | ||
| 910 | for folder in load_folders().await?.iter() { | 908 | for folder in load_folders().await?.iter() { |
| 911 | let folder = js_sys::Array::from(&folder); | 909 | let folder = js_sys::Array::from(&folder); |
| 912 | let root = folder.get(0).as_string().unwrap_or_default(); | 910 | let root = folder.get(0).as_string().unwrap_or_default(); |
| ... | @@ -916,6 +914,10 @@ pub async fn start( | ... | @@ -916,6 +914,10 @@ pub async fn start( |
| 916 | let state = open(fonts) | 914 | let state = open(fonts) |
| 917 | .await | 915 | .await |
| 918 | .map_err(|error| JsValue::from_str(&error.to_string()))?; | 916 | .map_err(|error| JsValue::from_str(&error.to_string()))?; |
| 917 | // Text no run tags is in the browser's language, as typed in it. | ||
| 918 | if let Some(spelling) = &state.spelling { | ||
| 919 | spelling.untagged(canvas::language::lcid(&input_language())); | ||
| 920 | } | ||
| 919 | STATE.with_borrow_mut(|slot| *slot = Some(state)); | 921 | STATE.with_borrow_mut(|slot| *slot = Some(state)); |
| 920 | attach(module); | 922 | attach(module); |
| 921 | request_frame(); | 923 | request_frame(); |
| ... | @@ -1423,6 +1425,20 @@ pub fn font_arrived(name: String, bytes: Vec<u8>) { | ... | @@ -1423,6 +1425,20 @@ pub fn font_arrived(name: String, bytes: Vec<u8>) { |
| 1423 | }))); | 1425 | }))); |
| 1424 | } | 1426 | } |
| 1425 | 1427 | ||
| 1428 | /// Spelling dictionary `name` arrived as its affix and word files: words in its languages | ||
| 1429 | /// are checked again. | ||
| 1430 | #[wasm_bindgen] | ||
| 1431 | pub fn dictionary_arrived(name: String, affix: String, words: String) { | ||
| 1432 | crate::spell::arrived(&name, &affix, &words); | ||
| 1433 | send(UserEvent::Then(Box::new(|state| { | ||
| 1434 | if let Some(spelling) = &state.spelling { | ||
| 1435 | spelling.recheck(); | ||
| 1436 | } | ||
| 1437 | state.window.request_redraw(); | ||
| 1438 | Ok(()) | ||
| 1439 | }))); | ||
| 1440 | } | ||
| 1441 | |||
| 1426 | /// Parks the text area while the page takes no text, keeping it writable for the | 1442 | /// Parks the text area while the page takes no text, keeping it writable for the |
| 1427 | /// interface's fields. | 1443 | /// interface's fields. |
| 1428 | fn park_input(state: &State) { | 1444 | fn park_input(state: &State) { |
crates/snowbound/web/glue.js+18-8| ... | @@ -287,19 +287,29 @@ export function pickNotebook() { | ... | @@ -287,19 +287,29 @@ export function pickNotebook() { |
| 287 | .catch((error) => error.name === "AbortError" || console.error("Opening the folder", error)); | 287 | .catch((error) => error.name === "AbortError" || console.error("Opening the folder", error)); |
| 288 | } | 288 | } |
| 289 | 289 | ||
| 290 | // The bytes at `url`, inflated where it ends in .gz. | ||
| 291 | async function inflated(url) { | ||
| 292 | const response = await fetch(url); | ||
| 293 | if (!response.ok) throw new Error(`${url}: ${response.status}`); | ||
| 294 | const body = url.endsWith(".gz") | ||
| 295 | ? new Response(response.body.pipeThrough(new DecompressionStream("gzip"))) | ||
| 296 | : response; | ||
| 297 | return body.arrayBuffer(); | ||
| 298 | } | ||
| 299 | |||
| 290 | export function fetchFont(name) { | 300 | export function fetchFont(name) { |
| 291 | fetch(`fonts/${name}`) | 301 | inflated(`fonts/${name}`) |
| 292 | .then((response) => { | ||
| 293 | if (!response.ok) return Promise.reject(response.status); | ||
| 294 | const body = name.endsWith(".gz") | ||
| 295 | ? new Response(response.body.pipeThrough(new DecompressionStream("gzip"))) | ||
| 296 | : response; | ||
| 297 | return body.arrayBuffer(); | ||
| 298 | }) | ||
| 299 | .then((data) => wasm.font_arrived(name, new Uint8Array(data))) | 302 | .then((data) => wasm.font_arrived(name, new Uint8Array(data))) |
| 300 | .catch((error) => console.warn("Font", name, error)); | 303 | .catch((error) => console.warn("Font", name, error)); |
| 301 | } | 304 | } |
| 302 | 305 | ||
| 306 | export function fetchDictionary(name) { | ||
| 307 | const text = new TextDecoder(); | ||
| 308 | Promise.all(["aff", "dic"].map((kind) => inflated(`dictionaries/${name}.${kind}.gz`))) | ||
| 309 | .then(([affix, words]) => wasm.dictionary_arrived(name, text.decode(affix), text.decode(words))) | ||
| 310 | .catch((error) => console.warn("Dictionary", name, error)); | ||
| 311 | } | ||
| 312 | |||
| 303 | export function requestFrame() { | 313 | export function requestFrame() { |
| 304 | if (!framePending) { | 314 | if (!framePending) { |
| 305 | framePending = true; | 315 | framePending = true; |
crates/snowbound/web/index.html+5-4| ... | @@ -83,13 +83,14 @@ | ... | @@ -83,13 +83,14 @@ |
| 83 | } | 83 | } |
| 84 | try { | 84 | try { |
| 85 | const fetched = (url) => bytes(url).then((data) => new Uint8Array(data)); | 85 | const fetched = (url) => bytes(url).then((data) => new Uint8Array(data)); |
| 86 | const [, fonts, dictionary] = await Promise.all([ | 86 | const [, fonts, dictionaries] = await Promise.all([ |
| 87 | init({ module_or_path: module }), | 87 | init({ module_or_path: module }), |
| 88 | Promise.all(FONTS.map((name) => fetched(`fonts/${name}.ttf`))), | 88 | Promise.all(FONTS.map((name) => fetched(`fonts/${name}.ttf`))), |
| 89 | // Spelling waits for none: without the dictionary, words go unmarked. | 89 | // The spelling dictionaries the site has, fetched as pages need them; without the |
| 90 | Promise.all(["aff", "dic"].map((kind) => fetched(`dictionaries/en_US.${kind}`))).catch(() => []), | 90 | // list, words go unmarked. |
| 91 | fetch("dictionaries/index.json").then((response) => response.json()).catch(() => []), | ||
| 91 | ]); | 92 | ]); |
| 92 | await snowbound.start(snowbound, fonts, dictionary); | 93 | await snowbound.start(snowbound, fonts, dictionaries); |
| 93 | status.remove(); | 94 | status.remove(); |
| 94 | console.info(`Snowbound loaded in ${Math.round(performance.now())} ms`); | 95 | console.info(`Snowbound loaded in ${Math.round(performance.now())} ms`); |
| 95 | } catch (error) { | 96 | } catch (error) { |
tools/release_web.py+48-5| ... | @@ -8,6 +8,7 @@ static folder and publishes it to the share's web/ folder. | ... | @@ -8,6 +8,7 @@ static folder and publishes it to the share's web/ folder. |
| 8 | import argparse | 8 | import argparse |
| 9 | import gzip | 9 | import gzip |
| 10 | import hashlib | 10 | import hashlib |
| 11 | import json | ||
| 11 | import os | 12 | import os |
| 12 | from pathlib import Path | 13 | from pathlib import Path |
| 13 | import shutil | 14 | import shutil |
| ... | @@ -28,6 +29,22 @@ EMOJI_URL = 'https://github.com/googlefonts/noto-emoji/raw/v2.051/fonts/Noto-COL | ... | @@ -28,6 +29,22 @@ EMOJI_URL = 'https://github.com/googlefonts/noto-emoji/raw/v2.051/fonts/Noto-COL |
| 28 | EMOJI_SHA256 = '0ae57fe58645638523ba35f388d93739d292539a9acb84df5700c81b1e1a28d2' | 29 | EMOJI_SHA256 = '0ae57fe58645638523ba35f388d93739d292539a9acb84df5700c81b1e1a28d2' |
| 29 | # As `EMOJI` in src/web.rs names it. | 30 | # As `EMOJI` in src/web.rs names it. |
| 30 | EMOJI_FONT = 'Noto-COLRv1.ttf.gz' | 31 | EMOJI_FONT = 'Noto-COLRv1.ttf.gz' |
| 32 | # Spelling dictionaries: the name the page asks for, the nixpkgs hunspellDicts package, its | ||
| 33 | # files' stem, and the SPDX licences it comes under, each checked to allow redistribution | ||
| 34 | # (the LGPL's text with the GPL's, which it amends). Add a row to offer another language. | ||
| 35 | DICTIONARIES = [ | ||
| 36 | ('en_US', 'en_US', 'en_US', ['BSD-3-Clause']), | ||
| 37 | ('en_GB', 'en_GB-ise', 'en_GB', ['BSD-3-Clause']), | ||
| 38 | ('de_DE', 'de_DE', 'de_DE', ['GPL-2.0-only', 'GPL-3.0-only']), | ||
| 39 | ('fr_FR', 'fr-moderne', 'fr-moderne', ['MPL-2.0']), | ||
| 40 | ('es_ES', 'es_ES', 'es_ES', ['GPL-3.0-only', 'LGPL-3.0-only', 'MPL-1.1']), | ||
| 41 | ('pt_BR', 'pt_BR', 'pt_BR', ['LGPL-3.0-only', 'GPL-3.0-only']), | ||
| 42 | ('pt_PT', 'pt_PT', 'pt_PT', ['GPL-2.0-only', 'LGPL-2.1-only', 'MPL-1.1']), | ||
| 43 | ('it_IT', 'it_IT', 'it_IT', ['GPL-3.0-only']), | ||
| 44 | ('nl_NL', 'nl_NL', 'nl_NL', ['BSD-3-Clause', 'CC-BY-3.0']), | ||
| 45 | ('sv_SE', 'sv_SE', 'sv_SE', ['LGPL-3.0-only', 'GPL-3.0-only']), | ||
| 46 | ('ru_RU', 'ru_RU', 'ru_RU', ['MPL-2.0', 'LGPL-3.0-only', 'GPL-3.0-only']), | ||
| 47 | ] | ||
| 31 | # Size over speed where it costs little: the module is most of the first load. | 48 | # Size over speed where it costs little: the module is most of the first load. |
| 32 | PROFILE = ['--config', 'profile.release.opt-level="s"'] | 49 | PROFILE = ['--config', 'profile.release.opt-level="s"'] |
| 33 | 50 | ||
| ... | @@ -75,11 +92,7 @@ def build(out): | ... | @@ -75,11 +92,7 @@ def build(out): |
| 75 | bound = out / 'snowbound_web_bg.wasm' | 92 | bound = out / 'snowbound_web_bg.wasm' |
| 76 | run([tool('wasm-opt', 'binaryen'), '-Oz', '--strip-debug', '--strip-producers', bound, '-o', bound]) | 93 | run([tool('wasm-opt', 'binaryen'), '-Oz', '--strip-debug', '--strip-producers', bound, '-o', bound]) |
| 77 | shutil.copy(WEB / 'index.html', out) | 94 | shutil.copy(WEB / 'index.html', out) |
| 78 | # Hunspell's American English dictionary (SCOWL, BSD-3-Clause), for spelling. | 95 | dictionaries(out / 'dictionaries') |
| 79 | (out / 'dictionaries').mkdir() | ||
| 80 | store = Path(tool('hunspell', 'hunspellDicts.en_US', store=True)) | ||
| 81 | for name in ('share/hunspell/en_US.aff', 'share/hunspell/en_US.dic', 'share/doc/hunspell-dict-en-us-wordlist.txt'): | ||
| 82 | shutil.copy(store / name, out / 'dictionaries') | ||
| 83 | (out / 'fonts').mkdir() | 96 | (out / 'fonts').mkdir() |
| 84 | for font in sorted(FONTS.glob('*')): | 97 | for font in sorted(FONTS.glob('*')): |
| 85 | if font.suffix in ('.ttf', '.txt'): | 98 | if font.suffix in ('.ttf', '.txt'): |
| ... | @@ -90,6 +103,36 @@ def build(out): | ... | @@ -90,6 +103,36 @@ def build(out): |
| 90 | print(f'{path.stat().st_size:>12,} {path.relative_to(out)}') | 103 | print(f'{path.stat().st_size:>12,} {path.relative_to(out)}') |
| 91 | 104 | ||
| 92 | 105 | ||
| 106 | def dictionaries(folder): | ||
| 107 | """The Hunspell dictionaries in DICTIONARIES as UTF-8, gzipped for the glue to inflate, | ||
| 108 | each beside its readme, the texts of the licences they come under in licenses/, and their | ||
| 109 | names in index.json for the page to pick from (`pick` in canvas/src/spelling.rs).""" | ||
| 110 | (folder / 'licenses').mkdir(parents=True) | ||
| 111 | texts = Path(tool('spdx', 'spdx-license-list-data.text', store=True)) / 'text' | ||
| 112 | for name, package, stem, licenses in DICTIONARIES: | ||
| 113 | store = Path(tool('hunspell', f'hunspellDicts.{package}', store=True)) | ||
| 114 | affix = (store / f'share/hunspell/{stem}.aff').read_bytes() | ||
| 115 | # Hunspell names the files' encoding in the affix file's SET line. | ||
| 116 | encoding = next((line.split()[1] for line in affix.decode('latin-1').splitlines() | ||
| 117 | if line.startswith('SET ')), 'UTF-8') | ||
| 118 | for kind in ('aff', 'dic'): | ||
| 119 | text = (store / f'share/hunspell/{stem}.{kind}').read_bytes().decode(encoding) | ||
| 120 | # One encoding and one line ending for spellbook, which reads only UTF-8. An | ||
| 121 | # 8-bit file's flags are its characters by default, as FLAG UTF-8 keeps them. | ||
| 122 | utf8 = 'SET UTF-8' if encoding == 'UTF-8' or 'FLAG ' in text else 'SET UTF-8\nFLAG UTF-8' | ||
| 123 | text = ''.join((utf8 if line.startswith('SET ') else line) + '\n' | ||
| 124 | for line in text.splitlines()) | ||
| 125 | (folder / f'{name}.{kind}.gz').write_bytes(gzip.compress(text.encode(), 9, mtime=0)) | ||
| 126 | # Where a package has no readme, its affix file's header comment names the licences. | ||
| 127 | readme = list((store / 'share/doc').glob('*.txt')) | ||
| 128 | notice = readme[0].read_bytes() if readme else \ | ||
| 129 | ''.join(line + '\n' for line in affix.decode(encoding).splitlines() if line.startswith('#')).encode() | ||
| 130 | (folder / f'{name}.txt').write_bytes(notice + f'\nLicences: {", ".join(licenses)} (licenses/)\n'.encode()) | ||
| 131 | for license in licenses: | ||
| 132 | shutil.copyfile(texts / f'{license}.txt', folder / 'licenses' / f'{license}.txt') | ||
| 133 | (folder / 'index.json').write_text(json.dumps([name for name, *_ in DICTIONARIES])) | ||
| 134 | |||
| 135 | |||
| 93 | def fallbacks(fonts): | 136 | def fallbacks(fonts): |
| 94 | """Noto's faces for scripts the bundled ones lack, which the page fetches as it needs them | 137 | """Noto's faces for scripts the bundled ones lack, which the page fetches as it needs them |
| 95 | (`FALLBACKS` in src/web.rs), under the SIL Open Font License.""" | 138 | (`FALLBACKS` in src/web.rs), under the SIL Open Font License.""" |