| author | |
| committer | |
| log | f837389ba02268ce5ee180e09bba96a337be406b |
| tree | 44c9676ca81879e4006f58c1b30dadbe14ef43a4 |
| parent | 21adbb5dfae140e40626cefd52ed377a44977c7f |
| signature | Signed by SSH key SHA256:52mNGHRsVFBDED9IAX5pe+LRWUefqTbxEReunq21QvU |
Dictionaries are fetched the first time a word in their language is
checked, and text no run tags counts as the browser's language. The site
offers en_US, en_GB, de_DE, fr_FR, es_ES, pt_BR, pt_PT, it_IT, nl_NL,
sv_SE and ru_RU from nixpkgs' Hunspell packages, converted to UTF-8 and
gzipped, each with its readme and its licences' texts; a row in
release_web.py's DICTIONARIES adds another. A bare pt picks pt_BR, as
Windows does.
Assisted-by: claude-opus-5.57 files changed, 233 insertions(+), 73 deletions(-)
arc/platforms.md+3-2| ... | ... | @@ -177,8 +177,9 @@ keyboard, the toolbar and the macOS menu bar all run commands from it. |
| 177 | 177 | browser keeps its own window and tab chords. Dialogs are the browser's; Insert, and Open |
| 178 | 178 | without a folder to give, ask for files to copy in; printing downloads the PDF. Servers and |
| 179 | 179 | recording are still to come. |
| 180 | - Spelling is Hunspell's American English dictionary through `spellbook`, fetched beside the | |
| 181 | module; Add to Dictionary keeps its words in the browser's files. Text the bundled faces | |
| 180 | - Spelling is Hunspell's dictionaries through `spellbook`, each fetched beside the module the | |
| 181 | first time a word in its language is checked, text no run tags counting as the browser's | |
| 182 | language; Add to Dictionary keeps its words in the browser's files. Text the bundled faces | |
| 182 | 183 | lack takes Noto's, fetched by script the first time a page holds it (a CJK face cut to |
| 183 | 184 | the national standards' characters by `tools/web/subset_cjk.py`; emoji in Noto Color |
| 184 | 185 | Emoji's COLRv1 outlines, which `draw` paints itself), and the page is laid out again. |
crates/canvas/src/spelling.rs+71-12| ... | ... | @@ -8,6 +8,7 @@ use onestore::page::text::Paragraph; |
| 8 | 8 | use std::collections::{HashMap, HashSet}; |
| 9 | 9 | use std::hash::{DefaultHasher, Hash, Hasher}; |
| 10 | 10 | use std::ops::Range; |
| 11 | use std::sync::atomic::{AtomicU32, Ordering}; | |
| 11 | 12 | use std::sync::{Arc, Mutex, mpsc}; |
| 12 | 13 | |
| 13 | 14 | /// Checked paragraphs kept before the cache starts over. |
| ... | ... | @@ -36,6 +37,7 @@ pub fn pick(language: u32, available: &[String]) -> Option<&str> { |
| 36 | 37 | // A bare tag is the locale Windows picks, most often the language's own country. |
| 37 | 38 | let home = match bare { |
| 38 | 39 | "en" => "US".to_owned(), |
| 40 | "pt" => "BR".to_owned(), | |
| 39 | 41 | _ => bare.to_uppercase(), |
| 40 | 42 | }; |
| 41 | 43 | let wanted = if region.is_empty() { |
| ... | ... | @@ -70,7 +72,8 @@ pub(crate) struct Mark { |
| 70 | 72 | /// A word of a paragraph: its bytes, its language, and whether its spelling is checked. |
| 71 | 73 | struct Word { |
| 72 | 74 | range: Range<usize>, |
| 73 | language: u32, | |
| 75 | /// The run's language, an LCID; `None` where untagged. | |
| 76 | language: Option<u32>, | |
| 74 | 77 | checked: bool, |
| 75 | 78 | } |
| 76 | 79 | |
| ... | ... | @@ -157,9 +160,7 @@ fn words(paragraph: &Paragraph) -> Vec<Word> { |
| 157 | 160 | } |
| 158 | 161 | words.push(Word { |
| 159 | 162 | range: at + offset..at + end, |
| 160 | language: format(at + offset) | |
| 161 | .language | |
| 162 | .unwrap_or(crate::language::EN_US), | |
| 163 | language: format(at + offset).language, | |
| 163 | 164 | checked: word.chars().any(char::is_lowercase), |
| 164 | 165 | }); |
| 165 | 166 | } |
| ... | ... | @@ -203,7 +204,8 @@ fn repeated(text: &str, words: &[Word]) -> Vec<bool> { |
| 203 | 204 | |
| 204 | 205 | /// The marks `dictionary` gives each of `paragraphs`, asking it once: each word repeating |
| 205 | 206 | /// the one before it, and each other checked word it finds misspelled. |
| 206 | fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary) -> Vec<Vec<Mark>> { | |
| 207 | /// Checks `paragraphs`, reading text no run tags as language `untagged`. | |
| 208 | fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary, untagged: u32) -> Vec<Vec<Mark>> { | |
| 207 | 209 | let words: Vec<(Vec<Word>, Vec<bool>)> = paragraphs |
| 208 | 210 | .iter() |
| 209 | 211 | .map(|paragraph| { |
| ... | ... | @@ -220,7 +222,10 @@ fn check(paragraphs: &[&Paragraph], dictionary: &dyn Dictionary) -> Vec<Vec<Mark |
| 220 | 222 | .iter() |
| 221 | 223 | .zip(repeated) |
| 222 | 224 | .filter(|(word, repeated)| word.checked && !**repeated) |
| 223 | .map(|(word, _)| (&paragraph.text()[word.range.clone()], word.language)) | |
| 225 | .map(|(word, _)| { | |
| 226 | let language = word.language.unwrap_or(untagged); | |
| 227 | (&paragraph.text()[word.range.clone()], language) | |
| 228 | }) | |
| 224 | 229 | }) |
| 225 | 230 | .collect(); |
| 226 | 231 | let mut misspelled = dictionary.misspelled(&asked).into_iter(); |
| ... | ... | @@ -309,6 +314,8 @@ impl State { |
| 309 | 314 | |
| 310 | 315 | struct Shared { |
| 311 | 316 | dictionary: Box<dyn Dictionary>, |
| 317 | /// The language text no run tags is checked in, an LCID: US English, as OneNote's. | |
| 318 | untagged: AtomicU32, | |
| 312 | 319 | state: Mutex<State>, |
| 313 | 320 | } |
| 314 | 321 | |
| ... | ... | @@ -324,6 +331,7 @@ impl Spelling { |
| 324 | 331 | pub fn new(dictionary: Box<dyn Dictionary>, redraw: std::task::Waker) -> Self { |
| 325 | 332 | let shared = Arc::new(Shared { |
| 326 | 333 | dictionary, |
| 334 | untagged: AtomicU32::new(crate::language::EN_US), | |
| 327 | 335 | state: Mutex::default(), |
| 328 | 336 | }); |
| 329 | 337 | let (jobs, receiver) = mpsc::channel::<(u64, Paragraph)>(); |
| ... | ... | @@ -341,7 +349,8 @@ impl Spelling { |
| 341 | 349 | .collect(); |
| 342 | 350 | let paragraphs: Vec<&Paragraph> = |
| 343 | 351 | jobs.iter().map(|(_, paragraph)| paragraph).collect(); |
| 344 | let marks = check(&paragraphs, &*worker.dictionary); | |
| 352 | let untagged = worker.untagged.load(Ordering::Relaxed); | |
| 353 | let marks = check(&paragraphs, &*worker.dictionary, untagged); | |
| 345 | 354 | let mut state = worker.state.lock().unwrap(); |
| 346 | 355 | for ((key, paragraph), marks) in jobs.iter().zip(marks) { |
| 347 | 356 | state.pending.remove(key); |
| ... | ... | @@ -379,17 +388,30 @@ impl Spelling { |
| 379 | 388 | if let Some(marks) = self.shared.state.lock().unwrap().marks(key, paragraph) { |
| 380 | 389 | return marks; |
| 381 | 390 | } |
| 382 | let marks = check(&[paragraph], &*self.shared.dictionary).remove(0); | |
| 391 | let untagged = self.shared.untagged.load(Ordering::Relaxed); | |
| 392 | let marks = check(&[paragraph], &*self.shared.dictionary, untagged).remove(0); | |
| 383 | 393 | let mut state = self.shared.state.lock().unwrap(); |
| 384 | 394 | state.keep(key, paragraph, marks); |
| 385 | 395 | state.marks(key, paragraph).unwrap_or_default() |
| 386 | 396 | } |
| 387 | 397 | |
| 398 | /// Checks every paragraph again when next asked, as after a dictionary arrived. | |
| 399 | pub fn recheck(&self) { | |
| 400 | self.shared.state.lock().unwrap().checked.clear(); | |
| 401 | } | |
| 402 | ||
| 403 | /// Checks text no run tags in `language`, an LCID, as a browser's text is in its own. | |
| 404 | pub fn untagged(&self, language: u32) { | |
| 405 | self.shared.untagged.store(language, Ordering::Relaxed); | |
| 406 | self.recheck(); | |
| 407 | } | |
| 408 | ||
| 388 | 409 | /// Corrections for `word` in `language`, best first. |
| 389 | 410 | pub(crate) fn suggest(&self, word: &str, language: Option<u32>) -> Vec<String> { |
| 390 | self.shared | |
| 391 | .dictionary | |
| 392 | .suggest(word, language.unwrap_or(crate::language::EN_US)) | |
| 411 | self.shared.dictionary.suggest( | |
| 412 | word, | |
| 413 | language.unwrap_or(self.shared.untagged.load(Ordering::Relaxed)), | |
| 414 | ) | |
| 393 | 415 | } |
| 394 | 416 | |
| 395 | 417 | /// Ignore: leaves `word` unmarked everywhere until the app quits. |
| ... | ... | @@ -470,10 +492,13 @@ pub(crate) mod tests { |
| 470 | 492 | let hunspell: Vec<String> = ["en_AU", "en_US", "fr_FR"].map(String::from).into(); |
| 471 | 493 | assert_eq!(pick(1033, &hunspell), Some("en_US")); |
| 472 | 494 | assert_eq!(pick(1036, &hunspell), Some("fr_FR")); |
| 495 | let portuguese: Vec<String> = ["pt_PT", "pt_BR"].map(String::from).into(); | |
| 496 | assert_eq!(pick(1046, &portuguese), Some("pt_BR")); | |
| 497 | assert_eq!(pick(2070, &portuguese), Some("pt_PT")); | |
| 473 | 498 | } |
| 474 | 499 | |
| 475 | 500 | fn marked(paragraph: &Paragraph) -> Vec<(&str, bool)> { |
| 476 | check(&[paragraph], &Fake) | |
| 501 | check(&[paragraph], &Fake, crate::language::EN_US) | |
| 477 | 502 | .remove(0) |
| 478 | 503 | .into_iter() |
| 479 | 504 | .map(|mark| (&paragraph.text()[mark.range], mark.repeated)) |
| ... | ... | @@ -583,4 +608,38 @@ pub(crate) mod tests { |
| 583 | 608 | .expect("the spelling thread wakes the host"); |
| 584 | 609 | assert_eq!(spelling.marks(&french).len(), 2); |
| 585 | 610 | } |
| 611 | ||
| 612 | #[test] | |
| 613 | fn arriving_dictionaries_and_untagged_languages_check_paragraphs_again() { | |
| 614 | struct Late(Arc<std::sync::atomic::AtomicBool>); | |
| 615 | impl Dictionary for Late { | |
| 616 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { | |
| 617 | match self.0.load(std::sync::atomic::Ordering::Relaxed) { | |
| 618 | true => Fake.misspelled(words), | |
| 619 | false => vec![false; words.len()], | |
| 620 | } | |
| 621 | } | |
| 622 | fn suggest(&self, _: &str, _: u32) -> Vec<String> { | |
| 623 | Vec::new() | |
| 624 | } | |
| 625 | fn learn(&self, _: &str) {} | |
| 626 | } | |
| 627 | let arrived = Arc::new(std::sync::atomic::AtomicBool::new(false)); | |
| 628 | let spelling = Spelling::new( | |
| 629 | Box::new(Late(Arc::clone(&arrived))), | |
| 630 | std::task::Waker::noop().clone(), | |
| 631 | ); | |
| 632 | let paragraph = Paragraph::new("Ths sentense here".into(), Format::default()); | |
| 633 | assert!(spelling.marks_now(&paragraph).is_empty()); | |
| 634 | arrived.store(true, std::sync::atomic::Ordering::Relaxed); | |
| 635 | assert!( | |
| 636 | spelling.marks_now(&paragraph).is_empty(), | |
| 637 | "kept until asked again" | |
| 638 | ); | |
| 639 | spelling.recheck(); | |
| 640 | assert_eq!(spelling.marks_now(&paragraph).len(), 2); | |
| 641 | // Untagged text is US English until the host says otherwise. | |
| 642 | spelling.untagged(1036); | |
| 643 | assert_eq!(spelling.marks_now(&paragraph).len(), 3); | |
| 644 | } | |
| 586 | 645 | } |
crates/snowbound/src/spell_web.rs+62-32| ... | ... | @@ -1,59 +1,89 @@ |
| 1 | //! Spelling in the browser, which gives pages no checker to ask: Hunspell's American English | |
| 2 | //! dictionary through `spellbook`, which `index.html` fetches beside the module, with the | |
| 3 | //! words Add to Dictionary learns kept in the browser's files. | |
| 1 | //! Spelling in the browser, which gives pages no checker to ask: Hunspell dictionaries through | |
| 2 | //! `spellbook`, each fetched beside the module the first time a word in its language is | |
| 3 | //! checked, with the words Add to Dictionary learns kept in the browser's files. | |
| 4 | 4 | |
| 5 | 5 | use canvas::spelling::{Dictionary, pick}; |
| 6 | use std::{io::Write, sync::Mutex}; | |
| 6 | use std::{collections::BTreeMap, io::Write, sync::Mutex}; | |
| 7 | 7 | |
| 8 | /// The dictionary's files, where `web::start` puts them, and the words learned. | |
| 9 | pub const AFFIX: &str = "/Dictionaries/en_US.aff"; | |
| 10 | pub const WORDS: &str = "/Dictionaries/en_US.dic"; | |
| 11 | 8 | const LEARNED: &str = "/Settings/dictionary.txt"; |
| 12 | 9 | |
| 13 | struct Checker(Mutex<spellbook::Dictionary>); | |
| 10 | /// The dictionaries the site has, as `en_US`, which `platform::start` lists. | |
| 11 | static AVAILABLE: Mutex<Vec<String>> = Mutex::new(Vec::new()); | |
| 12 | /// The dictionaries asked for, by name: `None` until one arrives, or where it failed to. | |
| 13 | static LOADED: Mutex<BTreeMap<String, Option<spellbook::Dictionary>>> = Mutex::new(BTreeMap::new()); | |
| 14 | 14 | |
| 15 | struct Checker; | |
| 16 | ||
| 17 | /// The checker, where the site has dictionaries. | |
| 15 | 18 | pub fn dictionary() -> Option<Box<dyn Dictionary>> { |
| 16 | let affix = notebook::fs::read_to_string(AFFIX).ok()?; | |
| 17 | let words = notebook::fs::read_to_string(WORDS).ok()?; | |
| 18 | let mut dictionary = spellbook::Dictionary::new(&affix, &words).ok()?; | |
| 19 | for word in notebook::fs::read_to_string(LEARNED) | |
| 20 | .unwrap_or_default() | |
| 21 | .lines() | |
| 22 | { | |
| 23 | let _ = dictionary.add(word); | |
| 19 | let available = AVAILABLE.lock().ok()?; | |
| 20 | (!available.is_empty()).then(|| Box::new(Checker) as Box<dyn Dictionary>) | |
| 21 | } | |
| 22 | ||
| 23 | /// Lists the dictionaries the site has. | |
| 24 | pub fn offer(names: Vec<String>) { | |
| 25 | if let Ok(mut available) = AVAILABLE.lock() { | |
| 26 | *available = names; | |
| 24 | 27 | } |
| 25 | Some(Box::new(Checker(Mutex::new(dictionary)))) | |
| 26 | 28 | } |
| 27 | 29 | |
| 28 | /// Whether the dictionary serves `language`, an LCID. | |
| 29 | fn serves(language: u32) -> bool { | |
| 30 | pick(language, &["en_US".to_owned()]).is_some() | |
| 30 | /// Dictionary `name` arrived as its affix and word files. | |
| 31 | pub fn arrived(name: &str, affix: &str, words: &str) { | |
| 32 | let dictionary = spellbook::Dictionary::new(affix, words) | |
| 33 | .ok() | |
| 34 | .map(|mut dictionary| { | |
| 35 | for word in notebook::fs::read_to_string(LEARNED) | |
| 36 | .unwrap_or_default() | |
| 37 | .lines() | |
| 38 | { | |
| 39 | let _ = dictionary.add(word); | |
| 40 | } | |
| 41 | dictionary | |
| 42 | }); | |
| 43 | if let Ok(mut loaded) = LOADED.lock() { | |
| 44 | loaded.insert(name.to_owned(), dictionary); | |
| 45 | } | |
| 46 | } | |
| 47 | ||
| 48 | /// Runs `check` with the dictionary serving `language`, an LCID, asking for it the first | |
| 49 | /// time; `None` until it arrives, or where none serves the language. | |
| 50 | fn with<T>(language: u32, check: impl FnOnce(&mut spellbook::Dictionary) -> T) -> Option<T> { | |
| 51 | let name = pick(language, &AVAILABLE.lock().ok()?)?.to_owned(); | |
| 52 | let mut loaded = LOADED.lock().ok()?; | |
| 53 | match loaded.get_mut(&name) { | |
| 54 | Some(dictionary) => dictionary.as_mut().map(check), | |
| 55 | None => { | |
| 56 | loaded.insert(name.clone(), None); | |
| 57 | crate::platform::fetch_dictionary(&name); | |
| 58 | None | |
| 59 | } | |
| 60 | } | |
| 31 | 61 | } |
| 32 | 62 | |
| 33 | 63 | impl Dictionary for Checker { |
| 34 | 64 | fn misspelled(&self, words: &[(&str, u32)]) -> Vec<bool> { |
| 35 | let Ok(dictionary) = self.0.lock() else { | |
| 36 | return vec![false; words.len()]; | |
| 37 | }; | |
| 38 | 65 | words |
| 39 | 66 | .iter() |
| 40 | .map(|(word, language)| serves(*language) && !dictionary.check(word)) | |
| 67 | .map(|(word, language)| { | |
| 68 | with(*language, |dictionary| !dictionary.check(word)).unwrap_or(false) | |
| 69 | }) | |
| 41 | 70 | .collect() |
| 42 | 71 | } |
| 43 | 72 | |
| 44 | 73 | fn suggest(&self, word: &str, language: u32) -> Vec<String> { |
| 45 | let mut suggestions = Vec::new(); | |
| 46 | if serves(language) | |
| 47 | && let Ok(dictionary) = self.0.lock() | |
| 48 | { | |
| 74 | with(language, |dictionary| { | |
| 75 | let mut suggestions = Vec::new(); | |
| 49 | 76 | dictionary.suggest(word, &mut suggestions); |
| 50 | } | |
| 51 | suggestions | |
| 77 | suggestions | |
| 78 | }) | |
| 79 | .unwrap_or_default() | |
| 52 | 80 | } |
| 53 | 81 | |
| 54 | 82 | fn learn(&self, word: &str) { |
| 55 | if let Ok(mut dictionary) = self.0.lock() { | |
| 56 | let _ = dictionary.add(word); | |
| 83 | if let Ok(mut loaded) = LOADED.lock() { | |
| 84 | for dictionary in loaded.values_mut().flatten() { | |
| 85 | let _ = dictionary.add(word); | |
| 86 | } | |
| 57 | 87 | } |
| 58 | 88 | let _ = notebook::fs::OpenOptions::new() |
| 59 | 89 | .create(true) |
crates/snowbound/src/web.rs+26-10| ... | ... | @@ -80,6 +80,9 @@ extern "C" { |
| 80 | 80 | /// Fetches `fonts/{name}` for `font_arrived`. |
| 81 | 81 | #[wasm_bindgen(js_name = fetchFont)] |
| 82 | 82 | fn fetch_font(name: &str); |
| 83 | /// Fetches dictionary `name`'s files for `dictionary_arrived`. | |
| 84 | #[wasm_bindgen(js_name = fetchDictionary)] | |
| 85 | pub fn fetch_dictionary(name: &str); | |
| 83 | 86 | #[wasm_bindgen(js_name = storeFiles)] |
| 84 | 87 | fn store_files(changes: js_sys::Array); |
| 85 | 88 | } |
| ... | ... | @@ -865,14 +868,14 @@ pub mod smb { |
| 865 | 868 | } |
| 866 | 869 | } |
| 867 | 870 | |
| 868 | /// Starts Snowbound on the page's canvas, `index.html` having fetched `fonts` and, where it | |
| 869 | /// could, the spelling dictionary's affix and word files. `module` is the module's own | |
| 870 | /// exports, which the glue calls. | |
| 871 | /// Starts Snowbound on the page's canvas, `index.html` having fetched `fonts` and the names | |
| 872 | /// of the spelling dictionaries the site has. `module` is the module's own exports, which | |
| 873 | /// the glue calls. | |
| 871 | 874 | #[wasm_bindgen] |
| 872 | 875 | pub async fn start( |
| 873 | 876 | module: JsValue, |
| 874 | 877 | fonts: Vec<js_sys::Uint8Array>, |
| 875 | dictionary: Vec<js_sys::Uint8Array>, | |
| 878 | dictionaries: Vec<String>, | |
| 876 | 879 | ) -> Result<(), JsValue> { |
| 877 | 880 | std::panic::set_hook(Box::new(|info| report(info))); |
| 878 | 881 | let window = web_sys::window().ok_or("No window")?; |
| ... | ... | @@ -901,12 +904,7 @@ pub async fn start( |
| 901 | 904 | canvas.set_width(host(|host| host.size.width)); |
| 902 | 905 | canvas.set_height(host(|host| host.size.height)); |
| 903 | 906 | restore(load_files().await?); |
| 904 | if let [affix, words] = dictionary.as_slice() { | |
| 905 | notebook::fs::restore("/Dictionaries", notebook::fs::Saved::Directory); | |
| 906 | for (path, file) in [(crate::spell::AFFIX, affix), (crate::spell::WORDS, words)] { | |
| 907 | notebook::fs::restore(path, notebook::fs::Saved::File(file.to_vec(), 0.0)); | |
| 908 | } | |
| 909 | } | |
| 907 | crate::spell::offer(dictionaries); | |
| 910 | 908 | for folder in load_folders().await?.iter() { |
| 911 | 909 | let folder = js_sys::Array::from(&folder); |
| 912 | 910 | let root = folder.get(0).as_string().unwrap_or_default(); |
| ... | ... | @@ -916,6 +914,10 @@ pub async fn start( |
| 916 | 914 | let state = open(fonts) |
| 917 | 915 | .await |
| 918 | 916 | .map_err(|error| JsValue::from_str(&error.to_string()))?; |
| 917 | // Text no run tags is in the browser's language, as typed in it. | |
| 918 | if let Some(spelling) = &state.spelling { | |
| 919 | spelling.untagged(canvas::language::lcid(&input_language())); | |
| 920 | } | |
| 919 | 921 | STATE.with_borrow_mut(|slot| *slot = Some(state)); |
| 920 | 922 | attach(module); |
| 921 | 923 | request_frame(); |
| ... | ... | @@ -1423,6 +1425,20 @@ pub fn font_arrived(name: String, bytes: Vec<u8>) { |
| 1423 | 1425 | }))); |
| 1424 | 1426 | } |
| 1425 | 1427 | |
| 1428 | /// Spelling dictionary `name` arrived as its affix and word files: words in its languages | |
| 1429 | /// are checked again. | |
| 1430 | #[wasm_bindgen] | |
| 1431 | pub fn dictionary_arrived(name: String, affix: String, words: String) { | |
| 1432 | crate::spell::arrived(&name, &affix, &words); | |
| 1433 | send(UserEvent::Then(Box::new(|state| { | |
| 1434 | if let Some(spelling) = &state.spelling { | |
| 1435 | spelling.recheck(); | |
| 1436 | } | |
| 1437 | state.window.request_redraw(); | |
| 1438 | Ok(()) | |
| 1439 | }))); | |
| 1440 | } | |
| 1441 | ||
| 1426 | 1442 | /// Parks the text area while the page takes no text, keeping it writable for the |
| 1427 | 1443 | /// interface's fields. |
| 1428 | 1444 | fn park_input(state: &State) { |
crates/snowbound/web/glue.js+18-8| ... | ... | @@ -287,19 +287,29 @@ export function pickNotebook() { |
| 287 | 287 | .catch((error) => error.name === "AbortError" || console.error("Opening the folder", error)); |
| 288 | 288 | } |
| 289 | 289 | |
| 290 | // The bytes at `url`, inflated where it ends in .gz. | |
| 291 | async function inflated(url) { | |
| 292 | const response = await fetch(url); | |
| 293 | if (!response.ok) throw new Error(`${url}: ${response.status}`); | |
| 294 | const body = url.endsWith(".gz") | |
| 295 | ? new Response(response.body.pipeThrough(new DecompressionStream("gzip"))) | |
| 296 | : response; | |
| 297 | return body.arrayBuffer(); | |
| 298 | } | |
| 299 | ||
| 290 | 300 | export function fetchFont(name) { |
| 291 | fetch(`fonts/${name}`) | |
| 292 | .then((response) => { | |
| 293 | if (!response.ok) return Promise.reject(response.status); | |
| 294 | const body = name.endsWith(".gz") | |
| 295 | ? new Response(response.body.pipeThrough(new DecompressionStream("gzip"))) | |
| 296 | : response; | |
| 297 | return body.arrayBuffer(); | |
| 298 | }) | |
| 301 | inflated(`fonts/${name}`) | |
| 299 | 302 | .then((data) => wasm.font_arrived(name, new Uint8Array(data))) |
| 300 | 303 | .catch((error) => console.warn("Font", name, error)); |
| 301 | 304 | } |
| 302 | 305 | |
| 306 | export function fetchDictionary(name) { | |
| 307 | const text = new TextDecoder(); | |
| 308 | Promise.all(["aff", "dic"].map((kind) => inflated(`dictionaries/${name}.${kind}.gz`))) | |
| 309 | .then(([affix, words]) => wasm.dictionary_arrived(name, text.decode(affix), text.decode(words))) | |
| 310 | .catch((error) => console.warn("Dictionary", name, error)); | |
| 311 | } | |
| 312 | ||
| 303 | 313 | export function requestFrame() { |
| 304 | 314 | if (!framePending) { |
| 305 | 315 | framePending = true; |
crates/snowbound/web/index.html+5-4| ... | ... | @@ -83,13 +83,14 @@ |
| 83 | 83 | } |
| 84 | 84 | try { |
| 85 | 85 | const fetched = (url) => bytes(url).then((data) => new Uint8Array(data)); |
| 86 | const [, fonts, dictionary] = await Promise.all([ | |
| 86 | const [, fonts, dictionaries] = await Promise.all([ | |
| 87 | 87 | init({ module_or_path: module }), |
| 88 | 88 | Promise.all(FONTS.map((name) => fetched(`fonts/${name}.ttf`))), |
| 89 | // Spelling waits for none: without the dictionary, words go unmarked. | |
| 90 | Promise.all(["aff", "dic"].map((kind) => fetched(`dictionaries/en_US.${kind}`))).catch(() => []), | |
| 89 | // The spelling dictionaries the site has, fetched as pages need them; without the | |
| 90 | // list, words go unmarked. | |
| 91 | fetch("dictionaries/index.json").then((response) => response.json()).catch(() => []), | |
| 91 | 92 | ]); |
| 92 | await snowbound.start(snowbound, fonts, dictionary); | |
| 93 | await snowbound.start(snowbound, fonts, dictionaries); | |
| 93 | 94 | status.remove(); |
| 94 | 95 | console.info(`Snowbound loaded in ${Math.round(performance.now())} ms`); |
| 95 | 96 | } catch (error) { |
tools/release_web.py+48-5| ... | ... | @@ -8,6 +8,7 @@ static folder and publishes it to the share's web/ folder. |
| 8 | 8 | import argparse |
| 9 | 9 | import gzip |
| 10 | 10 | import hashlib |
| 11 | import json | |
| 11 | 12 | import os |
| 12 | 13 | from pathlib import Path |
| 13 | 14 | import shutil |
| ... | ... | @@ -28,6 +29,22 @@ EMOJI_URL = 'https://github.com/googlefonts/noto-emoji/raw/v2.051/fonts/Noto-COL |
| 28 | 29 | EMOJI_SHA256 = '0ae57fe58645638523ba35f388d93739d292539a9acb84df5700c81b1e1a28d2' |
| 29 | 30 | # As `EMOJI` in src/web.rs names it. |
| 30 | 31 | EMOJI_FONT = 'Noto-COLRv1.ttf.gz' |
| 32 | # Spelling dictionaries: the name the page asks for, the nixpkgs hunspellDicts package, its | |
| 33 | # files' stem, and the SPDX licences it comes under, each checked to allow redistribution | |
| 34 | # (the LGPL's text with the GPL's, which it amends). Add a row to offer another language. | |
| 35 | DICTIONARIES = [ | |
| 36 | ('en_US', 'en_US', 'en_US', ['BSD-3-Clause']), | |
| 37 | ('en_GB', 'en_GB-ise', 'en_GB', ['BSD-3-Clause']), | |
| 38 | ('de_DE', 'de_DE', 'de_DE', ['GPL-2.0-only', 'GPL-3.0-only']), | |
| 39 | ('fr_FR', 'fr-moderne', 'fr-moderne', ['MPL-2.0']), | |
| 40 | ('es_ES', 'es_ES', 'es_ES', ['GPL-3.0-only', 'LGPL-3.0-only', 'MPL-1.1']), | |
| 41 | ('pt_BR', 'pt_BR', 'pt_BR', ['LGPL-3.0-only', 'GPL-3.0-only']), | |
| 42 | ('pt_PT', 'pt_PT', 'pt_PT', ['GPL-2.0-only', 'LGPL-2.1-only', 'MPL-1.1']), | |
| 43 | ('it_IT', 'it_IT', 'it_IT', ['GPL-3.0-only']), | |
| 44 | ('nl_NL', 'nl_NL', 'nl_NL', ['BSD-3-Clause', 'CC-BY-3.0']), | |
| 45 | ('sv_SE', 'sv_SE', 'sv_SE', ['LGPL-3.0-only', 'GPL-3.0-only']), | |
| 46 | ('ru_RU', 'ru_RU', 'ru_RU', ['MPL-2.0', 'LGPL-3.0-only', 'GPL-3.0-only']), | |
| 47 | ] | |
| 31 | 48 | # Size over speed where it costs little: the module is most of the first load. |
| 32 | 49 | PROFILE = ['--config', 'profile.release.opt-level="s"'] |
| 33 | 50 | |
| ... | ... | @@ -75,11 +92,7 @@ def build(out): |
| 75 | 92 | bound = out / 'snowbound_web_bg.wasm' |
| 76 | 93 | run([tool('wasm-opt', 'binaryen'), '-Oz', '--strip-debug', '--strip-producers', bound, '-o', bound]) |
| 77 | 94 | shutil.copy(WEB / 'index.html', out) |
| 78 | # Hunspell's American English dictionary (SCOWL, BSD-3-Clause), for spelling. | |
| 79 | (out / 'dictionaries').mkdir() | |
| 80 | store = Path(tool('hunspell', 'hunspellDicts.en_US', store=True)) | |
| 81 | for name in ('share/hunspell/en_US.aff', 'share/hunspell/en_US.dic', 'share/doc/hunspell-dict-en-us-wordlist.txt'): | |
| 82 | shutil.copy(store / name, out / 'dictionaries') | |
| 95 | dictionaries(out / 'dictionaries') | |
| 83 | 96 | (out / 'fonts').mkdir() |
| 84 | 97 | for font in sorted(FONTS.glob('*')): |
| 85 | 98 | if font.suffix in ('.ttf', '.txt'): |
| ... | ... | @@ -90,6 +103,36 @@ def build(out): |
| 90 | 103 | print(f'{path.stat().st_size:>12,} {path.relative_to(out)}') |
| 91 | 104 | |
| 92 | 105 | |
| 106 | def dictionaries(folder): | |
| 107 | """The Hunspell dictionaries in DICTIONARIES as UTF-8, gzipped for the glue to inflate, | |
| 108 | each beside its readme, the texts of the licences they come under in licenses/, and their | |
| 109 | names in index.json for the page to pick from (`pick` in canvas/src/spelling.rs).""" | |
| 110 | (folder / 'licenses').mkdir(parents=True) | |
| 111 | texts = Path(tool('spdx', 'spdx-license-list-data.text', store=True)) / 'text' | |
| 112 | for name, package, stem, licenses in DICTIONARIES: | |
| 113 | store = Path(tool('hunspell', f'hunspellDicts.{package}', store=True)) | |
| 114 | affix = (store / f'share/hunspell/{stem}.aff').read_bytes() | |
| 115 | # Hunspell names the files' encoding in the affix file's SET line. | |
| 116 | encoding = next((line.split()[1] for line in affix.decode('latin-1').splitlines() | |
| 117 | if line.startswith('SET ')), 'UTF-8') | |
| 118 | for kind in ('aff', 'dic'): | |
| 119 | text = (store / f'share/hunspell/{stem}.{kind}').read_bytes().decode(encoding) | |
| 120 | # One encoding and one line ending for spellbook, which reads only UTF-8. An | |
| 121 | # 8-bit file's flags are its characters by default, as FLAG UTF-8 keeps them. | |
| 122 | utf8 = 'SET UTF-8' if encoding == 'UTF-8' or 'FLAG ' in text else 'SET UTF-8\nFLAG UTF-8' | |
| 123 | text = ''.join((utf8 if line.startswith('SET ') else line) + '\n' | |
| 124 | for line in text.splitlines()) | |
| 125 | (folder / f'{name}.{kind}.gz').write_bytes(gzip.compress(text.encode(), 9, mtime=0)) | |
| 126 | # Where a package has no readme, its affix file's header comment names the licences. | |
| 127 | readme = list((store / 'share/doc').glob('*.txt')) | |
| 128 | notice = readme[0].read_bytes() if readme else \ | |
| 129 | ''.join(line + '\n' for line in affix.decode(encoding).splitlines() if line.startswith('#')).encode() | |
| 130 | (folder / f'{name}.txt').write_bytes(notice + f'\nLicences: {", ".join(licenses)} (licenses/)\n'.encode()) | |
| 131 | for license in licenses: | |
| 132 | shutil.copyfile(texts / f'{license}.txt', folder / 'licenses' / f'{license}.txt') | |
| 133 | (folder / 'index.json').write_text(json.dumps([name for name, *_ in DICTIONARIES])) | |
| 134 | ||
| 135 | ||
| 93 | 136 | def fallbacks(fonts): |
| 94 | 137 | """Noto's faces for scripts the bundled ones lack, which the page fetches as it needs them |
| 95 | 138 | (`FALLBACKS` in src/web.rs), under the SIL Open Font License.""" |