From f39bdfd6611fc0c8992b176cc5acf5de3d5d57a7 Mon Sep 17 00:00:00 2001 From: Muhammad Mominul Huque Date: Sat, 4 Oct 2025 01:06:07 +0600 Subject: [PATCH] Implement Bengali word suggestion for Fixed method Refactored Avro Phonetic support into the avro module --- benches/suggestions.rs | 150 ++++++++++++++++++++++-- examples/alloc.rs | 2 +- examples/{previous.rs => avro-regex.rs} | 0 examples/{example.rs => avro.rs} | 2 +- examples/bangla-regex.rs | 115 ++++++++++++++++++ generate/src/main.rs | 3 +- src/avro/mod.rs | 3 + src/{ => avro}/patterns.fst | Bin src/{ => avro}/suggest.rs | 8 +- src/{ => avro}/utils.rs | 0 src/bangla/mod.rs | 88 ++++++++++++++ src/fst.rs | 17 +++ src/lib.rs | 11 +- 13 files changed, 377 insertions(+), 22 deletions(-) rename examples/{previous.rs => avro-regex.rs} (100%) rename examples/{example.rs => avro.rs} (95%) create mode 100644 examples/bangla-regex.rs create mode 100644 src/avro/mod.rs rename src/{ => avro}/patterns.fst (100%) rename src/{ => avro}/suggest.rs (95%) rename src/{ => avro}/utils.rs (100%) create mode 100644 src/bangla/mod.rs diff --git a/benches/suggestions.rs b/benches/suggestions.rs index 102b37d..3ceead6 100644 --- a/benches/suggestions.rs +++ b/benches/suggestions.rs @@ -5,21 +5,21 @@ use criterion::{criterion_group, criterion_main, Criterion}; use okkhor::parser::Parser; use regex::Regex; -use upodesh::suggest::Suggest; +use upodesh::avro::Suggest; -fn upodesh_benchmark(c: &mut Criterion) { +fn upodesh_avro_benchmark(c: &mut Criterion) { let suggest = Suggest::new(); - c.bench_function("upodesh a", |b| b.iter(|| suggest.suggest(black_box("a")))); - c.bench_function("upodesh arO", |b| { + c.bench_function("upodesh avro a", |b| b.iter(|| suggest.suggest(black_box("a")))); + c.bench_function("upodesh avro arO", |b| { b.iter(|| suggest.suggest(black_box("arO"))) }); - c.bench_function("upodesh bistari", |b| { + c.bench_function("upodesh avro bistari", |b| { b.iter(|| suggest.suggest(black_box("bistari"))) }); } -fn regex_benchmark(c: &mut Criterion) { +fn regex_avro_benchmark(c: &mut Criterion) { let table: [(&str, &[&str]); 26] = [ ("a", &["a", "aa", "e", "oi", "o", "nya", "y"]), ("b", &["b", "bh"]), @@ -75,12 +75,140 @@ fn regex_benchmark(c: &mut Criterion) { .collect() }; - c.bench_function("regex a", |b| b.iter(|| suggest(black_box("a")))); - c.bench_function("regex arO", |b| b.iter(|| suggest(black_box("arO")))); - c.bench_function("regex bistari", |b| { + c.bench_function("regex avro a", |b| b.iter(|| suggest(black_box("a")))); + c.bench_function("regex avro arO", |b| b.iter(|| suggest(black_box("arO")))); + c.bench_function("regex avro bistari", |b| { b.iter(|| suggest(black_box("bistari"))) }); } -criterion_group!(benches, upodesh_benchmark, regex_benchmark); -criterion_main!(benches); +fn upodesh_bangla_benchmark(c: &mut Criterion) { + use upodesh::bangla::suggest; + c.bench_function("upodesh bangla আমা", |b| { + b.iter(|| suggest(black_box("আমা"))) + }); + c.bench_function("upodesh bangla কম্পি", |b| { + b.iter(|| suggest(black_box("কম্পি"))) + }); + c.bench_function("upodesh bangla কনট্রো", |b| { + b.iter(|| suggest(black_box("কনট্রো"))) + }); +} + +fn regex_bangla_benchmark(c: &mut Criterion) { + fn suggest(word: &str) -> Vec { + let table = match word.chars().next().unwrap_or_default() { + // Kars + 'া' => "aa", + 'ি' => "i", + 'ী' => "ii", + 'ু' => "u", + 'ূ' => "uu", + 'ৃ' => "rri", + 'ে' => "e", + 'ৈ' => "oi", + 'ো' => "o", + 'ৌ' => "ou", + // Vowels + 'অ' => "a", + 'আ' => "aa", + 'ই' => "i", + 'ঈ' => "ii", + 'উ' => "u", + 'ঊ' => "uu", + 'ঋ' => "rri", + 'এ' => "e", + 'ঐ' => "oi", + 'ও' => "o", + 'ঔ' => "ou", + // Consonants + 'ক' => "k", + 'খ' => "kh", + 'গ' => "g", + 'ঘ' => "gh", + 'ঙ' => "nga", + 'চ' => "c", + 'ছ' => "ch", + 'জ' => "j", + 'ঝ' => "jh", + 'ঞ' => "nya", + 'ট' => "tt", + 'ঠ' => "tth", + 'ড' => "dd", + 'ঢ' => "ddh", + 'ণ' => "nn", + 'ত' => "t", + 'থ' => "th", + 'দ' => "d", + 'ধ' => "dh", + 'ন' => "n", + 'প' => "p", + 'ফ' => "ph", + 'ব' => "b", + 'ভ' => "bh", + 'ম' => "m", + 'য' => "z", + 'র' => "r", + 'ল' => "l", + 'শ' => "sh", + 'ষ' => "ss", + 'স' => "s", + 'হ' => "h", + 'ড়' => "rr", + 'ঢ়' => "rrh", + 'য়' => "y", + 'ৎ' => "khandatta", + // Otherwise we don't have any suggestions to search from, so return from the function. + _ => return Vec::new(), + }; + + let word = clean_string(word); + + let need_chars_upto = match word.chars().count() { + 1 => 0, + 2..=3 => 1, + _ => 5, + }; + + let regex = format!( + "^{word}[অআইঈউঊঋএঐওঔঌৡািীুূৃেৈোৌকখগঘঙচছজঝঞটঠডঢণতথদধনপফবভমযরলশষসহৎড়ঢ়য়ংঃঁ\u{09CD}]{{0,{need_chars_upto}}}$" + ); + let rgx = Regex::new(®ex).unwrap(); + + let database: HashMap, RandomState> = + serde_json::from_slice(include_bytes!("../data/dictionary.json")).unwrap(); + + database + .get(table) + .unwrap() + .into_iter() + .filter(|i| rgx.is_match(i)) + .cloned() + .collect() + } + + fn clean_string(string: &str) -> String { + string + .chars() + .filter(|&c| !"|()[]{}^$*+?.~!@#%&-_='\";<>/\\,:`।\u{200C}".contains(c)) + .collect() + } + + c.bench_function("regex bangla আমা", |b| { + b.iter(|| suggest(black_box("আমা"))) + }); + c.bench_function("regex bangla কম্পি", |b| { + b.iter(|| suggest(black_box("কম্পি"))) + }); + c.bench_function("regex bangla কনট্রো", |b| { + b.iter(|| suggest(black_box("কনট্রো"))) + }); +} + +criterion_group!(benches_avro, upodesh_avro_benchmark, regex_avro_benchmark); +criterion_group!( + benches_bangla, + upodesh_bangla_benchmark, + regex_bangla_benchmark +); +criterion_main!(benches_avro, benches_bangla); diff --git a/examples/alloc.rs b/examples/alloc.rs index fe1e1b5..1eaef33 100644 --- a/examples/alloc.rs +++ b/examples/alloc.rs @@ -1,5 +1,5 @@ use peak_alloc::PeakAlloc; -use upodesh::suggest::Suggest; +use upodesh::avro::Suggest; #[global_allocator] static PEAK_ALLOC: PeakAlloc = PeakAlloc; diff --git a/examples/previous.rs b/examples/avro-regex.rs similarity index 100% rename from examples/previous.rs rename to examples/avro-regex.rs diff --git a/examples/example.rs b/examples/avro.rs similarity index 95% rename from examples/example.rs rename to examples/avro.rs index 6bee2ae..9fc9511 100644 --- a/examples/example.rs +++ b/examples/avro.rs @@ -1,5 +1,5 @@ use peak_alloc::PeakAlloc; -use upodesh::suggest::Suggest; +use upodesh::avro::Suggest; #[global_allocator] static PEAK_ALLOC: PeakAlloc = PeakAlloc; diff --git a/examples/bangla-regex.rs b/examples/bangla-regex.rs new file mode 100644 index 0000000..8246dfa --- /dev/null +++ b/examples/bangla-regex.rs @@ -0,0 +1,115 @@ +use std::collections::HashMap; + +use ahash::RandomState; +use regex::Regex; + +pub(crate) fn search_dictionary(word: &str) -> Vec { + let table = match word.chars().next().unwrap_or_default() { + // Kars + 'া' => "aa", + 'ি' => "i", + 'ী' => "ii", + 'ু' => "u", + 'ূ' => "uu", + 'ৃ' => "rri", + 'ে' => "e", + 'ৈ' => "oi", + 'ো' => "o", + 'ৌ' => "ou", + // Vowels + 'অ' => "a", + 'আ' => "aa", + 'ই' => "i", + 'ঈ' => "ii", + 'উ' => "u", + 'ঊ' => "uu", + 'ঋ' => "rri", + 'এ' => "e", + 'ঐ' => "oi", + 'ও' => "o", + 'ঔ' => "ou", + // Consonants + 'ক' => "k", + 'খ' => "kh", + 'গ' => "g", + 'ঘ' => "gh", + 'ঙ' => "nga", + 'চ' => "c", + 'ছ' => "ch", + 'জ' => "j", + 'ঝ' => "jh", + 'ঞ' => "nya", + 'ট' => "tt", + 'ঠ' => "tth", + 'ড' => "dd", + 'ঢ' => "ddh", + 'ণ' => "nn", + 'ত' => "t", + 'থ' => "th", + 'দ' => "d", + 'ধ' => "dh", + 'ন' => "n", + 'প' => "p", + 'ফ' => "ph", + 'ব' => "b", + 'ভ' => "bh", + 'ম' => "m", + 'য' => "z", + 'র' => "r", + 'ল' => "l", + 'শ' => "sh", + 'ষ' => "ss", + 'স' => "s", + 'হ' => "h", + 'ড়' => "rr", + 'ঢ়' => "rrh", + 'য়' => "y", + 'ৎ' => "khandatta", + // Otherwise we don't have any suggestions to search from, so return from the function. + _ => return Vec::new(), + }; + + let word = clean_string(word); + + let need_chars_upto = match word.chars().count() { + 1 => 0, + 2..=3 => 1, + _ => 5, + }; + + let regex = format!( + "^{word}[অআইঈউঊঋএঐওঔঌৡািীুূৃেৈোৌকখগঘঙচছজঝঞটঠডঢণতথদধনপফবভমযরলশষসহৎড়ঢ়য়ংঃঁ\u{09CD}]{{0,{need_chars_upto}}}$" + ); + let rgx = Regex::new(®ex).unwrap(); + + let database: HashMap, RandomState> = + serde_json::from_slice(include_bytes!("../data/dictionary.json")).unwrap(); + + database + .get(table) + .unwrap() + .into_iter() + .filter(|i| rgx.is_match(i)) + .cloned() + .collect() +} + +fn clean_string(string: &str) -> String { + string + .chars() + .filter(|&c| !"|()[]{}^$*+?.~!@#%&-_='\";<>/\\,:`।\u{200C}".contains(c)) + .collect() +} + +fn main() { + let Some(word) = std::env::args().nth(1) else { + eprintln!("Please provide a word"); + std::process::exit(1); + }; + + let mut suggestions = search_dictionary(&word); + + println!("Word: {}", word); + suggestions.sort(); + println!("Suggestions: [{}]", suggestions.join(", ")); +} diff --git a/generate/src/main.rs b/generate/src/main.rs index fa24d18..a8215c8 100644 --- a/generate/src/main.rs +++ b/generate/src/main.rs @@ -52,7 +52,7 @@ fn generate_words_fst() { fn generate_patterns_fst() { let root = PathBuf::from(var_os("CARGO_MANIFEST_DIR").unwrap()); let parent = root.parent().unwrap(); - let dest = parent.join("src").join("patterns.fst"); + let dest = parent.join("src").join("avro").join("patterns.fst"); let file = File::create(dest).expect("Failed to create patterns.fst"); let writer = BufWriter::new(file); @@ -80,7 +80,6 @@ fn generate_patterns_fst() { fn generate_regex_exploded_patterns(source: &str, dest: &str) { let file = File::create(dest).expect("Failed to create destination file"); - // let writer = BufWriter::new(file); let regex_patterns: HashMap = serde_json::from_slice(&read(source).expect("Failed to read source patterns file")) diff --git a/src/avro/mod.rs b/src/avro/mod.rs new file mode 100644 index 0000000..0ecd434 --- /dev/null +++ b/src/avro/mod.rs @@ -0,0 +1,3 @@ +mod utils; +mod suggest; +pub use suggest::Suggest; diff --git a/src/patterns.fst b/src/avro/patterns.fst similarity index 100% rename from src/patterns.fst rename to src/avro/patterns.fst diff --git a/src/suggest.rs b/src/avro/suggest.rs similarity index 95% rename from src/suggest.rs rename to src/avro/suggest.rs index 5d9fd7d..cd5d410 100644 --- a/src/suggest.rs +++ b/src/avro/suggest.rs @@ -3,9 +3,7 @@ use std::collections::{HashMap, HashSet}; use once_cell::sync::Lazy; use serde::Deserialize; -use crate::{fst::FstTree, utils::fix_string}; - -static WORDS: Lazy> = Lazy::new(|| FstTree::from_fst(include_bytes!("words.fst"))); +use crate::{fst::FstTree, avro::utils::fix_string, WORDS}; static PATTERNS: Lazy> = Lazy::new(|| FstTree::from_fst(include_bytes!("patterns.fst"))); @@ -24,8 +22,8 @@ pub struct Suggest { impl Suggest { pub fn new() -> Self { - let patterns_data = include_bytes!("../data/preprocessed-patterns.json"); - let common_data = include_bytes!("../data/source-common-patterns.json"); + let patterns_data = include_bytes!("../../data/preprocessed-patterns.json"); + let common_data = include_bytes!("../../data/source-common-patterns.json"); let patterns: HashMap = serde_json::from_slice(patterns_data).unwrap(); let common_suffixes = serde_json::from_slice(common_data).unwrap(); diff --git a/src/utils.rs b/src/avro/utils.rs similarity index 100% rename from src/utils.rs rename to src/avro/utils.rs diff --git a/src/bangla/mod.rs b/src/bangla/mod.rs new file mode 100644 index 0000000..056eacb --- /dev/null +++ b/src/bangla/mod.rs @@ -0,0 +1,88 @@ +use std::collections::HashSet; + +use once_cell::sync::Lazy; + +use crate::WORDS; + +const CHARS: [char; 61] = [ + 'অ', 'আ', 'ই', 'ঈ', 'উ', 'ঊ', 'ঋ', 'এ', 'ঐ', 'ও', 'ঔ', 'া', 'ি', 'ী', 'ু', 'ূ', 'ৃ', 'ে', 'ৈ', 'ো', + 'ৌ', 'ক', 'খ', 'গ', 'ঘ', 'ঙ', 'চ', 'ছ', 'জ', 'ঝ', 'ঞ', 'ট', 'ঠ', 'ড', 'ঢ', 'ণ', 'ত', 'থ', 'দ', + 'ধ', 'ন', 'প', 'ফ', 'ব', 'ভ', 'ম', 'য', 'র', 'ল', 'শ', 'ষ', 'স', 'হ', 'ৎ', 'ড়', 'ঢ়', 'য়', 'ং', + 'ঃ', 'ঁ', '্', +]; + +pub fn suggest(word: &str) -> Vec { + let words = Lazy::force(&WORDS); + + let need_chars_upto = match word.chars().count() { + 1 => 0, + 2..=3 => 1, + _ => 5, + }; + + let node = if let Some(n) = words.matching_node(word) { + n + } else { + return Vec::new(); + }; + + let mut nodes: Vec<_> = CHARS + .iter() + .filter_map(|&c| node.get_matching_node_by_char(c)) + .collect(); + + for _ in 0..(need_chars_upto - 1) { + let new_nodes: Vec<_> = nodes + .iter() + .flat_map(|node| { + CHARS + .iter() + .filter_map(|&c| node.get_matching_node_by_char(c)) + }) + .collect(); + + nodes.extend(new_nodes); + } + + let suggestions = nodes + .into_iter() + .filter_map(|n| n.get_word()) + .collect::>() + .into_iter() + .collect::>(); + + suggestions +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sort(mut vec: Vec) -> Vec { + vec.sort(); + vec + } + + #[test] + fn test_suggestions() { + assert_eq!(sort(sort(suggest("আমা"))), ["আমান", "আমার", "আমায়"]); + assert_eq!( + sort(sort(suggest("ই"))), + ["ইজ", "ইট", "ইন", "ইফ", "ইভ", "ইহ"] + ); + assert_eq!( + sort(sort(suggest("কম্পি"))), + [ + "কম্পিউটার", + "কম্পিউটিং", + "কম্পিউটেশন", + "কম্পিটিশন", + "কম্পিত", + "কম্পিতা" + ] + ); + assert_eq!(sort(sort(suggest("আইনস্"))), ["আইনস্টাইন"]); + assert_eq!(sort(sort(suggest("খ(১"))), Vec::::new()); + assert_eq!(sort(sort(suggest("1"))), Vec::::new()); + } +} diff --git a/src/fst.rs b/src/fst.rs index b6a99d1..04e29e5 100644 --- a/src/fst.rs +++ b/src/fst.rs @@ -101,6 +101,23 @@ impl<'a, D: AsRef<[u8]>> FstNode<'a, D> { }) } + pub fn get_matching_node_by_char(&self, suffix: char) -> Option> { + let mut node = self.node; + + match node.find_input(suffix as u8) { + Some(addr) => { + node = self.fst.node(node.transition_addr(addr)); + } + None => return None, + } + + Some(FstNode { + fst: self.fst, + node, + word: format!("{}{}", self.word, suffix), + }) + } + pub fn get_word(self) -> Option { if self.node.is_final() { Some(self.word) diff --git a/src/lib.rs b/src/lib.rs index 4d4f8a3..630c495 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,3 +1,10 @@ +use once_cell::sync::Lazy; + +use crate::fst::FstTree; + +/// The FST containing the valid Bengali words for suggestions. +static WORDS: Lazy> = Lazy::new(|| FstTree::from_fst(include_bytes!("words.fst"))); + mod fst; -pub mod suggest; -mod utils; +pub mod avro; +pub mod bangla;