Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
150 changes: 139 additions & 11 deletions benches/suggestions.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,21 +5,21 @@ use criterion::{criterion_group, criterion_main, Criterion};
use okkhor::parser::Parser;
use regex::Regex;

use upodesh::suggest::Suggest;
use upodesh::avro::Suggest;

fn upodesh_benchmark(c: &mut Criterion) {
fn upodesh_avro_benchmark(c: &mut Criterion) {
let suggest = Suggest::new();

c.bench_function("upodesh a", |b| b.iter(|| suggest.suggest(black_box("a"))));
c.bench_function("upodesh arO", |b| {
c.bench_function("upodesh avro a", |b| b.iter(|| suggest.suggest(black_box("a"))));
c.bench_function("upodesh avro arO", |b| {
b.iter(|| suggest.suggest(black_box("arO")))
});
c.bench_function("upodesh bistari", |b| {
c.bench_function("upodesh avro bistari", |b| {
b.iter(|| suggest.suggest(black_box("bistari")))
});
}

fn regex_benchmark(c: &mut Criterion) {
fn regex_avro_benchmark(c: &mut Criterion) {
let table: [(&str, &[&str]); 26] = [
("a", &["a", "aa", "e", "oi", "o", "nya", "y"]),
("b", &["b", "bh"]),
Expand Down Expand Up @@ -75,12 +75,140 @@ fn regex_benchmark(c: &mut Criterion) {
.collect()
};

c.bench_function("regex a", |b| b.iter(|| suggest(black_box("a"))));
c.bench_function("regex arO", |b| b.iter(|| suggest(black_box("arO"))));
c.bench_function("regex bistari", |b| {
c.bench_function("regex avro a", |b| b.iter(|| suggest(black_box("a"))));
c.bench_function("regex avro arO", |b| b.iter(|| suggest(black_box("arO"))));
c.bench_function("regex avro bistari", |b| {
b.iter(|| suggest(black_box("bistari")))
});
}

criterion_group!(benches, upodesh_benchmark, regex_benchmark);
criterion_main!(benches);
fn upodesh_bangla_benchmark(c: &mut Criterion) {
use upodesh::bangla::suggest;
c.bench_function("upodesh bangla আমা", |b| {
b.iter(|| suggest(black_box("আমা")))
});
c.bench_function("upodesh bangla কম্পি", |b| {
b.iter(|| suggest(black_box("কম্পি")))
});
c.bench_function("upodesh bangla কনট্রো", |b| {
b.iter(|| suggest(black_box("কনট্রো")))
});
}

fn regex_bangla_benchmark(c: &mut Criterion) {
fn suggest(word: &str) -> Vec<String> {
let table = match word.chars().next().unwrap_or_default() {
// Kars
'া' => "aa",
'ি' => "i",
'ী' => "ii",
'ু' => "u",
'ূ' => "uu",
'ৃ' => "rri",
'ে' => "e",
'ৈ' => "oi",
'ো' => "o",
'ৌ' => "ou",
// Vowels
'অ' => "a",
'আ' => "aa",
'ই' => "i",
'ঈ' => "ii",
'উ' => "u",
'ঊ' => "uu",
'ঋ' => "rri",
'এ' => "e",
'ঐ' => "oi",
'ও' => "o",
'ঔ' => "ou",
// Consonants
'ক' => "k",
'খ' => "kh",
'গ' => "g",
'ঘ' => "gh",
'ঙ' => "nga",
'চ' => "c",
'ছ' => "ch",
'জ' => "j",
'ঝ' => "jh",
'ঞ' => "nya",
'ট' => "tt",
'ঠ' => "tth",
'ড' => "dd",
'ঢ' => "ddh",
'ণ' => "nn",
'ত' => "t",
'থ' => "th",
'দ' => "d",
'ধ' => "dh",
'ন' => "n",
'প' => "p",
'ফ' => "ph",
'ব' => "b",
'ভ' => "bh",
'ম' => "m",
'য' => "z",
'র' => "r",
'ল' => "l",
'শ' => "sh",
'ষ' => "ss",
'স' => "s",
'হ' => "h",
'ড়' => "rr",
'ঢ়' => "rrh",
'য়' => "y",
'ৎ' => "khandatta",
// Otherwise we don't have any suggestions to search from, so return from the function.
_ => return Vec::new(),
};

let word = clean_string(word);

let need_chars_upto = match word.chars().count() {
1 => 0,
2..=3 => 1,
_ => 5,
};

let regex = format!(
"^{word}[অআইঈউঊঋএঐওঔঌৡািীুূৃেৈোৌকখগঘঙচছজঝঞটঠডঢণতথদধনপফবভমযরলশষসহৎড়ঢ়য়ংঃঁ\u{09CD}]{{0,{need_chars_upto}}}$"
);
let rgx = Regex::new(&regex).unwrap();

let database: HashMap<String, Vec<String>, RandomState> =
serde_json::from_slice(include_bytes!("../data/dictionary.json")).unwrap();

database
.get(table)
.unwrap()
.into_iter()
.filter(|i| rgx.is_match(i))
.cloned()
.collect()
}

fn clean_string(string: &str) -> String {
string
.chars()
.filter(|&c| !"|()[]{}^$*+?.~!@#%&-_='\";<>/\\,:`।\u{200C}".contains(c))
.collect()
}

c.bench_function("regex bangla আমা", |b| {
b.iter(|| suggest(black_box("আমা")))
});
c.bench_function("regex bangla কম্পি", |b| {
b.iter(|| suggest(black_box("কম্পি")))
});
c.bench_function("regex bangla কনট্রো", |b| {
b.iter(|| suggest(black_box("কনট্রো")))
});
}

criterion_group!(benches_avro, upodesh_avro_benchmark, regex_avro_benchmark);
criterion_group!(
benches_bangla,
upodesh_bangla_benchmark,
regex_bangla_benchmark
);
criterion_main!(benches_avro, benches_bangla);
2 changes: 1 addition & 1 deletion examples/alloc.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
use peak_alloc::PeakAlloc;
use upodesh::suggest::Suggest;
use upodesh::avro::Suggest;

#[global_allocator]
static PEAK_ALLOC: PeakAlloc = PeakAlloc;
Expand Down
File renamed without changes.
2 changes: 1 addition & 1 deletion examples/example.rs → examples/avro.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
use peak_alloc::PeakAlloc;
use upodesh::suggest::Suggest;
use upodesh::avro::Suggest;

#[global_allocator]
static PEAK_ALLOC: PeakAlloc = PeakAlloc;
Expand Down
115 changes: 115 additions & 0 deletions examples/bangla-regex.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,115 @@
use std::collections::HashMap;

use ahash::RandomState;
use regex::Regex;

pub(crate) fn search_dictionary(word: &str) -> Vec<String> {
let table = match word.chars().next().unwrap_or_default() {
// Kars
'া' => "aa",
'ি' => "i",
'ী' => "ii",
'ু' => "u",
'ূ' => "uu",
'ৃ' => "rri",
'ে' => "e",
'ৈ' => "oi",
'ো' => "o",
'ৌ' => "ou",
// Vowels
'অ' => "a",
'আ' => "aa",
'ই' => "i",
'ঈ' => "ii",
'উ' => "u",
'ঊ' => "uu",
'ঋ' => "rri",
'এ' => "e",
'ঐ' => "oi",
'ও' => "o",
'ঔ' => "ou",
// Consonants
'ক' => "k",
'খ' => "kh",
'গ' => "g",
'ঘ' => "gh",
'ঙ' => "nga",
'চ' => "c",
'ছ' => "ch",
'জ' => "j",
'ঝ' => "jh",
'ঞ' => "nya",
'ট' => "tt",
'ঠ' => "tth",
'ড' => "dd",
'ঢ' => "ddh",
'ণ' => "nn",
'ত' => "t",
'থ' => "th",
'দ' => "d",
'ধ' => "dh",
'ন' => "n",
'প' => "p",
'ফ' => "ph",
'ব' => "b",
'ভ' => "bh",
'ম' => "m",
'য' => "z",
'র' => "r",
'ল' => "l",
'শ' => "sh",
'ষ' => "ss",
'স' => "s",
'হ' => "h",
'ড়' => "rr",
'ঢ়' => "rrh",
'য়' => "y",
'ৎ' => "khandatta",
// Otherwise we don't have any suggestions to search from, so return from the function.
_ => return Vec::new(),
};

let word = clean_string(word);

let need_chars_upto = match word.chars().count() {
1 => 0,
2..=3 => 1,
_ => 5,
};

let regex = format!(
"^{word}[অআইঈউঊঋএঐওঔঌৡািীুূৃেৈোৌকখগঘঙচছজঝঞটঠডঢণতথদধনপফবভমযরলশষসহৎড়ঢ়য়ংঃঁ\u{09CD}]{{0,{need_chars_upto}}}$"
);
let rgx = Regex::new(&regex).unwrap();

let database: HashMap<String, Vec<String>, RandomState> =
serde_json::from_slice(include_bytes!("../data/dictionary.json")).unwrap();

database
.get(table)
.unwrap()
.into_iter()
.filter(|i| rgx.is_match(i))
.cloned()
.collect()
}

fn clean_string(string: &str) -> String {
string
.chars()
.filter(|&c| !"|()[]{}^$*+?.~!@#%&-_='\";<>/\\,:`।\u{200C}".contains(c))
.collect()
}

fn main() {
let Some(word) = std::env::args().nth(1) else {
eprintln!("Please provide a word");
std::process::exit(1);
};

let mut suggestions = search_dictionary(&word);

println!("Word: {}", word);
suggestions.sort();
println!("Suggestions: [{}]", suggestions.join(", "));
}
3 changes: 1 addition & 2 deletions generate/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,7 @@ fn generate_words_fst() {
fn generate_patterns_fst() {
let root = PathBuf::from(var_os("CARGO_MANIFEST_DIR").unwrap());
let parent = root.parent().unwrap();
let dest = parent.join("src").join("patterns.fst");
let dest = parent.join("src").join("avro").join("patterns.fst");

let file = File::create(dest).expect("Failed to create patterns.fst");
let writer = BufWriter::new(file);
Expand Down Expand Up @@ -80,7 +80,6 @@ fn generate_patterns_fst() {

fn generate_regex_exploded_patterns(source: &str, dest: &str) {
let file = File::create(dest).expect("Failed to create destination file");
// let writer = BufWriter::new(file);

let regex_patterns: HashMap<String, RegexBlock> =
serde_json::from_slice(&read(source).expect("Failed to read source patterns file"))
Expand Down
3 changes: 3 additions & 0 deletions src/avro/mod.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
mod utils;
mod suggest;
pub use suggest::Suggest;
File renamed without changes.
8 changes: 3 additions & 5 deletions src/suggest.rs → src/avro/suggest.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,7 @@ use std::collections::{HashMap, HashSet};
use once_cell::sync::Lazy;
use serde::Deserialize;

use crate::{fst::FstTree, utils::fix_string};

static WORDS: Lazy<FstTree<&[u8]>> = Lazy::new(|| FstTree::from_fst(include_bytes!("words.fst")));
use crate::{fst::FstTree, avro::utils::fix_string, WORDS};

static PATTERNS: Lazy<FstTree<&[u8]>> =
Lazy::new(|| FstTree::from_fst(include_bytes!("patterns.fst")));
Expand All @@ -24,8 +22,8 @@ pub struct Suggest {

impl Suggest {
pub fn new() -> Self {
let patterns_data = include_bytes!("../data/preprocessed-patterns.json");
let common_data = include_bytes!("../data/source-common-patterns.json");
let patterns_data = include_bytes!("../../data/preprocessed-patterns.json");
let common_data = include_bytes!("../../data/source-common-patterns.json");

let patterns: HashMap<String, Block> = serde_json::from_slice(patterns_data).unwrap();
let common_suffixes = serde_json::from_slice(common_data).unwrap();
Expand Down
File renamed without changes.
Loading