word model and lists
This commit is contained in:
@@ -0,0 +1,149 @@
|
||||
pub mod blocklist;
|
||||
pub mod index;
|
||||
pub mod loader;
|
||||
pub mod word;
|
||||
|
||||
use std::collections::HashSet;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::error::Result;
|
||||
|
||||
use self::blocklist::Blocklist;
|
||||
use self::index::WordIndex;
|
||||
use self::loader::load_wordlist;
|
||||
use self::word::Word;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct Dictionary {
|
||||
pub words: Vec<Word>,
|
||||
pub index: WordIndex,
|
||||
}
|
||||
|
||||
impl Dictionary {
|
||||
pub fn load(wordlist_path: &Path, blocklist_path: Option<&Path>, min_length: usize) -> Result<Self> {
|
||||
let bl = match blocklist_path {
|
||||
Some(p) if p.exists() => Blocklist::load(p)?,
|
||||
_ => Blocklist::empty(),
|
||||
};
|
||||
|
||||
let words = load_wordlist(wordlist_path, &bl, min_length)?;
|
||||
let index = WordIndex::build(&words);
|
||||
|
||||
Ok(Self { words, index })
|
||||
}
|
||||
|
||||
/// Remove words that have no clues and boost the rest.
|
||||
pub fn boost_from_clue_words(&mut self, clue_words: &HashSet<String>) {
|
||||
self.words.retain(|w| clue_words.contains(&w.text));
|
||||
for word in &mut self.words {
|
||||
word.frequency = word.frequency.max(0.8);
|
||||
}
|
||||
self.index = WordIndex::build(&self.words);
|
||||
}
|
||||
|
||||
/// Set word frequencies based on Broda quality scores (0-100 scale).
|
||||
/// target: 0=very easy, 1=easy, 2=medium, 3=hard
|
||||
/// Higher Broda score = better/more common word.
|
||||
pub fn apply_difficulty(&mut self, word_scores: &std::collections::HashMap<String, f64>, target: u8) {
|
||||
let min_score = match target {
|
||||
0 => 70.0, // very easy: only clean common words
|
||||
1 => 60.0, // easy
|
||||
2 => 45.0, // medium
|
||||
_ => 30.0, // hard: allows crossword-ese
|
||||
};
|
||||
|
||||
for word in &mut self.words {
|
||||
if let Some(&score) = word_scores.get(&word.text) {
|
||||
if score < min_score {
|
||||
word.frequency *= 0.1; // heavily penalize below threshold
|
||||
} else {
|
||||
// boost proportional to score quality
|
||||
let boost = (score / 100.0).max(0.5);
|
||||
word.frequency = word.frequency.max(boost);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Boost words that match a theme. Theme words get very high frequency
|
||||
/// so the solver strongly prefers them, especially for longer slots.
|
||||
/// Returns the count of theme words found in the dictionary.
|
||||
pub fn boost_theme(&mut self, theme_words: &HashSet<String>) -> usize {
|
||||
let mut count = 0;
|
||||
for word in &mut self.words {
|
||||
if word.frequency > 0.0 && theme_words.contains(&word.text) {
|
||||
word.frequency = 2.0;
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
count
|
||||
}
|
||||
|
||||
/// Boost a single word's frequency (for pack integration).
|
||||
pub fn boost_word(&mut self, word: &str, freq: f64) {
|
||||
for w in &mut self.words {
|
||||
if w.text == word {
|
||||
w.frequency = w.frequency.max(freq);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Penalize words that users consistently rate poorly in puzzles.
|
||||
/// Quality scores are averages (1.0-5.0). Words below 2.0 get penalized.
|
||||
pub fn apply_quality(&mut self, word_quality: &std::collections::HashMap<String, f64>) {
|
||||
for word in &mut self.words {
|
||||
if let Some(&quality) = word_quality.get(&word.text) {
|
||||
if quality < 2.0 {
|
||||
word.frequency *= 0.3;
|
||||
} else if quality > 4.0 {
|
||||
word.frequency = word.frequency.max(0.85);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn word_count(&self) -> usize {
|
||||
self.words.len()
|
||||
}
|
||||
|
||||
pub fn get(&self, idx: usize) -> &Word {
|
||||
&self.words[idx]
|
||||
}
|
||||
|
||||
pub fn candidates(&self, length: usize, constraints: &[(usize, u8)]) -> Vec<usize> {
|
||||
self.index.candidates(length, constraints)
|
||||
}
|
||||
|
||||
pub fn candidates_filtered(
|
||||
&self,
|
||||
length: usize,
|
||||
constraints: &[(usize, u8)],
|
||||
min_freq: f64,
|
||||
) -> Vec<usize> {
|
||||
self.index
|
||||
.candidates(length, constraints)
|
||||
.into_iter()
|
||||
.filter(|&i| self.words[i].frequency >= min_freq)
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn load_dictionary() {
|
||||
let path = Path::new("../data/wordlists/enable.txt");
|
||||
if !path.exists() {
|
||||
return;
|
||||
}
|
||||
let dict = Dictionary::load(path, None, 3).unwrap();
|
||||
assert!(dict.word_count() > 100_000);
|
||||
|
||||
// should be able to find common 3-letter words
|
||||
let cats = dict.candidates(3, &[(0, b'C'), (1, b'A'), (2, b'T')]);
|
||||
assert!(!cats.is_empty());
|
||||
assert_eq!(dict.get(cats[0]).text, "CAT");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct Word {
|
||||
pub text: String,
|
||||
pub frequency: f64,
|
||||
pub length: usize,
|
||||
}
|
||||
|
||||
impl Word {
|
||||
pub fn new(text: String, frequency: f64) -> Self {
|
||||
let length = text.len();
|
||||
Self {
|
||||
text,
|
||||
frequency,
|
||||
length,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn char_at(&self, pos: usize) -> u8 {
|
||||
self.text.as_bytes()[pos]
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user