init
This commit is contained in:
@@ -0,0 +1 @@
|
|||||||
|
/target
|
||||||
Generated
+7
@@ -0,0 +1,7 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "word-count"
|
||||||
|
version = "0.1.0"
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[package]
|
||||||
|
name = "word-count"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
+101
@@ -0,0 +1,101 @@
|
|||||||
|
|
||||||
|
// Write some functions to:
|
||||||
|
// 1. tokenize a string
|
||||||
|
// 2. manually make a map of tokens:their_counts
|
||||||
|
// 3. analyze the token structure:
|
||||||
|
// 3.a. count the words
|
||||||
|
// 3.b. count the unique words
|
||||||
|
// 3.c. get the first n most frequently used words
|
||||||
|
// 3.d. count the average word length
|
||||||
|
// 4. (pretty) print the info
|
||||||
|
|
||||||
|
const PUNC: &str = ",.?:;'~|!@#$%^&*\"";
|
||||||
|
|
||||||
|
|
||||||
|
// todo: what about CTRL chars?
|
||||||
|
// todo: what about feeding Bytes?
|
||||||
|
// todo: return Result for whitespace input
|
||||||
|
fn tokenize(input: &str) -> Vec<String> {
|
||||||
|
// filter punctuation
|
||||||
|
let mut s = String::new();
|
||||||
|
for c in input.chars() {
|
||||||
|
if !PUNC.contains(c) {
|
||||||
|
s.push(c.to_lowercase().next().unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// convert to vec of strings
|
||||||
|
s.split_whitespace().map(|l| l.to_string()).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
// todo: can the input type be better?
|
||||||
|
fn count_words(input: &[String]) -> Vec<(String,usize)> {
|
||||||
|
let mut counts : Vec<(String, usize)> = Vec::new();
|
||||||
|
for w in input {
|
||||||
|
match counts.iter_mut().find(|(c,_)| c == w) {
|
||||||
|
Some((_, count)) => *count += 1,
|
||||||
|
_ => counts.push((w.to_string(), 1)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
counts.sort_by(|a,b| b.1.cmp(&a.1));
|
||||||
|
counts
|
||||||
|
}
|
||||||
|
|
||||||
|
fn analyze_word_count(input: &[(String,usize)]) -> usize {
|
||||||
|
let mut count = 0;
|
||||||
|
input.iter().for_each(|(_,c)| { count+=c });
|
||||||
|
count
|
||||||
|
}
|
||||||
|
|
||||||
|
fn analyze_unique_word_count(input: &[(String,usize)]) -> usize {
|
||||||
|
input.len()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn average_word_length(input: &[(String, usize)]) -> usize {
|
||||||
|
let mut len = 0;
|
||||||
|
let mut tokens = 0;
|
||||||
|
input.iter().for_each(|(s,c)| {
|
||||||
|
len += s.len() * c;
|
||||||
|
tokens += c;
|
||||||
|
});
|
||||||
|
|
||||||
|
len/tokens
|
||||||
|
}
|
||||||
|
|
||||||
|
fn top_n_most_used(input: &[(String, usize)], n: usize) -> &[(String, usize)] {
|
||||||
|
let n = n.min(input.len());
|
||||||
|
&input[..n]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn print(input: &[(String, usize)]) {
|
||||||
|
let counts = input;
|
||||||
|
let wc = analyze_word_count(&counts);
|
||||||
|
|
||||||
|
// todo: clean this up a bit
|
||||||
|
println!("Word Count % of total");
|
||||||
|
println!("---------------------");
|
||||||
|
for (w, c) in counts {
|
||||||
|
println!("{} {} {}", w, c, *c as f64/(wc as f64));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let tokens = tokenize("Hello, world! hello");
|
||||||
|
assert_eq!(tokens.len(), 3);
|
||||||
|
|
||||||
|
let counts = count_words(&tokens);
|
||||||
|
let wc = analyze_word_count(&counts);
|
||||||
|
assert_eq!(wc, 3);
|
||||||
|
|
||||||
|
let uwc = analyze_unique_word_count(&counts);
|
||||||
|
assert_eq!(uwc, 2);
|
||||||
|
|
||||||
|
let avg_len = average_word_length(&counts);
|
||||||
|
assert_eq!(avg_len, 5);
|
||||||
|
|
||||||
|
let top_n = top_n_most_used(&counts, 4);
|
||||||
|
assert_eq!(top_n[0].0, "hello");
|
||||||
|
|
||||||
|
print(&counts);
|
||||||
|
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user