updated comments, added tests
This commit is contained in:
+58
-10
@@ -11,14 +11,14 @@
|
|||||||
// 4. (pretty) print the info
|
// 4. (pretty) print the info
|
||||||
|
|
||||||
|
|
||||||
// todo: what about CTRL chars?
|
|
||||||
// todo: what about feeding Bytes?
|
|
||||||
// todo: return Result for whitespace input
|
// todo: return Result for whitespace input
|
||||||
|
// CTRL chars are preserved in tokens, though whitespace ones will get dropped
|
||||||
|
// bytes cannot be fed since they would be &[u8]; str ensures valid UTF-8
|
||||||
|
// hypenated words are preserved
|
||||||
pub fn tokenize(input: &str) -> Vec<String> {
|
pub fn tokenize(input: &str) -> Vec<String> {
|
||||||
// filter punctuation
|
|
||||||
let mut s = String::new();
|
let mut s = String::new();
|
||||||
for c in input.chars() {
|
for c in input.chars() {
|
||||||
if !c.is_ascii_punctuation() {
|
if !c.is_ascii_punctuation() { // todo: unicode punctuation not handled
|
||||||
// to_lowercase on a char will always yield at least one char
|
// to_lowercase on a char will always yield at least one char
|
||||||
// if a char lowercases to two chars then we don't truncate by using
|
// if a char lowercases to two chars then we don't truncate by using
|
||||||
// a for loop
|
// a for loop
|
||||||
@@ -35,7 +35,7 @@ pub fn tokenize(input: &str) -> Vec<String> {
|
|||||||
// todo: can the input type be better?
|
// todo: can the input type be better?
|
||||||
// total runtime complexity worst-case O(n log n)
|
// total runtime complexity worst-case O(n log n)
|
||||||
pub fn create_frequency_vec(input: &[String]) -> Vec<(String,usize)> {
|
pub fn create_frequency_vec(input: &[String]) -> Vec<(String,usize)> {
|
||||||
let mut sorted = input.to_vec();
|
let mut sorted = input.to_vec(); // copy: O(n)
|
||||||
sorted.sort(); // O(n log n)
|
sorted.sort(); // O(n log n)
|
||||||
|
|
||||||
let mut counts : Vec<(String, usize)> = Vec::new();
|
let mut counts : Vec<(String, usize)> = Vec::new();
|
||||||
@@ -66,7 +66,10 @@ pub fn average_word_length(input: &[(String, usize)]) -> f64 {
|
|||||||
let mut len = 0;
|
let mut len = 0;
|
||||||
let mut tokens = 0;
|
let mut tokens = 0;
|
||||||
input.iter().for_each(|(s,c)| {
|
input.iter().for_each(|(s,c)| {
|
||||||
len += s.len() * c;
|
// this will work for Unicode Scalar value characters
|
||||||
|
// ie: graphemes with multiple code points (like some emojis) will break
|
||||||
|
// but not going to pull in unicode-segmentation just to get this right
|
||||||
|
len += s.chars().count() * c;
|
||||||
tokens += c;
|
tokens += c;
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -119,8 +122,7 @@ pub fn print_analysis(input: &[(String, usize)]) {
|
|||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
println!("playing around with vecs and slices");
|
println!("playing around with vecs and slices");
|
||||||
println!("use `cargo test`");
|
println!("use `cargo test`]\n");
|
||||||
println!("");
|
|
||||||
test_print();
|
test_print();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -135,7 +137,7 @@ pub mod test {
|
|||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
pub fn empty_tests() {
|
pub fn empty() {
|
||||||
let tokens = tokenize("");
|
let tokens = tokenize("");
|
||||||
assert_eq!(tokens.len(), 0);
|
assert_eq!(tokens.len(), 0);
|
||||||
|
|
||||||
@@ -158,7 +160,7 @@ pub mod test {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
pub fn basic_tests() {
|
pub fn basic() {
|
||||||
let tokens = tokenize("Hello, world! hello");
|
let tokens = tokenize("Hello, world! hello");
|
||||||
assert_eq!(tokens.len(), 3);
|
assert_eq!(tokens.len(), 3);
|
||||||
|
|
||||||
@@ -179,4 +181,50 @@ pub mod test {
|
|||||||
|
|
||||||
print_analysis(&freq_vec);
|
print_analysis(&freq_vec);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
pub fn unicode_and_ctrl() {
|
||||||
|
let tokens = tokenize("Hello\u{0000} world! hello ß");
|
||||||
|
assert_eq!(tokens.len(), 4);
|
||||||
|
|
||||||
|
let freq_vec = create_frequency_vec(&tokens);
|
||||||
|
assert_eq!(freq_vec.len(), 4);
|
||||||
|
|
||||||
|
let wc = analyze_word_count(&freq_vec);
|
||||||
|
assert_eq!(wc, 4);
|
||||||
|
|
||||||
|
let uwc = analyze_unique_word_count(&freq_vec);
|
||||||
|
assert_eq!(uwc, 4);
|
||||||
|
|
||||||
|
let avg_len = average_word_length(&freq_vec);
|
||||||
|
assert_eq!(avg_len, 4.25);
|
||||||
|
|
||||||
|
let top_n = top_n_most_used(&freq_vec, 4);
|
||||||
|
assert_eq!(top_n[0].0, "hello");
|
||||||
|
|
||||||
|
print_analysis(&freq_vec);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
pub fn unicode_2() {
|
||||||
|
let tokens = tokenize("café");
|
||||||
|
assert_eq!(tokens.len(), 1);
|
||||||
|
|
||||||
|
let freq_vec = create_frequency_vec(&tokens);
|
||||||
|
assert_eq!(freq_vec.len(), 1);
|
||||||
|
|
||||||
|
let wc = analyze_word_count(&freq_vec);
|
||||||
|
assert_eq!(wc, 1);
|
||||||
|
|
||||||
|
let uwc = analyze_unique_word_count(&freq_vec);
|
||||||
|
assert_eq!(uwc, 1);
|
||||||
|
|
||||||
|
let avg_len = average_word_length(&freq_vec);
|
||||||
|
assert_eq!(avg_len, 4.0);
|
||||||
|
|
||||||
|
let top_n = top_n_most_used(&freq_vec, 4);
|
||||||
|
assert_eq!(top_n[0].0, "café");
|
||||||
|
|
||||||
|
print_analysis(&freq_vec);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user