-
Notifications
You must be signed in to change notification settings - Fork 3
/
tokenizer.js
49 lines (44 loc) · 2.13 KB
/
tokenizer.js
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
const commonInsignificantWords = ['a', 'an', 'the', 'and', 'but', 'or', 'for', 'nor', 'on',
'at', 'to', 'from', 'by', 'we', 'of', 'as', 'do', 'up', 'if', 'i', 'you', 'are', 'they',
'it', 'our', 'be', 'is', 'in', 'my', 'with', 'have', 'has', 'no', 'how', 'was', 'very',
'this', 'he', 'that', 'it\'s', 'cunt', 'fuck', 'like', 'not', 'your', 'don\'t', 'she',
'his', 'her', 'just', 'when', 'so', 'got', 'get', 'what', 'why', 'who', 'how', 'would',
'should', 'could', 'some', 'can', 'you\'re', 'about', 'which', 'had', 'want', 'made' ]
module.exports = function(reddit) {
return {
findTopWords: (subreddit, callback) => {
var tokens = {};
reddit.getSubreddit(subreddit)
.getHot({limit: 100})
.map(data => data.title)
.then(data => {
data.forEach(title => {
var words = title.split(' ');
words.forEach(element => {
element = element.toLowerCase();
tokens[element] = tokens[element] ? tokens[element]+1 : 1;
});
});
var objs = Object.keys(tokens).map(toke => {
if (tokens[toke] > 1) {
return {
token: toke,
count: tokens[toke]
}
} else {
return false;
}
})
.filter(obj => obj)
.filter(obj => obj.token.length > 1)
.filter(obj => commonInsignificantWords.indexOf(obj.token) === -1)
.sort( (a,b) => b.count - a.count)
.slice(0, 20);
var words = objs.map(obj => obj.token);
var counts = objs.map(obj => obj.count);
if (callback) callback(words, counts);
else console.log(objs);
});
}
}
}