-
Notifications
You must be signed in to change notification settings - Fork 3
/
slack-wordcloud.r
46 lines (33 loc) · 1.3 KB
/
slack-wordcloud.r
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
#Obviously these need to be installed!
library(jsonlite)
library(tm)
library(wordcloud)
output.filename <- "~/Desktop/wordcloud.png"
files <- list.files('.',"*.json", recursive=T)
json <- sapply(files, fromJSON)
texts <- sapply(json, function(f){if ('subtype' %in% names(f)) f$text[is.na(f$subtype)] else f$text})
flat <- unlist(texts)
removeUsernames <- function(doc) {
gsub("<@[0-9a-zA-Z]*>", "", doc)
}
removeCharacters <- function (doc, characters) {
pattern <- paste(characters, collapse = "|")
gsub(pattern, "", doc, perl = TRUE)
}
corpus <- Corpus(VectorSource(flat))
# Remove slack username identifiers
corpus <- tm_map(corpus, removeUsernames)
# general cleaning
corpus <- tm_map(corpus, stripWhitespace)
corpus <- tm_map(corpus, tolower)
corpus <- tm_map(corpus, removePunctuation)
corpus <- tm_map(corpus, removeNumbers)
# Remove a unicode apostrophe
corpus <- tm_map(corpus, removeCharacters, c('\u2019'))
# Remove some stopwords
corpus <- tm_map(corpus, removeWords, stopwords('english'))
corpus <- tm_map(corpus, removeWords, c('just', 'like', 'can', 'get'))
# Generate the wordcloud and save it to a png
png(output.filename, height=800, width=800)
wordcloud(corpus, scale=c(4,1), max.words=300, random.order=FALSE, rot.per=0.35, use.r.layout=FALSE, colors=brewer.pal(8, "Dark2"))
dev.off()