Repository navigation
Expand file tree
/
Copy pathEDA.Rmd
More file actions
97 lines (77 loc) · 2.11 KB
/
Copy pathEDA.Rmd
File metadata and controls
97 lines (77 loc) · 2.11 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
---
title: "R Notebook"
output: html_notebook
---
```{r}
library(tidyverse)
library(tidytext)
```
```{r}
en_US_news <- read_delim("final/en_US/en_US.news.txt", "\t",
escape_double = FALSE, trim_ws = TRUE,
col_names = "news")
```
```{r}
en_US_news %>%
# slice(1:100) %>%
mutate(obs = row_number()) %>%
unnest_tokens(word, news) %>%
count(word, sort = T) %>%
anti_join(stop_words) %>%
mutate(word = reorder(word, n)) %>%
slice(1:20) %>%
ggplot(aes(word, n)) +
geom_col() + xlab(NULL) + coord_flip()
```
```{r}
en_US_news
```
```{r}
en_US_blogs <- read_delim("final/en_US/en_US.blogs.txt", "\t",
escape_double = FALSE, col_names = "blogs", trim_ws = TRUE)
en_US_twitter <- read_delim("final/en_US/en_US.twitter.txt", "\t",
escape_double = FALSE, col_names = "twits", trim_ws = TRUE)
```
```{r}
en_US_blogs %>% mutate(l = nchar(blogs)) %>% arrange(desc(l)) %>% slice(1)
```
```{r}
en_US_twitter %>%
#slice(1:100) %>%
mutate(obs = row_number()) %>%
unnest_tokens(word, twits) %>%
filter(word %in% c("love", "hate")) %>%
group_by(obs, word) %>% summarise(n = n()) %>% ungroup() %>%
group_by(word) %>% summarise(n = n())
```
```{r}
99997/17662
```
```{r}
en_US_twitter %>%
filter(str_detect(twits, "A computer once beat me at chess, but it was no match for me at kickboxing"))
```
```{r}
en_US_twitter %>%
# slice(1:100) %>%
mutate(obs = row_number()) %>%
unnest_tokens(word, twits) %>%
count(word, sort = T) %>%
anti_join(stop_words, by = "word") %>%
mutate(word = reorder(word, n)) %>%
slice(1:20) %>%
ggplot(aes(word, n)) +
geom_col() + xlab(NULL) + coord_flip()
```
```{r}
en_US_twitter %>%
# slice(1:100) %>%
mutate(obs = row_number()) %>%
unnest_tokens(word, twits, token = "ngrams", n = 3) %>%
count(word, sort = T) %>%
anti_join(stop_words, by = "word") %>%
mutate(word = reorder(word, n)) %>%
slice(1:20) %>%
ggplot(aes(word, n)) +
geom_col() + xlab(NULL) + coord_flip()
```