15-sentiment

Author
Affiliation

Prof Amanda Luby

Carleton College
Stat 220 - Fall 2026

library(tidyverse)
library(tidytext) # functions for doing text analysis

Load the Data

Random (?) sample of 26,882 reviews of coursera courses (Source: Kaggle)

en_coursera_reviews <- read_csv("https://stat220-f26.github.io/data/en_coursera_sample.csv")
en_coursera_reviews
# A tibble: 26,882 × 5
   CourseId                    Review                      Label cld2  review_id
   <chr>                       <chr>                       <dbl> <chr>     <dbl>
 1 nurture-market-strategies   It would be better if the …     1 en            1
 2 nand2tetris2                Superb course. Great prese…     5 en            2
 3 schedule-projects           Excellent course!               5 en            3
 4 teaching-english-capstone-2 I'd recommend this course …     5 en            4
 5 machine-learning            This course was so effecti…     5 en            5
 6 python-network-data         Words cannot describe how …     5 en            6
 7 clinical-trials             Great course!                   5 en            7
 8 python-genomics             I didn't know anything abo…     3 en            8
 9 strategic-management        Loved everything about thi…     5 en            9
10 script-writing              No significant instruction…     1 en           10
# ℹ 26,872 more rows

Load Sentiment data

“bing”

bing_sentiments = get_sentiments("bing") %>%
  slice_sample(n = 20)

“afinn”

# A tibble: 20 × 2
   word          value
   <chr>         <dbl>
 1 weary            -2
 2 comforting        2
 3 novel             2
 4 substantially     1
 5 costly           -2
 6 appreciated       2
 7 piteous          -2
 8 excite            3
 9 revered           2
10 flu              -2
11 cancelling       -1
12 benefits          2
13 beautify          3
14 unconcerned      -2
15 fortunate         2
16 vulnerability    -2
17 goddamn          -3
18 restore           1
19 menace           -2
20 enlightens        2

Sentiment of each review

bing_review_scores <- en_coursera_reviews %>%
  unnest_tokens(word, Review) %>% 
  inner_join(bing_sentiments, by = "word") %>%
  group_by(review_id) %>%
  summarize(
    sum = (sum(sentiment == "positive") - sum(sentiment == "negative"))
  )

bing_review_scores
# A tibble: 6 × 2
  review_id   sum
      <dbl> <int>
1       766    -1
2      3768     1
3     13782    -1
4     15401    -1
5     15988    -1
6     23226    -1