###########################################################
#
# R script for analyzing survey responses in "A People's History of MDLS" project.
# Summer Mengarelli, smengare@nd.edu
#
# AI Disclosure: No part of this script was generated or aided by generative AI.
# 
# Reference: Much of the code for the analyses included below was adapted from: https://www.tidytextmining.com/ 
#
###########################################################

library(tidyverse)
library(tidytext)
library(tm)
library(topicmodels)
library(stringr)
library(ggthemes)
library(igraph)
library(ggraph)
library(cowplot)

# Sentiment Analysis ----------------------------------------------------------------
## Step 1: Load in survey responses and concatenate relevant responses per respondent
# resp <- read.csv() [should be uncommented and filepath provided in parentheses]
resp <- resp %>%
  unite(response, first_learn, attend_multiple, professional_impact,
        specific_impact, connections, anything_else, sep = " ") %>%
  select(respondent_id, response)

## Step 2: Make data tidy
tidy.resp <- resp %>%
  unnest_tokens(word, response) %>% # tokenize
  anti_join(stop_words, by = "word") # remove stop words

## Step 3: Assess for joy and negative emotions using the 'nrc' sentiment lexicon
joy <- get_sentiments("nrc") %>% filter(sentiment == "joy")

resp.joy <- tidy.resp %>% inner_join(joy) %>%
  count(word, sort = TRUE) %>%
  mutate(emotion = "Joy")

neg.emos <- c("anger", "sadness", "fear", "disgust")
neg.intens <- get_sentiments("nrc") %>% filter(sentiment %in% neg.emos)

resp.neg <- tidy.resp %>% inner_join(neg.intens) %>%
  count(word, sort = TRUE) %>%
  mutate(emotion = "Anger/Fear/Disgust/Sadness")

nrc <- bind_rows(resp.joy, resp.neg)

### VISUALIZE: Negative and joyful words mirrored
nrc <- nrc %>%  mutate(emotion_val = ifelse(emotion =="Joy",n,n*-1))

nrc %>% filter(n > 2) %>%
  ggplot(aes(x=word, y=emotion_val, fill=emotion)) +
  geom_bar(stat="identity",position="identity") +
  scale_fill_manual(values = c("#CC79A7", "#009E73")) +
  coord_flip()+
  ggtitle(" ")+
  geom_hline(yintercept=0)+
  xlab(" ")+
  ylab("Word Occurrence")+
  scale_y_continuous(breaks = pretty(nrc$emotion_val),labels = abs(pretty(nrc$emotion_val))) +
  theme_hc() +
  theme(legend.title=element_blank()) +
  theme(legend.position = "top")
# Shout out to Luis D. Verde Arregoitia for the code this plot was adapted from: https://luisdva.github.io/rstats/Diverging-bar-plots/

## Step 4: Assess for negative and positive using the "bing" sentiment lexicon
resp.bing <- tidy.resp %>%
  inner_join(get_sentiments("bing")) %>%
  count(word, sentiment, sort = TRUE) %>%
  ungroup()

### VISUALIZE: Negative and positive words using bing
resp.bing %>%
  group_by(sentiment) %>%
  slice_max(n, n = 10) %>% 
  ungroup() %>%
  mutate(word = reorder(word, n)) %>%
  ggplot(aes(n, word, fill = sentiment)) +
  geom_col(show.legend = FALSE) +
  facet_wrap(~sentiment, scales = "free_y") +
  scale_fill_manual(values = c("#CC79A7", "#009E73")) +
  labs(x = " ", y = " ") +
  theme_cowplot()

# Bigrams ----------------------------------------------------------------
## Step 1: Generate and count bigrams, collapsing all respondents' responses together
resp.bigrams <- resp %>%
  unnest_tokens(bigram, response, token = "ngrams", n = 2) %>%
  filter(!is.na(bigram)) %>%
  separate(bigram, c("word1", "word2"), sep = " ") %>%
  filter(!word1 %in% stop_words$word & word1 != "na") %>%
  filter(!word2 %in% stop_words$word & word2 != "na") %>%
  count(word1, word2, sort = TRUE)

## Step 2: Create bigram graph object using igraph package
bigram_graph <- resp.bigrams %>%
  filter(n > 1) %>%
  graph_from_data_frame()

## Step 3: Visualize graph using ggraph package
set.seed(2020)
a <- grid::arrow(type = "closed", length = unit(.05, "inches"))

ggraph(bigram_graph, layout = "fr") +
  geom_edge_link(aes(edge_alpha = n, label = n), vjust = 1, show.legend = FALSE,
                 arrow = a, end_cap = circle(.2, 'inches')) +
  geom_node_point(color = "#009E73", size = 10) +
  geom_node_text(aes(label = name), vjust = 1, hjust = 1) +
  theme_void() 

# Topic Modeling ----------------------------------------------------------------
## Step 1: Read in respondent data organized per question
# quest <- read.csv() [should be uncommented and filepath provided in parentheses]

## Step 2: Concatenate all responses per question into a single "responses" column and tidy
quest$question <- as.factor(quest$question)
quest <- quest %>% filter(!is.na(response)) %>% select(question, response) %>%
  group_by(question) %>%
  summarize(responses = paste(response, collapse = " "))

tidy.q <- quest %>%
  unnest_tokens(word, responses) %>% # tokenize
  anti_join(stop_words, by = "word") # remove stop words

stops <- c("mdls", "symposium", "data", "librarian") # remove custom stop words; this is subjective
tidy.q <- tidy.q %>% filter(!(word %in% stops))

## Step 3: Create DocumentTermMatrix
quest.dtm <- tidy.q %>%
  count(question, word) %>%
  cast_dtm(question, word, n)

## Step 4: Run the LDA topic model. 'k' should be tested; determines resulting number of topics
q.lda <- LDA(quest.dtm, k = 4, control = list(seed = 1234))
q.topics <- tidy(q.lda, matrix = "beta")

##  Step 5: Pull out top 10 terms for each topic and visualize them
q.terms <- q.topics %>%
  group_by(topic) %>%
  slice_max(beta, n = 10, with_ties = FALSE) %>% 
  ungroup() %>%
  arrange(topic, -beta)

q.terms %>%
  mutate(term = reorder_within(term, beta, topic)) %>%
  ggplot(aes(beta, term, fill = factor(topic))) +
  geom_col(show.legend = FALSE) +
  facet_wrap(~ topic, scales = "free") +
  scale_y_reordered()
# I didn't find this to be meaningful, and increasing or decreasing 'k' did not seem to improve interpretability

# The End! ----------------------------------------------------------------