##################################################################
# Chapter 17: Using Topic Models to Glean Customer Insights      #
# Structural Topic Model of Apple Watch Product Reviews          #
##################################################################

# This script reproduces the full Apple Watch product review topic modeling example in R:
#   Step 1. Look at a sample of the data
#   Step 2. Process the documents (clean and stem the review text)
#   Step 3. Determine the number of topics and estimate the topic model
#   Step 4. See which topics relate to high vs. low ratings
#   Step 5. Identify the top 3 (positive) and bottom 3 (negative) topics
#   Step 6. Visualize the topics (word clouds)
# Run it from top to bottom. Each step prints its result to the console or Plots pane.

## Install Packages (if needed - only installs packages that are missing)
packages <- c("stm", "tm", "Rtsne", "rsvd", "geometry", "SnowballC", "wordcloud")
new_packages <- packages[!(packages %in% installed.packages()[, "Package"])]
if (length(new_packages) > 0) install.packages(new_packages)

## Load Packages and Set Seed
library(stm)
library(tm)
library(Rtsne)
library(rsvd)
library(geometry)
library(SnowballC)
library(wordcloud)
set.seed(1)

## Read in the Product Reviews Data
reviews <- read.csv(file.choose())  ## Choose the file reviews_data.csv


##################################################################
# Step 1: Look at a Sample of the Data                           #
##################################################################

head(reviews)
table(reviews$review_star)

## Make Sure the Text is Encoded Properly (works on both Mac and PC)
reviews$review_text <- iconv(reviews$review_text, from = "", to = "UTF-8", sub = " ")


##################################################################
# Step 2: Process Documents                                      #
##################################################################

## (lowercase, remove stop words, punctuation, and numbers, stem words,
##  and remove the custom context-specific words below)
customwords <- c("watch", "iphone", "apple")
processed <- textProcessor(reviews$review_text, metadata = reviews,
                           customstopwords = customwords)
out <- prepDocuments(processed$documents, processed$vocab, processed$meta)
docs <- out$documents
vocab <- out$vocab
meta <- out$meta


##################################################################
# Step 3: Determine Number of Topics and Estimate the Topic      #
#         Model                                                  #
##################################################################

## (K = 0 lets the algorithm select the number of topics; takes significant time)
reviewsFit <- stm(documents = out$documents, vocab = out$vocab, K = 0, seed = 1,
                  prevalence =~ review_star, data = out$meta, init.type = "Spectral")

## See How Many Topics
num_topics <- reviewsFit$settings$dim$K
num_topics


##################################################################
# Step 4: See Which Topics Relate to High vs. Low Ratings        #
##################################################################

out$meta$rating <- as.factor(out$meta$review_star)
prep <- estimateEffect(1:num_topics ~ rating, reviewsFit, meta = out$meta,
                       uncertainty = "Global")
effects <- plot(prep, covariate = "rating", topics = c(1:num_topics), model = reviewsFit,
                method = "difference", cov.value1 = 5, cov.value2 = 1,
                xlab = "Lower Rating ... Higher Rating",
                main = "Relationship between Topic and Rating",
                labeltype = "custom", custom.labels = c(1:num_topics))


##################################################################
# Step 5: Identify the Top 3 (Positive) and Bottom 3 (Negative)  #
#         Topics                                                 #
##################################################################

## (difference in topic prevalence between 5-star and 1-star reviews)
topic_diff <- data.frame(topic = effects$topics, difference = unlist(effects$means))
topic_diff <- topic_diff[order(topic_diff$difference, decreasing = TRUE), ]
positive_topics <- head(topic_diff$topic, 3)
negative_topics <- tail(topic_diff$topic, 3)[3:1]  ## most negative first
positive_topics
negative_topics

## Top Words in the Positive and Negative Topics
labelTopics(reviewsFit, topics = positive_topics)
labelTopics(reviewsFit, topics = negative_topics)


##################################################################
# Step 6: Visualize Topics                                       #
##################################################################

## Each word cloud is drawn as its own plot (use the arrows in the Plots pane
## to page through them). If a warning says a word "could not be fit on page",
## make the Plots pane larger and rerun this step.

## Positive Topics - Top 3
for (k in positive_topics) {
  cloud(reviewsFit, topic = k, scale = c(3, 0.5))
  title(main = paste("Positive Topic", k))
}

## Negative Topics - Bottom 3
for (k in negative_topics) {
  cloud(reviewsFit, topic = k, scale = c(3, 0.5))
  title(main = paste("Negative Topic", k))
}

