##############################################################
# Chapter 16 - Using Marketing Experiments to Optimize the   #
#              Marketing Mix                                 #
# Chestnut Ridge Email Campaign Causal Random Forest         #
# (R)                                                        #
##############################################################

# This script shows another way to estimate the effect of the Chestnut Ridge email
# campaign in R. It uses the same data as the propensity score matching example
# (propensity_score_matching.R), but estimates the effect with a causal random forest.
# NOTE: This approach is NOT covered in the book's examples. It is provided as an
# optional extension for comparison with the propensity score matching results.
#   Step 1. Install packages (if needed), load packages, and set seed
#   Step 2. Read in the email marketing campaign data
#   Step 3. Train a causal random forest
#   Step 4. Estimate the average treatment effect of the email
# Run it from top to bottom. Each step prints its result to the console or Plots pane.


##############################################################
# Step 1: Install packages (if needed), load packages, and   #
#         set seed                                           #
##############################################################

packages <- c("grf")
new_packages <- packages[!(packages %in% installed.packages()[, "Package"])]
if (length(new_packages) > 0) install.packages(new_packages)

library(grf)
set.seed(1)


##############################################################
# Step 2: Read in the email marketing campaign data          #
##############################################################

data <- read.csv(file.choose())  ## Choose the file retail_psmatch.csv

## Number of customers
nrow(data)


##############################################################
# Step 3: Train a causal random forest                       #
##############################################################

## X: customer characteristics (the same covariates used for matching)
## W: treatment (1 = received the email, 0 = no email)
## Y: outcome (purchase amount)
x_vars <- c('revenue', 'number_of_orders', 'number_of_orders2', 'recency_days')
X <- data[, x_vars]
W <- data$email
Y <- data$purchase_amt

email.forest <- causal_forest(X, Y, W)


##############################################################
# Step 4: Estimate the average treatment effect of the email #
##############################################################

## Average treatment effect (ATE) on the full sample, with its standard error
average_treatment_effect(email.forest, target.sample = "all")
