##########################################################
# Perceptual and Preference Mapping in Python            #
# Chestnut Ridge Competitive Positioning Example         #
##########################################################

# This script reproduces the full Chestnut Ridge perceptual and preference mapping example in Python:
#   Step 1. Principal components analysis (PCA) on consumer perceptions
#   Step 2. Perceptual map data (attribute vectors and brand positions)
#   Step 3. Perceptual map of retailers and attributes
#   Step 4. Joint perceptual and preference map (adds consumer preferences)
#   Step 5. Market share predictions (first choice and share of preference rules)
# Run it from top to bottom. Each step prints its result or shows a plot.
# The "# %%" lines split the script into cells you can run one at a time
# in VS Code, Spyder, or PyCharm.

# %%
## Install Packages (if needed - only installs packages that are missing)
## Each entry is {name used with import: name used with pip install}
import sys, subprocess, importlib.util
packages = {"pandas": "pandas", "numpy": "numpy", "matplotlib": "matplotlib", "sklearn": "scikit-learn", "adjustText": "adjustText"}
new_packages = [pip_name for import_name, pip_name in packages.items() if importlib.util.find_spec(import_name) is None]
if new_packages: subprocess.check_call([sys.executable, "-m", "pip", "install", *new_packages])

## Load Packages
import pandas as pd                               # data manipulation
import numpy as np                                # matrix calculations
import matplotlib.pyplot as plt                   # plots and maps
from sklearn.decomposition import PCA             # principal components analysis
from adjustText import adjust_text                # labels that do not overlap on the maps
from tkinter import Tk, filedialog                # file-choosing dialog box

## Function to choose a file with a dialog box
## (the Python version of file.choose() in R - no need to set the working directory)
def file_choose(title="Choose a file"):
    root = Tk()
    root.withdraw()                     # hide the empty Tk window
    root.attributes("-topmost", True)   # bring the dialog to the front
    path = filedialog.askopenfilename(title=title, filetypes=[("CSV files", "*.csv"), ("All files", "*.*")])
    root.destroy()
    if not path:
        raise SystemExit("No file was chosen.")
    return path

# Import Data
per = pd.read_csv(file_choose("Choose perceptions.csv"))   ## Choose perceptions.csv file
pref = pd.read_csv(file_choose("Choose preferences.csv"))  ## Choose preferences.csv file


# %%
#####################################################
# Step 1: Principal Components Analysis             #
#####################################################

# The seven attributes (all columns except Brand)
attributes = per.columns[1:].tolist()

# Standardize each attribute (subtract the mean, divide by the standard deviation)
# so all are measured on the same scale
per_scaled = (per[attributes] - per[attributes].mean()) / per[attributes].std()

# Run PCA on the perception ratings
pca = PCA()
scores = pca.fit_transform(per_scaled)
loadings = pca.components_.T          # one row per attribute, one column per component

# The direction of each component is arbitrary (flipping its sign gives the same map,
# mirrored). Orient the map as in the book: Price points left on Factor 1 and
# Convenience points up on Factor 2.
flip = np.ones(len(attributes))
if loadings[attributes.index("Price"), 0] > 0:
    flip[0] = -1
if loadings[attributes.index("Convenience"), 1] < 0:
    flip[1] = -1
loadings = loadings * flip
scores = scores * flip

# Share of the variance in perceptions explained by each principal component
# The first two components are the two dimensions (Factor 1 and Factor 2) of the map
sdev = np.sqrt(pca.explained_variance_)
importance = pd.DataFrame({"Standard deviation": sdev,
                           "Proportion of Variance": pca.explained_variance_ratio_,
                           "Cumulative Proportion": pca.explained_variance_ratio_.cumsum()},
                          index=[f"PC{i}" for i in range(1, len(sdev) + 1)]).T
print("\nImportance of Components")
print(importance.round(4).to_string())


# %%
#####################################################
# Step 2: Perceptual Map Data                       #
#####################################################

# Attribute vectors: the loadings of each attribute on the first two components.
# Multiplying by the standard deviation of each component gives the correlation
# between the attribute and the component, so each vector runs from -1 to 1.
attribute_vectors = pd.DataFrame({"attribute": attributes,
                                  "factor1": loadings[:, 0] * sdev[0],
                                  "factor2": loadings[:, 1] * sdev[1]})

print("\nAttribute Vectors (Loadings on Factor 1 and Factor 2)")
print(attribute_vectors.round(2).to_string(index=False))

# Brand positions: the component scores of each retailer, rescaled so the
# retailer farthest from the center on each dimension sits at -1 or 1.
# This puts the brands on the same scale as the attribute vectors.
brand_positions = pd.DataFrame({"brand": per["Brand"],
                                "score1": scores[:, 0] / np.abs(scores[:, 0]).max(),
                                "score2": scores[:, 1] / np.abs(scores[:, 1]).max()})

print("\nBrand Positions (Scores on Factor 1 and Factor 2)")
print(brand_positions.round(2).to_string(index=False))


# %%
#####################################################
# Step 3: Perceptual Map                            #
#####################################################

def draw_map(ax):
    """Draw the attribute vectors and retailer positions shared by both maps."""
    ax.axhline(0, color="0.85", zorder=0)
    ax.axvline(0, color="0.85", zorder=0)
    labels = []
    # Attribute vectors drawn from the origin
    for _, row in attribute_vectors.iterrows():
        ax.annotate("", xy=(row["factor1"], row["factor2"]), xytext=(0, 0),
                    arrowprops=dict(arrowstyle="->", color="0.4"))
        labels.append(ax.text(row["factor1"], row["factor2"], row["attribute"],
                              fontsize=10, color="0.2"))
    # Retailer positions
    ax.scatter(brand_positions["score1"], brand_positions["score2"],
               s=60, color="firebrick", label="Retailer", zorder=3)
    for _, row in brand_positions.iterrows():
        labels.append(ax.text(row["score1"], row["score2"], row["brand"],
                              fontsize=10, fontweight="bold", color="firebrick"))
    ax.set_xlim(-1.1, 1.1)
    ax.set_ylim(-1.1, 1.1)
    ax.set_aspect("equal")
    ax.set_xlabel("Factor 1")
    ax.set_ylabel("Factor 2")
    return labels

fig, ax = plt.subplots(figsize=(8, 8))
labels = draw_map(ax)
ax.set_title("Perceptual Map of Retailers and Attributes")
adjust_text(labels, ax=ax)   # move labels so they do not overlap
plt.show()

# Retailers close together are perceived as similar. A retailer located far along
# an attribute vector is perceived as strong on that attribute.


# %%
#####################################################
# Step 4: Joint Perceptual and Preference Map       #
#####################################################

# Each respondent's preference vector is the weighted sum of the brand positions,
# using that respondent's 0-10 preference ratings as weights. Respondents who like
# a retailer are pulled toward that retailer's position on the map.
preference_matrix = pref[brand_positions["brand"]].to_numpy()
brand_matrix = brand_positions[["score1", "score2"]].to_numpy()
respondent_points = preference_matrix @ brand_matrix

# Rescale so the respondent farthest from the center on each dimension sits at -1 or 1
respondent_points = respondent_points / np.abs(respondent_points).max(axis=0)

respondent_preferences = pd.DataFrame({"respondent": pref["Respondents"],
                                       "score1": respondent_points[:, 0],
                                       "score2": respondent_points[:, 1]})

# Each point marks the tip of a respondent's preference vector. The vectors
# themselves are not drawn to reduce clutter.
fig, ax = plt.subplots(figsize=(8, 8.5))
ax.scatter(respondent_preferences["score1"], respondent_preferences["score2"],
           s=25, color="steelblue", alpha=0.7, label="Consumer Preference", zorder=2)
labels = draw_map(ax)
ax.set_title("Joint Perceptual and Preference Map of Retailers and Attributes")
ax.legend(loc="upper center", bbox_to_anchor=(0.5, -0.08), ncol=2, frameon=False)
adjust_text(labels, ax=ax)
plt.show()

# Look for areas of the map with many consumer preferences. Retailers located in
# or moving toward those areas are positioned to attract more consumer preference.


# %%
#####################################################
# Step 5: Market Share Predictions                  #
#####################################################

# First choice rule: each respondent buys only from the retailer they rate highest.
# If a respondent gives the same top rating to more than one retailer, each of
# those retailers gets credit for that respondent.
first_choice = preference_matrix == preference_matrix.max(axis=1, keepdims=True)
first_choice_share = first_choice.sum(axis=0) / first_choice.sum()

# Share of preference rule (logit): each respondent spreads their purchases across
# retailers in proportion to exp(preference rating). A retailer's market share is
# the average of these proportions across respondents.
logit_share = np.exp(preference_matrix) / np.exp(preference_matrix).sum(axis=1, keepdims=True)
logit_share_share = logit_share.mean(axis=0)

market_share = pd.DataFrame({"brand": brand_positions["brand"],
                             "first_choice": first_choice_share,
                             "logit_share": logit_share_share})

print("\nPredicted Market Share by Retailer")
print(market_share.round(3).to_string(index=False))



