Lesson 39 · Market Research Analytics in Python
Master Brand Awareness & Perception Metrics Using Python for Market Research
In this lesson, we will explore how businesses track and analyze customer awareness of their brand as well as perceptions and attitudes. Understanding brand…
- CourseMarket Research Analytics in Python
- Lesson39 of 56
- Video27 min
- FormatJupyter notebook · 24 code cells
What you'll learn
Data
No separate download needed — the notebook creates or downloads everything it uses.
📓 Full notebook
Download .ipynbBrand Awareness and Brand Perception Metrics#
- In this lesson, we will explore how businesses track and analyze customer awareness of their brand as well as perceptions and attitudes.
- Understanding brand awareness and perception is crucial for market research and guides marketing strategy.
- You will practice loading, analyzing, and visualizing real survey and feedback data to interpret how people view a brand.
- By the end, you will be able to generate actionable insights from survey datasets, customer NPS, and open-ended feedback.
import pandas as pd
import numpy as np
import openml
import matplotlib.pyplot as plt
import seaborn as sns
import warnings
warnings.filterwarnings('ignore')
Understanding Brand Perception Data#
- Brand perception data can come from customer surveys, NPS responses, and open feedback.
- Surveys usually include demographics, scaled ratings (like 1-10), and sometimes open-ended comments.
- Net Promoter Score (NPS) asks customers how likely they are to recommend a brand to others.
- Open-ended feedback lets customers describe their brand experience in their own words.
- Beginners sometimes confuse scale directions, group customers incorrectly, or skip cleaning missing data.
np.random.seed(42)
df = pd.DataFrame({'CustomerID': range(1,501), 'Age': np.random.randint(18,70,500), 'Region': np.random.choice(['North','South','East','West'],500), 'NPS_Score': np.random.randint(0,11,500)})
print(df.shape)
print(df.head(3))
plt.figure(figsize=(7,4))
sns.histplot(df['NPS_Score'], bins=11, kde=False, color='skyblue')
plt.title('Distribution of NPS Scores')
plt.xlabel('NPS Score (0-10)')
plt.ylabel('Number of Customers')
plt.tight_layout()
plt.show()
# Classify NPS responses
def nps_category(score):
if score >= 9:
return 'Promoter'
elif score >= 7:
return 'Passive'
else:
return 'Detractor'
df['NPS_Type'] = df['NPS_Score'].apply(nps_category)
counts = df['NPS_Type'].value_counts()
print(counts)
n_promoters = counts.get('Promoter', 0)
n_detractors = counts.get('Detractor', 0)
n_total = len(df)
nps_score = ((n_promoters - n_detractors) / n_total) * 100
print(f"The company's Net Promoter Score (NPS) is: {nps_score:.2f}")
plt.figure(figsize=(8,4))
sns.countplot(data=df, x='Region', hue='NPS_Type', palette='coolwarm')
plt.title('NPS Segments by Region')
plt.xlabel('Region')
plt.ylabel('Customer Count')
plt.tight_layout()
plt.show()
df_feedback = pd.DataFrame({'CustomerID':[1,2,3,4,5], 'Feedback':['Great service and friendly staff','Delivery was slow and packaging was poor','Excellent quality, will buy again','Customer support needs improvement','Good value for money']})
print(df_feedback.head(3))
keywords = {}
for text in df_feedback['Feedback']:
for word in text.lower().split():
keywords[word] = keywords.get(word, 0) + 1
sorted_keywords = sorted(keywords.items(), key=lambda x: x[1], reverse=True)
print('Most common words:')
for word, count in sorted_keywords[:5]:
print(f'{word}: {count}')
dataset = openml.datasets.get_dataset(1461)
df_marketing, _, _, _ = dataset.get_data(dataset_format='dataframe')
df_marketing.columns = ['age','job','marital','education','default','balance','housing','loan','contact','day','month','duration','campaign','pdays','previous','poutcome','response']
print(df_marketing.shape)
print(df_marketing.head(3))
response_rate = df_marketing['response'].value_counts(normalize=True) * 100
print('Campaign Response Rate (%):')
print(response_rate.round(2))
plt.figure(figsize=(8,4))
sns.barplot(x='education', y='response', data=df_marketing.replace({'response': {'yes': 1, 'no': 0}}), estimator=np.mean)
plt.title('Campaign Response Rate by Education Level')
plt.xlabel('Education Level')
plt.ylabel('Response Rate')
plt.tight_layout()
plt.show()
merged = pd.merge(df, df_marketing[['response']].iloc[:len(df)], left_index=True, right_index=True)
xtab = pd.crosstab(merged['NPS_Type'], merged['response'], normalize='index') * 100
print('Cross-tabulation: NPS Segment vs. Campaign Response')
print(xtab.round(1))
df['Age_Group'] = pd.cut(df['Age'], bins=[17,30,45,60,70], labels=['18-30','31-45','46-60','61-70'])
plt.figure(figsize=(7,4))
sns.boxplot(x='Age_Group', y='NPS_Score', data=df, palette='pastel')
plt.title('Distribution of NPS Scores by Age Group')
plt.xlabel('Age Group')
plt.ylabel('NPS Score')
plt.tight_layout()
plt.show()
positive_words = ['great', 'excellent', 'good', 'friendly']
negative_words = ['poor', 'slow', 'needs improvement', 'bad']
def simple_sentiment(text):
text = text.lower()
if any(w in text for w in positive_words):
return 'Positive'
elif any(w in text for w in negative_words):
return 'Negative'
else:
return 'Neutral'
df_feedback['Sentiment'] = df_feedback['Feedback'].apply(simple_sentiment)
print(df_feedback[['Feedback','Sentiment']])
np.random.seed(42)
dates = pd.date_range('2022-01-01', periods=12, freq='M')
trend_data = pd.DataFrame({'Month': dates})
trend_data['Mean_NPS'] = np.random.normal(7, 1, 12).clip(0,10)
plt.figure(figsize=(8,4))
plt.plot(trend_data['Month'], trend_data['Mean_NPS'], marker='o', color='navy')
plt.title('NPS Trend Over Time')
plt.xlabel('Month')
plt.ylabel('Mean NPS Score')
plt.xticks(rotation=45)
plt.tight_layout()
plt.show()
import openml
dataset = openml.datasets.get_dataset(42178)
df_satisfaction, _, _, _ = dataset.get_data(dataset_format='dataframe')
factors = ['OnlineSecurity', 'TechSupport', 'StreamingTV', 'OnlineBackup', 'Churn']
for col in factors:
df_satisfaction[col] = df_satisfaction[col].astype(str)
df_satisfaction[col] = df_satisfaction[col].map({'Yes': 1, 'No': 0, 'No internet service': 0, 'No phone service': 0, 'nan': 0, 'False': 0, 'True':1}).fillna(0)
df_satisfaction['Brand_Perception_Index'] = df_satisfaction[factors[:-1]].mean(axis=1) * (1 - df_satisfaction['Churn'])
print(df_satisfaction[['Brand_Perception_Index']].head(5))
print('Index ranges from 0 (poor perception) to 1 (very positive perception and retention)')
with open('brand_perception_report.txt', 'w') as f:
f.write('Brand Perception Analysis Report\n')
f.write(f'NPS: {nps_score:.2f}\n')
f.write(f'Highest NPS Region: {df.groupby("Region")["NPS_Score"].mean().idxmax()}\n')
f.write('Sample Brand Perception Indices (first 5):\n')
for idx, val in enumerate(df_satisfaction["Brand_Perception_Index"].head(5)):
f.write(f'Customer {idx+1}: {val:.2f}\n')
df_missing = df.copy()
df_missing.loc[5:10, 'NPS_Score'] = np.nan
missing_count = df_missing['NPS_Score'].isnull().sum()
print(f'Missing NPS responses: {missing_count}')
df_missing['NPS_Score'] = df_missing['NPS_Score'].fillna(df['NPS_Score'].mean())
print('Filled missing scores with mean NPS.')
def mistake_nps_group(score):
if score > 8:
return 'Promoter'
elif 6 < score <= 8:
return 'Passive'
else:
return 'Detractor'
df['Wrong_NPS_Type'] = df['NPS_Score'].apply(mistake_nps_group)
error_count = (df['NPS_Type'] != df['Wrong_NPS_Type']).sum()
print(f'Number of records affected by grouping error: {error_count}')
region_sum = df.groupby('Region')['NPS_Score'].sum()
region_mean = df.groupby('Region')['NPS_Score'].mean()
print('Total NPS by Region (should use mean for insight):')
print(region_sum)
print('Correct: Mean NPS by Region:')
print(region_mean)
crosstab = pd.crosstab(df['Age_Group'], df['NPS_Type'], normalize='index') * 100
print('NPS Distribution by Age Group (%)')
print(crosstab.round(1))
index_mean = df_satisfaction['Brand_Perception_Index'].mean()
index_trend = df_satisfaction.groupby('SeniorCitizen')['Brand_Perception_Index'].mean()
print(f'Mean Brand Perception Index: {index_mean:.2f}')
print('Index by SeniorCitizen:')
print(index_trend.round(2))
# Calculate mean NPS and highest sentiment from all inputs
overall_nps = df['NPS_Score'].mean()
sentiment_counts = df_feedback['Sentiment'].value_counts()
if sentiment_counts.get('Negative',0) > sentiment_counts.get('Positive',0):
rec = 'Address complaints in support and delivery for brand improvement.'
else:
rec = 'Maintain current service strengths while promoting positive attributes.'
print(f'Average NPS: {overall_nps:.2f}')
print(f'Feedback Sentiment Breakdown: {dict(sentiment_counts)}')
print(f'Recommendation: {rec}')
Found this useful?
All lessons, notebooks and datasets here are free. If they helped you, a coffee keeps new lessons coming.



