Introduction
In our previous post, we covered the theory behind descriptive statistics — the formulas, notation, and intuition. Here, we put that theory into practice, computing each measure in Python using scipy.stats, with support from numpy and pandas.
What it covers:
- Numeric data — mean, median, mode, range, variance, std. deviation, IQR, CV, quartiles, skewness, kurtosis
- Categorical data — frequency, relative/percentage frequency, cumulative frequency
- Bivariate relationships — covariance, Pearson and Spearman correlation, Contigency table
Python code:
Variables & parameters declaration:
import numpy as np import pandas as pd from scipy import stats # ===================================================================== # CONFIG — every variable you might want to change lives here # ===================================================================== # — Data source ————————————————— DATA_PATH = “/dataset.csv” # path to the CSV to analyze DATE_COL = “date_col” # if you have date in your column use it otherwise drop it. # — Column roles ————————————————— # Numeric / continuous columns NUMERIC_COLS = “num_col” # Categorical / coded columns CATEGORICAL_COLS = “cat_col” # — Statistical parameters ——————————————- PERCENTILE_PROBS = [0.10,0.25,0.50,0.75] # percentile positions IQR_OUTLIER_MULTIPLIER = 1.5 # multiplier k in [Q1 – k*IQR, Q3 + k*IQR] SAMPLE_DDOF = 1 # ddof for sample variance/std (n – 1) POPULATION_DDOF = 0 # ddof for population variance/std (N) # — Display / formatting ——————————————— ROUND_DECIMALS = 2 # decimal places for most printed tables # Categorical pairs to cross-tabulate # Leave as None to auto-generate all unique pairs from CATEGORICAL_COLS CATEGORICAL_PAIRS = None # or e.g.[(“season”, “weathersit”), (“yr”, “workingday”)]
import numpy as np
import pandas as pd
from scipy import stats
# =====================================================================
# CONFIG -- every variable you might want to change lives here
# =====================================================================
# --- Data source ---------------------------------------------------
DATA_PATH = "/dataset.csv" # path to the CSV to analyze
DATE_COL = "date_col" # if you have date in your column use it otherwise drop it.
# --- Column roles ---------------------------------------------------
# Numeric / continuous columns
NUMERIC_COLS = "num_col"
# Categorical / coded columns
CATEGORICAL_COLS = "cat_col"
# --- Statistical parameters -------------------------------------------
PERCENTILE_PROBS = [0.10,0.25,0.50,0.75] # percentile positions
IQR_OUTLIER_MULTIPLIER = 1.5 # multiplier k in [Q1 - k*IQR, Q3 + k*IQR]
SAMPLE_DDOF = 1 # ddof for sample variance/std (n - 1)
POPULATION_DDOF = 0 # ddof for population variance/std (N)
# --- Display / formatting ---------------------------------------------
ROUND_DECIMALS = 2 # decimal places for most printed tables
# Categorical pairs to cross-tabulate
# Leave as None to auto-generate all unique pairs from CATEGORICAL_COLS
CATEGORICAL_PAIRS = None # or e.g.[("season", "weathersit"), ("yr", "workingday")]
Code block:
Show code
# ============================================================================ # END CONFIG # ============================================================================ pd.set_option(“display.width”, 120) pd.set_option(“display.max_columns”, 20) RULE = “=” * 78 def coeff_variation(s): return (s.std(ddof=SAMPLE_DDOF) / s.mean()) * 100 if s.mean() != 0 else np.nan def cramers_v(confusion_matrix): “””Cramér’s V: 0 = no association, 1 = perfect association.””” chi2 = stats.chi2_contingency(confusion_matrix)[0] n = confusion_matrix.sum().sum() r, k = confusion_matrix.shape return np.sqrt((chi2 / n) / (min(r – 1, k – 1))) # ————————————————————————– # 0. LOAD DATA + METADATA # ————————————————————————– # First, peek at the header only, so we know which columns actually exist # before deciding whether DATE_COL can be used. header_cols = pd.read_csv(DATA_PATH, nrows=0).columns.tolist() date_col_available = bool(DATE_COL) and (DATE_COL in header_cols) if date_col_available: df = pd.read_csv(DATA_PATH, parse_dates=[DATE_COL]) else: df = pd.read_csv(DATA_PATH) if DATE_COL: print(f”[Note] DATE_COL ‘{DATE_COL}’ not found in the data – ” f”skipping date parsing / date-range summary.\n”) # Keep only the numeric/categorical columns that actually exist in the data, # so the script degrades gracefully if the CSV doesn’t match the config. NUMERIC_COLS = [c for c in NUMERIC_COLS if c in df.columns] CATEGORICAL_COLS = [c for c in CATEGORICAL_COLS if c in df.columns] print(RULE) print(“0. METADATA”) print(RULE) meta = pd.DataFrame({ “dtype”: df.dtypes.astype(str), “non_null_count”: df.notna().sum(), “missing_count”: df.isna().sum(), “missing_%”: (df.isna().mean() * 100).round(ROUND_DECIMALS), “n_unique”: df.nunique(), }) print(f”Shape (rows, columns): {df.shape}”) if date_col_available: print(f”Date range: {df[DATE_COL].min().date()} to {df[DATE_COL].max().date()}”) else: print(“Date range: N/A (no valid date column present)”) print(“\nColumn metadata:”) print(meta) print(f”\nNumeric / count columns treated as continuous : {NUMERIC_COLS}”) print(f”Categorical (coded) columns : {CATEGORICAL_COLS}”) # ————————————————————————– # 1. NUMERIC / COUNT DATA # ————————————————————————– if NUMERIC_COLS: print(“\n” + RULE) print(“1.1 CENTRAL TENDENCY (Mean, Median, Mode)”) print(RULE) central = pd.DataFrame({ “mean”: df[NUMERIC_COLS].mean(), “median”: df[NUMERIC_COLS].median(), “mode”: df[NUMERIC_COLS].apply(lambda s: s.mode().iloc[0]), }) print(central.round(ROUND_DECIMALS)) print(“\n” + RULE) print(“1.2 SPREAD / DISPERSION”) print(RULE) spread = pd.DataFrame({ “minimum”: df[NUMERIC_COLS].min(), “maximum”: df[NUMERIC_COLS].max(), “range”: df[NUMERIC_COLS].max() – df[NUMERIC_COLS].min(), “sample_variance”: df[NUMERIC_COLS].var(ddof=SAMPLE_DDOF), “population_variance”: df[NUMERIC_COLS].var(ddof=POPULATION_DDOF), “sample_std”: df[NUMERIC_COLS].std(ddof=SAMPLE_DDOF), “population_std”: df[NUMERIC_COLS].std(ddof=POPULATION_DDOF), “IQR”: df[NUMERIC_COLS].quantile(0.25) – df[NUMERIC_COLS].quantile(0.75), “CV_%”: df[NUMERIC_COLS].apply(coeff_variation), }) print(spread.round(ROUND_DECIMALS)) print(“\n” + RULE) print(“1.3 POSITION & SHAPE (Percentile, Skewness, Kurtosis)”) print(RULE) # Quantiles: rows = NUMERIC_COLS, columns = each probability percentiles_df = df[NUMERIC_COLS].quantile(PERCENTILE_PROBS).T percentiles_df.columns = [f”P{int(p*100)}” for p in PERCENTILE_PROBS] # e.g. P25, P50, P75 # Stats: one value per column stats_df = pd.DataFrame({ “skewness_g1”: df[NUMERIC_COLS].apply(lambda s: stats.skew(s, bias=True)), “kurtosis_raw”: df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=False)), “excess_kurtosis”: df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=True)), }) position = pd.concat([percentiles_df, stats_df], axis=1) print(position.round(ROUND_DECIMALS)) # Outlier fences from IQR rule (Q1 – k*IQR, Q3 + k*IQR) print(f”\nOutlier fences (Q1 – {IQR_OUTLIER_MULTIPLIER}*IQR, Q3 + {IQR_OUTLIER_MULTIPLIER}*IQR) and outlier counts:”) for col in NUMERIC_COLS: q1, q3 = df[col].quantile([0.25, 0.75]) iqr = q3 – q1 lower, upper = q1 – IQR_OUTLIER_MULTIPLIER * iqr, q3 + IQR_OUTLIER_MULTIPLIER * iqr n_out = ((df[col] < lower) | (df[col] > upper)).sum() print(f” {col:12s}: lower={lower:9.3f} upper={upper:9.3f} outliers={n_out}”) print(“\n” + RULE) print(“1.4 BIVARIATE (Covariance, Pearson r, Spearman rho)”) print(RULE) print(“\nSample covariance matrix:”) print(df[NUMERIC_COLS].cov().round(ROUND_DECIMALS)) print(“\nPearson correlation matrix:”) pearson_corr = df[NUMERIC_COLS].corr(method=”pearson”) print(pearson_corr.round(ROUND_DECIMALS)) print(“\nSpearman rank correlation matrix:”) spearman_corr = df[NUMERIC_COLS].corr(method=”spearman”) print(spearman_corr.round(ROUND_DECIMALS)) else: print(“\n[Note] No numeric columns available – skipping section 1 (Numeric / Count data).”) # ————————————————————————– # 2. CATEGORICAL DATA # ————————————————————————– if CATEGORICAL_COLS: print(“\n” + RULE) print(“2.1 FREQUENCY ANALYSIS (categorical / coded columns)”) print(RULE) for col in CATEGORICAL_COLS: freq = df[col].value_counts().sort_index() rel = freq / len(df) pct = rel * 100 cum_freq = freq.cumsum() cum_pct = pct.cumsum() n_levels = df[col].nunique(dropna=True) tbl = pd.DataFrame({ “abs_freq_f”: freq.round(ROUND_DECIMALS), “rel_freq_p”: rel.round(ROUND_DECIMALS), “pct_freq_%”: pct.round(ROUND_DECIMALS), “cum_freq_F”: cum_freq.round(ROUND_DECIMALS), “cum_rel_freq_%”: cum_pct.round(ROUND_DECIMALS), }) print(f”\n— {col} ({n_levels} level{‘s’ if n_levels != 1 else ”}) —“) print(tbl) # ———————————————————————- # 2.2 BIVARIATE CATEGORICAL (Contingency tables, Chi-square, Cramer’s V) # ———————————————————————- print(“\n” + RULE) print(“2.2 CONTINGENCY TABLES (categorical x categorical)”) print(RULE) import itertools if CATEGORICAL_PAIRS is not None: pairs = CATEGORICAL_PAIRS elif len(CATEGORICAL_COLS) >= 2: pairs = list(itertools.combinations(CATEGORICAL_COLS, 2)) else: pairs = [] if not pairs: print(“\n[Note] Fewer than 2 categorical columns available – skipping contingency tables.”) for col1, col2 in pairs: print(f”\n— {col1} x {col2} —“) freq = pd.crosstab(df[col1], df[col2]) row_pct = pd.crosstab(df[col1], df[col2], normalize=”index”) * 100 col_pct = pd.crosstab(df[col1], df[col2], normalize=”columns”) * 100 print(“Cell format: Frequency (Row %, Col %)\n”) combined = freq.astype(str) + ” (” + row_pct.round(ROUND_DECIMALS).astype(str) + “%, ” \ + col_pct.round(ROUND_DECIMALS).astype(str) + “%)” print(combined) # Chi-square test of independence chi2, p_val, dof, expected = stats.chi2_contingency(freq) v = cramers_v(freq) print(f”\nChi-square = {chi2:.{ROUND_DECIMALS}f}, Cramer’s V = {v:.{ROUND_DECIMALS}f}”) else: print(“\n[Note] No categorical columns available – skipping section 2 (Categorical data).”) print(“\nDone.”)
# ==========================================================================
# END CONFIG
# ==========================================================================
pd.set_option("display.width", 120)
pd.set_option("display.max_columns", 20)
RULE = "=" * 78
def coeff_variation(s):
return (s.std(ddof=SAMPLE_DDOF) / s.mean()) * 100 if s.mean() != 0 else np.nan
def cramers_v(confusion_matrix):
"""Cramér's V: 0 = no association, 1 = perfect association."""
chi2 = stats.chi2_contingency(confusion_matrix)[0]
n = confusion_matrix.sum().sum()
r, k = confusion_matrix.shape
return np.sqrt((chi2 / n) / (min(r - 1, k - 1)))
# --------------------------------------------------------------------------
# 0. LOAD DATA + METADATA
# --------------------------------------------------------------------------
# First, peek at the header only, so we know which columns actually exist
# before deciding whether DATE_COL can be used.
header_cols = pd.read_csv(DATA_PATH, nrows=0).columns.tolist()
date_col_available = bool(DATE_COL) and (DATE_COL in header_cols)
if date_col_available:
df = pd.read_csv(DATA_PATH, parse_dates=[DATE_COL])
else:
df = pd.read_csv(DATA_PATH)
if DATE_COL:
print(f"[Note] DATE_COL '{DATE_COL}' not found in the data - "
f"skipping date parsing / date-range summary.\n")
# Keep only the numeric/categorical columns that actually exist in the data,
# so the script degrades gracefully if the CSV doesn't match the config.
NUMERIC_COLS = [c for c in NUMERIC_COLS if c in df.columns]
CATEGORICAL_COLS = [c for c in CATEGORICAL_COLS if c in df.columns]
print(RULE)
print("0. METADATA")
print(RULE)
meta = pd.DataFrame({
"dtype": df.dtypes.astype(str),
"non_null_count": df.notna().sum(),
"missing_count": df.isna().sum(),
"missing_%": (df.isna().mean() * 100).round(ROUND_DECIMALS),
"n_unique": df.nunique(),
})
print(f"Shape (rows, columns): {df.shape}")
if date_col_available:
print(f"Date range: {df[DATE_COL].min().date()} to {df[DATE_COL].max().date()}")
else:
print("Date range: N/A (no valid date column present)")
print("\nColumn metadata:")
print(meta)
print(f"\nNumeric / count columns treated as continuous : {NUMERIC_COLS}")
print(f"Categorical (coded) columns : {CATEGORICAL_COLS}")
# --------------------------------------------------------------------------
# 1. NUMERIC / COUNT DATA
# --------------------------------------------------------------------------
if NUMERIC_COLS:
print("\n" + RULE)
print("1.1 CENTRAL TENDENCY (Mean, Median, Mode)")
print(RULE)
central = pd.DataFrame({
"mean": df[NUMERIC_COLS].mean(),
"median": df[NUMERIC_COLS].median(),
"mode": df[NUMERIC_COLS].apply(lambda s: s.mode().iloc[0]),
})
print(central.round(ROUND_DECIMALS))
print("\n" + RULE)
print("1.2 SPREAD / DISPERSION")
print(RULE)
spread = pd.DataFrame({
"minimum": df[NUMERIC_COLS].min(),
"maximum": df[NUMERIC_COLS].max(),
"range": df[NUMERIC_COLS].max() - df[NUMERIC_COLS].min(),
"sample_variance": df[NUMERIC_COLS].var(ddof=SAMPLE_DDOF),
"population_variance": df[NUMERIC_COLS].var(ddof=POPULATION_DDOF),
"sample_std": df[NUMERIC_COLS].std(ddof=SAMPLE_DDOF),
"population_std": df[NUMERIC_COLS].std(ddof=POPULATION_DDOF),
"IQR": df[NUMERIC_COLS].quantile(0.25) - df[NUMERIC_COLS].quantile(0.75),
"CV_%": df[NUMERIC_COLS].apply(coeff_variation),
})
print(spread.round(ROUND_DECIMALS))
print("\n" + RULE)
print("1.3 POSITION & SHAPE (Percentile, Skewness, Kurtosis)")
print(RULE)
# Quantiles: rows = NUMERIC_COLS, columns = each probability
percentiles_df = df[NUMERIC_COLS].quantile(PERCENTILE_PROBS).T
percentiles_df.columns = [f"P{int(p*100)}" for p in PERCENTILE_PROBS] # e.g. P25, P50, P75
# Stats: one value per column
stats_df = pd.DataFrame({
"skewness_g1": df[NUMERIC_COLS].apply(lambda s: stats.skew(s, bias=True)),
"kurtosis_raw": df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=False)),
"excess_kurtosis": df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=True)),
})
position = pd.concat([percentiles_df, stats_df], axis=1)
print(position.round(ROUND_DECIMALS))
# Outlier fences from IQR rule (Q1 - k*IQR, Q3 + k*IQR)
print(f"\nOutlier fences (Q1 - {IQR_OUTLIER_MULTIPLIER}*IQR, Q3 + {IQR_OUTLIER_MULTIPLIER}*IQR) and outlier counts:")
for col in NUMERIC_COLS:
q1, q3 = df[col].quantile([0.25, 0.75])
iqr = q3 - q1
lower, upper = q1 - IQR_OUTLIER_MULTIPLIER * iqr, q3 + IQR_OUTLIER_MULTIPLIER * iqr
n_out = ((df[col] < lower) | (df[col] > upper)).sum()
print(f" {col:12s}: lower={lower:9.3f} upper={upper:9.3f} outliers={n_out}")
print("\n" + RULE)
print("1.4 BIVARIATE (Covariance, Pearson r, Spearman rho)")
print(RULE)
print("\nSample covariance matrix:")
print(df[NUMERIC_COLS].cov().round(ROUND_DECIMALS))
print("\nPearson correlation matrix:")
pearson_corr = df[NUMERIC_COLS].corr(method="pearson")
print(pearson_corr.round(ROUND_DECIMALS))
print("\nSpearman rank correlation matrix:")
spearman_corr = df[NUMERIC_COLS].corr(method="spearman")
print(spearman_corr.round(ROUND_DECIMALS))
else:
print("\n[Note] No numeric columns available - skipping section 1 (Numeric / Count data).")
# --------------------------------------------------------------------------
# 2. CATEGORICAL DATA
# --------------------------------------------------------------------------
if CATEGORICAL_COLS:
print("\n" + RULE)
print("2.1 FREQUENCY ANALYSIS (categorical / coded columns)")
print(RULE)
for col in CATEGORICAL_COLS:
freq = df[col].value_counts().sort_index()
rel = freq / len(df)
pct = rel * 100
cum_freq = freq.cumsum()
cum_pct = pct.cumsum()
n_levels = df[col].nunique(dropna=True)
tbl = pd.DataFrame({
"abs_freq_f": freq.round(ROUND_DECIMALS),
"rel_freq_p": rel.round(ROUND_DECIMALS),
"pct_freq_%": pct.round(ROUND_DECIMALS),
"cum_freq_F": cum_freq.round(ROUND_DECIMALS),
"cum_rel_freq_%": cum_pct.round(ROUND_DECIMALS),
})
print(f"\n--- {col} ({n_levels} level{'s' if n_levels != 1 else ''}) ---")
print(tbl)
# ----------------------------------------------------------------------
# 2.2 BIVARIATE CATEGORICAL (Contingency tables, Chi-square, Cramer's V)
# ----------------------------------------------------------------------
print("\n" + RULE)
print("2.2 CONTINGENCY TABLES (categorical x categorical)")
print(RULE)
import itertools
if CATEGORICAL_PAIRS is not None:
pairs = CATEGORICAL_PAIRS
elif len(CATEGORICAL_COLS) >= 2:
pairs = list(itertools.combinations(CATEGORICAL_COLS, 2))
else:
pairs = []
if not pairs:
print("\n[Note] Fewer than 2 categorical columns available - skipping contingency tables.")
for col1, col2 in pairs:
print(f"\n--- {col1} x {col2} ---")
freq = pd.crosstab(df[col1], df[col2])
row_pct = pd.crosstab(df[col1], df[col2], normalize="index") * 100
col_pct = pd.crosstab(df[col1], df[col2], normalize="columns") * 100
print("Cell format: Frequency (Row %, Col %)\n")
combined = freq.astype(str) + " (" + row_pct.round(ROUND_DECIMALS).astype(str) + "%, " \
+ col_pct.round(ROUND_DECIMALS).astype(str) + "%)"
print(combined)
# Chi-square test of independence
chi2, p_val, dof, expected = stats.chi2_contingency(freq)
v = cramers_v(freq)
print(f"\nChi-square = {chi2:.{ROUND_DECIMALS}f}, Cramer's V = {v:.{ROUND_DECIMALS}f}")
else:
print("\n[Note] No categorical columns available - skipping section 2 (Categorical data).")
print("\nDone.")
Usage:
The following contents are available in the folder –
- Dataset
- The above Python code
- Numerical Analysis
- Visual analysis