Descriptive Statistics in Python: A Practical Analytical Guide

In our previous post, we covered the theory behind descriptive statistics — the formulas, notation, and intuition. Here, we put that theory into practice, computing each measure in Python using scipy.stats, with support from numpy and pandas.

What it covers:

  1. Numeric data — mean, median, mode, range, variance, std. deviation, IQR, CV, quartiles, skewness, kurtosis
  2. Categorical data — frequency, relative/percentage frequency, cumulative frequency
  3. Bivariate relationships — covariance, Pearson and Spearman correlation, Contigency table

import numpy as np import pandas as pd from scipy import stats # ===================================================================== # CONFIG — every variable you might want to change lives here # ===================================================================== # — Data source ————————————————— DATA_PATH = “/dataset.csv” # path to the CSV to analyze DATE_COL = “date_col” # if you have date in your column use it otherwise drop it. # — Column roles ————————————————— # Numeric / continuous columns NUMERIC_COLS = “num_col” # Categorical / coded columns CATEGORICAL_COLS = “cat_col” # — Statistical parameters ——————————————- PERCENTILE_PROBS = [0.10,0.25,0.50,0.75] # percentile positions IQR_OUTLIER_MULTIPLIER = 1.5 # multiplier k in [Q1 – k*IQR, Q3 + k*IQR] SAMPLE_DDOF = 1 # ddof for sample variance/std (n – 1) POPULATION_DDOF = 0 # ddof for population variance/std (N) # — Display / formatting ——————————————— ROUND_DECIMALS = 2 # decimal places for most printed tables # Categorical pairs to cross-tabulate # Leave as None to auto-generate all unique pairs from CATEGORICAL_COLS CATEGORICAL_PAIRS = None # or e.g.[(“season”, “weathersit”), (“yr”, “workingday”)]

import numpy as np
import pandas as pd
from scipy import stats

# =====================================================================
# CONFIG  --  every variable you might want to change lives here
# =====================================================================

# --- Data source ---------------------------------------------------
DATA_PATH = "/dataset.csv"   # path to the CSV to analyze
DATE_COL = "date_col"        # if you have date in your column use it otherwise drop it.

# --- Column roles ---------------------------------------------------
# Numeric / continuous columns 
NUMERIC_COLS = "num_col"

# Categorical / coded columns 
CATEGORICAL_COLS = "cat_col"

# --- Statistical parameters -------------------------------------------
PERCENTILE_PROBS = [0.10,0.25,0.50,0.75]  # percentile positions

IQR_OUTLIER_MULTIPLIER = 1.5   # multiplier k in [Q1 - k*IQR, Q3 + k*IQR]

SAMPLE_DDOF = 1                 # ddof for sample variance/std (n - 1)
POPULATION_DDOF = 0             # ddof for population variance/std (N)

# --- Display / formatting ---------------------------------------------
ROUND_DECIMALS = 2               # decimal places for most printed tables

# Categorical pairs to cross-tabulate 
# Leave as None to auto-generate all unique pairs from CATEGORICAL_COLS
CATEGORICAL_PAIRS =  None   # or e.g.[("season", "weathersit"), ("yr", "workingday")]
Show code

# ============================================================================ # END CONFIG # ============================================================================ pd.set_option(“display.width”, 120) pd.set_option(“display.max_columns”, 20) RULE = “=” * 78 def coeff_variation(s): return (s.std(ddof=SAMPLE_DDOF) / s.mean()) * 100 if s.mean() != 0 else np.nan def cramers_v(confusion_matrix): “””Cramér’s V: 0 = no association, 1 = perfect association.””” chi2 = stats.chi2_contingency(confusion_matrix)[0] n = confusion_matrix.sum().sum() r, k = confusion_matrix.shape return np.sqrt((chi2 / n) / (min(r – 1, k – 1))) # ————————————————————————– # 0. LOAD DATA + METADATA # ————————————————————————– # First, peek at the header only, so we know which columns actually exist # before deciding whether DATE_COL can be used. header_cols = pd.read_csv(DATA_PATH, nrows=0).columns.tolist() date_col_available = bool(DATE_COL) and (DATE_COL in header_cols) if date_col_available: df = pd.read_csv(DATA_PATH, parse_dates=[DATE_COL]) else: df = pd.read_csv(DATA_PATH) if DATE_COL: print(f”[Note] DATE_COL ‘{DATE_COL}’ not found in the data – ” f”skipping date parsing / date-range summary.\n”) # Keep only the numeric/categorical columns that actually exist in the data, # so the script degrades gracefully if the CSV doesn’t match the config. NUMERIC_COLS = [c for c in NUMERIC_COLS if c in df.columns] CATEGORICAL_COLS = [c for c in CATEGORICAL_COLS if c in df.columns] print(RULE) print(“0. METADATA”) print(RULE) meta = pd.DataFrame({ “dtype”: df.dtypes.astype(str), “non_null_count”: df.notna().sum(), “missing_count”: df.isna().sum(), “missing_%”: (df.isna().mean() * 100).round(ROUND_DECIMALS), “n_unique”: df.nunique(), }) print(f”Shape (rows, columns): {df.shape}”) if date_col_available: print(f”Date range: {df[DATE_COL].min().date()} to {df[DATE_COL].max().date()}”) else: print(“Date range: N/A (no valid date column present)”) print(“\nColumn metadata:”) print(meta) print(f”\nNumeric / count columns treated as continuous : {NUMERIC_COLS}”) print(f”Categorical (coded) columns : {CATEGORICAL_COLS}”) # ————————————————————————– # 1. NUMERIC / COUNT DATA # ————————————————————————– if NUMERIC_COLS: print(“\n” + RULE) print(“1.1 CENTRAL TENDENCY (Mean, Median, Mode)”) print(RULE) central = pd.DataFrame({ “mean”: df[NUMERIC_COLS].mean(), “median”: df[NUMERIC_COLS].median(), “mode”: df[NUMERIC_COLS].apply(lambda s: s.mode().iloc[0]), }) print(central.round(ROUND_DECIMALS)) print(“\n” + RULE) print(“1.2 SPREAD / DISPERSION”) print(RULE) spread = pd.DataFrame({ “minimum”: df[NUMERIC_COLS].min(), “maximum”: df[NUMERIC_COLS].max(), “range”: df[NUMERIC_COLS].max() – df[NUMERIC_COLS].min(), “sample_variance”: df[NUMERIC_COLS].var(ddof=SAMPLE_DDOF), “population_variance”: df[NUMERIC_COLS].var(ddof=POPULATION_DDOF), “sample_std”: df[NUMERIC_COLS].std(ddof=SAMPLE_DDOF), “population_std”: df[NUMERIC_COLS].std(ddof=POPULATION_DDOF), “IQR”: df[NUMERIC_COLS].quantile(0.25) – df[NUMERIC_COLS].quantile(0.75), “CV_%”: df[NUMERIC_COLS].apply(coeff_variation), }) print(spread.round(ROUND_DECIMALS)) print(“\n” + RULE) print(“1.3 POSITION & SHAPE (Percentile, Skewness, Kurtosis)”) print(RULE) # Quantiles: rows = NUMERIC_COLS, columns = each probability percentiles_df = df[NUMERIC_COLS].quantile(PERCENTILE_PROBS).T percentiles_df.columns = [f”P{int(p*100)}” for p in PERCENTILE_PROBS] # e.g. P25, P50, P75 # Stats: one value per column stats_df = pd.DataFrame({ “skewness_g1”: df[NUMERIC_COLS].apply(lambda s: stats.skew(s, bias=True)), “kurtosis_raw”: df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=False)), “excess_kurtosis”: df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=True)), }) position = pd.concat([percentiles_df, stats_df], axis=1) print(position.round(ROUND_DECIMALS)) # Outlier fences from IQR rule (Q1 – k*IQR, Q3 + k*IQR) print(f”\nOutlier fences (Q1 – {IQR_OUTLIER_MULTIPLIER}*IQR, Q3 + {IQR_OUTLIER_MULTIPLIER}*IQR) and outlier counts:”) for col in NUMERIC_COLS: q1, q3 = df[col].quantile([0.25, 0.75]) iqr = q3 – q1 lower, upper = q1 – IQR_OUTLIER_MULTIPLIER * iqr, q3 + IQR_OUTLIER_MULTIPLIER * iqr n_out = ((df[col] < lower) | (df[col] > upper)).sum() print(f” {col:12s}: lower={lower:9.3f} upper={upper:9.3f} outliers={n_out}”) print(“\n” + RULE) print(“1.4 BIVARIATE (Covariance, Pearson r, Spearman rho)”) print(RULE) print(“\nSample covariance matrix:”) print(df[NUMERIC_COLS].cov().round(ROUND_DECIMALS)) print(“\nPearson correlation matrix:”) pearson_corr = df[NUMERIC_COLS].corr(method=”pearson”) print(pearson_corr.round(ROUND_DECIMALS)) print(“\nSpearman rank correlation matrix:”) spearman_corr = df[NUMERIC_COLS].corr(method=”spearman”) print(spearman_corr.round(ROUND_DECIMALS)) else: print(“\n[Note] No numeric columns available – skipping section 1 (Numeric / Count data).”) # ————————————————————————– # 2. CATEGORICAL DATA # ————————————————————————– if CATEGORICAL_COLS: print(“\n” + RULE) print(“2.1 FREQUENCY ANALYSIS (categorical / coded columns)”) print(RULE) for col in CATEGORICAL_COLS: freq = df[col].value_counts().sort_index() rel = freq / len(df) pct = rel * 100 cum_freq = freq.cumsum() cum_pct = pct.cumsum() n_levels = df[col].nunique(dropna=True) tbl = pd.DataFrame({ “abs_freq_f”: freq.round(ROUND_DECIMALS), “rel_freq_p”: rel.round(ROUND_DECIMALS), “pct_freq_%”: pct.round(ROUND_DECIMALS), “cum_freq_F”: cum_freq.round(ROUND_DECIMALS), “cum_rel_freq_%”: cum_pct.round(ROUND_DECIMALS), }) print(f”\n— {col} ({n_levels} level{‘s’ if n_levels != 1 else ”}) —“) print(tbl) # ———————————————————————- # 2.2 BIVARIATE CATEGORICAL (Contingency tables, Chi-square, Cramer’s V) # ———————————————————————- print(“\n” + RULE) print(“2.2 CONTINGENCY TABLES (categorical x categorical)”) print(RULE) import itertools if CATEGORICAL_PAIRS is not None: pairs = CATEGORICAL_PAIRS elif len(CATEGORICAL_COLS) >= 2: pairs = list(itertools.combinations(CATEGORICAL_COLS, 2)) else: pairs = [] if not pairs: print(“\n[Note] Fewer than 2 categorical columns available – skipping contingency tables.”) for col1, col2 in pairs: print(f”\n— {col1} x {col2} —“) freq = pd.crosstab(df[col1], df[col2]) row_pct = pd.crosstab(df[col1], df[col2], normalize=”index”) * 100 col_pct = pd.crosstab(df[col1], df[col2], normalize=”columns”) * 100 print(“Cell format: Frequency (Row %, Col %)\n”) combined = freq.astype(str) + ” (” + row_pct.round(ROUND_DECIMALS).astype(str) + “%, ” \ + col_pct.round(ROUND_DECIMALS).astype(str) + “%)” print(combined) # Chi-square test of independence chi2, p_val, dof, expected = stats.chi2_contingency(freq) v = cramers_v(freq) print(f”\nChi-square = {chi2:.{ROUND_DECIMALS}f}, Cramer’s V = {v:.{ROUND_DECIMALS}f}”) else: print(“\n[Note] No categorical columns available – skipping section 2 (Categorical data).”) print(“\nDone.”)

# ==========================================================================
# END CONFIG
# ==========================================================================

pd.set_option("display.width", 120)
pd.set_option("display.max_columns", 20)

RULE = "=" * 78

def coeff_variation(s):
    return (s.std(ddof=SAMPLE_DDOF) / s.mean()) * 100 if s.mean() != 0 else np.nan

def cramers_v(confusion_matrix):
    """Cramér's V: 0 = no association, 1 = perfect association."""
    chi2 = stats.chi2_contingency(confusion_matrix)[0]
    n = confusion_matrix.sum().sum()
    r, k = confusion_matrix.shape
    return np.sqrt((chi2 / n) / (min(r - 1, k - 1)))

# --------------------------------------------------------------------------
# 0. LOAD DATA + METADATA
# --------------------------------------------------------------------------

# First, peek at the header only, so we know which columns actually exist
# before deciding whether DATE_COL can be used.
header_cols = pd.read_csv(DATA_PATH, nrows=0).columns.tolist()

date_col_available = bool(DATE_COL) and (DATE_COL in header_cols)

if date_col_available:
    df = pd.read_csv(DATA_PATH, parse_dates=[DATE_COL])
else:
    df = pd.read_csv(DATA_PATH)
    if DATE_COL:
        print(f"[Note] DATE_COL '{DATE_COL}' not found in the data - "
              f"skipping date parsing / date-range summary.\n")

# Keep only the numeric/categorical columns that actually exist in the data,
# so the script degrades gracefully if the CSV doesn't match the config.
NUMERIC_COLS = [c for c in NUMERIC_COLS if c in df.columns]
CATEGORICAL_COLS = [c for c in CATEGORICAL_COLS if c in df.columns]

print(RULE)
print("0. METADATA")
print(RULE)

meta = pd.DataFrame({
    "dtype": df.dtypes.astype(str),
    "non_null_count": df.notna().sum(),
    "missing_count": df.isna().sum(),
    "missing_%": (df.isna().mean() * 100).round(ROUND_DECIMALS),
    "n_unique": df.nunique(),
})
print(f"Shape (rows, columns): {df.shape}")

if date_col_available:
    print(f"Date range: {df[DATE_COL].min().date()}  to  {df[DATE_COL].max().date()}")
else:
    print("Date range: N/A (no valid date column present)")

print("\nColumn metadata:")
print(meta)

print(f"\nNumeric / count columns treated as continuous  : {NUMERIC_COLS}")
print(f"Categorical (coded) columns                    : {CATEGORICAL_COLS}")

# --------------------------------------------------------------------------
# 1. NUMERIC / COUNT DATA
# --------------------------------------------------------------------------
if NUMERIC_COLS:
    print("\n" + RULE)
    print("1.1  CENTRAL TENDENCY  (Mean, Median, Mode)")
    print(RULE)

    central = pd.DataFrame({
        "mean": df[NUMERIC_COLS].mean(),
        "median": df[NUMERIC_COLS].median(),
        "mode": df[NUMERIC_COLS].apply(lambda s: s.mode().iloc[0]),
    })
    print(central.round(ROUND_DECIMALS))

    print("\n" + RULE)
    print("1.2  SPREAD / DISPERSION")
    print(RULE)

    spread = pd.DataFrame({
        "minimum": df[NUMERIC_COLS].min(),
        "maximum": df[NUMERIC_COLS].max(),
        "range": df[NUMERIC_COLS].max() - df[NUMERIC_COLS].min(),
        "sample_variance": df[NUMERIC_COLS].var(ddof=SAMPLE_DDOF),
        "population_variance": df[NUMERIC_COLS].var(ddof=POPULATION_DDOF),
        "sample_std": df[NUMERIC_COLS].std(ddof=SAMPLE_DDOF),
        "population_std": df[NUMERIC_COLS].std(ddof=POPULATION_DDOF),
        "IQR": df[NUMERIC_COLS].quantile(0.25) - df[NUMERIC_COLS].quantile(0.75),
        "CV_%": df[NUMERIC_COLS].apply(coeff_variation),
    })
    print(spread.round(ROUND_DECIMALS))

    print("\n" + RULE)
    print("1.3  POSITION & SHAPE  (Percentile, Skewness, Kurtosis)")
    print(RULE)

    # Quantiles: rows = NUMERIC_COLS, columns = each probability
    percentiles_df = df[NUMERIC_COLS].quantile(PERCENTILE_PROBS).T
    percentiles_df.columns = [f"P{int(p*100)}" for p in PERCENTILE_PROBS]  # e.g. P25, P50, P75

    # Stats: one value per column
    stats_df = pd.DataFrame({
        "skewness_g1": df[NUMERIC_COLS].apply(lambda s: stats.skew(s, bias=True)),
        "kurtosis_raw": df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=False)),
        "excess_kurtosis": df[NUMERIC_COLS].apply(lambda s: stats.kurtosis(s, bias=True, fisher=True)),
    })

    position = pd.concat([percentiles_df, stats_df], axis=1)
    print(position.round(ROUND_DECIMALS))

    # Outlier fences from IQR rule (Q1 - k*IQR, Q3 + k*IQR)
    print(f"\nOutlier fences (Q1 - {IQR_OUTLIER_MULTIPLIER}*IQR, Q3 + {IQR_OUTLIER_MULTIPLIER}*IQR) and outlier counts:")
    for col in NUMERIC_COLS:
        q1, q3 = df[col].quantile([0.25, 0.75])
        iqr = q3 - q1
        lower, upper = q1 - IQR_OUTLIER_MULTIPLIER * iqr, q3 + IQR_OUTLIER_MULTIPLIER * iqr
        n_out = ((df[col] < lower) | (df[col] > upper)).sum()
        print(f"  {col:12s}: lower={lower:9.3f}  upper={upper:9.3f}  outliers={n_out}")

    print("\n" + RULE)
    print("1.4  BIVARIATE  (Covariance, Pearson r, Spearman rho)")
    print(RULE)

    print("\nSample covariance matrix:")
    print(df[NUMERIC_COLS].cov().round(ROUND_DECIMALS))

    print("\nPearson correlation matrix:")
    pearson_corr = df[NUMERIC_COLS].corr(method="pearson")
    print(pearson_corr.round(ROUND_DECIMALS))

    print("\nSpearman rank correlation matrix:")
    spearman_corr = df[NUMERIC_COLS].corr(method="spearman")
    print(spearman_corr.round(ROUND_DECIMALS))
else:
    print("\n[Note] No numeric columns available - skipping section 1 (Numeric / Count data).")

# --------------------------------------------------------------------------
# 2. CATEGORICAL DATA
# --------------------------------------------------------------------------
if CATEGORICAL_COLS:
    print("\n" + RULE)
    print("2.1  FREQUENCY ANALYSIS (categorical / coded columns)")
    print(RULE)

    for col in CATEGORICAL_COLS:
        freq = df[col].value_counts().sort_index()
        rel = freq / len(df)
        pct = rel * 100
        cum_freq = freq.cumsum()
        cum_pct = pct.cumsum()
        n_levels = df[col].nunique(dropna=True)
        tbl = pd.DataFrame({
            "abs_freq_f": freq.round(ROUND_DECIMALS),
            "rel_freq_p": rel.round(ROUND_DECIMALS),
            "pct_freq_%": pct.round(ROUND_DECIMALS),
            "cum_freq_F": cum_freq.round(ROUND_DECIMALS),
            "cum_rel_freq_%": cum_pct.round(ROUND_DECIMALS),
        })
        print(f"\n--- {col} ({n_levels} level{'s' if n_levels != 1 else ''}) ---")
        print(tbl)

    # ----------------------------------------------------------------------
    # 2.2  BIVARIATE CATEGORICAL  (Contingency tables, Chi-square, Cramer's V)
    # ----------------------------------------------------------------------
    print("\n" + RULE)
    print("2.2  CONTINGENCY TABLES (categorical x categorical)")
    print(RULE)

    import itertools

    if CATEGORICAL_PAIRS is not None:
        pairs = CATEGORICAL_PAIRS
    elif len(CATEGORICAL_COLS) >= 2:
        pairs = list(itertools.combinations(CATEGORICAL_COLS, 2))
    else:
        pairs = []

    if not pairs:
        print("\n[Note] Fewer than 2 categorical columns available - skipping contingency tables.")

    for col1, col2 in pairs:
        print(f"\n--- {col1} x {col2} ---")

        freq = pd.crosstab(df[col1], df[col2])
        row_pct = pd.crosstab(df[col1], df[col2], normalize="index") * 100
        col_pct = pd.crosstab(df[col1], df[col2], normalize="columns") * 100

        print("Cell format: Frequency (Row %, Col %)\n")
        combined = freq.astype(str) + " (" + row_pct.round(ROUND_DECIMALS).astype(str) + "%, " \
                   + col_pct.round(ROUND_DECIMALS).astype(str) + "%)"
        print(combined)

        # Chi-square test of independence
        chi2, p_val, dof, expected = stats.chi2_contingency(freq)
        v = cramers_v(freq)
        print(f"\nChi-square = {chi2:.{ROUND_DECIMALS}f}, Cramer's V = {v:.{ROUND_DECIMALS}f}")
        
else:
    print("\n[Note] No categorical columns available - skipping section 2 (Categorical data).")

print("\nDone.")

The following contents are available in the folder –

  1. Dataset
  2. The above Python code
  3. Numerical Analysis
  4. Visual analysis

Scroll to Top