Evaluating K-Means Initialization Strategies on the Scikit-Learn Digits Dataset

import numpy as np
from sklearn.datasets import load_digits
from sklearn.cluster import KMeans
from sklearn.decomposition import PCA
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import make_pipeline
from sklearn import metrics
import time

# Load feature matrix and ground truth labels
digit_features, true_labels = load_digits(return_X_y=True)
num_samples, num_attributes = digit_features.shape
unique_clusters = len(np.unique(true_labels))

def evaluate_clustering(model_instance, method_label, features, targets):
    """Run model fitting and compute statistical quality scores."""
    start_timestamp = time.perf_counter()
    
    # Wrap scaler and classifier in a sequential workflow
    processed_model = make_pipeline(StandardScaler(), model_instance)
    processed_model.fit(features)
    
    elapsed_seconds = time.perf_counter() - start_timestamp
    inertia_val = processed_model[-1].inertia_
    predicted_labels = processed_model[-1].labels_

    # Compile performance statistics
    stats_report = {
        "method": method_label,
        "fit_duration_s": elapsed_seconds,
        "inertia": inertia_val,
        "homogeneity": metrics.homogeneity_score(targets, predicted_labels),
        "completeness": metrics.completeness_score(targets, predicted_labels),
        "v_measure": metrics.v_measure_score(targets, predicted_labels),
        "adjusted_rand": metrics.adjusted_rand_score(targets, predicted_labels),
        "mutual_info": metrics.adjusted_mutual_info_score(targets, predicted_labels),
        "silhouette": metrics.silhouette_score(features, predicted_labels, sample_size=300, metric="euclidean")
    }
    return stats_report
# Define initialization configurations to test
config_set = [
    {"init": "k-means++", "clusters": unique_clusters, "init_iter": 4, "seed": 42, "label": "k-means++"},
    {"init": "random", "clusters": unique_clusters, "init_iter": 4, "seed": 42, "label": "uniform-random"},
    {"init": "k-means++", "clusters": unique_clusters, "init_iter": 1, "seed": None, "label": "PCA-derived"}
]

print(f'{"Strategy":<15} | {"Time(s)":<8} | {"Inertia":<10} | {"Homo":<6} | {"Comp":<6} | {"V-Meas":<7} | {"ARI":<7} | {"AMI":<7} | {"Sil":<7}')
print("-" * 90)

results_table = []

# Execute first two standard initializations
for cfg in config_set[:2]:
    estimator = KMeans(**{k:v for k,v in cfg.items() if k != 'label'}, random_state=cfg.get('seed'))
    report = evaluate_clustering(estimator, cfg['label'], digit_features, true_labels)
    results_table.append(report)

# Execute PCA-based initialization
pca_initial_centers = PCA(n_components=unique_clusters).fit(digit_features).components_
pca_estimator = KMeans(init=pca_initial_centers, n_clusters=unique_clusters, n_init=1)
pca_report = evaluate_clustering(pca_estimator, "PCA-derived", digit_features, true_labels)
results_table.append(pca_report)

# Print formatted output
for r in results_table:
    print(f"{r['method']:<15} | {r['fit_duration_s']:<8.3f} | {r['inertia']:<10.0f} | {r['homogeneity']:<6.3f} | {r['completeness']:<6.3f} | {r['v_measure']:<7.3f} | {r['adjusted_rand']:<7.3f} | {r['mutual_info']:<7.3f} | {r['silhouette']:<7.3f}")
import matplotlib.pyplot as plt

# Reduce dimensionality for spatial representation
latent_space = PCA(n_components=2).fit_transform(digit_features)

# Train final clustering model on lowered dimensions
space_model = KMeans(init="k-means++", n_clusters=unique_clusters, n_init=4, random_state=10)
space_model.fit(latent_space)

# Configure decision boundary grid
step_size = 0.02
x_range = np.arange(latent_space[:, 0].min() - 1, latent_space[:, 0].max() + 1, step_size)
y_range = np.arange(latent_space[:, 1].min() - 1, latent_space[:, 1].max() + 1, step_size)
coord_grid = np.meshgrid(x_range, y_range)
flat_coords = np.c_[coord_grid[0].ravel(), coord_grid[1].ravel()]

# Predict clusters across the grid
grid_predictions = space_model.predict(flat_coords).reshape(coord_grid[0].shape)

fig, ax = plt.subplots(figsize=(10, 8))
ax.contourf(coord_grid[0], coord_grid[1], grid_predictions, cmap=plt.cm.Paired, alpha=0.6)
ax.scatter(latent_space[:, 0], latent_space[:, 1], c=true_labels, cmap=plt.cm.nipy_spectral, s=5, edgecolors='k')
ax.scatter(space_model.cluster_centers_[:, 0], space_model.cluster_centers_[:, 1], 
           marker='X', s=100, linewidths=2, color='white', label='Centroids')
ax.set_title("Decision Regions for Digit Clustering (PCA Latent Space)")
ax.legend()
plt.show()

Evaluation Metrics Framework

Clustering quality is quantified using standardized validation indices that compare algorithm assignments against known ground truth:

  • Homogeneity Score: Measures whether each cluster contains only members of a single class. Values range from 0 to 1, where 1 indicates perfect separation.
  • Completeness Score: Assesses whether all members of a given class are assigned to the same cluster. Ranges from 0 to 1.
  • V-Measure: Provides the harmonic mean of homogeneity and completeness, balancing both aspects equally.
  • Adjusted Rand Index (ARI): Evaluates pairwise similarity between true labels and predicted clusters, corrected for random chance. Values span roughly -1 to 1, with higher scores indicating stronger alignment.
  • Adjusted Mutual Information (AMI): Quantifies the mutual dependence between two labelings while adjusting for expected random co-occurence.
  • Silhouette Coefficient: Computes how similar each sample is to its own cluster compared to other clusters. Ranges from -1 to 1; values closer to 1 denote well-separated, cohesive groups.

All metrics operate on a normalized scale, facilitating direct comparison across different initialization tactics and dataset subsets.

Computational Efficiency Observations

Initializing centroids via principle components distributes starting centers along directions of maximum variance. This geometrically informed seeding typically requires fewer expectation-maximization iterations to reach convergence. Benchmark outputs consistently demonstrate reduced fitting durations for the PCA-derived strategy compared to uniform-random sampling, highlighting dimensionality reduction as a practical optimization for accelerating unsupervised learning pipelines without sacrificing cluster cohesion.

Tags: machine-learning Clustering k-means scikit-learn dimensionality-reduction

Posted on Mon, 28 Sep 2026 16:13:30 +0000 by daxxy