Example the LPCA and KPCA algorithms and compare with sklearn:
- TSNE
- KMeans
- PCA
- KCPA
# Import Data
# - Iris import
import pandas as pd
data_iris = pd.read_csv(filepath_or_buffer='https://archive.ics.uci.edu/ml/machine-learning-databases/iris/iris.data',
header= None,
sep= ',')
data_iris.columns=['sepal_len', 'sepal_wid', 'petal_len', 'petal_wid', 'class']
data_iris.dropna(how="all", inplace=True) # drops the empty line at file-end
data_iris.info(verbose=True)
X_iris = data_iris.iloc[:,0:4].values
y_iris = data_iris.iloc[:,4].values
print("\nShape of X_iris: {} Shape of y_iris: {} \n".format(X_iris.shape, y_iris.shape))
# - Bank data import from file
data_bank = pd.read_csv('data/bank.csv',
sep=';')
# Columns is build in file no need to add to data
data_bank.dropna(how="all", inplace=True) # drops the empty line at file-end
data_bank.info(verbose=True)
X_bank = data_bank.iloc[:,0:16].values
y_bank = data_bank.iloc[:,16].values
print("\nShape of X_bank: {} Shape of y_bank: {} \n".format(X_bank.shape, y_bank.shape))<class 'pandas.DataFrame'>
RangeIndex: 150 entries, 0 to 149
Data columns (total 5 columns):
# Column Non-Null Count Dtype
--- ------ -------------- -----
0 sepal_len 150 non-null float64
1 sepal_wid 150 non-null float64
2 petal_len 150 non-null float64
3 petal_wid 150 non-null float64
4 class 150 non-null str
dtypes: float64(4), str(1)
memory usage: 6.0 KB
Shape of X_iris: (150, 4) Shape of y_iris: (150,)
<class 'pandas.DataFrame'>
RangeIndex: 4521 entries, 0 to 4520
Data columns (total 17 columns):
# Column Non-Null Count Dtype
--- ------ -------------- -----
0 age 4521 non-null int64
1 job 4521 non-null str
2 marital 4521 non-null str
3 education 4521 non-null str
4 default 4521 non-null str
5 balance 4521 non-null int64
6 housing 4521 non-null str
7 loan 4521 non-null str
8 contact 4521 non-null str
9 day 4521 non-null int64
10 month 4521 non-null str
11 duration 4521 non-null int64
12 campaign 4521 non-null int64
13 pdays 4521 non-null int64
14 previous 4521 non-null int64
15 poutcome 4521 non-null str
16 y 4521 non-null str
dtypes: int64(7), str(10)
memory usage: 600.6 KB
Shape of X_bank: (4521, 16) Shape of y_bank: (4521,)
# Preprocesin Data
import numpy as np
from sklearn.compose import ColumnTransformer
from sklearn.preprocessing import OneHotEncoder, LabelEncoder
categorical_columns = [1, 2, 3, 4, 6, 7, 8, 10, 15]
ct = ColumnTransformer(
transformers=[
(
"encoder",
OneHotEncoder(
sparse_output=False,
handle_unknown="ignore",
dtype=np.float64,
),
categorical_columns,
)
],
remainder="passthrough",
)
X_bank = ct.fit_transform(X_bank).astype(
np.float64,
copy=False,
)
X_bank = np.float64(ct.fit_transform(X_bank))
print("\n After Preprocesing data shape is: ")
print("Shape of X_bank: {} Shape of Y_bank: {}".format(X_bank.shape, y_bank.shape)) After Preprocesing data shape is:
Shape of X_bank: (4521, 60) Shape of Y_bank: (4521,)
from sklearn.manifold import TSNE
from sklearn.cluster import KMeans
from sklearn.decomposition import PCA
from pca.pca import LPCA # OWN Implementation
from kpca.kpca import Clustering # successor
import time
_s = time.time()
X_tsne_iris = TSNE(n_components=2, n_jobs=-1).fit_transform(X_iris)
X_tsne_bank = TSNE(n_components=2, n_jobs=-1).fit_transform(X_bank)
X_KMns_iris = KMeans(n_clusters=2, random_state=0).fit_transform(X_iris)
X_KMns_bank = KMeans(n_clusters=2, random_state=0).fit_transform(X_bank)
X_pca_iris = PCA(n_components=2, random_state=0).fit_transform(X_iris)
X_pca_bank = PCA(n_components=2, random_state=0).fit_transform(X_bank)
entropy = 0.35
dt = 0.05
iris_lpca = LPCA(entropy=entropy, dt=dt, gpu=False, seed=42)
X_s_iris, m_w_iris = iris_lpca.l_pca(X_iris, C=False)
iris_lpca_c = LPCA(entropy=entropy, dt=dt, gpu=False, seed=42)
X_s_C_iris, m_w_C_iris, C_x_iris = iris_lpca_c.l_pca( X_iris, C=True)
X_s_bank, m_w_bank = LPCA(entropy, dt, gpu=False).l_pca(X_bank, C=False)
X_s_C_bank, m_w_C_bank, C_x_bank = LPCA(entropy, dt, gpu=False).l_pca(X_bank, C=True)
# Prepare for KPCA
import torch
X_iris_t = torch.as_tensor(
X_iris,
dtype=torch.float64,
)
kpca_iris = Clustering(
p_components=0.99,
n_cluster=3,
max_iter=300,
n_init=25,
)
X_kpca_iris = kpca_iris.kpca(
X_iris_t,
return_labels=False,
)
y_kpca_iris = X_kpca_iris.argmin(dim=1)
print("Labels:", y_kpca_iris.shape)
print("Distances:", X_kpca_iris.shape)
X_bank_t = torch.as_tensor(
X_bank,
dtype=torch.float64,
)
kpca_bank = Clustering(
p_components=0.99,
n_cluster=2,
max_iter=300,
n_init=25,
)
X_kpca_bank = kpca_bank.kpca(
X_bank_t,
return_labels=False,
)
y_kpca_bank = X_kpca_bank.argmin(dim=1)
print("Labels:", y_kpca_bank.shape)
print("Distances:", X_kpca_bank.shape)
_time = time.time() - _s
print("\n Finish in {}s".format(_time))Labels: torch.Size([150])
Distances: torch.Size([150, 3])
Labels: torch.Size([4521])
Distances: torch.Size([4521, 2])
Finish in 6.376402378082275s
a = [1, 2, 3, 4, 5]
print(C_x_bank.shape)torch.Size([14, 14])
import matplotlib.pyplot as plt
import numpy as np
import torch as t
def to_numpy(x) -> np.ndarray:
"""
Converts a NumPy array, CPU tensor, or CUDA tensor to a NumPy array.
"""
if isinstance(x, t.Tensor):
return x.detach().cpu().numpy()
return np.asarray(x)
def plot_print(
X,
y,
name_plot: str,
category: list[str],
colors: list[str],
) -> None:
X_np = to_numpy(X)
y_np = to_numpy(y)
if X_np.ndim != 2:
raise ValueError(
f"X must be a 2D matrix, received shape={X_np.shape}"
)
if X_np.shape[1] < 2:
raise ValueError(
f"At least 2 columns are required for the plot, "
f"received shape={X_np.shape}"
)
if X_np.shape[0] != y_np.shape[0]:
raise ValueError(
f"X and y contain different numbers of samples: "
f"X={X_np.shape[0]}, y={y_np.shape[0]}"
)
with plt.style.context("seaborn-v0_8-whitegrid"):
fig, ax = plt.subplots(figsize=(6, 4))
for label, color in zip(category, colors):
mask = y_np == label
ax.scatter(
X_np[mask, 0],
X_np[mask, 1],
label=label,
color=color,
alpha=0.75,
s=30,
)
ax.set_title(name_plot)
ax.set_xlabel("Component 1")
ax.set_ylabel("Component 2")
ax.legend()
fig.tight_layout()
plt.show()
plt.close(fig)
category_iris = [
"Iris-setosa",
"Iris-virginica",
"Iris-versicolor",
]
colors_iris = ["blue", "red", "green"]
plot_print(
X_tsne_iris,
y_iris,
"TSNE - IRIS DataSet",
category_iris,
colors_iris,
)
plot_print(
X_KMns_iris,
y_iris,
"KMeans - IRIS DataSet",
category_iris,
colors_iris,
)
plot_print(
X_pca_iris,
y_iris,
"PCA - IRIS DataSet",
category_iris,
colors_iris,
)
# LPCA without C
X_lpca_iris = X_s_iris @ m_w_iris
plot_print(
X_lpca_iris,
y_iris,
"LPCA - IRIS DataSet C=False",
category_iris,
colors_iris,
)
# LPCA with C
X_lpca_c_iris_pre = X_s_C_iris @ m_w_C_iris
print("IRIS before C:", tuple(X_lpca_c_iris_pre.shape))
print("IRIS C:", tuple(C_x_iris.shape))
X_lpca_c_iris = X_lpca_c_iris_pre @ C_x_iris
plot_print(
X_lpca_c_iris,
y_iris,
"LPCA - IRIS DataSet C=True",
category_iris,
colors_iris,
)
print("\nFor Bank Dataset\n")
category_bank = ["no", "yes"]
colors_bank = ["blue", "green"]
plot_print(
X_tsne_bank,
y_bank,
"TSNE - Bank DataSet",
category_bank,
colors_bank,
)
plot_print(
X_KMns_bank,
y_bank,
"KMeans - Bank DataSet",
category_bank,
colors_bank,
)
plot_print(
X_pca_bank,
y_bank,
"PCA - Bank DataSet",
category_bank,
colors_bank,
)
# LPCA without C
X_lpca_bank = X_s_bank @ m_w_bank
plot_print(
X_lpca_bank,
y_bank,
"LPCA - Bank DataSet C=False",
category_bank,
colors_bank,
)
# LPCA with C
X_lpca_c_bank_pre = X_s_C_bank @ m_w_C_bank
print("BANK before C:", tuple(X_lpca_c_bank_pre.shape))
print("BANK C:", tuple(C_x_bank.shape))
X_lpca_c_bank = X_lpca_c_bank_pre @ C_x_bank
plot_print(
X_lpca_c_bank,
y_bank,
"LPCA - Bank DataSet C=True",
category_bank,
colors_bank,
)
# KPCA
plot_print(
X_kpca_iris,
y_iris,
"KPCA - IRIS DataSet",
category_iris,
colors_iris,
)
plot_print(
X_kpca_bank,
y_bank,
"KPCA - Bank DataSet",
category_bank,
colors_bank,
)
plot_print(
X_kpca_iris,
y_kpca_iris,
"KPCA - IRIS Predicted Clusters",
[0, 1, 2],
colors_iris,
)IRIS before C: (150, 2)
IRIS C: (2, 2)
For Bank Dataset
BANK before C: (4521, 14)
BANK C: (14, 14)
import math
import re
import time
from pathlib import Path
import matplotlib.pyplot as plt
import numpy as np
import torch
from PIL import Image
from sklearn.cluster import KMeans
from sklearn.decomposition import PCA
from sklearn.manifold import TSNE
from sklearn.preprocessing import StandardScaler
from pca.pca import LPCA
from kpca.kpca import Clustering as KPCA
# ============================================================
# CONFIGURATION
# ============================================================
IMAGE_PATH = "image/test_image.jpg"
N_CLUSTERS = 5
RANDOM_STATE = 42
# t-SNE is computationally expensive for tens of thousands of pixels.
MAX_PROCESSING_PIXELS = 96_000
# Importance of pixel position relative to color:
#
# 0.0 -> RGB only
# 0.25 -> slight consideration of position
# 0.75 -> a reasonable starting value for objects
# 1.5 -> strong emphasis on spatial continuity
SPATIAL_WEIGHT = 2.0
# LPCA parameters
LPCA_ENTROPY = 0.35
LPCA_DT = 0.05
LPCA_MIN_TOLERANCE = 1
# KPCA parameters
KPCA_COMPONENTS = 0.99
KPCA_MAX_ITER = 300
KPCA_N_INIT = 25
USE_GPU = True
SAVE_MASKS = True
OUTPUT_DIRECTORY = Path("image_clustering_results")
# ============================================================
# UTILITIES
# ============================================================
def to_numpy(x) -> np.ndarray:
"""
Converts a CPU/CUDA tensor or another value to NumPy.
"""
if isinstance(x, torch.Tensor):
return x.detach().cpu().numpy()
return np.asarray(x)
def sanitize_filename(name: str) -> str:
"""
Creates a safe directory or file name.
"""
name = name.strip().lower()
name = re.sub(r"[^a-z0-9]+", "_", name)
return name.strip("_")
def load_rgb_image(
image_path: str | Path,
) -> Image.Image:
"""
Loads an image and enforces RGB mode.
"""
image_path = Path(image_path)
if not image_path.exists():
raise FileNotFoundError(
f"Image not found: {image_path.resolve()}"
)
return Image.open(image_path).convert("RGB")
def resize_for_processing(
image: Image.Image,
max_pixels: int,
) -> Image.Image:
"""
Resizes the image while preserving its aspect ratio if the number
of pixels exceeds max_pixels.
"""
width, height = image.size
pixel_count = width * height
if pixel_count <= max_pixels:
return image.copy()
scale = math.sqrt(max_pixels / pixel_count)
new_width = max(2, int(round(width * scale)))
new_height = max(2, int(round(height * scale)))
return image.resize(
(new_width, new_height),
Image.Resampling.LANCZOS,
)
def create_pixel_features(
image: Image.Image,
spatial_weight: float,
) -> np.ndarray:
"""
Creates pixel features:
[R, G, B, X, Y]
RGB values are standardized, while the X/Y coordinates
are normalized to the range [-1, 1].
spatial_weight controls the influence of the pixel position.
"""
image_array = np.asarray(
image,
dtype=np.float64,
) / 255.0
height, width, channels = image_array.shape
if channels != 3:
raise ValueError(
f"Expected RGB, received shape={image_array.shape}"
)
rgb = image_array.reshape(-1, 3)
rgb_scaled = StandardScaler().fit_transform(rgb)
y_grid, x_grid = np.mgrid[0:height, 0:width]
if width > 1:
x_normalized = (
2.0 * x_grid.astype(np.float64) / (width - 1)
) - 1.0
else:
x_normalized = np.zeros_like(
x_grid,
dtype=np.float64,
)
if height > 1:
y_normalized = (
2.0 * y_grid.astype(np.float64) / (height - 1)
) - 1.0
else:
y_normalized = np.zeros_like(
y_grid,
dtype=np.float64,
)
coordinates = np.column_stack(
(
x_normalized.reshape(-1),
y_normalized.reshape(-1),
)
)
coordinates *= spatial_weight
features = np.concatenate(
(rgb_scaled, coordinates),
axis=1,
)
return np.ascontiguousarray(
features,
dtype=np.float64,
)
def create_kmeans() -> KMeans:
"""
Returns an identically configured KMeans instance for all
external representations.
"""
return KMeans(
n_clusters=N_CLUSTERS,
n_init=10,
max_iter=300,
random_state=RANDOM_STATE,
)
def validate_labels(
labels,
expected_samples: int,
) -> np.ndarray:
"""
Validates and normalizes cluster labels.
"""
labels_np = to_numpy(labels).reshape(-1).astype(
np.int64,
copy=False,
)
if labels_np.shape[0] != expected_samples:
raise ValueError(
"Invalid number of labels: "
f"received {labels_np.shape[0]}, "
f"expected {expected_samples}."
)
if np.any(labels_np < 0):
raise ValueError(
"Cluster labels cannot be negative."
)
return labels_np
def relabel_clusters_by_size(
labels: np.ndarray,
n_clusters: int,
) -> np.ndarray:
"""
Orders clusters from largest to smallest.
This ensures that mask 1 in each row represents the largest
cluster. It does not affect the clustering result itself.
"""
counts = np.bincount(
labels,
minlength=n_clusters,
)
old_labels_in_new_order = np.argsort(-counts)
mapping = np.empty(
n_clusters,
dtype=np.int64,
)
for new_label, old_label in enumerate(
old_labels_in_new_order
):
mapping[old_label] = new_label
return mapping[labels]
def timed_clustering(
method_name: str,
function,
) -> tuple[np.ndarray, float]:
"""
Runs the method and measures its execution time.
"""
start = time.perf_counter()
labels = function()
elapsed = time.perf_counter() - start
labels = validate_labels(
labels,
expected_samples=PROCESSING_SAMPLE_COUNT,
)
labels = relabel_clusters_by_size(
labels,
n_clusters=N_CLUSTERS,
)
print(
f"{method_name:<24} "
f"{elapsed:>10.4f} s | "
f"clusters: {np.unique(labels).size}"
)
return labels, elapsed
# ============================================================
# CLUSTERING METHODS
# ============================================================
def cluster_raw_kmeans(
features: np.ndarray,
) -> np.ndarray:
"""
KMeans applied directly to RGB+XY.
"""
return create_kmeans().fit_predict(features)
def cluster_pca_kmeans(
features: np.ndarray,
) -> np.ndarray:
"""
PCA reduced to two dimensions, followed by KMeans.
"""
transformed = PCA(
n_components=2,
random_state=RANDOM_STATE,
).fit_transform(features)
return create_kmeans().fit_predict(transformed)
def cluster_tsne_kmeans(
features: np.ndarray,
) -> np.ndarray:
"""
t-SNE reduced to two dimensions, followed by KMeans.
"""
sample_count = features.shape[0]
if sample_count < 5:
raise ValueError(
"t-SNE requires a larger number of pixels."
)
perplexity = min(
30.0,
max(2.0, (sample_count - 1) / 20.0),
)
# Perplexity must be lower than the number of samples.
perplexity = min(
perplexity,
sample_count - 1.0,
)
transformed = TSNE(
n_components=2,
perplexity=perplexity,
init="pca",
learning_rate="auto",
max_iter=1000,
random_state=RANDOM_STATE,
n_jobs=-1,
).fit_transform(features)
return create_kmeans().fit_predict(transformed)
def cluster_lpca_kmeans(
features: np.ndarray,
use_correlation: bool,
) -> np.ndarray:
"""
LPCA or LPCA+C, followed by KMeans.
"""
model = LPCA(
entropy=LPCA_ENTROPY,
dt=LPCA_DT,
min_tolerance=LPCA_MIN_TOLERANCE,
gpu=USE_GPU,
seed=RANDOM_STATE,
)
with torch.no_grad():
if use_correlation:
x_reduced, projection, correlation = model.l_pca(
features,
C=True,
)
transformed = (
x_reduced
@ projection
@ correlation
)
else:
x_reduced, projection = model.l_pca(
features,
C=False,
)
transformed = x_reduced @ projection
transformed_np = to_numpy(transformed)
if transformed_np.ndim != 2:
raise ValueError(
"LPCA returned an invalid representation: "
f"{transformed_np.shape}"
)
if transformed_np.shape[1] == 0:
raise ValueError(
"LPCA did not preserve any components."
)
return create_kmeans().fit_predict(transformed_np)
def cluster_kpca(
features: np.ndarray,
) -> np.ndarray:
device = torch.device(
"cuda"
if USE_GPU and torch.cuda.is_available()
else "cpu"
)
x_tensor = torch.as_tensor(
features,
dtype=torch.float64,
device=device,
)
model = KPCA(
p_components=KPCA_COMPONENTS,
n_cluster=N_CLUSTERS,
max_iter=KPCA_MAX_ITER,
n_init=KPCA_N_INIT,
)
with torch.no_grad():
labels = model.kpca(
x_tensor,
return_labels=True,
)
return to_numpy(labels)
# ============================================================
# MASKS
# ============================================================
def labels_to_masks(
labels: np.ndarray,
processing_size: tuple[int, int],
original_size: tuple[int, int],
n_clusters: int,
) -> list[np.ndarray]:
"""
Converts pixel labels into separate binary masks.
processing_size and original_size use the following format:
(width, height)
"""
processing_width, processing_height = processing_size
labels_image = labels.reshape(
processing_height,
processing_width,
)
masks: list[np.ndarray] = []
for cluster_idx in range(n_clusters):
small_mask = (
labels_image == cluster_idx
).astype(np.uint8) * 255
mask_image = Image.fromarray(
small_mask,
mode="L",
)
if processing_size != original_size:
mask_image = mask_image.resize(
original_size,
Image.Resampling.NEAREST,
)
mask = np.asarray(mask_image) > 127
masks.append(mask)
return masks
def save_masks(
method_name: str,
masks: list[np.ndarray],
output_directory: Path,
) -> None:
"""
Saves the masks generated by a given method to a separate directory.
"""
method_directory = (
output_directory / sanitize_filename(method_name)
)
method_directory.mkdir(
parents=True,
exist_ok=True,
)
for index, mask in enumerate(masks, start=1):
mask_image = Image.fromarray(
mask.astype(np.uint8) * 255,
mode="L",
)
mask_image.save(
method_directory / f"mask_{index}.png"
)
def plot_original_images(
original_image: Image.Image,
processing_image: Image.Image,
) -> None:
"""
Displays the original image and the version used for processing.
"""
fig, axes = plt.subplots(
1,
2,
figsize=(10, 5),
)
axes[0].imshow(original_image)
axes[0].set_title(
f"Original: {original_image.width}×"
f"{original_image.height}"
)
axes[0].axis("off")
axes[1].imshow(processing_image)
axes[1].set_title(
f"Processing: {processing_image.width}×"
f"{processing_image.height}"
)
axes[1].axis("off")
fig.tight_layout()
plt.show()
plt.close(fig)
def plot_masks_grid(
masks_by_method: dict[str, list[np.ndarray]],
times_by_method: dict[str, float],
n_clusters: int,
) -> None:
"""
Displays one row of masks for each method.
"""
method_names = list(masks_by_method.keys())
row_count = len(method_names)
fig, axes = plt.subplots(
row_count,
n_clusters,
figsize=(
3.0 * n_clusters,
2.8 * row_count,
),
squeeze=False,
)
for row_index, method_name in enumerate(method_names):
masks = masks_by_method[method_name]
for cluster_index in range(n_clusters):
axis = axes[row_index, cluster_index]
mask = masks[cluster_index]
axis.imshow(
mask,
cmap="gray",
vmin=0,
vmax=1,
)
mask_percentage = (
100.0 * float(mask.mean())
)
axis.set_title(
f"Mask {cluster_index + 1}\n"
f"{mask_percentage:.1f}% of image"
)
axis.set_xticks([])
axis.set_yticks([])
if cluster_index == 0:
axis.set_ylabel(
f"{method_name}\n"
f"{times_by_method[method_name]:.3f} s",
rotation=0,
ha="right",
va="center",
labelpad=15,
fontsize=10,
fontweight="bold",
)
fig.suptitle(
f"Method comparison — {n_clusters} clusters",
fontsize=16,
)
fig.tight_layout(
rect=(0.02, 0.01, 1.0, 0.97)
)
plt.show()
plt.close(fig)
# ============================================================
# IMAGE LOADING AND PREPARATION
# ============================================================
original_image = load_rgb_image(IMAGE_PATH)
processing_image = resize_for_processing(
original_image,
max_pixels=MAX_PROCESSING_PIXELS,
)
print("Original image:", original_image.size)
print("Image used for processing:", processing_image.size)
plot_original_images(
original_image,
processing_image,
)
pixel_features = create_pixel_features(
processing_image,
spatial_weight=SPATIAL_WEIGHT,
)
PROCESSING_SAMPLE_COUNT = pixel_features.shape[0]
print("Number of pixels:", PROCESSING_SAMPLE_COUNT)
print("Number of features:", pixel_features.shape[1])
print()
# ============================================================
# RUNNING ALL METHODS
# ============================================================
labels_by_method: dict[str, np.ndarray] = {}
times_by_method: dict[str, float] = {}
labels_by_method["KMeans"], times_by_method["KMeans"] = (
timed_clustering(
"KMeans",
lambda: cluster_raw_kmeans(pixel_features),
)
)
labels_by_method["PCA + KMeans"], times_by_method[
"PCA + KMeans"
] = timed_clustering(
"PCA + KMeans",
lambda: cluster_pca_kmeans(pixel_features),
)
labels_by_method["t-SNE + KMeans"], times_by_method[
"t-SNE + KMeans"
] = timed_clustering(
"t-SNE + KMeans",
lambda: cluster_tsne_kmeans(pixel_features),
)
labels_by_method["LPCA + KMeans"], times_by_method[
"LPCA + KMeans"
] = timed_clustering(
"LPCA + KMeans",
lambda: cluster_lpca_kmeans(
pixel_features,
use_correlation=False,
),
)
labels_by_method["LPCA+C + KMeans"], times_by_method[
"LPCA+C + KMeans"
] = timed_clustering(
"LPCA+C + KMeans",
lambda: cluster_lpca_kmeans(
pixel_features,
use_correlation=True,
),
)
labels_by_method["KPCA"], times_by_method["KPCA"] = (
timed_clustering(
"KPCA",
lambda: cluster_kpca(pixel_features),
)
)
# ============================================================
# MASK GENERATION
# ============================================================
masks_by_method: dict[str, list[np.ndarray]] = {}
for method_name, labels in labels_by_method.items():
masks = labels_to_masks(
labels=labels,
processing_size=processing_image.size,
original_size=original_image.size,
n_clusters=N_CLUSTERS,
)
masks_by_method[method_name] = masks
if SAVE_MASKS:
save_masks(
method_name=method_name,
masks=masks,
output_directory=OUTPUT_DIRECTORY,
)
# ============================================================
# DISPLAYING THE COMPARISON
# ============================================================
plot_masks_grid(
masks_by_method=masks_by_method,
times_by_method=times_by_method,
n_clusters=N_CLUSTERS,
)
if SAVE_MASKS:
print(
"\nMasks saved to:",
OUTPUT_DIRECTORY.resolve(),
)Original image: (4032, 2268)
Image used for processing: (413, 232)
Number of pixels: 95816
Number of features: 5
KMeans 1.3218 s | clusters: 5
PCA + KMeans 1.1825 s | clusters: 5
t-SNE + KMeans 128.6031 s | clusters: 5
LPCA + KMeans 1.2219 s | clusters: 5
LPCA+C + KMeans 1.1848 s | clusters: 5
KPCA 0.5130 s | clusters: 5
Masks saved to: /home/euuki/github_project/aktualne/PCA_pytorch_implementation/image_clustering_results














