For every single data point in the dataset, calculate its Euclidean distance to all three centroids,Assign each data point to its closest centroid to form three temporary clusters.
5. Update centroid positions
Recalculate the position of each centroid by taking the mathematical mean of all coordinates assigned to that specific cluster
6. Repeat until convergence
Go back to Step 4 and reassign all points based on the newly calculated centroid coordinates. Repeat the assignment and update steps iteratively. The loop finishes when:
- Centroids stop moving to new positions.
- No data point changes its cluster assignment.
- The maximum number of preset iterations is reached.
7. Evaluate and interpret
Once converged, your 150 flowers are split into three distinct mathematically optimized buckets. You can evaluate the quality of your clustering by tracking metrics like the Silhouette Score or by cross-referencing your final clusters with the original species labels to check the model's structural accuracy
from sklearn.cluster import KMeans
import numpy as np
import matplotlib.pyplot as plt
import pandas as pd
Iris = pd.read_csv(r'C:\Users\admin\Downloads\Iris.csv')
x1 = np.array(Iris['SepalLengthCm'])
x2 = np.array(Iris['PetalWidthCm'])
plt.plot()
plt.title('Dataset')
plt.scatter(x1, x2)
plt.show()
Iris
output:
| Id | SepalLengthCm | SepalWidthCm | PetalLengthCm | PetalWidthCm | Species |
|---|
| 0 | 1 | 5.1 | 3.5 | 1.4 | 0.2 | Iris-setosa |
|---|
| 1 | 2 | 4.9 | 3.0 | 1.4 | 0.2 | Iris-setosa |
|---|
| 2 | 3 | 4.7 | 3.2 | 1.3 | 0.2 | Iris-setosa |
|---|
| 3 | 4 | 4.6 | 3.1 | 1.5 | 0.2 | Iris-setosa |
|---|
| 4 | 5 | 5.0 | 3.6 | 1.4 | 0.2 | Iris-setosa |
|---|
| ... | ... | ... | ... | ... | ... | ... |
|---|
| 145 | 146 | 6.7 | 3.0 | 5.2 | 2.3 | Iris-virginica |
|---|
| 146 | 147 | 6.3 | 2.5 | 5.0 | 1.9 | Iris-virginica |
|---|
| 147 | 148 | 6.5 | 3.0 | 5.2 | 2.0 | Iris-virginica |
|---|
| 148 | 149 | 6.2 | 3.4 | 5.4 | 2.3 | Iris-virginica |
|---|
| 149 | 150 | 5.9 | 3.0 | 5.1 | 1.8 | Iris-virginica |
|---|
150 rows × 6 columns
Iris.head()
output:
| Id | SepalLengthCm | SepalWidthCm | PetalLengthCm | PetalWidthCm | Species |
|---|
| 0 | 1 | 5.1 | 3.5 | 1.4 | 0.2 | Iris-setosa |
|---|
| 1 | 2 | 4.9 | 3.0 | 1.4 | 0.2 | Iris-setosa |
|---|
| 2 | 3 | 4.7 | 3.2 | 1.3 | 0.2 | Iris-setosa |
|---|
| 3 | 4 | 4.6 | 3.1 | 1.5 | 0.2 | Iris-setosa |
|---|
| 4 | 5 | 5.0 | 3.6 | 1.4 | 0.2 | Iris-setosa |
|---|
Iris.tail()
output:
len(Iris)
output:
150
Iris.shape
Iris.columns
output:
Index(['Id', 'SepalLengthCm', 'SepalWidthCm', 'PetalLengthCm', 'PetalWidthCm',
'Species'],
dtype='object')
for i,col in enumerate(Iris.columns):
print(f'Column number {1+i} is {col}')
output:
Column number 1 is Id
Column number 2 is SepalLengthCm
Column number 3 is SepalWidthCm
Column number 4 is PetalLengthCm
Column number 5 is PetalWidthCm
Column number 6 is Species
Iris.dtypes
output:
Id int64
SepalLengthCm float64
SepalWidthCm float64
PetalLengthCm float64
PetalWidthCm float64
Species object
dtype: object
Iris.info()
output:
<class 'pandas.core.frame.DataFrame'>
RangeIndex: 150 entries, 0 to 149
Data columns (total 6 columns):
# Column Non-Null Count Dtype
--- ------ -------------- -----
0 Id 150 non-null int64
1 SepalLengthCm 150 non-null float64
2 SepalWidthCm 150 non-null float64
3 PetalLengthCm 150 non-null float64
4 PetalWidthCm 150 non-null float64
5 Species 150 non-null object
dtypes: float64(4), int64(1), object(1)
memory usage: 7.2+ KB
Iris.isna().sum()
output:
Id 0
SepalLengthCm 0
SepalWidthCm 0
PetalLengthCm 0
PetalWidthCm 0
Species 0
dtype: int64
Iris.isnull().sum()
output:
Id 0
SepalLengthCm 0
SepalWidthCm 0
PetalLengthCm 0
PetalWidthCm 0
Species 0
dtype: int64
Iris['Species'].value_counts()
output:
Species
Iris-setosa 50
Iris-versicolor 50
Iris-virginica 50
Name: count, dtype: int64
target_data = Iris.iloc[:,5]
target_data
output:
0 Iris-setosa
1 Iris-setosa
2 Iris-setosa
3 Iris-setosa
4 Iris-setosa
...
145 Iris-virginica
146 Iris-virginica
147 Iris-virginica
148 Iris-virginica
149 Iris-virginica
Name: Species, Length: 150, dtype: object
clustering_data = Iris.iloc[:,[1,2,3,4]]
clustering_data
output:
| SepalLengthCm | SepalWidthCm | PetalLengthCm | PetalWidthCm |
|---|
| 0 | 5.1 | 3.5 | 1.4 | 0.2 |
|---|
| 1 | 4.9 | 3.0 | 1.4 | 0.2 |
|---|
| 2 | 4.7 | 3.2 | 1.3 | 0.2 |
|---|
| 3 | 4.6 | 3.1 | 1.5 | 0.2 |
|---|
| 4 | 5.0 | 3.6 | 1.4 | 0.2 |
|---|
| ... | ... | ... | ... | ... |
|---|
| 145 | 6.7 | 3.0 | 5.2 | 2.3 |
|---|
| 146 | 6.3 | 2.5 | 5.0 | 1.9 |
|---|
| 147 | 6.5 | 3.0 | 5.2 | 2.0 |
|---|
| 148 | 6.2 | 3.4 | 5.4 | 2.3 |
|---|
| 149 | 5.9 | 3.0 | 5.1 | 1.8 |
|---|
150 rows × 4 columns
fig, ax = plt.subplots(figsize=(15,7))
sns.set(font_scale=1.5)
ax = sns.scatterplot(x=Iris['SepalLengthCm'],y=Iris['SepalWidthCm'], s=70, color='#f73434', edgecolor='#f73434', linewidth=0.3)
ax.set_ylabel('Sepal Width (in cm)')
ax.set_xlabel('Sepal Length (in cm)')
plt.title('Sepal Length vs Width', fontsize = 20)
plt.show()
output:
from sklearn.cluster import KMeans
wcss=[]
for i in range(1,11):
km = KMeans(i)
km.fit(clustering_data)
wcss.append(km.inertia_)
np.array(wcss)
output:
array([680.8244 , 152.36870648, 78.94506583, 57.34540932,
46.80170193, 44.81835983, 37.52130193, 32.69137106,
28.29091775, 26.66453557])
fig, ax = plt.subplots(figsize=(15,7))
ax = plt.plot(range(1,11),wcss, linewidth=2, color="red", marker ="8")
plt.axvline(x=3, ls='--')
plt.ylabel('WCSS')
plt.xlabel('No. of Clusters (k)')
plt.title('The Elbow Method', fontsize = 20)
plt.show()
output:
from sklearn.cluster import KMeans
kms = KMeans(n_clusters=3, init='k-means++')
kms.fit(clustering_data)
output:
clusters = clustering_data.copy()
clusters['Cluster_Prediction'] = kms.fit_predict(clustering_data)
clusters
output:
| SepalLengthCm | SepalWidthCm | PetalLengthCm | PetalWidthCm | Cluster_Prediction |
|---|
| 0 | 5.1 | 3.5 | 1.4 | 0.2 | 1 |
|---|
| 1 | 4.9 | 3.0 | 1.4 | 0.2 | 1 |
|---|
| 2 | 4.7 | 3.2 | 1.3 | 0.2 | 1 |
|---|
| 3 | 4.6 | 3.1 | 1.5 | 0.2 | 1 |
|---|
| 4 | 5.0 | 3.6 | 1.4 | 0.2 | 1 |
|---|
| ... | ... | ... | ... | ... | ... |
|---|
| 145 | 6.7 | 3.0 | 5.2 | 2.3 | 2 |
|---|
| 146 | 6.3 | 2.5 | 5.0 | 1.9 | 0 |
|---|
| 147 | 6.5 | 3.0 | 5.2 | 2.0 | 2 |
|---|
| 148 | 6.2 | 3.4 | 5.4 | 2.3 | 2 |
|---|
| 149 | 5.9 | 3.0 | 5.1 | 1.8 | 0 |
|---|
150 rows × 5 columns
kms.cluster_centers_
array([[5.9016129 , 2.7483871 , 4.39354839, 1.43387097],
[5.006 , 3.418 , 1.464 , 0.244 ],
[6.85 , 3.07368421, 5.74210526, 2.07105263]])
fig, ax = plt.subplots(figsize=(15,7))
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 0]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 0]['SepalWidthCm'],
s=70,edgecolor='teal', linewidth=0.3, c='teal', label='Iris-versicolor')
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 1]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 1]['SepalWidthCm'],
s=70,edgecolor='lime', linewidth=0.3, c='lime', label='Iris-setosa')
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 2]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 2]['SepalWidthCm'],
s=70,edgecolor='magenta', linewidth=0.3, c='magenta', label='Iris-virginica')
plt.scatter(x=kms.cluster_centers_[:, 0], y=kms.cluster_centers_[:, 1], s = 170, c = 'yellow', label = 'Centroids',edgecolor='black', linewidth=0.3)
plt.legend(loc='upper right')
plt.xlim(4,8)
plt.ylim(1.8,4.5)
ax.set_ylabel('Sepal Width (in cm)')
ax.set_xlabel('Sepal Length (in cm)')
plt.title('Clusters', fontsize = 20)
plt.show()
fig, ax = plt.subplots(figsize=(15,7))
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 0]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 0]['SepalWidthCm'],
s=70,edgecolor='teal', linewidth=0.3, c='teal', label='Iris-versicolor')
plt.scatter(x=kms.cluster_centers_[0, 0], y=kms.cluster_centers_[0, 1], s = 170, c = 'yellow', label = 'Centroids',edgecolor='black', linewidth=0.3)
plt.legend(loc='upper right')
plt.xlim(4,8)
plt.ylim(1.8,4.5)
ax.set_ylabel('Sepal Width (in cm)')
ax.set_xlabel('Sepal Length (in cm)')
plt.title('Individual Clusters', fontsize = 20)
plt.show()
output:
fig, ax = plt.subplots(figsize=(15,7))
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 1]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 1]['SepalWidthCm'],
s=70,edgecolor='lime', linewidth=0.3, c='lime', label='Iris-versicolor')
plt.scatter(x=kms.cluster_centers_[1, 0], y=kms.cluster_centers_[1, 1], s = 170, c = 'yellow', label = 'Centroids',edgecolor='black', linewidth=0.3)
plt.legend(loc='upper right')
plt.xlim(4,8)
plt.ylim(1.8,4.5)
ax.set_ylabel('Sepal Width (in cm)')
ax.set_xlabel('Sepal Length (in cm)')
plt.title('Individual Clusters', fontsize = 20)
plt.show()
output:
fig, ax = plt.subplots(figsize=(15,7))
plt.scatter(x=clusters[clusters['Cluster_Prediction'] == 2]['SepalLengthCm'],
y=clusters[clusters['Cluster_Prediction'] == 2]['SepalWidthCm'],
s=70,edgecolor='magenta', linewidth=0.3, c='magenta', label='Iris-versicolor')
output:
plt.scatter(x=kms.cluster_centers_[2, 0], y=kms.cluster_centers_[2, 1], s = 170, c = 'yellow', label = 'Centroids',edgecolor='black', linewidth=0.3)
plt.legend(loc='upper right')
plt.xlim(4,8)
plt.ylim(1.8,4.5)
ax.set_ylabel('Sepal Width (in cm)')
ax.set_xlabel('Sepal Length (in cm)')
plt.title('Individual Clusters', fontsize = 20)
plt.show()
output: