%%shell
jupyter nbconvert --to html k-means-clusters.ipynb
## %%shell
## jupyter nbconvert --to html k-means-clusters.ipynb
import os
import pandas as pd
import numpy as np
from scipy import spatial
import matplotlib.pyplot as plt
from sklearn.manifold import TSNE
## Collecting embedded data
## path = "/content/drive/MyDrive/Colab Notebooks/CA2data/"
embeddings_dict1 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/fruits", 'r') as f1:
for line in f1:
values = line.split()
word = values[0]
vector = np.asarray(values[1:], "float32")
embeddings_dict1[word] = vector
f1.close()
embeddings_dict2 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/animals", 'r') as f2:
for line in f2:
values = line.split()
word = values[0]
vector = np.asarray(values[1:], "float32")
embeddings_dict2[word] = vector
f2.close()
embeddings_dict3 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/veggies", 'r') as f3:
for line in f3:
values = line.split()
word = values[0]
vector = np.asarray(values[1:], "float32")
embeddings_dict3[word] = vector
f3.close()
embeddings_dict4 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/countries", 'r') as f4:
for line in f4:
values = line.split()
word = values[0]
vector = np.asarray(values[1:], "float32")
embeddings_dict4[word] = vector
f4.close()
## Defining merge function
def Merge(a, b, c, d):
merged = {**a, **b, **c, **d}
return merged;
## Merging the data
all_data = Merge(embeddings_dict1, embeddings_dict2, embeddings_dict3, embeddings_dict4)
# Checking data
# all_data
lists = sorted(all_data.items()) # sorted by key, return a list of tuples
x, y = zip(*lists) # unpack a list of pairs into two tuples
ax = plt.gca()
ax.set_xticks(ax.get_xticks()[::10])
plt.plot(x, y)
plt.show()
## plt.scatter(embeddings_dict1[:,0], embeddings_dict1[:,1], c=labels)
def thefunction(filedata, centroids, labels, cmap_name='seaice_2'):
df = pd.read_table('filedata',sep=' ', index_col= 0,
).iloc[:, 0:].dropna()
values = df.values.reshape((len(df),2))
centroids,labels = vq.kmeans2(values, 3, minit='points')
plt.plot()
plt.xlim([0, 10])
plt.ylim([0, 10])
plt.title('Dataset')
plt.scatter(centroids, labels)
plt.show()
# create new plot and data
plt.plot()
X = np.array(list(zip(centroids, labels))).reshape(len(centroids), 2)
colors = ['b', 'g', 'r']
markers = ['o', 'v', 's']
# KMeans algorithm
K = 3
kmeans_model = KMeans(n_clusters=K).fit(X)
plt.plot()
for i, l in enumerate(kmeans_model.labels_):
plt.plot(centroids[i], labels(df1)[i], color=colors[l], marker=markers[l],ls='None')
plt.xlim([0, 10])
plt.ylim([0, 10])
plt.show()
#calculate 2d indicators
def indic(data):
#alternatively you can calulate any other indicators
max = np.max(data, axis=1)
min = np.min(data, axis=1)
return max, min
x,y = indic(zip(*lists)) # unpack a list of pairs into two tuples
plt.scatter(x, y, marker='x')
plt.show()
pd.DataFrame(all_data).T.plot()
plt.show()
Data Processing
# Required library
from sklearn.cluster import KMeans
models = KMeans(n_clusters=4)
models
df = pd.DataFrame(columns=["test_items"])
df["values"] = all_data.values()
vectors = [np.array(f) for f in df["values"]]
vectors
models.fit(vectors)
models.inertia_
from gensim.models import Word2Vec
from nltk.cluster import KMeansClusterer
import nltk
from sklearn import cluster
from sklearn import metrics
# Example
# training data
## sentences = [['this', 'is', 'the', 'good', 'machine', 'learning', 'book'],
# ['this', 'is', 'another', 'book'],
# ['one', 'more', 'book'],
# ['this', 'is', 'the', 'new', 'post'],
# ['this', 'is', 'about', 'machine', 'learning', 'post'],
# ['and', 'this', 'is', 'the', 'last', 'post']]
# # training model
# model = Word2Vec(sentences, min_count=1)
# # get vector data
# X = model[model.wv.vocab]
# print (X)
NUM_CLUSTERS=4
kclusterer = KMeansClusterer(NUM_CLUSTERS, distance=nltk.cluster.util.cosine_distance, repeats=25)
assigned_clusters = kclusterer.cluster(all_data.values(), assign_clusters=True)
centroids = kclusterer._centroid
centroids
print (assigned_clusters)
words = list(all_data)
for i, word in enumerate(words):
print (word + ":" + str(assigned_clusters[i]))
labels = models.labels_
labels
df["test_items"] = all_data.keys()
df["clusters"] = assigned_clusters
df["centroids"] = df['clusters'].apply(lambda x: kclusterer.means()[x])
df
feature_matrix = df["values"] .to_numpy()
centroid = df['centroids'].to_numpy()
def nltk_inertia(feature_matrix, centroid):
sum_ = []
for i in range(feature_matrix.shape[0]):
sum_.append(np.sum((feature_matrix[i] - centroid[i])**2)) #here implementing inertia as given in the docs of scikit i.e sum of squared distance..
return sum(sum_)
nltk_inertia(feature_matrix, centroid)
Data Visualisation
x = df["test_items"]
y = df["clusters"]
ax = plt.gca()
ax.set_xticks(ax.get_xticks()[::10])
plt.scatter(x, y, marker='x')
df1 = df[df.clusters==0]
df2 = df[df.clusters==1]
df3 = df[df.clusters==2]
df4 = df[df.clusters==3]
ax = plt.gca()
plt.scatter(df1.test_items,df1.clusters,color='green')
plt.scatter(df2.test_items,df2.clusters,color='red')
plt.scatter(df3.test_items, df3.clusters,color='blue')
plt.scatter(df4.test_items, df4.clusters,color='black')
plt.xlabel('Age')
ax.set_xticks(ax.get_xticks()[::70])
## plt.axes(xscale='log', yscale='log')## plt.xticks(["veggie", "animals", "countries"])
## plt.locator_params(axis='x', nbins=5)
plt.ylabel('Income ($)')
plt.legend()
sum_ = []
k_rng = range(feature_matrix.shape[0])
for i in k_rng:
sum_.append(np.sum((feature_matrix[i] - centroid[i])**2)) #here implementing inertia as given in the docs of scikit i.e sum of squared distance..
sum_
plt.xlabel('K')
plt.ylabel('Sum of squared error')
plt.plot(k_rng,sum_)