In [57]:
%%shell
jupyter nbconvert --to html k-means-clusters.ipynb
[NbConvertApp] Converting notebook k-means-clusters.ipynb to html
[NbConvertApp] Writing 731468 bytes to k-means-clusters.html
Out[57]:

In [45]:
## %%shell
## jupyter nbconvert --to html k-means-clusters.ipynb
In [46]:
import os
import pandas as pd
import numpy as np
from scipy import spatial
import matplotlib.pyplot as plt
from sklearn.manifold import TSNE
In [48]:
## Collecting embedded data
## path = "/content/drive/MyDrive/Colab Notebooks/CA2data/"
embeddings_dict1 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/fruits", 'r') as f1:     
   for line in f1:        
      values = line.split()        
      word = values[0]        
      vector = np.asarray(values[1:], "float32")        
      embeddings_dict1[word] = vector
f1.close()
 
embeddings_dict2 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/animals", 'r') as f2:    
   for line in f2:        
      values = line.split()        
      word = values[0]        
      vector = np.asarray(values[1:], "float32")        
      embeddings_dict2[word] = vector
f2.close()
 
embeddings_dict3 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/veggies", 'r') as f3:    
   for line in f3:        
      values = line.split()        
      word = values[0]        
      vector = np.asarray(values[1:], "float32")        
      embeddings_dict3[word] = vector
f3.close()
embeddings_dict4 = {}
with open("/content/drive/MyDrive/Colab Notebooks/CA2data/countries", 'r') as f4:    
   for line in f4:        
      values = line.split()        
      word = values[0]        
      vector = np.asarray(values[1:], "float32")        
      embeddings_dict4[word] = vector
f4.close()
In [49]:
## Defining merge function

def Merge(a, b, c, d):
   merged = {**a, **b, **c, **d}
   return merged;
In [50]:
## Merging the data
 
all_data = Merge(embeddings_dict1, embeddings_dict2, embeddings_dict3, embeddings_dict4)
In [51]:
# Checking data
 
# all_data
In [52]:
lists = sorted(all_data.items()) # sorted by key, return a list of tuples
 
x, y = zip(*lists) # unpack a list of pairs into two tuples
ax = plt.gca()
ax.set_xticks(ax.get_xticks()[::10])
plt.plot(x, y)
plt.show()
In [ ]:
##  plt.scatter(embeddings_dict1[:,0], embeddings_dict1[:,1], c=labels)
In [ ]:
def thefunction(filedata, centroids, labels, cmap_name='seaice_2'):
        df = pd.read_table('filedata',sep=' ', index_col= 0,
).iloc[:, 0:].dropna()
        values = df.values.reshape((len(df),2))
        centroids,labels = vq.kmeans2(values, 3, minit='points')
        plt.plot()
        plt.xlim([0, 10])
        plt.ylim([0, 10])
        plt.title('Dataset')
        plt.scatter(centroids, labels)
        plt.show()
        #   create new plot and data
        plt.plot()
        X = np.array(list(zip(centroids, labels))).reshape(len(centroids), 2)
        colors = ['b', 'g', 'r']
        markers = ['o', 'v', 's']
        # KMeans algorithm
        K = 3
        kmeans_model = KMeans(n_clusters=K).fit(X)
        plt.plot()
        for i, l in enumerate(kmeans_model.labels_):
            plt.plot(centroids[i], labels(df1)[i], color=colors[l], marker=markers[l],ls='None')
            plt.xlim([0, 10])
            plt.ylim([0, 10])
            plt.show()
In [ ]:
#calculate 2d indicators
def indic(data):
    #alternatively you can calulate any other indicators
    max = np.max(data, axis=1)
    min = np.min(data, axis=1)
    return max, min
 
x,y = indic(zip(*lists)) # unpack a list of pairs into two tuples
plt.scatter(x, y, marker='x')
plt.show()
In [14]:
pd.DataFrame(all_data).T.plot()
plt.show()
In [ ]:

Data Processing

In [15]:
# Required library
from sklearn.cluster import KMeans
In [16]:
models = KMeans(n_clusters=4)
models
Out[16]:
KMeans(algorithm='auto', copy_x=True, init='k-means++', max_iter=300,
       n_clusters=4, n_init=10, n_jobs=None, precompute_distances='auto',
       random_state=None, tol=0.0001, verbose=0)
In [20]:
df = pd.DataFrame(columns=["test_items"])
 df["values"] = all_data.values()
In [ ]:
vectors = [np.array(f) for f in df["values"]]
vectors
In [22]:
models.fit(vectors)
Out[22]:
KMeans(algorithm='auto', copy_x=True, init='k-means++', max_iter=300,
       n_clusters=4, n_init=10, n_jobs=None, precompute_distances='auto',
       random_state=None, tol=0.0001, verbose=0)
In [23]:
models.inertia_
Out[23]:
9474.517538278536
In [24]:
from gensim.models import Word2Vec
from nltk.cluster import KMeansClusterer
import nltk
from sklearn import cluster
from sklearn import metrics
In [ ]:
# Example
# training data
 
## sentences = [['this', 'is', 'the', 'good', 'machine', 'learning', 'book'],
#             ['this', 'is',  'another', 'book'],
#             ['one', 'more', 'book'],
#             ['this', 'is', 'the', 'new', 'post'],
#           ['this', 'is', 'about', 'machine', 'learning', 'post'],  
#             ['and', 'this', 'is', 'the', 'last', 'post']]
 
 
# # training model
# model = Word2Vec(sentences, min_count=1)
 
# # get vector data
# X = model[model.wv.vocab]
# print (X)
In [26]:
NUM_CLUSTERS=4
kclusterer = KMeansClusterer(NUM_CLUSTERS, distance=nltk.cluster.util.cosine_distance, repeats=25)
assigned_clusters = kclusterer.cluster(all_data.values(), assign_clusters=True)
centroids = kclusterer._centroid
centroids
 
print (assigned_clusters)
[3, 3, 0, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 0, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 0, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 2, 0, 3, 3, 3, 3, 3, 3, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3, 0, 0, 0, 0, 0, 0, 3, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]
In [27]:
words = list(all_data)
for i, word in enumerate(words):  
    print (word + ":" + str(assigned_clusters[i]))
apple:3
apricot:3
avocado:0
banana:3
bilberry:3
blackberry:3
blackcurrant:3
blueberry:3
boysenberry:3
currant:3
cherry:3
cherimoya:3
cloudberry:3
coconut:3
cranberry:3
cucumber:0
damson:3
dragonfruit:3
durian:3
elderberry:3
feijoa:3
fig:3
gooseberry:3
grape:3
raisin:3
grapefruit:3
guava:3
honeyberry:3
huckleberry:3
jabuticaba:3
jackfruit:3
jambul:3
jujube:3
kiwano:3
kiwifruit:3
kumquat:3
lemon:0
lime:3
loquat:3
longan:3
lychee:3
mango:3
mangosteen:3
marionberry:3
melon:3
cantaloupe:3
honeydew:3
watermelon:3
mulberry:3
nectarine:3
nance:2
olive:0
orange:3
clementine:3
mandarine:3
tangerine:3
papaya:3
peach:3
elephant:2
leopard:2
dog:2
cat:2
aligator:2
ant:2
baboon:2
bear:2
bat:2
butterfly:2
camel:2
catfish:2
fish:2
cow:2
crow:2
boa:2
dolphin:2
donkey:2
eagle:2
falcon:2
fox:2
frog:2
gecko:2
giraffe:2
goat:2
gibbon:2
hampster:2
hawk:2
hare:2
horse:2
hummingbird:2
hippopotamus:2
iguana:2
jaguar:2
kangaroo:2
lion:2
leech:2
mouse:2
mosquito:2
owl:2
panda:2
penguin:2
parrot:2
peacock:2
rabbit:2
raven:2
shark:2
snake:2
spider:2
tiger:2
aubergine:0
amaranth:0
asparagus:0
legumes:0
beans:0
chickpeas:0
lentils:0
peas:0
broccoli:0
cabbage:0
kohlrabi:0
cauliflower:0
celery:0
endive:0
fiddleheads:0
frisee:0
fennel:0
greens:0
kale:0
spinach:0
anise:0
basil:0
caraway:0
cilantro:0
coriander:0
chamomile:0
dill:0
lavender:3
marjoram:0
oregano:0
parsley:0
rosemary:0
sage:0
thyme:0
lettuce:0
arugula:0
mushrooms:0
nettles:0
okra:0
onions:0
chives:0
garlic:0
leek:0
onion:0
shallot:0
peppers:0
habanero:0
paprika:0
radicchio:0
rhubarb:3
turnip:0
radish:0
courgette:0
delicata:0
pumpkin:0
potato:0
quandong:3
sunchokes:0
zucchini:0
afghanistan:1
albania:1
algeria:1
andorra:1
angola:1
argentina:1
armenia:1
aruba:1
australia:1
austria:1
azerbaijan:1
bahrain:1
bangladesh:1
barbados:1
belarus:1
belgium:1
belize:1
benin:1
bhutan:1
bolivia:1
botswana:1
brazil:1
brunei:1
bulgaria:1
burma:1
burundi:1
cambodia:1
cameroon:1
canada:1
chad:1
chile:1
china:1
colombia:1
comoros:1
croatia:1
cuba:1
curacao:1
cyprus:1
czechia:1
denmark:1
djibouti:1
dominica:1
ecuador:1
egypt:1
eritrea:1
estonia:1
ethiopia:1
fiji:1
finland:1
france:1
gabon:1
georgia:1
germany:1
ghana:1
greece:1
grenada:1
guatemala:1
guinea:1
guinea-bissau:1
guyana:1
haiti:1
honduras:1
hungary:1
iceland:1
india:1
indonesia:1
iran:1
iraq:1
ireland:1
israel:1
italy:1
jamaica:1
japan:1
kazakhstan:1
kenya:1
kiribati:1
kosovo:1
kuwait:1
kyrgyzstan:1
laos:1
latvia:1
lebanon:1
lesotho:1
liberia:1
libya:1
liechtenstein:1
lithuania:1
luxembourg:1
macau:1
macedonia:1
madagascar:1
malawi:1
malaysia:1
maldives:1
mali:1
malta:1
mauritania:1
mauritius:1
mexico:1
micronesia:1
moldova:1
monaco:1
mongolia:1
montenegro:1
morocco:1
mozambique:1
namibia:1
nauru:1
nepal:1
netherlands:1
nicaragua:1
niger:1
nigeria:1
norway:1
oman:1
pakistan:1
palau:1
panama:1
paraguay:1
peru:1
philippines:1
poland:1
portugal:1
qatar:1
romania:1
russia:1
rwanda:1
samoa:1
senegal:1
serbia:1
seychelles:1
singapore:1
slovakia:1
somalia:1
spain:1
sudan:1
suriname:1
swaziland:1
sweden:1
switzerland:1
syria:1
taiwan:1
tajikistan:1
tanzania:1
thailand:1
togo:1
tonga:1
tunisia:1
turkey:1
turkmenistan:1
tuvalu:1
uganda:1
ukraine:1
uruguay:1
uzbekistan:1
vanuatu:1
venezuela:1
vietnam:1
yemen:1
zambia:1
zimbabwe:1
In [31]:
labels = models.labels_
labels
Out[31]:
array([2, 2, 0, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 0, 2, 2, 2, 2, 2, 2,
       2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 0, 2, 2, 2, 2, 2, 2, 2,
       2, 2, 2, 2, 2, 2, 3, 0, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3,
       3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
       3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 0, 2,
       0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 2, 0, 0, 0,
       0, 2, 0, 2, 0, 0, 0, 0, 3, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0, 0,
       2, 0, 0, 2, 0, 0, 0, 2, 2, 0, 2, 2, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
       1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1],
      dtype=int32)
In [ ]:

In [32]:
df["test_items"] = all_data.keys()
df["clusters"] = assigned_clusters
df["centroids"] = df['clusters'].apply(lambda x: kclusterer.means()[x])
In [33]:
df
Out[33]:
test_items values clusters centroids
0 apple (-0.077424, -0.0072709, -0.17181, 0.18994, 0.7... 3 [-0.23876481, -0.10741026, 0.10374063, -0.1213...
1 apricot (-0.36757, -0.68801, -0.14844, -0.96095, 1.177... 3 [-0.23876481, -0.10741026, 0.10374063, -0.1213...
2 avocado (-0.29165, -0.43165, 0.26492, -1.1137, 1.0291,... 0 [-0.13114165, -0.102825135, -0.07042423, -0.42...
3 banana (-0.36379, -0.10368, 0.19613, -0.096709, 0.863... 3 [-0.23876481, -0.10741026, 0.10374063, -0.1213...
4 bilberry (-0.60602, 0.30315, 0.16475, -0.5083, 0.91074,... 3 [-0.23876481, -0.10741026, 0.10374063, -0.1213...
... ... ... ... ...
323 venezuela (0.017367, 0.50206, 0.62323, -0.031531, 0.2422... 1 [0.38400927, 0.31275502, 0.060346555, 0.076397...
324 vietnam (0.71465, 0.35513, 0.021552, 0.27957, -0.11784... 1 [0.38400927, 0.31275502, 0.060346555, 0.076397...
325 yemen (-0.081987, 0.47922, 0.24027, -0.17446, -0.103... 1 [0.38400927, 0.31275502, 0.060346555, 0.076397...
326 zambia (0.70537, 0.36108, 0.048266, 0.76742, -0.34548... 1 [0.38400927, 0.31275502, 0.060346555, 0.076397...
327 zimbabwe (0.33176, 0.67068, 0.39108, 0.654, -0.46691, 0... 1 [0.38400927, 0.31275502, 0.060346555, 0.076397...

328 rows × 4 columns

In [34]:
feature_matrix = df["values"] .to_numpy()
 centroid = df['centroids'].to_numpy()
 
 def nltk_inertia(feature_matrix, centroid):
     sum_ = []
     for i in range(feature_matrix.shape[0]):
         sum_.append(np.sum((feature_matrix[i] - centroid[i])**2))  #here implementing inertia as given in the docs of scikit i.e sum of squared distance..
 
     return sum(sum_)
 
 nltk_inertia(feature_matrix, centroid)
Out[34]:
9489.733222961426

Data Visualisation

In [35]:
x = df["test_items"]
y = df["clusters"] 
ax = plt.gca()
ax.set_xticks(ax.get_xticks()[::10])
plt.scatter(x, y, marker='x')
Out[35]:
<matplotlib.collections.PathCollection at 0x7f48090736d0>
In [36]:
df1 = df[df.clusters==0]
df2 = df[df.clusters==1]
df3 = df[df.clusters==2]
df4 = df[df.clusters==3]
In [37]:
ax = plt.gca()
plt.scatter(df1.test_items,df1.clusters,color='green')
plt.scatter(df2.test_items,df2.clusters,color='red')
plt.scatter(df3.test_items, df3.clusters,color='blue')
plt.scatter(df4.test_items, df4.clusters,color='black')
plt.xlabel('Age')
ax.set_xticks(ax.get_xticks()[::70])
## plt.axes(xscale='log', yscale='log')## plt.xticks(["veggie", "animals", "countries"]) 
## plt.locator_params(axis='x', nbins=5)
plt.ylabel('Income ($)')
plt.legend()
No handles with labels found to put in legend.
Out[37]:
<matplotlib.legend.Legend at 0x7f480b826710>
In [38]:
sum_ = []
k_rng = range(feature_matrix.shape[0])
for i in k_rng:
     sum_.append(np.sum((feature_matrix[i] - centroid[i])**2))  #here implementing inertia as given in the docs of scikit i.e sum of squared distance..
In [ ]:
sum_
In [40]:
plt.xlabel('K')
plt.ylabel('Sum of squared error')
plt.plot(k_rng,sum_)
Out[40]:
[<matplotlib.lines.Line2D at 0x7f481b8f8390>]