import numpy as np import pandas as pd import scipy import pylab import scipy.cluster.hierarchy as hierarchy from scipy.cluster.hierarchy import fcluster from scipy import ndimage from scipy.spatial import distance_matrix from matplotlib import pyplot as plt import matplotlib.cm as cm from sklearn import manifold, datasets from sklearn.cluster import AgglomerativeClustering from sklearn.metrics import silhouette_samples, silhouette_score from sklearn.preprocessing import MinMaxScaler filename = "AirQualityUCI.csv" #Read csv pdf = pd.read_csv(filename, delimiter=";", decimal=",", na_values=-200) print ("Shape of dataset: ", pdf.shape) # print(pdf) print ("Shape of dataset before cleaning: ", pdf.size) pdf[[ 'CO(GT)', 'PT08.S1(CO)', 'NMHC(GT)', 'C6H6(GT)', 'PT08.S2(NMHC)', 'NOx(GT)', 'NO2(GT)', 'PT08.S4(NO2)', 'PT08.S5(O3)', 'T', 'RH', 'AH']] = pdf[['CO(GT)', 'PT08.S1(CO)', 'NMHC(GT)', 'C6H6(GT)', 'PT08.S2(NMHC)', 'NOx(GT)', 'NO2(GT)', 'PT08.S4(NO2)', 'PT08.S5(O3)', 'T', 'RH', 'AH']].apply(pd.to_numeric, errors='coerce') pdf = pdf.dropna() pdf = pdf.reset_index(drop=True) print ("Shape of dataset after cleaning: ", pdf.size) print(pdf) featureset = pdf[['CO(GT)', 'PT08.S1(CO)', 'NMHC(GT)', 'C6H6(GT)', 'PT08.S2(NMHC)', 'NOx(GT)', 'NO2(GT)', 'PT08.S4(NO2)', 'PT08.S5(O3)', 'T', 'RH', 'AH']] x = featureset.values #returns a numpy array min_max_scaler = MinMaxScaler() feature_mtx = min_max_scaler.fit_transform(x) feature_mtx [0:5] leng = feature_mtx.shape[0] D = scipy.zeros([leng,leng]) for i in range(leng): for j in range(leng): D[i,j] = scipy.spatial.distance.euclidean(feature_mtx[i], feature_mtx[j]) Z = hierarchy.linkage(D, 'single') max_d = 3 clusters = fcluster(Z, max_d, criterion='distance') clusters # k = 5 # clusters = fcluster(Z, k, criterion='maxclust') # clusters fig = pylab.figure(figsize=(50,100)) def llf(id): return '[%s]' % (pdf['Time'][id] ) dendro = hierarchy.dendrogram(Z, leaf_label_func=llf, leaf_rotation=0 , leaf_font_size =10, orientation = 'right') plt.show()