k-means算法的Python实现

 #coding=utf-8

 import codecs

 import numpy

 from numpy import *

 import pylab

 def loadDataSet(fileName):

     dataMat = []

     fr = codecs.open(fileName)

     for line in fr.readlines():

         curLine = line.strip().split('\t')

         fltLine = map(float, curLine)

         dataMat.append(fltLine)

     return dataMat    

 def distMeasure(vecA, vecB):

     #print vecA

     dist = sqrt(sum(power(vecA - vecB, 2)))

     return dist

 def kMeansInitCentroids(X, K):

     """

     KMEANSINITCENTROIDS This function initializes K centroids that are to be

     used in K-Means on the dataset X

     centroids = KMEANSINITCENTROIDS(X, K) returns K initial centroids to be

     used with the K-Means on the dataset X.

     """

     n = shape(X)[1]

     centroids = mat(zeros((K,n)))

     for j in range(n):

         #print X[:,j]

         minJ = min(X[:,j])

         rangeJ = float(max(array(X)[:,j]) - minJ)

         centroids[:,j] = minJ + rangeJ * random.rand(K,1)

     return centroids

 def findClosestCentroids(X, centroids):

     """

     FINDCLOSESTCENTROIDS computes the centroid memberships for every example

     idx = FINDCLOSESTCENTROIDS (X, centroids) returns the closest centroids

     in idx for a dataset X where each row is a single example. idx = m x 1

     vector of centroid assignments (i.e. each entry in range [1..K])

     """

     # 数据总量

     m = shape(X)[0]

     K = shape(centroids)[0]

     clusterAssment = mat(zeros((m,2)))#create mat to assign data points

                                       #to a centroid, also holds SE of each point

     #centroids = createCent(dataSet, k)

     clusterChanged = True

     while clusterChanged:

         clusterChanged = False

         for i in range(m):#for each data point assign it to the closest centroid

             minDist = inf; minIndex = -1

             # k个中间数据（质心）都与数据i进行欧氏比较，选择距离最近的第minIndex类

             for j in range(K):

                 distJI = distMeasure(centroids[j,:],X[i,:])

                 if distJI < minDist:

                     minDist = distJI; minIndex = j

             if clusterAssment[i,0] != minIndex: clusterChanged = True

             clusterAssment[i,:] = minIndex,minDist**2

     return clusterAssment

 def computeCentroids(X, clusterAssment, K):

     """

     COMPUTECENTROIDS returs the new centroids by computing the means of the

     data points assigned to each centroid.

     centroids = COMPUTECENTROIDS(X, idx, K) returns the new centroids by

     computing the means of the data points assigned to each centroid. It is

     given a dataset X where each row is a single data point, a vector

     idx of centroid assignments (i.e. each entry in range [1..K]) for each

     example, and K, the number of centroids. You should return a matrix

     centroids, where each row of centroids is the mean of the data points

     assigned to it.

     """

     n = shape(X)[1]

     centroids = mat(zeros((K,n)))

     for centroid in range(K):#recalculate centroids

         # nonzero会产生两个array，第一个非零的为序号列表

         ptsInClust = X[nonzero(clusterAssment[:,0].A==centroid)[0]]#get all the point in this cluster

         #print 'ererer:',ptsInClust,'dfdf'

         centroids[centroid,:] = mean(ptsInClust, axis=0) #assign centroid to mean

     return centroids

 def show(dataSet, k, centroids, clusterAssment):

     from matplotlib import pyplot as plt

     numSamples, dim = dataSet.shape

     mark = ['or', 'ob', 'og', 'ok', '^r', '+r', 'sr', 'dr', '<r', 'pr']

     print type(dataSet)

     for i in xrange(numSamples):

         markIndex = int(clusterAssment[i, 0])

         plt.plot(dataSet[i, 0], dataSet[i, 1], mark[markIndex])

     mark = ['Dr', 'Db', 'Dg', 'Dk', '^b', '+b', 'sb', 'db', '<b', 'pb']

     for i in range(k):

         plt.plot(centroids[i, 0], centroids[i, 1], mark[i], markersize = 12)

     plt.show()

 def runkMeans(X, initial_centroids,max_iters, plot_progress):

     """

     RUNKMEANS runs the K-Means algorithm on data matrix X, where each row of X

     is a single example

     [centroids, idx] = RUNKMEANS(X, initial_centroids, max_iters, ...

     plot_progress) runs the K-Means algorithm on data matrix X, where each

     row of X is a single example. It uses initial_centroids used as the

     initial centroids. max_iters specifies the total number of interactions

     of K-Means to execute. plot_progress is a true/false flag that

     indicates if the function should also plot its progress as the

     learning happens. This is set to false by default. runkMeans returns

     centroids, a Kxn matrix of the computed centroids and idx, a m x 1

     vector of centroid assignments (i.e. each entry in range [1..K]).

     """

     (m,n) = shape(X)

     K = shape(initial_centroids)[0]

     centroids = initial_centroids

     clusterAssment = zeros((m,2))

     #Run K-Means

     for i in range(max_iters):

         clusterAssment = findClosestCentroids(X, centroids)

         centroids = computeCentroids(X, clusterAssment, K);

     return centroids, clusterAssment

 def main():

     K =5

     max_iters = 10

     dataSet =  loadDataSet('E://PythonSpace//TextClustering//data//test2.txt')

     X = array(dataSet)

     X = (X - mean(X)) / std(X)

     initial_centroids = kMeansInitCentroids(X, K)

     myCentroids, clusterAssment = runkMeans(X, initial_centroids, max_iters,False);

     print "-------------------------------------"

     show(X, K, myCentroids, clusterAssment)

 main()

参考了Andrew Ng的Machine Learning Assignment(https://github.com/rieder91/MachineLearning/blob/master/Exercise%207/ex7/runkMeans.m)

以及博文http://www.cnblogs.com/MrLJC/p/4127553.html

运行结果：

k-means算法的Python实现的更多相关文章

Fuzzy C Means 算法及其 Python 实现——写得很清楚，见原文
Fuzzy C Means 算法及其 Python 实现转自:http://note4code.com/2015/04/14/fuzzy-c-means-%E7%AE%97%E6%B3%95%E5% ...
分类算法——k最近邻算法（Python实现）（文末附工程源代码）
kNN算法原理 k最近邻(k-Nearest Neighbor)算法是比较简单的机器学习算法.它采用测量不同特征值之间的距离方法进行分类,思想很简单:如果一个样本在特征空间中的k个最近邻(最相似)的样 ...
KNN 与 K - Means 算法比较
KNN K-Means 1.分类算法聚类算法 2.监督学习非监督学习 3.数据类型:喂给它的数据集是带label的数据,已经是完全正确的数据喂给它的数据集是无label的数据,是杂乱无章的,经过 ...
K－means算法
K-means算法很简单,它属于无监督学习算法中的聚类算法中的一种方法吧,利用欧式距离进行聚合啦. 解决的问题如图所示哈:有一堆没有标签的训练样本,并且它们可以潜在地分为K类,我们怎么把它们划分呢? ...
Python实现kNN（k邻近算法）
Python实现kNN(k邻近算法) 运行环境 Pyhton3 numpy科学计算模块计算过程 st=>start: 开始 op1=>operation: 读入数据 op2=>op ...
机器学习算法与Python实践之（五）k均值聚类（k-means）
机器学习算法与Python实践这个系列主要是参考<机器学习实战>这本书.因为自己想学习Python,然后也想对一些机器学习算法加深下了解,所以就想通过Python来实现几个比较常用的机器学 ...
机器学习算法与Python实践之（六）二分k均值聚类
http://blog.csdn.net/zouxy09/article/details/17590137 机器学习算法与Python实践之(六)二分k均值聚类 zouxy09@qq.com http ...
用Python从零开始实现K近邻算法
KNN算法的定义: KNN通过测量不同样本的特征值之间的距离进行分类.它的思路是:如果一个样本在特征空间中的k个最相似(即特征空间中最邻近)的样本中的大多数属于某一个类别,则该样本也属于这个类别.K通 ...
机器学习 Python实践-K近邻算法
机器学习K近邻算法的实现主要是参考<机器学习实战>这本书. 一.K近邻(KNN)算法 K最近邻(k-Nearest Neighbour,KNN)分类算法,理解的思路是:如果一个样本在特征空 ...
K均值算法-python实现
测试数据展示: #coding:utf-8__author__ = 'similarface''''实现K均值算法算法摘要:-----------------------------输入:所有数据点 ...

随机推荐

Ansible6：Playbook简单使用【转】
ansbile-playbook是一系列ansible命令的集合,利用yaml 语言编写.playbook命令根据自上而下的顺序依次执行.同时,playbook开创了很多特性,它可以允许你传输某个命令 ...
Heartbeat+DRBD+MySQL高可用方案【转】
转自Heartbeat+DRBD+MySQL高可用方案 - yayun - 博客园 http://www.cnblogs.com/gomysql/p/3674030.html 1.方案简介本方案采用 ...
Nginx配置IP白名单和黑名单
白名单设置,访问根目录 location / { allow 123.34.22.155; allow ; deny all; } 黑名单设置,访问根目录 location / { deny 123. ...
hostent h_addr_list
struct hostent { char FAR * h_name; /* official name of host */ char FAR * FAR * h_aliases; /* alias ...
Local declaration of 'XXX' hides instance variable
今天调试程序遇到这么一个警告! Local declaration of 'XXX' hides instance variable 遇到这种原因,是因为本地变量跟函数参数变量同名.改变其一即可.
jave学习1--基础介绍
java 技术主要分为三个部分: jave SE基础知识. 对于各个程序的开发语言都包含的基本数据类型,循环控制,数组,方法等. jave SE的面向对象部分. 所有的面向对象的概念,为最终的接口准备 ...
jz2440 环境搭建遇到的问题
已解决:
一个设置 material design icon的插件工具
一个设置 material design icon的插件工具 github地址:https://github.com/konifar/android-material-design-icon-gene ...
C++：预处理指令
Preprocessor directives 预处理器指令预处理器指令是指那些包含在我们代码中的预处理器语句行,这些预处理器语句不是真正的代码语句,但是他们指导程序如何进行编译.这些语句总是以 ‘ ...
SEO优化之主页上加上nofollow
<a href=http://www.主页.cn/ rel=”nofollow”>这里是锚文字</a> <光年日志分析系统>来分析抓取比较多的是哪个网页,没用的no ...

k-means算法的Python实现

k-means算法的Python实现的更多相关文章

随机推荐

热门专题