From d4095e5cbe3fdee52b8890c9d2aa15fc6df153e0 Mon Sep 17 00:00:00 2001 From: Marin Mlinarevic Date: Thu, 10 Dec 2020 13:33:24 +0100 Subject: [PATCH 1/2] Import specific methods and add random samples for testing --- week09/improvement/clustering.py | 4 ++-- week09/improvement/samples.csv | 12 ++++++++++++ 2 files changed, 14 insertions(+), 2 deletions(-) create mode 100644 week09/improvement/samples.csv diff --git a/week09/improvement/clustering.py b/week09/improvement/clustering.py index d6359ea..def02c4 100644 --- a/week09/improvement/clustering.py +++ b/week09/improvement/clustering.py @@ -1,7 +1,7 @@ """This code implements the k-means clustering algorithm for 2 dimensions.""" -from math import * -from random import * +from math import sqrt +from random import randrange k=3 diff --git a/week09/improvement/samples.csv b/week09/improvement/samples.csv new file mode 100644 index 0000000..2e4b0d5 --- /dev/null +++ b/week09/improvement/samples.csv @@ -0,0 +1,12 @@ +342,4454 +34,5 +34,34 +45,56 +453,67 +34,6 +7,45 +45,7 +45,6 +65654,456 +54,564 +456,56 \ No newline at end of file From 9dffb2c04e6710033767e253382c1b996f6eafcf Mon Sep 17 00:00:00 2001 From: Marin Mlinarevic Date: Thu, 10 Dec 2020 13:43:32 +0100 Subject: [PATCH 2/2] Changed variable k to num_clusters and adapted code to use the variable --- week09/improvement/clustering.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/week09/improvement/clustering.py b/week09/improvement/clustering.py index def02c4..16ccf00 100644 --- a/week09/improvement/clustering.py +++ b/week09/improvement/clustering.py @@ -3,13 +3,13 @@ from math import sqrt from random import randrange -k=3 +num_clusters=3 lines = open('samples.csv', 'r').readlines() ps=[] for line in lines: ps.append(tuple(map(float, line.strip().split(',')))) -m=[ps[randrange(len(ps))], ps[randrange(len(ps))], ps[randrange(len(ps))]] +m=[ps[randrange(len(ps))] for i in range(num_clusters)] alloc=[None]*len(ps) n=0 @@ -21,12 +21,12 @@ d[1]=sqrt((p[0]-m[1][0])**2 + (p[1]-m[1][1])**2) d[2]=sqrt((p[0]-m[2][0])**2 + (p[1]-m[2][1])**2) alloc[i]=d.index(min(d)) - for i in range(3): + for i in range(num_clusters): alloc_ps=[p for j, p in enumerate(ps) if alloc[j] == i] new_mean=(sum([a[0] for a in alloc_ps]) / len(alloc_ps), sum([a[1] for a in alloc_ps]) / len(alloc_ps)) m[i]=new_mean n=n+1 -for i in range(3): +for i in range(num_clusters): alloc_ps=[p for j, p in enumerate(ps) if alloc[j] == i] print("Cluster " + str(i) + " is centred at " + str(m[i]) + " and has " + str(len(alloc_ps)) + " points.") \ No newline at end of file