344 KiB
344 KiB
In [1]:
import sys
sys.path.insert(1, '../')
from utils.utils import *
from utils.dataset4learners import *
psource(DataSet)class DataSet: """ A data set for a machine learning problem. It has the following fields: d.examples A list of examples. Each one is a list of attribute values. d.attrs A list of integers to index into an example, so example[attr] gives a value. Normally the same as range(len(d.examples[0])). d.attr_names Optional list of mnemonic names for corresponding attrs. d.target The attribute that a learning algorithm will try to predict. By default the final attribute. d.inputs The list of attrs without the target. d.values A list of lists: each sublist is the set of possible values for the corresponding attribute. If initially None, it is computed from the known examples by self.set_problem. If not None, an erroneous value raises ValueError. d.distance A function from a pair of examples to a non-negative number. Should be symmetric, etc. Defaults to mean_boolean_error since that can handle any field types. d.name Name of the data set (for output display only). d.source URL or other source where the data came from. d.exclude A list of attribute indexes to exclude from d.inputs. Elements of this list can either be integers (attrs) or attr_names. Normally, you call the constructor and you're done; then you just access fields like d.examples and d.target and d.inputs. """ def __init__(self, examples=None, attrs=None, attr_names=None, target=-1, inputs=None, values=None, distance=mean_boolean_error, name='', source='', exclude=()): """ Accepts any of DataSet's fields. Examples can also be a string or file from which to parse examples using parse_csv. Optional parameter: exclude, as documented in .set_problem(). >>> DataSet(examples='1, 2, 3') <DataSet(): 1 examples, 3 attributes> """ self.name = name self.source = source self.values = values self.distance = distance self.got_values_flag = bool(values) # initialize .examples from string or list or data directory if isinstance(examples, str): self.examples = parse_csv(examples) elif examples is None: self.examples = parse_csv(open_data(name + '.csv').read()) else: self.examples = examples # attrs are the indices of examples, unless otherwise stated. if self.examples is not None and attrs is None: attrs = list(range(len(self.examples[0]))) self.attrs = attrs # initialize .attr_names from string, list, or by default if isinstance(attr_names, str): self.attr_names = attr_names.split() else: self.attr_names = attr_names or attrs self.set_problem(target, inputs=inputs, exclude=exclude) def set_problem(self, target, inputs=None, exclude=()): """ Set (or change) the target and/or inputs. This way, one DataSet can be used multiple ways. inputs, if specified, is a list of attributes, or specify exclude as a list of attributes to not use in inputs. Attributes can be -n .. n, or an attr_name. Also computes the list of possible values, if that wasn't done yet. """ self.target = self.attr_num(target) exclude = list(map(self.attr_num, exclude)) if inputs: self.inputs = remove_all(self.target, inputs) else: self.inputs = [a for a in self.attrs if a != self.target and a not in exclude] if not self.values: self.update_values() self.check_me() def check_me(self): """Check that my fields make sense.""" assert len(self.attr_names) == len(self.attrs) assert self.target in self.attrs assert self.target not in self.inputs assert set(self.inputs).issubset(set(self.attrs)) if self.got_values_flag: # only check if values are provided while initializing DataSet list(map(self.check_example, self.examples)) def add_example(self, example): """Add an example to the list of examples, checking it first.""" self.check_example(example) self.examples.append(example) def check_example(self, example): """Raise ValueError if example has any invalid values.""" if self.values: for a in self.attrs: if example[a] not in self.values[a]: raise ValueError('Bad value {} for attribute {} in {}' .format(example[a], self.attr_names[a], example)) def attr_num(self, attr): """Returns the number used for attr, which can be a name, or -n .. n-1.""" if isinstance(attr, str): return self.attr_names.index(attr) elif attr < 0: return len(self.attrs) + attr else: return attr def update_values(self): self.values = list(map(unique, zip(*self.examples))) def sanitize(self, example): """Return a copy of example, with non-input attributes replaced by None.""" return [attr_i if i in self.inputs else None for i, attr_i in enumerate(example)][:-1] def classes_to_numbers(self, classes=None): """Converts class names to numbers.""" if not classes: # if classes were not given, extract them from values classes = sorted(self.values[self.target]) for item in self.examples: item[self.target] = classes.index(item[self.target]) def remove_examples(self, value=''): """Remove examples that contain given value.""" self.examples = [x for x in self.examples if value not in x] self.update_values() def split_values_by_classes(self): """Split values into buckets according to their class.""" buckets = defaultdict(lambda: []) target_names = self.values[self.target] for v in self.examples: item = [a for a in v if a not in target_names] # remove target from item buckets[v[self.target]].append(item) # add item to bucket of its class return buckets def find_means_and_deviations(self): """ Finds the means and standard deviations of self.dataset. means : a dictionary for each class/target. Holds a list of the means of the features for the class. deviations: a dictionary for each class/target. Holds a list of the sample standard deviations of the features for the class. """ target_names = self.values[self.target] feature_numbers = len(self.inputs) item_buckets = self.split_values_by_classes() means = defaultdict(lambda: [0] * feature_numbers) deviations = defaultdict(lambda: [0] * feature_numbers) for t in target_names: # find all the item feature values for item in class t features = [[] for _ in range(feature_numbers)] for item in item_buckets[t]: for i in range(feature_numbers): features[i].append(item[i]) # calculate means and deviations fo the class for i in range(feature_numbers): means[t][i] = mean(features[i]) deviations[t][i] = stdev(features[i]) return means, deviations def __repr__(self): return '<DataSet({}): {:d} examples, {:d} attributes>'.format(self.name, len(self.examples), len(self.attrs))
In [2]:
iris = DataSet(name="iris")In [3]:
print(iris.examples[0])
print(iris.inputs)[5.1, 3.5, 1.4, 0.2, 'setosa'] [0, 1, 2, 3]
In [4]:
iris2 = DataSet(name="iris",exclude=[1])
print(iris2.inputs)[0, 2, 3]
In [5]:
print(iris.examples[:3])[[5.1, 3.5, 1.4, 0.2, 'setosa'], [4.9, 3.0, 1.4, 0.2, 'setosa'], [4.7, 3.2, 1.3, 0.2, 'setosa']]
In [6]:
print("attrs:", iris.attrs)
print("attrnames (by default same as attrs):", iris.attr_names)
print("target:", iris.target)
print("inputs:", iris.inputs)attrs: [0, 1, 2, 3, 4] attrnames (by default same as attrs): [0, 1, 2, 3, 4] target: 4 inputs: [0, 1, 2, 3]
In [7]:
print(iris.values[0])[4.7, 5.5, 5.0, 4.9, 5.1, 4.6, 5.4, 4.4, 4.8, 4.3, 5.8, 7.0, 7.1, 4.5, 5.9, 5.6, 6.9, 6.5, 6.4, 6.6, 6.0, 6.1, 7.6, 7.4, 7.9, 5.7, 5.3, 5.2, 6.3, 6.7, 6.2, 6.8, 7.3, 7.2, 7.7]
In [8]:
print("name:", iris.name)
print("source:", iris.source)name: iris source:
In [9]:
print(iris.values[iris.target])['virginica', 'versicolor', 'setosa']
In [10]:
print("Sanitized:",iris.sanitize(iris.examples[0]))
print("Original:",iris.examples[0])Sanitized: [5.1, 3.5, 1.4, 0.2] Original: [5.1, 3.5, 1.4, 0.2, 'setosa']
In [11]:
iris2 = DataSet(name="iris")
iris2.remove_examples("virginica")
print(iris2.values[iris2.target])['versicolor', 'setosa']
In [12]:
print("Class of first example:",iris2.examples[0][iris2.target])
iris2.classes_to_numbers()
print("Class of first example:",iris2.examples[0][iris2.target])Class of first example: setosa Class of first example: 0
In [13]:
means, deviations = iris.find_means_and_deviations()
print("Setosa feature means:", means["setosa"])
print("Versicolor mean for first feature:", means["versicolor"][0])
print("Setosa feature deviations:", deviations["setosa"])
print("Virginica deviation for second feature:",deviations["virginica"][1])Setosa feature means: [5.006, 3.418, 1.464, 0.244] Versicolor mean for first feature: 5.936 Setosa feature deviations: [0.3524896872134513, 0.38102439795469095, 0.17351115943644546, 0.10720950308167838] Virginica deviation for second feature: 0.32249663817263746
In [14]:
import matplotlib.pyplot as plt
def show_iris(i=0, j=1, k=2):
"""Plots the iris dataset in a 3D plot.
The three axes are given by i, j and k,
which correspond to three of the four iris features."""
plt.rcParams.update(plt.rcParamsDefault)
fig = plt.figure()
ax = fig.add_subplot(111, projection='3d')
iris = DataSet(name="iris")
buckets = iris.split_values_by_classes()
features = ["Sepal Length", "Sepal Width", "Petal Length", "Petal Width"]
f1, f2, f3 = features[i], features[j], features[k]
a_setosa = [v[i] for v in buckets["setosa"]]
b_setosa = [v[j] for v in buckets["setosa"]]
c_setosa = [v[k] for v in buckets["setosa"]]
a_virginica = [v[i] for v in buckets["virginica"]]
b_virginica = [v[j] for v in buckets["virginica"]]
c_virginica = [v[k] for v in buckets["virginica"]]
a_versicolor = [v[i] for v in buckets["versicolor"]]
b_versicolor = [v[j] for v in buckets["versicolor"]]
c_versicolor = [v[k] for v in buckets["versicolor"]]
for c, m, sl, sw, pl in [('b', 's', a_setosa, b_setosa, c_setosa),
('g', '^', a_virginica, b_virginica, c_virginica),
('r', 'o', a_versicolor, b_versicolor, c_versicolor)]:
ax.scatter(sl, sw, pl, c=c, marker=m)
ax.set_xlabel(f1)
ax.set_ylabel(f2)
ax.set_zlabel(f3)
plt.show()
In [15]:
iris = DataSet(name="iris")
show_iris()
show_iris(0, 1, 3)
show_iris(1, 2, 3)In [16]:
def manhattan_distance(X, Y):
return sum([abs(x - y) for x, y in zip(X, Y)])
distance = manhattan_distance([1,2], [3,4])
print("Manhattan Distance between (1,2) and (3,4) is", distance)Manhattan Distance between (1,2) and (3,4) is 4
In [17]:
def euclidean_distance(X, Y):
return math.sqrt(sum([(x - y)**2 for x, y in zip(X,Y)]))
distance = euclidean_distance([1,2], [3,4])
print("Euclidean Distance between (1,2) and (3,4) is", distance)Euclidean Distance between (1,2) and (3,4) is 2.8284271247461903
In [18]:
def hamming_distance(X, Y):
return sum(x != y for x, y in zip(X, Y))
distance = hamming_distance(['a','b','c'], ['a','b','b'])
print("Hamming Distance between 'abc' and 'abb' is", distance)Hamming Distance between 'abc' and 'abb' is 1
In [19]:
def mean_boolean_error(X, Y):
return mean(int(x != y) for x, y in zip(X, Y))
distance = mean_boolean_error([1,2,3], [1,4,5])
print("Mean Boolean Error Distance between (1,2,3) and (1,4,5) is", distance)Mean Boolean Error Distance between (1,2,3) and (1,4,5) is 0.6666666666666666
In [20]:
def mean_error(X, Y):
return mean([abs(x - y) for x, y in zip(X, Y)])
distance = mean_error([1,0,5], [3,10,5])
print("Mean Error Distance between (1,0,5) and (3,10,5) is", distance)Mean Error Distance between (1,0,5) and (3,10,5) is 4
In [21]:
def ms_error(X, Y):
return mean([(x - y)**2 for x, y in zip(X, Y)])
distance = ms_error([1,0,5], [3,10,5])
print("Mean Square Distance between (1,0,5) and (3,10,5) is", distance)Mean Square Distance between (1,0,5) and (3,10,5) is 34.666666666666664
In [22]:
def rms_error(X, Y):
return math.sqrt(ms_error(X, Y))
distance = rms_error([1,0,5], [3,10,5])
print("Root of Mean Error Distance between (1,0,5) and (3,10,5) is", distance)Root of Mean Error Distance between (1,0,5) and (3,10,5) is 5.887840577551898