From bb069153bebd890395661cc7165a4f71fd85cb65 Mon Sep 17 00:00:00 2001 From: Xi Xu Date: Wed, 16 Oct 2024 20:12:44 +0800 Subject: [PATCH] something changes --- 02_decision_tree/DecisionTreeLearner_1.py | 140 ------------------ 02_decision_tree/DecisionTreeLearner_2.py | 132 ----------------- 02_decision_tree/DecisionTreeLearner_3.py | 121 --------------- 02_decision_tree/DecisionTreeLearner_4.py | 34 ----- ...earRegression_1.py => LinearRegression.py} | 0 03_linear_regression/LinearRegression_2.py | 131 ---------------- 6 files changed, 558 deletions(-) delete mode 100644 02_decision_tree/DecisionTreeLearner_1.py delete mode 100644 02_decision_tree/DecisionTreeLearner_2.py delete mode 100644 02_decision_tree/DecisionTreeLearner_3.py delete mode 100644 02_decision_tree/DecisionTreeLearner_4.py rename 03_linear_regression/{LinearRegression_1.py => LinearRegression.py} (100%) delete mode 100644 03_linear_regression/LinearRegression_2.py diff --git a/02_decision_tree/DecisionTreeLearner_1.py b/02_decision_tree/DecisionTreeLearner_1.py deleted file mode 100644 index 32da42d..0000000 --- a/02_decision_tree/DecisionTreeLearner_1.py +++ /dev/null @@ -1,140 +0,0 @@ -import sys -sys.path.insert(1, '../') -from utils.utils import * - - -class DecisionFork: - """ - A fork of a decision tree holds an attribute to test, and a dict - of branches, one for each of the attribute's values. - """ - - def __init__(self, attr, attr_name=None, default_child=None, branches=None): - """Initialize by saying what attribute this node tests.""" - self.attr = attr - self.attr_name = attr_name or attr - self.default_child = default_child - self.branches = branches or {} - - def __call__(self, example): - """Given an example, classify it using the attribute and the branches.""" - attr_val = example[self.attr] - if attr_val in self.branches: - return self.branches[attr_val](example) - else: - # return default class when attribute is unknown - return self.default_child(example) - - def add(self, val, subtree): - """Add a branch. If self.attr = val, go to the given subtree.""" - self.branches[val] = subtree - - def display(self, indent=0): - name = self.attr_name - print('Test', name) - for (val, subtree) in self.branches.items(): - print(' ' * 4 * indent, name, '=', val, '==>', end=' ') - subtree.display(indent + 1) - - def __repr__(self): - return 'DecisionFork({0!r}, {1!r}, {2!r})'.format(self.attr, self.attr_name, self.branches) - - -class DecisionLeaf: - """A leaf of a decision tree holds just a result.""" - - def __init__(self, result): - self.result = result - - def __call__(self, example): - return self.result - - def display(self): - print('RESULT =', self.result) - - def __repr__(self): - return repr(self.result) - - -class DecisionTreeLearner: - """DecisionTreeLearner: based on information gain""" - - def __init__(self, dataset): - self.dataset = dataset - self.tree = self.decision_tree_learning(dataset.examples, dataset.inputs) - - def decision_tree_learning(self, examples, attrs, parent_examples=()): - if len(examples) == 0: - return self.plurality_value(parent_examples) - if self.all_same_class(examples): - return DecisionLeaf(examples[0][self.dataset.target]) - if len(attrs) == 0: - return self.plurality_value(examples) - A = self.choose_attribute(attrs, examples) - tree = DecisionFork(A, self.dataset.attr_names[A], self.plurality_value(examples)) - for (v_k, exs) in self.split_by(A, examples): - subtree = self.decision_tree_learning(exs, remove_all(A, attrs), examples) - tree.add(v_k, subtree) - return tree - - def plurality_value(self, examples): - """ - Return the most popular target value for this set of examples. - (If target is binary, this is the majority; otherwise plurality). - """ - popular = argmax_random_tie(self.dataset.values[self.dataset.target], - key=lambda v: self.count(self.dataset.target, v, examples)) - return DecisionLeaf(popular) - - def count(self, attr, val, examples): - """Count the number of examples that have example[attr] = val.""" - return sum(e[attr] == val for e in examples) - - def all_same_class(self, examples): - """Are all these examples in the same target class?""" - class0 = examples[0][self.dataset.target] - return all(e[self.dataset.target] == class0 for e in examples) - - def choose_attribute(self, attrs, examples): - """Choose the attribute with the highest information gain.""" - return argmax_random_tie(attrs, key=lambda a: self.information_gain(a, examples)) - - def information_gain(self, attr, examples): - """Return the expected reduction in entropy from splitting by attr.""" - - def I(examples): - return information_content([self.count(self.dataset.target, v, examples) - for v in self.dataset.values[self.dataset.target]]) - - n = len(examples) - remainder = sum((len(examples_i) / n) * I(examples_i) - for (v, examples_i) in self.split_by(attr, examples)) - return I(examples) - remainder - - def split_by(self, attr, examples): - """Return a list of (val, examples) pairs for each val of attr.""" - return [(v, [e for e in examples if e[attr] == v]) for v in self.dataset.values[attr]] - - def predict(self, x): - return self.tree(x) - - def __call__(self, x): - return self.predict(x) - - -def information_content(values): - """Number of bits to represent the probability distribution in values.""" - raise NotImplementedError - - -if __name__ == "__main__": - from utils.dataset4learners import * - - iris = DataSet(name="iris") - DTL = DecisionTreeLearner(iris) - print(f'DTL.predict([5, 3, 1, 0.1]): {DTL.predict([5, 3, 1, 0.1])}') - assert DTL.predict([5, 3, 1, 0.1]) == 'setosa' - print(f'DTL.predict([6, 5, 3, 1.5]): {DTL.predict([6, 5, 3, 1.5])}') - assert DTL.predict([6, 5, 3, 1.5]) == 'versicolor' - print(f'DTL.predict([7.5, 4, 6, 2]): {DTL.predict([7.5, 4, 6, 2])}') - assert DTL.predict([7.5, 4, 6, 2]) == 'virginica' diff --git a/02_decision_tree/DecisionTreeLearner_2.py b/02_decision_tree/DecisionTreeLearner_2.py deleted file mode 100644 index 49d4448..0000000 --- a/02_decision_tree/DecisionTreeLearner_2.py +++ /dev/null @@ -1,132 +0,0 @@ -import sys -sys.path.insert(1, '../') -from utils.utils import * - - -class DecisionFork: - """ - A fork of a decision tree holds an attribute to test, and a dict - of branches, one for each of the attribute's values. - """ - - def __init__(self, attr, attr_name=None, default_child=None, branches=None): - """Initialize by saying what attribute this node tests.""" - self.attr = attr - self.attr_name = attr_name or attr - self.default_child = default_child - self.branches = branches or {} - - def __call__(self, example): - """Given an example, classify it using the attribute and the branches.""" - attr_val = example[self.attr] - if attr_val in self.branches: - return self.branches[attr_val](example) - else: - # return default class when attribute is unknown - return self.default_child(example) - - def add(self, val, subtree): - """Add a branch. If self.attr = val, go to the given subtree.""" - self.branches[val] = subtree - - def display(self, indent=0): - name = self.attr_name - print('Test', name) - for (val, subtree) in self.branches.items(): - print(' ' * 4 * indent, name, '=', val, '==>', end=' ') - subtree.display(indent + 1) - - def __repr__(self): - return 'DecisionFork({0!r}, {1!r}, {2!r})'.format(self.attr, self.attr_name, self.branches) - - -class DecisionLeaf: - """A leaf of a decision tree holds just a result.""" - - def __init__(self, result): - self.result = result - - def __call__(self, example): - return self.result - - def display(self): - print('RESULT =', self.result) - - def __repr__(self): - return repr(self.result) - - -class DecisionTreeLearner: - """DecisionTreeLearner: based on information gain""" - - def __init__(self, dataset): - self.dataset = dataset - self.tree = self.decision_tree_learning(dataset.examples, dataset.inputs) - - def decision_tree_learning(self, examples, attrs, parent_examples=()): - if len(examples) == 0: - return self.plurality_value(parent_examples) - if self.all_same_class(examples): - return DecisionLeaf(examples[0][self.dataset.target]) - if len(attrs) == 0: - return self.plurality_value(examples) - A = self.choose_attribute(attrs, examples) - tree = DecisionFork(A, self.dataset.attr_names[A], self.plurality_value(examples)) - for (v_k, exs) in self.split_by(A, examples): - subtree = self.decision_tree_learning(exs, remove_all(A, attrs), examples) - tree.add(v_k, subtree) - return tree - - def plurality_value(self, examples): - """ - Return the most popular target value for this set of examples. - (If target is binary, this is the majority; otherwise plurality). - """ - popular = argmax_random_tie(self.dataset.values[self.dataset.target], - key=lambda v: self.count(self.dataset.target, v, examples)) - return DecisionLeaf(popular) - - def count(self, attr, val, examples): - """Count the number of examples that have example[attr] = val.""" - return sum(e[attr] == val for e in examples) - - def all_same_class(self, examples): - """Are all these examples in the same target class?""" - class0 = examples[0][self.dataset.target] - return all(e[self.dataset.target] == class0 for e in examples) - - def choose_attribute(self, attrs, examples): - """Choose the attribute with the highest information gain.""" - return argmax_random_tie(attrs, key=lambda a: self.information_gain(a, examples)) - - def information_gain(self, attr, examples): - """Return the expected reduction in entropy from splitting by attr.""" - raise NotImplementedError - - def split_by(self, attr, examples): - """Return a list of (val, examples) pairs for each val of attr.""" - return [(v, [e for e in examples if e[attr] == v]) for v in self.dataset.values[attr]] - - def predict(self, x): - return self.tree(x) - - def __call__(self, x): - return self.predict(x) - - -def information_content(values): - """Number of bits to represent the probability distribution in values.""" - raise NotImplementedError - - -if __name__ == "__main__": - from utils.dataset4learners import * - - iris = DataSet(name="iris") - DTL = DecisionTreeLearner(iris) - print(f'DTL.predict([5, 3, 1, 0.1]): {DTL.predict([5, 3, 1, 0.1])}') - assert DTL.predict([5, 3, 1, 0.1]) == 'setosa' - print(f'DTL.predict([6, 5, 3, 1.5]): {DTL.predict([6, 5, 3, 1.5])}') - assert DTL.predict([6, 5, 3, 1.5]) == 'versicolor' - print(f'DTL.predict([7.5, 4, 6, 2]): {DTL.predict([7.5, 4, 6, 2])}') - assert DTL.predict([7.5, 4, 6, 2]) == 'virginica' diff --git a/02_decision_tree/DecisionTreeLearner_3.py b/02_decision_tree/DecisionTreeLearner_3.py deleted file mode 100644 index 568fc24..0000000 --- a/02_decision_tree/DecisionTreeLearner_3.py +++ /dev/null @@ -1,121 +0,0 @@ -import sys -sys.path.insert(1, '../') -from utils.utils import * - - -class DecisionFork: - """ - A fork of a decision tree holds an attribute to test, and a dict - of branches, one for each of the attribute's values. - """ - - def __init__(self, attr, attr_name=None, default_child=None, branches=None): - """Initialize by saying what attribute this node tests.""" - self.attr = attr - self.attr_name = attr_name or attr - self.default_child = default_child - self.branches = branches or {} - - def __call__(self, example): - """Given an example, classify it using the attribute and the branches.""" - attr_val = example[self.attr] - if attr_val in self.branches: - return self.branches[attr_val](example) - else: - # return default class when attribute is unknown - return self.default_child(example) - - def add(self, val, subtree): - """Add a branch. If self.attr = val, go to the given subtree.""" - self.branches[val] = subtree - - def display(self, indent=0): - name = self.attr_name - print('Test', name) - for (val, subtree) in self.branches.items(): - print(' ' * 4 * indent, name, '=', val, '==>', end=' ') - subtree.display(indent + 1) - - def __repr__(self): - return 'DecisionFork({0!r}, {1!r}, {2!r})'.format(self.attr, self.attr_name, self.branches) - - -class DecisionLeaf: - """A leaf of a decision tree holds just a result.""" - - def __init__(self, result): - self.result = result - - def __call__(self, example): - return self.result - - def display(self): - print('RESULT =', self.result) - - def __repr__(self): - return repr(self.result) - - -class DecisionTreeLearner: - """DecisionTreeLearner: based on information gain""" - - def __init__(self, dataset): - self.dataset = dataset - self.tree = self.decision_tree_learning(dataset.examples, dataset.inputs) - - def decision_tree_learning(self, examples, attrs, parent_examples=()): - raise NotImplementedError - - def plurality_value(self, examples): - """ - Return the most popular target value for this set of examples. - (If target is binary, this is the majority; otherwise plurality). - """ - popular = argmax_random_tie(self.dataset.values[self.dataset.target], - key=lambda v: self.count(self.dataset.target, v, examples)) - return DecisionLeaf(popular) - - def count(self, attr, val, examples): - """Count the number of examples that have example[attr] = val.""" - return sum(e[attr] == val for e in examples) - - def all_same_class(self, examples): - """Are all these examples in the same target class?""" - class0 = examples[0][self.dataset.target] - return all(e[self.dataset.target] == class0 for e in examples) - - def choose_attribute(self, attrs, examples): - """Choose the attribute with the highest information gain.""" - return argmax_random_tie(attrs, key=lambda a: self.information_gain(a, examples)) - - def information_gain(self, attr, examples): - """Return the expected reduction in entropy from splitting by attr.""" - raise NotImplementedError - - def split_by(self, attr, examples): - """Return a list of (val, examples) pairs for each val of attr.""" - return [(v, [e for e in examples if e[attr] == v]) for v in self.dataset.values[attr]] - - def predict(self, x): - return self.tree(x) - - def __call__(self, x): - return self.predict(x) - - -def information_content(values): - """Number of bits to represent the probability distribution in values.""" - raise NotImplementedError - - -if __name__ == "__main__": - from utils.dataset4learners import * - - iris = DataSet(name="iris") - DTL = DecisionTreeLearner(iris) - print(f'DTL.predict([5, 3, 1, 0.1]): {DTL.predict([5, 3, 1, 0.1])}') - assert DTL.predict([5, 3, 1, 0.1]) == 'setosa' - print(f'DTL.predict([6, 5, 3, 1.5]): {DTL.predict([6, 5, 3, 1.5])}') - assert DTL.predict([6, 5, 3, 1.5]) == 'versicolor' - print(f'DTL.predict([7.5, 4, 6, 2]): {DTL.predict([7.5, 4, 6, 2])}') - assert DTL.predict([7.5, 4, 6, 2]) == 'virginica' diff --git a/02_decision_tree/DecisionTreeLearner_4.py b/02_decision_tree/DecisionTreeLearner_4.py deleted file mode 100644 index c820f18..0000000 --- a/02_decision_tree/DecisionTreeLearner_4.py +++ /dev/null @@ -1,34 +0,0 @@ -import sys -sys.path.insert(1, '../') -from utils.utils import * - - -class DecisionFork: - """ - A fork of a decision tree holds an attribute to test, and a dict - of branches, one for each of the attribute's values. - """ - raise NotImplementedError - - -class DecisionLeaf: - """A leaf of a decision tree holds just a result.""" - raise NotImplementedError - - -class DecisionTreeLearner: - """DecisionTreeLearner: based on information gain""" - raise NotImplementedError - - -if __name__ == "__main__": - from utils.dataset4learners import * - - iris = DataSet(name="iris") - DTL = DecisionTreeLearner(iris) - print(f'DTL.predict([5, 3, 1, 0.1]): {DTL.predict([5, 3, 1, 0.1])}') - assert DTL.predict([5, 3, 1, 0.1]) == 'setosa' - print(f'DTL.predict([6, 5, 3, 1.5]): {DTL.predict([6, 5, 3, 1.5])}') - assert DTL.predict([6, 5, 3, 1.5]) == 'versicolor' - print(f'DTL.predict([7.5, 4, 6, 2]): {DTL.predict([7.5, 4, 6, 2])}') - assert DTL.predict([7.5, 4, 6, 2]) == 'virginica' diff --git a/03_linear_regression/LinearRegression_1.py b/03_linear_regression/LinearRegression.py similarity index 100% rename from 03_linear_regression/LinearRegression_1.py rename to 03_linear_regression/LinearRegression.py diff --git a/03_linear_regression/LinearRegression_2.py b/03_linear_regression/LinearRegression_2.py deleted file mode 100644 index 58065fe..0000000 --- a/03_linear_regression/LinearRegression_2.py +++ /dev/null @@ -1,131 +0,0 @@ -import numpy as np - - -class LinearRegression: - def solve(self, lr, nepoch): - raise NotImplementedError - - -class LinearRegressionLS(LinearRegression): - """ - solve linear regression problem via least squares - """ - X: np.array - Y: np.array - w: np.array - - def __init__(self, ylist): - num_data = len(ylist) - self.X = self.homogeneous([x for x in range(num_data)]) - self.Y = np.array(ylist).reshape(num_data, 1) - self.w = np.random.rand(2) - - def homogeneous(self, xlist): - """ build homogeneous coordinates """ - raise NotImplementedError - - def linout(self, xlist): - """ linear output for given data """ - raise NotImplementedError - - def loss_sq(self, X, Y): - """ loss function: (half) sum of square errors """ - raise NotImplementedError - - def solve(self, lr, nepoch): - """ form normal equation """ - raise NotImplementedError - - -class LinearRegressionGD1(LinearRegression): - """ - solve linear regression problem via gradient descent, - using single weight vector: homogeneous coordinates - """ - X: np.array - Y: np.array - w: np.array - - def __init__(self, ylist): - num_data = len(ylist) - self.X = self.homogeneous([x for x in range(num_data)]) - self.Y = np.array(ylist).reshape(num_data, 1) - self.w = np.random.rand(2) - - def homogeneous(self, xlist): - """ build homogeneous coordinates """ - raise NotImplementedError - - def linout(self, xlist): - """ linear output for given data """ - raise NotImplementedError - - def loss_sq(self, X, Y): - """ loss function: (half) sum of square errors """ - raise NotImplementedError - - def gd(self, lr): - """ gradient descent update """ - raise NotImplementedError - - def solve(self, lr, nepoch): - """ iterative solver """ - for epoch in range(num_epochs): - self.gd(lr) - print(f'epoch {epoch + 1}, loss {self.loss_sq(self.X, self.Y)}') - - -class LinearRegressionGD2(LinearRegression): - """ - solve linear regression problem via gradient descent, - using two weights: w, b - """ - X: np.array - Y: np.array - w: np.array - b: np.array - - def __init__(self, ylist): - num_data = len(ylist) - self.X = np.array([x for x in range(num_data)]).reshape(-1, 1) - self.Y = np.array(ylist).reshape(num_data, 1) - self.w = np.random.rand(1) - self.b = np.min(ylist) - - def linout(self, xlist): - """ linear output for given data """ - raise NotImplementedError - - def loss_sq(self, X, Y): - """ loss function: (half) sum of square errors """ - raise NotImplementedError - - def gd(self, lr): - """ gradient descent update """ - raise NotImplementedError - - def solve(self, lr, nepoch): - """ iterative solver """ - for epoch in range(num_epochs): - self.gd(lr) - print(f'epoch {epoch + 1}, loss {self.loss_sq(self.X, self.Y)}') - - -if __name__ == "__main__": - hp = [14213, 13448, 13870, 16192, 16415, 21501, 25910, 24866, 28981, 32926, 36741, 40974] - hp = [x / 10000. for x in hp] - - lr = 0.001 - num_epochs = 10 - - ls = LinearRegressionLS(hp) - ls.solve(lr, num_epochs) - print(f'next prediction (LS): {ls.linout([len(hp)])}') - - gd1 = LinearRegressionGD1(hp) - gd1.solve(lr, num_epochs) - print(f'next prediction year (GD1): {gd1.linout([len(hp)])}') - - gd2 = LinearRegressionGD1(hp) - gd2.solve(lr, num_epochs) - print(f'next prediction year (GD2): {gd2.linout([len(hp)])}')