The oneR algorithm returns a rule list that splits on only one (usually continuous) feature It works by building a greedy rule list using only one feature at a time, and then returning the rule list with the highest accuracy
Expand source code
'''The oneR algorithm returns a rule list that splits on only one (usually continuous) feature
It works by building a greedy rule list using only one feature at a time, and then returning
the rule list with the highest accuracy
'''
import numpy as np
from imodels import GreedyRuleListClassifier
from imodels.util.progress import progress_iter
from imodels.util.arguments import check_binary_target, check_fit_arguments, check_two_classes
class OneRClassifier(GreedyRuleListClassifier):
def __init__(self, max_depth=5, class_weight=None, criterion='gini',
verbose=0):
self.max_depth = max_depth
self.class_weight = class_weight
self.criterion = criterion
self.verbose = verbose
self._estimator_type = 'classifier'
def fit(self, X, y, feature_names=None):
"""Fit oneR
"""
check_binary_target(self, y)
check_two_classes(self, y)
X, y, feature_names = check_fit_arguments(self, X, y, feature_names)
ms = []
accs = np.zeros(X.shape[1])
for col_idx in progress_iter(range(X.shape[1]), verbose=self.verbose,
desc='scoring features'):
x = X[:, col_idx].reshape(-1, 1)
m = GreedyRuleListClassifier(max_depth=self.max_depth, class_weight=self.class_weight,
criterion=self.criterion)
feat_names_single = [self.feature_names_[col_idx]]
m.fit(x, y, feature_names=feat_names_single)
accs[col_idx] = np.mean(m.predict(x) == y)
ms.append(m)
# print('acc', feat_names_single[0], f'{accs[col_idx]:0.2f}')
col_idx_best = np.argmax(accs)
self.rules_ = ms[col_idx_best].rules_
self.complexity_ = len(self.rules_)
# need to adjust index_col since was fitted with only 1 col
for rule in self.rules_:
if 'index_col' in rule:
rule['index_col'] += col_idx_best
self.depth = len(self.rules_)
return self
Classes
class OneRClassifier (max_depth=5, class_weight=None, criterion='gini', verbose=0)-
Base class for all estimators in scikit-learn.
Inheriting from this class provides default implementations of:
- setting and getting parameters used by
GridSearchCVand friends; - textual and HTML representation displayed in terminals and IDEs;
- estimator serialization;
- parameters validation;
- data validation;
- feature names validation.
Read more in the :ref:
User Guide <rolling_your_own_estimator>.Notes
All estimators should specify all the parameters that can be set at the class level in their
__init__as explicit keyword arguments (no*argsor**kwargs).Examples
>>> import numpy as np >>> from sklearn.base import BaseEstimator >>> class MyEstimator(BaseEstimator): ... def __init__(self, *, param=1): ... self.param = param ... def fit(self, X, y=None): ... self.is_fitted_ = True ... return self ... def predict(self, X): ... return np.full(shape=X.shape[0], fill_value=self.param) >>> estimator = MyEstimator(param=2) >>> estimator.get_params() {'param': 2} >>> X = np.array([[1, 2], [2, 3], [3, 4]]) >>> y = np.array([1, 0, 1]) >>> estimator.fit(X, y).predict(X) array([2, 2, 2]) >>> estimator.set_params(param=3).fit(X, y).predict(X) array([3, 3, 3])Parameters
max_depth- Maximum depth the list can achieve
class_weight:dict, 'balanced'orNone- Weights of the classes when choosing each split, keyed by the original labels; passed on to the sklearn stump that finds the split
criterion:str- Criterion used to split 'gini', 'entropy', or 'log_loss'
Expand source code
class OneRClassifier(GreedyRuleListClassifier): def __init__(self, max_depth=5, class_weight=None, criterion='gini', verbose=0): self.max_depth = max_depth self.class_weight = class_weight self.criterion = criterion self.verbose = verbose self._estimator_type = 'classifier' def fit(self, X, y, feature_names=None): """Fit oneR """ check_binary_target(self, y) check_two_classes(self, y) X, y, feature_names = check_fit_arguments(self, X, y, feature_names) ms = [] accs = np.zeros(X.shape[1]) for col_idx in progress_iter(range(X.shape[1]), verbose=self.verbose, desc='scoring features'): x = X[:, col_idx].reshape(-1, 1) m = GreedyRuleListClassifier(max_depth=self.max_depth, class_weight=self.class_weight, criterion=self.criterion) feat_names_single = [self.feature_names_[col_idx]] m.fit(x, y, feature_names=feat_names_single) accs[col_idx] = np.mean(m.predict(x) == y) ms.append(m) # print('acc', feat_names_single[0], f'{accs[col_idx]:0.2f}') col_idx_best = np.argmax(accs) self.rules_ = ms[col_idx_best].rules_ self.complexity_ = len(self.rules_) # need to adjust index_col since was fitted with only 1 col for rule in self.rules_: if 'index_col' in rule: rule['index_col'] += col_idx_best self.depth = len(self.rules_) return selfAncestors
- GreedyRuleListClassifier
- sklearn.base.BaseEstimator
- sklearn.utils._estimator_html_repr._HTMLDocumentationLinkMixin
- sklearn.utils._metadata_requests._MetadataRequester
- RuleList
- RulesMixin
- TextMixin
- sklearn.base.ClassifierMixin
Methods
def fit(self, X, y, feature_names=None)-
Fit oneR
Expand source code
def fit(self, X, y, feature_names=None): """Fit oneR """ check_binary_target(self, y) check_two_classes(self, y) X, y, feature_names = check_fit_arguments(self, X, y, feature_names) ms = [] accs = np.zeros(X.shape[1]) for col_idx in progress_iter(range(X.shape[1]), verbose=self.verbose, desc='scoring features'): x = X[:, col_idx].reshape(-1, 1) m = GreedyRuleListClassifier(max_depth=self.max_depth, class_weight=self.class_weight, criterion=self.criterion) feat_names_single = [self.feature_names_[col_idx]] m.fit(x, y, feature_names=feat_names_single) accs[col_idx] = np.mean(m.predict(x) == y) ms.append(m) # print('acc', feat_names_single[0], f'{accs[col_idx]:0.2f}') col_idx_best = np.argmax(accs) self.rules_ = ms[col_idx_best].rules_ self.complexity_ = len(self.rules_) # need to adjust index_col since was fitted with only 1 col for rule in self.rules_: if 'index_col' in rule: rule['index_col'] += col_idx_best self.depth = len(self.rules_) return self
Inherited members
- setting and getting parameters used by