sklearn_model_validation: stacking_ensembles.py comparison

comparison stacking_ensembles.py @ 19:efbec977a47d draft

planemo upload for repository https://github.com/bgruening/galaxytools/tree/master/tools/sklearn commit 60f0fbc0eafd7c11bc60fb6c77f2937782efd8a9-dirty

author	bgruening
date	Fri, 09 Aug 2019 07:26:09 -0400
parents	cf9aa11b91c8
children	887e0aaa482e

comparison

equal deleted inserted replaced

-:492d34a75de6
+:efbec977a47d
 import argparse
+import ast
 import json
+import mlxtend.regressor
+import mlxtend.classifier
 import pandas as pd
 import pickle
-import xgboost
+import sklearn
+import sys
 import warnings
-from sklearn import (cluster, compose, decomposition, ensemble,
+from sklearn import ensemble
-feature_extraction, feature_selection,
-gaussian_process, kernel_approximation, metrics,
-model_selection, naive_bayes, neighbors,
-pipeline, preprocessing, svm, linear_model,
-tree, discriminant_analysis)
-from sklearn.model_selection._split import check_cv
-from feature_selectors import (DyRFE, DyRFECV,
-MyPipeline, MyimbPipeline)
-from iraps_classifier import (IRAPSCore, IRAPSClassifier,
-BinarizeTargetClassifier,
-BinarizeTargetRegressor)
-from preprocessors import Z_RandomOverSampler
-from utils import load_model, get_cv, get_estimator, get_search_params
-from mlxtend.regressor import StackingCVRegressor, StackingRegressor
+from galaxy_ml.utils import (load_model, get_cv, get_estimator,
-from mlxtend.classifier import StackingCVClassifier, StackingClassifier
+get_search_params)
 warnings.filterwarnings('ignore')
 N_JOBS = int(__import__('os').environ.get('GALAXY_SLOTS', 1))
 File path for params output
 """
 with open(inputs_path, 'r') as param_handler:
 params = json.load(param_handler)
+estimator_type = params['algo_selection']['estimator_type']
+# get base estimators
 base_estimators = []
 for idx, base_file in enumerate(base_paths.split(',')):
 if base_file and base_file != 'None':
 with open(base_file, 'rb') as handler:
 model = load_model(handler)
 else:
 estimator_json = (params['base_est_builder'][idx]
 ['estimator_selector'])
 model = get_estimator(estimator_json)
-base_estimators.append(model)
-if meta_path:
+if estimator_type.startswith('sklearn'):
-with open(meta_path, 'rb') as f:
+named = model.__class__.__name__.lower()
-meta_estimator = load_model(f)
+named = 'base_%d_%s' % (idx, named)
-else:
+base_estimators.append((named, model))
-estimator_json = params['meta_estimator']['estimator_selector']
+else:
-meta_estimator = get_estimator(estimator_json)
+base_estimators.append(model)
+# get meta estimator, if applicable
+if estimator_type.startswith('mlxtend'):
+if meta_path:
+with open(meta_path, 'rb') as f:
+meta_estimator = load_model(f)
+else:
+estimator_json = (params['algo_selection']
+['meta_estimator']['estimator_selector'])
+meta_estimator = get_estimator(estimator_json)
 options = params['algo_selection']['options']
 cv_selector = options.pop('cv_selector', None)
 if cv_selector:
 splitter, groups = get_cv(cv_selector)
 options['cv'] = splitter
 # set n_jobs
 options['n_jobs'] = N_JOBS
-if params['algo_selection']['estimator_type'] == 'StackingCVClassifier':
+weights = options.pop('weights', None)
-ensemble_estimator = StackingCVClassifier(
+if weights:
+options['weights'] = ast.literal_eval(weights)
+mod_and_name = estimator_type.split('_')
+mod = sys.modules[mod_and_name[0]]
+klass = getattr(mod, mod_and_name[1])
+if estimator_type.startswith('sklearn'):
+options['n_jobs'] = N_JOBS
+ensemble_estimator = klass(base_estimators, **options)
+elif mod == mlxtend.classifier:
+ensemble_estimator = klass(
 classifiers=base_estimators,
 meta_classifier=meta_estimator,
 **options)
-elif params['algo_selection']['estimator_type'] == 'StackingClassifier':
-ensemble_estimator = StackingClassifier(
-classifiers=base_estimators,
-meta_classifier=meta_estimator,
-**options)
-elif params['algo_selection']['estimator_type'] == 'StackingCVRegressor':
-ensemble_estimator = StackingCVRegressor(
-regressors=base_estimators,
-meta_regressor=meta_estimator,
-**options)
 else:
-ensemble_estimator = StackingRegressor(
+ensemble_estimator = klass(
 regressors=base_estimators,
 meta_regressor=meta_estimator,
 **options)
 print(ensemble_estimator)

Mercurial > repos > bgruening > sklearn_model_validation

comparison stacking_ensembles.py @ 19:efbec977a47d draft