From 926eab10951e3728619de360c5a697ac775e328b Mon Sep 17 00:00:00 2001 From: Matthias Feurer Date: Tue, 1 Oct 2019 13:16:15 +0200 Subject: [PATCH 1/3] Add example * adds example for Feurer et al. (2015) * removes the stub for Fusi et al. (2018) as they actually perform the same task. I can't create an example, though, as they used regression datasets for classification (and OpenML by now forbids creating such tasks). --- .../40_paper/2015_neurips_feurer_example.py | 78 ++++++++++++++++++- .../40_paper/2018_neurips_fusi_example.py | 17 ---- 2 files changed, 75 insertions(+), 20 deletions(-) delete mode 100644 examples/40_paper/2018_neurips_fusi_example.py diff --git a/examples/40_paper/2015_neurips_feurer_example.py b/examples/40_paper/2015_neurips_feurer_example.py index 106d120df..5bc7a3b32 100644 --- a/examples/40_paper/2015_neurips_feurer_example.py +++ b/examples/40_paper/2015_neurips_feurer_example.py @@ -10,9 +10,81 @@ ~~~~~~~~~~~ | Efficient and Robust Automated Machine Learning -| Matthias Feurer, Aaron Klein, Katharina Eggensperger, Jost Springenberg, Manuel Blum and Frank Hutter +| Matthias Feurer, Aaron Klein, Katharina Eggensperger, Jost Springenberg, Manuel Blum and Frank Hutter # noqa F401 | In *Advances in Neural Information Processing Systems 28*, 2015 | Available at http://papers.nips.cc/paper/5872-efficient-and-robust-automated-machine-learning.pdf - -This is currently a placeholder. """ + +import pandas as pd + +import openml + +#################################################################################################### +# List of dataset IDs given in the supplementary material of Feurer et al.: +# https://papers.nips.cc/paper/5872-efficient-and-robust-automated-machine-learning-supplemental.zip +dataset_ids = [ + 3, 6, 12, 14, 16, 18, 21, 22, 23, 24, 26, 28, 30, 31, 32, 36, 38, 44, 46, + 57, 60, 179, 180, 181, 182, 184, 185, 273, 293, 300, 351, 354, 357, 389, + 390, 391, 392, 393, 395, 396, 398, 399, 401, 554, 679, 715, 718, 720, 722, + 723, 727, 728, 734, 735, 737, 740, 741, 743, 751, 752, 761, 772, 797, 799, + 803, 806, 807, 813, 816, 819, 821, 822, 823, 833, 837, 843, 845, 846, 847, + 849, 866, 871, 881, 897, 901, 903, 904, 910, 912, 913, 914, 917, 923, 930, + 934, 953, 958, 959, 962, 966, 971, 976, 977, 978, 979, 980, 991, 993, 995, + 1000, 1002, 1018, 1019, 1020, 1021, 1036, 1040, 1041, 1049, 1050, 1053, + 1056, 1067, 1068, 1069, 1111, 1112, 1114, 1116, 1119, 1120, 1128, 1130, + 1134, 1138, 1139, 1142, 1146, 1161, 1166, +] + +#################################################################################################### +# The dataset IDs could be used directly to load the dataset and split the data into a training +# and a test set. However, to be reproducible, we will first obtain the respective tasks from +# OpenML, which define both the target feature and the train/test split. +# +# .. note:: +# Please do not use datasets but the respective tasks as basis for a paper. + +# active tasks +tasks_a = openml.tasks.list_tasks( + task_type_id=1, + status='active', + output_format='dataframe', +) + +# query only those with holdout as the resampling startegy. +tasks_a = tasks_a[(tasks_a.estimation_procedure == "33% Holdout set")] + +# deactivated tasks +tasks_d = openml.tasks.list_tasks( + task_type_id=1, + status='deactivated', + output_format='dataframe', +) +tasks_d = tasks_d[(tasks_d.estimation_procedure == "33% Holdout set")] + +tasks = pd.concat([tasks_a, tasks_d], sort=True) + +task_ids = [] +for did in dataset_ids: + tasks_ = list(tasks.query("did == {}".format(did)).tid) + if len(tasks_) >= 1: # if there are more than one task, take the lowest one. + task_id = min(tasks_) + else: + raise ValueError(did) + + # Optional - Check that the task has the same target attribute as the + # dataset default target attribute + # (disabled for this example as it needs to run fast to be rendered online) + # task = openml.tasks.get_task(task_id) + # dataset = task.get_dataset() + # if task.target_name != dataset.default_target_attribute: + # raise ValueError( + # (task.target_name, dataset.default_target_attribute) + # ) + + task_ids.append(task_id) + +assert len(task_ids) == 140 +task_ids.sort() + +# These are the tasks to work with: +print(task_ids) diff --git a/examples/40_paper/2018_neurips_fusi_example.py b/examples/40_paper/2018_neurips_fusi_example.py deleted file mode 100644 index 656f617fa..000000000 --- a/examples/40_paper/2018_neurips_fusi_example.py +++ /dev/null @@ -1,17 +0,0 @@ -""" -Fusi et al. (2018) -================== - -A tutorial on how to get the datasets used in the paper introducing *Probabilistic Matrix -Factorization for Automated Machine Learning* by Fusi et al.. - -Publication -~~~~~~~~~~~ - -| Probabilistic Matrix Factorization for Automated Machine Learning -| Nicolo Fusi and Rishit Sheth and Melih Elibol -| In *Advances in Neural Information Processing Systems 31*, 2018 -| Available at http://papers.nips.cc/paper/7595-probabilistic-matrix-factorization-for-automated-machine-learning.pdf - -This is currently a placeholder. -""" From 57b689172899ded6d2b0e882105890992baf58c6 Mon Sep 17 00:00:00 2001 From: Matthias Feurer Date: Wed, 2 Oct 2019 09:14:39 +0200 Subject: [PATCH 2/3] warn users of using dataset IDs, simplify code --- .../40_paper/2015_neurips_feurer_example.py | 25 ++++++++----------- 1 file changed, 10 insertions(+), 15 deletions(-) diff --git a/examples/40_paper/2015_neurips_feurer_example.py b/examples/40_paper/2015_neurips_feurer_example.py index 5bc7a3b32..0ea9cc83a 100644 --- a/examples/40_paper/2015_neurips_feurer_example.py +++ b/examples/40_paper/2015_neurips_feurer_example.py @@ -41,27 +41,22 @@ # OpenML, which define both the target feature and the train/test split. # # .. note:: -# Please do not use datasets but the respective tasks as basis for a paper. +# It is discouraged to work directly on datasets and only provide dataset IDs in a paper as +# this does not allow reproducibility (unclear splitting). Please do not use datasets but the +# respective tasks as basis for a paper and publish task IDS. This example is only given to +# showcase the use OpenML-Python for a published paper and as a warning on how not to do it. +# Please check the `OpenML documentation of tasks `_ if you +# want to learn more about them. # active tasks -tasks_a = openml.tasks.list_tasks( - task_type_id=1, - status='active', +tasks = openml.tasks.list_tasks( + task_type_id=openml.tasks.TaskTypeEnum.SUPERVISED_CLASSIFICATION, + status='all', output_format='dataframe', ) # query only those with holdout as the resampling startegy. -tasks_a = tasks_a[(tasks_a.estimation_procedure == "33% Holdout set")] - -# deactivated tasks -tasks_d = openml.tasks.list_tasks( - task_type_id=1, - status='deactivated', - output_format='dataframe', -) -tasks_d = tasks_d[(tasks_d.estimation_procedure == "33% Holdout set")] - -tasks = pd.concat([tasks_a, tasks_d], sort=True) +tasks = tasks[(tasks.estimation_procedure == "33% Holdout set")] task_ids = [] for did in dataset_ids: From 811dfc6347309366cda4bb0fbb3703d62cb8bbc3 Mon Sep 17 00:00:00 2001 From: Matthias Feurer Date: Wed, 2 Oct 2019 09:22:10 +0200 Subject: [PATCH 3/3] improve documentation of the example --- examples/40_paper/2015_neurips_feurer_example.py | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/examples/40_paper/2015_neurips_feurer_example.py b/examples/40_paper/2015_neurips_feurer_example.py index 0ea9cc83a..6f0de618e 100644 --- a/examples/40_paper/2015_neurips_feurer_example.py +++ b/examples/40_paper/2015_neurips_feurer_example.py @@ -48,20 +48,24 @@ # Please check the `OpenML documentation of tasks `_ if you # want to learn more about them. -# active tasks +#################################################################################################### +# This lists both active and inactive tasks (because of ``status='all'``). Unfortunately, +# this is necessary as some of the datasets contain issues found after the publication and became +# deactivated, which also deactivated the tasks on them. More information on active or inactive +# datasets can be found in the `online docs `_. tasks = openml.tasks.list_tasks( task_type_id=openml.tasks.TaskTypeEnum.SUPERVISED_CLASSIFICATION, status='all', output_format='dataframe', ) -# query only those with holdout as the resampling startegy. -tasks = tasks[(tasks.estimation_procedure == "33% Holdout set")] +# Query only those with holdout as the resampling startegy. +tasks = tasks.query('estimation_procedure == "33% Holdout set"') task_ids = [] for did in dataset_ids: tasks_ = list(tasks.query("did == {}".format(did)).tid) - if len(tasks_) >= 1: # if there are more than one task, take the lowest one. + if len(tasks_) >= 1: # if there are multiple task, take the one with lowest ID (oldest). task_id = min(tasks_) else: raise ValueError(did)