diff --git a/TODO.txt b/TODO.txt index 7a36ae1..945fbc9 100644 --- a/TODO.txt +++ b/TODO.txt @@ -15,7 +15,6 @@ scale each value by per-class thresholds, i.e., [0.33*0.1, 0.33*1, 0.33*1]/sum." - This functionality should be accessible via sampling protocols and evaluation functions - [TODO] document confidence in manuals -- [TODO] Test the return_type="index" in protocols and finish the "distributing_samples.py" example - [TODO] add ensemble methods SC-MQ, MC-SQ, MC-MQ - [TODO] add HistNetQ - [TODO] add CDE-iteration and Bayes-CDE methods diff --git a/examples/distributing_samples.py b/examples/distributing_samples.py deleted file mode 100644 index 76a9731..0000000 --- a/examples/distributing_samples.py +++ /dev/null @@ -1,38 +0,0 @@ -""" -Imagine we want to generate many samples out of a collection, that we want to distribute for others to run their -own experiments in the very same test samples. One naive solution would come down to applying a given protocol to -our collection (say the artificial prevalence protocol on the 'academic-success' UCI dataset), store all those samples -on disk and make them available online. Distributing many such samples is undesirable. -In this example, we generate the indexes that allow anyone to regenerate the samples out of the original collection. -""" - -import quapy as qp -from quapy.method.aggregative import PACC -from quapy.protocol import UPP - -data = qp.datasets.fetch_UCIMulticlassDataset('academic-success') -train, test = data.train_test - -# let us train a quantifier to check whether we can actually replicate the results -quantifier = PACC() -quantifier.fit(train) - -# let us simulate our experimental results -protocol = UPP(test, sample_size=100, repeats=100, random_state=0) -our_mae = qp.evaluation.evaluate(quantifier, protocol=protocol, error_metric='mae') - -print(f'We have obtained a MAE={our_mae:.3f}') - -# let us distribute the indexes; we specify that we want the indexes, not the samples -protocol = UPP(test, sample_size=100, repeats=100, random_state=0, return_type='index') -indexes = protocol.samples_parameters() - -# Imagine we distribute the indexes; now we show how to replicate our experiments. -from quapy.protocol import ProtocolFromIndex -data = qp.datasets.fetch_UCIMulticlassDataset('academic-success') -train, test = data.train_test -protocol = ProtocolFromIndex(data=test, indexes=indexes) -their_mae = qp.evaluation.evaluate(quantifier, protocol=protocol, error_metric='mae') - -print(f'Another lab obtains a MAE={our_mae:.3f}') - diff --git a/examples/ensembles.py b/examples/ensembles.py deleted file mode 100644 index 84aeb2c..0000000 --- a/examples/ensembles.py +++ /dev/null @@ -1,56 +0,0 @@ -from sklearn.exceptions import ConvergenceWarning -from sklearn.linear_model import LogisticRegression -from sklearn.naive_bayes import MultinomialNB -from sklearn.neighbors import KNeighborsClassifier -from statsmodels.sandbox.distributions.genpareto import quant - -import quapy as qp -from quapy.protocol import UPP -from quapy.method.aggregative import PACC, DMy, EMQ, KDEyML -from quapy.method.meta import SCMQ, MCMQ, MCSQ -import warnings -warnings.filterwarnings("ignore", category=DeprecationWarning) -warnings.filterwarnings("ignore", category=ConvergenceWarning) - -qp.environ["SAMPLE_SIZE"]=100 - -def train_and_test_model(quantifier, train, test): - quantifier.fit(train) - report = qp.evaluation.evaluation_report(quantifier, UPP(test), error_metrics=['mae', 'mrae']) - print(quantifier.__class__.__name__) - print(report.mean(numeric_only=True)) - - -quantifiers = [ - PACC(), - DMy(), - EMQ(), - KDEyML() -] - -classifier = LogisticRegression() - -dataset_name = qp.datasets.UCI_MULTICLASS_DATASETS[0] -data = qp.datasets.fetch_UCIMulticlassDataset(dataset_name) -train, test = data.train_test - -scmq = SCMQ(classifier, quantifiers) - -train_and_test_model(scmq, train, test) - -# for quantifier in quantifiers: -# train_and_test_model(quantifier, train, test) - -classifiers = [ - LogisticRegression(), - KNeighborsClassifier(), - # MultinomialNB() -] - -mcmq = MCMQ(classifiers, quantifiers) - -train_and_test_model(mcmq, train, test) - -mcsq = MCSQ(classifiers, PACC()) - -train_and_test_model(mcsq, train, test) \ No newline at end of file