Compare commits
4 Commits
| Author | SHA1 | Date |
|---|---|---|
|
|
a387ede701 | |
|
|
e6323bcd0b | |
|
|
b3d5be0ac2 | |
|
|
e0269ac625 |
|
|
@ -17,7 +17,6 @@ for facilitating the analysis and interpretation of the experimental results.
|
||||||
|
|
||||||
* Version 0.2.1 is released! major changes can be consulted [here](CHANGE_LOG.txt).
|
* Version 0.2.1 is released! major changes can be consulted [here](CHANGE_LOG.txt).
|
||||||
* The developer API documentation is available [here](https://hlt-isti.github.io/QuaPy/index.html)
|
* The developer API documentation is available [here](https://hlt-isti.github.io/QuaPy/index.html)
|
||||||
* Manuals are available [here](https://hlt-isti.github.io/QuaPy/manuals.html)
|
|
||||||
|
|
||||||
### Installation
|
### Installation
|
||||||
|
|
||||||
|
|
|
||||||
1
TODO.txt
1
TODO.txt
|
|
@ -15,7 +15,6 @@ scale each value by per-class thresholds, i.e., [0.33*0.1, 0.33*1, 0.33*1]/sum."
|
||||||
- This functionality should be accessible via sampling protocols and evaluation functions
|
- This functionality should be accessible via sampling protocols and evaluation functions
|
||||||
|
|
||||||
- [TODO] document confidence in manuals
|
- [TODO] document confidence in manuals
|
||||||
- [TODO] Test the return_type="index" in protocols and finish the "distributing_samples.py" example
|
|
||||||
- [TODO] add ensemble methods SC-MQ, MC-SQ, MC-MQ
|
- [TODO] add ensemble methods SC-MQ, MC-SQ, MC-MQ
|
||||||
- [TODO] add HistNetQ
|
- [TODO] add HistNetQ
|
||||||
- [TODO] add CDE-iteration and Bayes-CDE methods
|
- [TODO] add CDE-iteration and Bayes-CDE methods
|
||||||
|
|
|
||||||
|
|
@ -1,38 +0,0 @@
|
||||||
"""
|
|
||||||
Imagine we want to generate many samples out of a collection, that we want to distribute for others to run their
|
|
||||||
own experiments in the very same test samples. One naive solution would come down to applying a given protocol to
|
|
||||||
our collection (say the artificial prevalence protocol on the 'academic-success' UCI dataset), store all those samples
|
|
||||||
on disk and make them available online. Distributing many such samples is undesirable.
|
|
||||||
In this example, we generate the indexes that allow anyone to regenerate the samples out of the original collection.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import quapy as qp
|
|
||||||
from quapy.method.aggregative import PACC
|
|
||||||
from quapy.protocol import UPP
|
|
||||||
|
|
||||||
data = qp.datasets.fetch_UCIMulticlassDataset('academic-success')
|
|
||||||
train, test = data.train_test
|
|
||||||
|
|
||||||
# let us train a quantifier to check whether we can actually replicate the results
|
|
||||||
quantifier = PACC()
|
|
||||||
quantifier.fit(train)
|
|
||||||
|
|
||||||
# let us simulate our experimental results
|
|
||||||
protocol = UPP(test, sample_size=100, repeats=100, random_state=0)
|
|
||||||
our_mae = qp.evaluation.evaluate(quantifier, protocol=protocol, error_metric='mae')
|
|
||||||
|
|
||||||
print(f'We have obtained a MAE={our_mae:.3f}')
|
|
||||||
|
|
||||||
# let us distribute the indexes; we specify that we want the indexes, not the samples
|
|
||||||
protocol = UPP(test, sample_size=100, repeats=100, random_state=0, return_type='index')
|
|
||||||
indexes = protocol.samples_parameters()
|
|
||||||
|
|
||||||
# Imagine we distribute the indexes; now we show how to replicate our experiments.
|
|
||||||
from quapy.protocol import ProtocolFromIndex
|
|
||||||
data = qp.datasets.fetch_UCIMulticlassDataset('academic-success')
|
|
||||||
train, test = data.train_test
|
|
||||||
protocol = ProtocolFromIndex(data=test, indexes=indexes)
|
|
||||||
their_mae = qp.evaluation.evaluate(quantifier, protocol=protocol, error_metric='mae')
|
|
||||||
|
|
||||||
print(f'Another lab obtains a MAE={our_mae:.3f}')
|
|
||||||
|
|
||||||
|
|
@ -1,56 +0,0 @@
|
||||||
from sklearn.exceptions import ConvergenceWarning
|
|
||||||
from sklearn.linear_model import LogisticRegression
|
|
||||||
from sklearn.naive_bayes import MultinomialNB
|
|
||||||
from sklearn.neighbors import KNeighborsClassifier
|
|
||||||
from statsmodels.sandbox.distributions.genpareto import quant
|
|
||||||
|
|
||||||
import quapy as qp
|
|
||||||
from quapy.protocol import UPP
|
|
||||||
from quapy.method.aggregative import PACC, DMy, EMQ, KDEyML
|
|
||||||
from quapy.method.meta import SCMQ, MCMQ, MCSQ
|
|
||||||
import warnings
|
|
||||||
warnings.filterwarnings("ignore", category=DeprecationWarning)
|
|
||||||
warnings.filterwarnings("ignore", category=ConvergenceWarning)
|
|
||||||
|
|
||||||
qp.environ["SAMPLE_SIZE"]=100
|
|
||||||
|
|
||||||
def train_and_test_model(quantifier, train, test):
|
|
||||||
quantifier.fit(train)
|
|
||||||
report = qp.evaluation.evaluation_report(quantifier, UPP(test), error_metrics=['mae', 'mrae'])
|
|
||||||
print(quantifier.__class__.__name__)
|
|
||||||
print(report.mean(numeric_only=True))
|
|
||||||
|
|
||||||
|
|
||||||
quantifiers = [
|
|
||||||
PACC(),
|
|
||||||
DMy(),
|
|
||||||
EMQ(),
|
|
||||||
KDEyML()
|
|
||||||
]
|
|
||||||
|
|
||||||
classifier = LogisticRegression()
|
|
||||||
|
|
||||||
dataset_name = qp.datasets.UCI_MULTICLASS_DATASETS[0]
|
|
||||||
data = qp.datasets.fetch_UCIMulticlassDataset(dataset_name)
|
|
||||||
train, test = data.train_test
|
|
||||||
|
|
||||||
scmq = SCMQ(classifier, quantifiers)
|
|
||||||
|
|
||||||
train_and_test_model(scmq, train, test)
|
|
||||||
|
|
||||||
# for quantifier in quantifiers:
|
|
||||||
# train_and_test_model(quantifier, train, test)
|
|
||||||
|
|
||||||
classifiers = [
|
|
||||||
LogisticRegression(),
|
|
||||||
KNeighborsClassifier(),
|
|
||||||
# MultinomialNB()
|
|
||||||
]
|
|
||||||
|
|
||||||
mcmq = MCMQ(classifiers, quantifiers)
|
|
||||||
|
|
||||||
train_and_test_model(mcmq, train, test)
|
|
||||||
|
|
||||||
mcsq = MCSQ(classifiers, PACC())
|
|
||||||
|
|
||||||
train_and_test_model(mcsq, train, test)
|
|
||||||
|
|
@ -1,254 +0,0 @@
|
||||||
from scipy.sparse import csc_matrix, csr_matrix
|
|
||||||
from sklearn.base import BaseEstimator, TransformerMixin
|
|
||||||
from sklearn.feature_extraction.text import TfidfTransformer, TfidfVectorizer, CountVectorizer
|
|
||||||
import numpy as np
|
|
||||||
from joblib import Parallel, delayed
|
|
||||||
import sklearn
|
|
||||||
import math
|
|
||||||
from scipy.stats import t
|
|
||||||
|
|
||||||
|
|
||||||
class ContTable:
|
|
||||||
def __init__(self, tp=0, tn=0, fp=0, fn=0):
|
|
||||||
self.tp=tp
|
|
||||||
self.tn=tn
|
|
||||||
self.fp=fp
|
|
||||||
self.fn=fn
|
|
||||||
|
|
||||||
def get_d(self): return self.tp + self.tn + self.fp + self.fn
|
|
||||||
|
|
||||||
def get_c(self): return self.tp + self.fn
|
|
||||||
|
|
||||||
def get_not_c(self): return self.tn + self.fp
|
|
||||||
|
|
||||||
def get_f(self): return self.tp + self.fp
|
|
||||||
|
|
||||||
def get_not_f(self): return self.tn + self.fn
|
|
||||||
|
|
||||||
def p_c(self): return (1.0*self.get_c())/self.get_d()
|
|
||||||
|
|
||||||
def p_not_c(self): return 1.0-self.p_c()
|
|
||||||
|
|
||||||
def p_f(self): return (1.0*self.get_f())/self.get_d()
|
|
||||||
|
|
||||||
def p_not_f(self): return 1.0-self.p_f()
|
|
||||||
|
|
||||||
def p_tp(self): return (1.0*self.tp) / self.get_d()
|
|
||||||
|
|
||||||
def p_tn(self): return (1.0*self.tn) / self.get_d()
|
|
||||||
|
|
||||||
def p_fp(self): return (1.0*self.fp) / self.get_d()
|
|
||||||
|
|
||||||
def p_fn(self): return (1.0*self.fn) / self.get_d()
|
|
||||||
|
|
||||||
def tpr(self):
|
|
||||||
c = 1.0*self.get_c()
|
|
||||||
return self.tp / c if c > 0.0 else 0.0
|
|
||||||
|
|
||||||
def fpr(self):
|
|
||||||
_c = 1.0*self.get_not_c()
|
|
||||||
return self.fp / _c if _c > 0.0 else 0.0
|
|
||||||
|
|
||||||
|
|
||||||
def __ig_factor(p_tc, p_t, p_c):
|
|
||||||
den = p_t * p_c
|
|
||||||
if den != 0.0 and p_tc != 0:
|
|
||||||
return p_tc * math.log(p_tc / den, 2)
|
|
||||||
else:
|
|
||||||
return 0.0
|
|
||||||
|
|
||||||
|
|
||||||
def information_gain(cell):
|
|
||||||
return __ig_factor(cell.p_tp(), cell.p_f(), cell.p_c()) + \
|
|
||||||
__ig_factor(cell.p_fp(), cell.p_f(), cell.p_not_c()) +\
|
|
||||||
__ig_factor(cell.p_fn(), cell.p_not_f(), cell.p_c()) + \
|
|
||||||
__ig_factor(cell.p_tn(), cell.p_not_f(), cell.p_not_c())
|
|
||||||
|
|
||||||
|
|
||||||
def squared_information_gain(cell):
|
|
||||||
return information_gain(cell)**2
|
|
||||||
|
|
||||||
|
|
||||||
def posneg_information_gain(cell):
|
|
||||||
ig = information_gain(cell)
|
|
||||||
if cell.tpr() < cell.fpr():
|
|
||||||
return -ig
|
|
||||||
else:
|
|
||||||
return ig
|
|
||||||
|
|
||||||
|
|
||||||
def pos_information_gain(cell):
|
|
||||||
if cell.tpr() < cell.fpr():
|
|
||||||
return 0
|
|
||||||
else:
|
|
||||||
return information_gain(cell)
|
|
||||||
|
|
||||||
def pointwise_mutual_information(cell):
|
|
||||||
return __ig_factor(cell.p_tp(), cell.p_f(), cell.p_c())
|
|
||||||
|
|
||||||
|
|
||||||
def gss(cell):
|
|
||||||
return cell.p_tp()*cell.p_tn() - cell.p_fp()*cell.p_fn()
|
|
||||||
|
|
||||||
|
|
||||||
def chi_square(cell):
|
|
||||||
den = cell.p_f() * cell.p_not_f() * cell.p_c() * cell.p_not_c()
|
|
||||||
if den==0.0: return 0.0
|
|
||||||
num = gss(cell)**2
|
|
||||||
return num / den
|
|
||||||
|
|
||||||
|
|
||||||
def conf_interval(xt, n):
|
|
||||||
if n>30:
|
|
||||||
z2 = 3.84145882069 # norm.ppf(0.5+0.95/2.0)**2
|
|
||||||
else:
|
|
||||||
z2 = t.ppf(0.5 + 0.95 / 2.0, df=max(n-1,1)) ** 2
|
|
||||||
p = (xt + 0.5 * z2) / (n + z2)
|
|
||||||
amplitude = 0.5 * z2 * math.sqrt((p * (1.0 - p)) / (n + z2))
|
|
||||||
return p, amplitude
|
|
||||||
|
|
||||||
|
|
||||||
def strength(minPosRelFreq, minPos, maxNeg):
|
|
||||||
if minPos > maxNeg:
|
|
||||||
return math.log(2.0 * minPosRelFreq, 2.0)
|
|
||||||
else:
|
|
||||||
return 0.0
|
|
||||||
|
|
||||||
|
|
||||||
#set cancel_features=True to allow some features to be weighted as 0 (as in the original article)
|
|
||||||
#however, for some extremely imbalanced dataset caused all documents to be 0
|
|
||||||
def conf_weight(cell, cancel_features=False):
|
|
||||||
c = cell.get_c()
|
|
||||||
not_c = cell.get_not_c()
|
|
||||||
tp = cell.tp
|
|
||||||
fp = cell.fp
|
|
||||||
|
|
||||||
pos_p, pos_amp = conf_interval(tp, c)
|
|
||||||
neg_p, neg_amp = conf_interval(fp, not_c)
|
|
||||||
|
|
||||||
min_pos = pos_p-pos_amp
|
|
||||||
max_neg = neg_p+neg_amp
|
|
||||||
den = (min_pos + max_neg)
|
|
||||||
minpos_relfreq = min_pos / (den if den != 0 else 1)
|
|
||||||
|
|
||||||
str_tplus = strength(minpos_relfreq, min_pos, max_neg);
|
|
||||||
|
|
||||||
if str_tplus == 0 and not cancel_features:
|
|
||||||
return 1e-20
|
|
||||||
|
|
||||||
return str_tplus
|
|
||||||
|
|
||||||
|
|
||||||
def get_tsr_matrix(cell_matrix, tsr_score_funtion):
|
|
||||||
nC = len(cell_matrix)
|
|
||||||
nF = len(cell_matrix[0])
|
|
||||||
tsr_matrix = [[tsr_score_funtion(cell_matrix[c,f]) for f in range(nF)] for c in range(nC)]
|
|
||||||
return np.array(tsr_matrix)
|
|
||||||
|
|
||||||
|
|
||||||
def feature_label_contingency_table(positive_document_indexes, feature_document_indexes, nD):
|
|
||||||
tp_ = len(positive_document_indexes & feature_document_indexes)
|
|
||||||
fp_ = len(feature_document_indexes - positive_document_indexes)
|
|
||||||
fn_ = len(positive_document_indexes - feature_document_indexes)
|
|
||||||
tn_ = nD - (tp_ + fp_ + fn_)
|
|
||||||
return ContTable(tp=tp_, tn=tn_, fp=fp_, fn=fn_)
|
|
||||||
|
|
||||||
|
|
||||||
def category_tables(feature_sets, category_sets, c, nD, nF):
|
|
||||||
return [feature_label_contingency_table(category_sets[c], feature_sets[f], nD) for f in range(nF)]
|
|
||||||
|
|
||||||
|
|
||||||
def get_supervised_matrix(coocurrence_matrix, label_matrix, n_jobs=-1):
|
|
||||||
"""
|
|
||||||
Computes the nC x nF supervised matrix M where Mcf is the 4-cell contingency table for feature f and class c.
|
|
||||||
Efficiency O(nF x nC x log(S)) where S is the sparse factor
|
|
||||||
"""
|
|
||||||
|
|
||||||
nD, nF = coocurrence_matrix.shape
|
|
||||||
nD2, nC = label_matrix.shape
|
|
||||||
|
|
||||||
if nD != nD2:
|
|
||||||
raise ValueError('Number of rows in coocurrence matrix shape %s and label matrix shape %s is not consistent' %
|
|
||||||
(coocurrence_matrix.shape,label_matrix.shape))
|
|
||||||
|
|
||||||
def nonzero_set(matrix, col):
|
|
||||||
return set(matrix[:, col].nonzero()[0])
|
|
||||||
|
|
||||||
if isinstance(coocurrence_matrix, csr_matrix):
|
|
||||||
coocurrence_matrix = csc_matrix(coocurrence_matrix)
|
|
||||||
feature_sets = [nonzero_set(coocurrence_matrix, f) for f in range(nF)]
|
|
||||||
category_sets = [nonzero_set(label_matrix, c) for c in range(nC)]
|
|
||||||
cell_matrix = Parallel(n_jobs=n_jobs, backend="threading")(
|
|
||||||
delayed(category_tables)(feature_sets, category_sets, c, nD, nF) for c in range(nC)
|
|
||||||
)
|
|
||||||
return np.array(cell_matrix)
|
|
||||||
|
|
||||||
|
|
||||||
class TSRweighting(BaseEstimator,TransformerMixin):
|
|
||||||
"""
|
|
||||||
Supervised Term Weighting function based on any Term Selection Reduction (TSR) function (e.g., information gain,
|
|
||||||
chi-square, etc.) or, more generally, on any function that could be computed on the 4-cell contingency table for
|
|
||||||
each category-feature pair.
|
|
||||||
The supervised_4cell_matrix is a `(n_classes, n_words)` matrix containing the 4-cell contingency tables
|
|
||||||
for each class-word pair, and can be pre-computed (e.g., during the feature selection phase) and passed as an
|
|
||||||
argument.
|
|
||||||
When `n_classes>1`, i.e., in multiclass scenarios, a global_policy is used in order to determine a
|
|
||||||
single feature-score which informs about its relevance. Accepted policies include "max" (takes the max score
|
|
||||||
across categories), "ave" and "wave" (take the average, or weighted average, across all categories -- weights
|
|
||||||
correspond to the class prevalence), and "sum" (which sums all category scores).
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, tsr_function, global_policy='max', supervised_4cell_matrix=None, sublinear_tf=True, norm='l2', min_df=3, n_jobs=-1):
|
|
||||||
if global_policy not in ['max', 'ave', 'wave', 'sum']: raise ValueError('Global policy should be in {"max", "ave", "wave", "sum"}')
|
|
||||||
self.tsr_function = tsr_function
|
|
||||||
self.global_policy = global_policy
|
|
||||||
self.supervised_4cell_matrix = supervised_4cell_matrix
|
|
||||||
self.sublinear_tf = sublinear_tf
|
|
||||||
self.norm = norm
|
|
||||||
self.min_df = min_df
|
|
||||||
self.n_jobs = n_jobs
|
|
||||||
|
|
||||||
def fit(self, X, y):
|
|
||||||
self.count_vectorizer = CountVectorizer(min_df=self.min_df)
|
|
||||||
X = self.count_vectorizer.fit_transform(X)
|
|
||||||
|
|
||||||
self.tf_vectorizer = TfidfTransformer(
|
|
||||||
norm=None, use_idf=False, smooth_idf=False, sublinear_tf=self.sublinear_tf
|
|
||||||
).fit(X)
|
|
||||||
|
|
||||||
if len(y.shape) == 1:
|
|
||||||
y = np.expand_dims(y, axis=1)
|
|
||||||
|
|
||||||
nD, nC = y.shape
|
|
||||||
nF = len(self.tf_vectorizer.get_feature_names_out())
|
|
||||||
|
|
||||||
if self.supervised_4cell_matrix is None:
|
|
||||||
self.supervised_4cell_matrix = get_supervised_matrix(X, y, n_jobs=self.n_jobs)
|
|
||||||
else:
|
|
||||||
if self.supervised_4cell_matrix.shape != (nC, nF):
|
|
||||||
raise ValueError("Shape of supervised information matrix is inconsistent with X and y")
|
|
||||||
|
|
||||||
tsr_matrix = get_tsr_matrix(self.supervised_4cell_matrix, self.tsr_function)
|
|
||||||
|
|
||||||
if self.global_policy == 'ave':
|
|
||||||
self.global_tsr_vector = np.average(tsr_matrix, axis=0)
|
|
||||||
elif self.global_policy == 'wave':
|
|
||||||
category_prevalences = [sum(y[:,c])*1.0/nD for c in range(nC)]
|
|
||||||
self.global_tsr_vector = np.average(tsr_matrix, axis=0, weights=category_prevalences)
|
|
||||||
elif self.global_policy == 'sum':
|
|
||||||
self.global_tsr_vector = np.sum(tsr_matrix, axis=0)
|
|
||||||
elif self.global_policy == 'max':
|
|
||||||
self.global_tsr_vector = np.amax(tsr_matrix, axis=0)
|
|
||||||
return self
|
|
||||||
|
|
||||||
def fit_transform(self, X, y):
|
|
||||||
return self.fit(X,y).transform(X)
|
|
||||||
|
|
||||||
def transform(self, X):
|
|
||||||
if not hasattr(self, 'global_tsr_vector'): raise NameError('TSRweighting: transform method called before fit.')
|
|
||||||
X = self.count_vectorizer.transform(X)
|
|
||||||
tf_X = self.tf_vectorizer.transform(X).toarray()
|
|
||||||
weighted_X = np.multiply(tf_X, self.global_tsr_vector)
|
|
||||||
if self.norm is not None and self.norm!='none':
|
|
||||||
weighted_X = sklearn.preprocessing.normalize(weighted_X, norm=self.norm, axis=1, copy=False)
|
|
||||||
return csr_matrix(weighted_X)
|
|
||||||
|
|
@ -1,208 +0,0 @@
|
||||||
from scipy.sparse import issparse
|
|
||||||
from sklearn.decomposition import TruncatedSVD
|
|
||||||
from sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer
|
|
||||||
from sklearn.linear_model import LogisticRegression
|
|
||||||
from sklearn.preprocessing import StandardScaler
|
|
||||||
|
|
||||||
import quapy as qp
|
|
||||||
from data import LabelledCollection
|
|
||||||
import numpy as np
|
|
||||||
|
|
||||||
from experimental_non_aggregative.custom_vectorizers import *
|
|
||||||
from method._kdey import KDEBase
|
|
||||||
from protocol import APP
|
|
||||||
from quapy.method.aggregative import HDy, DistributionMatchingY
|
|
||||||
from quapy.method.base import BaseQuantifier
|
|
||||||
from scipy import optimize
|
|
||||||
import pandas as pd
|
|
||||||
import quapy.functional as F
|
|
||||||
|
|
||||||
|
|
||||||
# TODO: explore the bernoulli (term presence/absence) variant
|
|
||||||
# TODO: explore the multinomial (term frequency) variant
|
|
||||||
# TODO: explore the multinomial + length normalization variant
|
|
||||||
# TODO: consolidate the TSR-variant (e.g., using information gain) variant;
|
|
||||||
# - works better with the idf?
|
|
||||||
# - works better with length normalization?
|
|
||||||
# - etc
|
|
||||||
|
|
||||||
class DxS(BaseQuantifier):
|
|
||||||
def __init__(self, vectorizer=None, divergence='topsoe'):
|
|
||||||
self.vectorizer = vectorizer
|
|
||||||
self.divergence = divergence
|
|
||||||
|
|
||||||
# def __as_distribution(self, instances):
|
|
||||||
# return np.asarray(instances.sum(axis=0) / instances.sum()).flatten()
|
|
||||||
|
|
||||||
def __as_distribution(self, instances):
|
|
||||||
dist = instances.mean(axis=0)
|
|
||||||
return np.asarray(dist).flatten()
|
|
||||||
|
|
||||||
def fit(self, text_instances, labels):
|
|
||||||
|
|
||||||
classes = np.unique(labels)
|
|
||||||
|
|
||||||
if self.vectorizer is not None:
|
|
||||||
text_instances = self.vectorizer.fit_transform(text_instances, y=labels)
|
|
||||||
|
|
||||||
distributions = []
|
|
||||||
for class_i in classes:
|
|
||||||
distributions.append(self.__as_distribution(text_instances[labels == class_i]))
|
|
||||||
|
|
||||||
self.validation_distribution = np.asarray(distributions)
|
|
||||||
|
|
||||||
return self
|
|
||||||
|
|
||||||
def predict(self, text_instances):
|
|
||||||
if self.vectorizer is not None:
|
|
||||||
text_instances = self.vectorizer.transform(text_instances)
|
|
||||||
|
|
||||||
test_distribution = self.__as_distribution(text_instances)
|
|
||||||
divergence = qp.functional.get_divergence(self.divergence)
|
|
||||||
n_classes, n_feats = self.validation_distribution.shape
|
|
||||||
|
|
||||||
def match(prev):
|
|
||||||
prev = np.expand_dims(prev, axis=0)
|
|
||||||
mixture_distribution = (prev @ self.validation_distribution).flatten()
|
|
||||||
return divergence(test_distribution, mixture_distribution)
|
|
||||||
|
|
||||||
# the initial point is set as the uniform distribution
|
|
||||||
uniform_distribution = np.full(fill_value=1 / n_classes, shape=(n_classes,))
|
|
||||||
|
|
||||||
# solutions are bounded to those contained in the unit-simplex
|
|
||||||
bounds = tuple((0, 1) for x in range(n_classes)) # values in [0,1]
|
|
||||||
constraints = ({'type': 'eq', 'fun': lambda x: 1 - sum(x)}) # values summing up to 1
|
|
||||||
r = optimize.minimize(match, x0=uniform_distribution, method='SLSQP', bounds=bounds, constraints=constraints)
|
|
||||||
return r.x
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
class KDExML(BaseQuantifier, KDEBase):
|
|
||||||
|
|
||||||
def __init__(self, bandwidth=0.1, standardize=False):
|
|
||||||
self._check_bandwidth(bandwidth)
|
|
||||||
self.bandwidth = bandwidth
|
|
||||||
self.standardize = standardize
|
|
||||||
|
|
||||||
def fit(self, X, y):
|
|
||||||
classes = sorted(np.unique(y))
|
|
||||||
|
|
||||||
if self.standardize:
|
|
||||||
self.scaler = StandardScaler()
|
|
||||||
X = self.scaler.fit_transform(X)
|
|
||||||
|
|
||||||
if issparse(X):
|
|
||||||
X = X.toarray()
|
|
||||||
|
|
||||||
self.mix_densities = self.get_mixture_components(X, y, classes, self.bandwidth)
|
|
||||||
return self
|
|
||||||
|
|
||||||
def predict(self, X):
|
|
||||||
"""
|
|
||||||
Searches for the mixture model parameter (the sought prevalence values) that maximizes the likelihood
|
|
||||||
of the data (i.e., that minimizes the negative log-likelihood)
|
|
||||||
|
|
||||||
:param X: instances in the sample
|
|
||||||
:return: a vector of class prevalence estimates
|
|
||||||
"""
|
|
||||||
epsilon = 1e-10
|
|
||||||
if issparse(X):
|
|
||||||
X = X.toarray()
|
|
||||||
n_classes = len(self.mix_densities)
|
|
||||||
if self.standardize:
|
|
||||||
X = self.scaler.transform(X)
|
|
||||||
test_densities = [self.pdf(kde_i, X) for kde_i in self.mix_densities]
|
|
||||||
|
|
||||||
def neg_loglikelihood(prev):
|
|
||||||
test_mixture_likelihood = sum(prev_i * dens_i for prev_i, dens_i in zip (prev, test_densities))
|
|
||||||
test_loglikelihood = np.log(test_mixture_likelihood + epsilon)
|
|
||||||
return -np.sum(test_loglikelihood)
|
|
||||||
|
|
||||||
return F.optim_minimize(neg_loglikelihood, n_classes)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
|
|
||||||
qp.environ['SAMPLE_SIZE'] = 250
|
|
||||||
qp.environ['N_JOBS'] = -1
|
|
||||||
min_df = 10
|
|
||||||
# dataset = 'imdb'
|
|
||||||
repeats = 10
|
|
||||||
error = 'mae'
|
|
||||||
|
|
||||||
div = 'topsoe'
|
|
||||||
|
|
||||||
# generates tuples (dataset, method, method_name)
|
|
||||||
# (the dataset is needed for methods that process the dataset differently)
|
|
||||||
def gen_methods():
|
|
||||||
|
|
||||||
for dataset in qp.datasets.REVIEWS_SENTIMENT_DATASETS:
|
|
||||||
|
|
||||||
data = qp.datasets.fetch_reviews(dataset, tfidf=False)
|
|
||||||
|
|
||||||
# bernoulli_vectorizer = CountVectorizer(min_df=min_df, binary=True)
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=bernoulli_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-Bernoulli'
|
|
||||||
#
|
|
||||||
# multinomial_vectorizer = CountVectorizer(min_df=min_df, binary=False)
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=multinomial_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-multinomial'
|
|
||||||
#
|
|
||||||
# tf_vectorizer = TfidfVectorizer(sublinear_tf=False, use_idf=False, min_df=min_df, norm=None)
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=tf_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-TF'
|
|
||||||
#
|
|
||||||
# logtf_vectorizer = TfidfVectorizer(sublinear_tf=True, use_idf=False, min_df=min_df, norm=None)
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=logtf_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-logTF'
|
|
||||||
#
|
|
||||||
# tfidf_vectorizer = TfidfVectorizer(use_idf=True, min_df=min_df, norm=None)
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=tfidf_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-TFIDF'
|
|
||||||
#
|
|
||||||
# tfidf_vectorizer = TfidfVectorizer(use_idf=True, min_df=min_df, norm='l2')
|
|
||||||
# dxs = DxS(divergence=div, vectorizer=tfidf_vectorizer)
|
|
||||||
# yield data, dxs, 'DxS-TFIDF-l2'
|
|
||||||
|
|
||||||
tsr_vectorizer = TSRweighting(tsr_function=information_gain, min_df=min_df, norm='l2')
|
|
||||||
dxs = DxS(divergence=div, vectorizer=tsr_vectorizer)
|
|
||||||
yield data, dxs, 'DxS-TFTSR-l2'
|
|
||||||
|
|
||||||
data = qp.datasets.fetch_reviews(dataset, tfidf=True, min_df=min_df)
|
|
||||||
|
|
||||||
kdex = KDExML()
|
|
||||||
reduction = TruncatedSVD(n_components=100, random_state=0)
|
|
||||||
red_data = qp.data.preprocessing.instance_transformation(data, transformer=reduction, inplace=False)
|
|
||||||
yield red_data, kdex, 'KDEx'
|
|
||||||
|
|
||||||
hdy = HDy(LogisticRegression())
|
|
||||||
yield data, hdy, 'HDy'
|
|
||||||
|
|
||||||
# dm = DistributionMatchingY(LogisticRegression(), divergence=div, nbins=5)
|
|
||||||
# yield data, dm, 'DM-5b'
|
|
||||||
#
|
|
||||||
# dm = DistributionMatchingY(LogisticRegression(), divergence=div, nbins=10)
|
|
||||||
# yield data, dm, 'DM-10b'
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
result_path = 'results.csv'
|
|
||||||
with open(result_path, 'wt') as csv:
|
|
||||||
csv.write(f'Method\tDataset\tMAE\tMRAE\n')
|
|
||||||
for data, quantifier, quant_name in gen_methods():
|
|
||||||
quantifier.fit(*data.training.Xy)
|
|
||||||
report = qp.evaluation.evaluation_report(quantifier, APP(data.test, repeats=repeats), error_metrics=['mae','mrae'], verbose=True)
|
|
||||||
means = report.mean(numeric_only=True)
|
|
||||||
csv.write(f'{quant_name}\t{data.name}\t{means["mae"]:.5f}\t{means["mrae"]:.5f}\n')
|
|
||||||
|
|
||||||
df = pd.read_csv(result_path, sep='\t')
|
|
||||||
# print(df)
|
|
||||||
|
|
||||||
pv = df.pivot_table(index='Method', columns="Dataset", values=["MAE", "MRAE"])
|
|
||||||
print(pv)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Loading…
Reference in New Issue