Formatting (#68)

Fix most flake/lint errors and ignore a few others Signed-off-by: abigailt <abigailt@il.ibm.com>
2026-06-26 15:49:37 +02:00 · 2022-12-25 15:13:57 +02:00 · 2022-12-25 15:13:57 +02:00 · d52fcd0041
commit d52fcd0041
parent b47ba24906
16 changed files with 91 additions and 92 deletions
--- a/apt/init.py
+++ b/apt/init.py
@ -6,4 +6,4 @@ from apt import anonymization
 from apt import minimization
 from apt import utils

-__version__ = "0.1.0"
+__version__ = "0.1.0"
--- a/apt/anonymization/anonymizer.py
+++ b/apt/anonymization/anonymizer.py
@ -68,7 +68,7 @@ class Anonymize:

        if dataset.features_names is not None:
            self.features_names = dataset.features_names
-        else: # if no names provided, use numbers instead
+        else:  # if no names provided, use numbers instead
            self.features_names = self.features

        if not set(self.quasi_identifiers).issubset(set(self.features_names)):
--- a/apt/minimization/minimizer.py
+++ b/apt/minimization/minimizer.py
@ -16,7 +16,7 @@ from sklearn.utils.validation import check_is_fitted
 from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor
 from sklearn.model_selection import train_test_split

-from apt.utils.datasets import ArrayDataset, Data, DATA_PANDAS_NUMPY_TYPE
+from apt.utils.datasets import ArrayDataset, DATA_PANDAS_NUMPY_TYPE
 from apt.utils.models import Model, SklearnRegressor, ModelOutputType, SklearnClassifier


@ -268,14 +268,14 @@ class GeneralizeToRepresentative(BaseEstimator, MetaEstimatorMixin, TransformerM
            if self.encoder is None:
                numeric_features = [f for f in self._features if f not in self.categorical_features]
                numeric_transformer = Pipeline(
-                        steps=[('imputer', SimpleImputer(strategy='constant', fill_value=0))]
+                    steps=[('imputer', SimpleImputer(strategy='constant', fill_value=0))]
                )
                categorical_transformer = OneHotEncoder(handle_unknown="ignore", sparse=False)
                self.encoder = ColumnTransformer(
-                        transformers=[
-                            ("num", numeric_transformer, numeric_features),
-                            ("cat", categorical_transformer, self.categorical_features),
-                        ]
+                    transformers=[
+                        ("num", numeric_transformer, numeric_features),
+                        ("cat", categorical_transformer, self.categorical_features),
+                    ]
                )
                self.encoder.fit(x)

@ -345,7 +345,6 @@ class GeneralizeToRepresentative(BaseEstimator, MetaEstimatorMixin, TransformerM
                        print('Pruned tree to level: %d, new relative accuracy: %f' % (level, accuracy))
                        level += 1

-
            # if accuracy below threshold, improve accuracy by removing features from generalization
            elif accuracy < self.target_accuracy:
                print('Improving accuracy')
@ -599,8 +598,8 @@ class GeneralizeToRepresentative(BaseEstimator, MetaEstimatorMixin, TransformerM
                            new_cell['ranges'][feature]['end'] = right_cell['ranges'][feature]['start']
                        for feature in left_cell['categories'].keys():
                            new_cell['categories'][feature] = \
-                                list(set(left_cell['categories'][feature]) |
-                                     set(right_cell['categories'][feature]))
+                                list(set(left_cell['categories'][feature])
+                                     | set(right_cell['categories'][feature]))
                        for feature in left_cell['untouched']:
                            if feature in right_cell['untouched']:
                                new_cell['untouched'].append(feature)
@ -707,8 +706,8 @@ class GeneralizeToRepresentative(BaseEstimator, MetaEstimatorMixin, TransformerM
            for feature in self._features:
                # if feature has a representative value in the cell and should not be left untouched,
                # take the representative value
-                if feature in cells[i]['representative'] and ('untouched' not in cells[i] or
-                                                              feature not in cells[i]['untouched']):
+                if feature in cells[i]['representative'] \
+                        and ('untouched' not in cells[i] or feature not in cells[i]['untouched']):
                    representatives.loc[i, feature] = cells[i]['representative'][feature]
                # else, drop the feature (removes from representatives columns that do not have a
                # representative value or should remain untouched)
--- a/apt/utils/dataset_utils.py
+++ b/apt/utils/dataset_utils.py
@ -24,7 +24,7 @@ def _load_iris(test_set_size: float = 0.3):

    # Split training and test sets
    x_train, x_test, y_train, y_test = model_selection.train_test_split(data, labels, test_size=test_set_size,
-                                                                                random_state=18, stratify=labels)
+                                                                        random_state=18, stratify=labels)

    return (x_train, y_train), (x_test, y_test)

@ -94,9 +94,6 @@ def get_german_credit_dataset_pd(test_set: float = 0.3):
    x_test = test.drop(["label"], axis=1)
    y_test = test.loc[:, "label"]

-    categorical_features = ["Existing_checking_account", "Credit_history", "Purpose", "Savings_account",
-                            "Present_employment_since", "Personal_status_sex", "debtors", "Property",
-                            "Other_installment_plans", "Housing", "Job"]
    x_train.reset_index(drop=True, inplace=True)
    y_train.reset_index(drop=True, inplace=True)
    x_test.reset_index(drop=True, inplace=True)
--- a/apt/utils/models/keras_model.py
+++ b/apt/utils/models/keras_model.py
@ -1,11 +1,9 @@
 from typing import Optional

 import numpy as np
-from sklearn.preprocessing import OneHotEncoder

 import tensorflow as tf
 from tensorflow import keras
-tf.compat.v1.disable_eager_execution()

 from sklearn.metrics import mean_squared_error

@ -16,6 +14,8 @@ from art.utils import check_and_transform_label_format
 from art.estimators.classification.keras import KerasClassifier as ArtKerasClassifier
 from art.estimators.regression.keras import KerasRegressor as ArtKerasRegressor

+tf.compat.v1.disable_eager_execution()
+

 class KerasModel(Model):
    """
@ -23,7 +23,6 @@ class KerasModel(Model):
    """


-
 class KerasClassifier(KerasModel):
    """
    Wrapper class for keras classification models.
--- a/apt/utils/models/model.py
+++ b/apt/utils/models/model.py
@ -1,5 +1,5 @@
 from abc import ABCMeta, abstractmethod
-from typing import Any, Optional, Callable, Tuple, Union
+from typing import Any, Optional, Callable, Tuple, Union, TYPE_CHECKING
 from enum import Enum, auto
 import numpy as np

@ -7,6 +7,9 @@ from apt.utils.datasets import Dataset, Data, OUTPUT_DATA_ARRAY_TYPE
 from art.estimators.classification import BlackBoxClassifier
 from art.utils import check_and_transform_label_format

+if TYPE_CHECKING:
+    import torch
+

 class ModelOutputType(Enum):
    CLASSIFIER_PROBABILITIES = auto()  # vector of probabilities
--- a/apt/utils/models/pytorch_model.py
+++ b/apt/utils/models/pytorch_model.py
@ -8,7 +8,7 @@ import numpy as np
 import torch
 from torch.utils.data import DataLoader, TensorDataset

-from art.utils import check_and_transform_label_format, logger
+from art.utils import check_and_transform_label_format
 from apt.utils.datasets.datasets import PytorchData
 from apt.utils.models import Model, ModelOutputType
 from apt.utils.datasets import OUTPUT_DATA_ARRAY_TYPE
--- a/apt/utils/models/sklearn_model.py
+++ b/apt/utils/models/sklearn_model.py
@ -1,6 +1,5 @@
 from typing import Optional

-from sklearn.preprocessing import OneHotEncoder
 from sklearn.base import BaseEstimator

 from apt.utils.models import Model, ModelOutputType, get_nb_classes, check_correct_model_output
--- a/apt/utils/models/xgboost_model.py
+++ b/apt/utils/models/xgboost_model.py
@ -38,7 +38,7 @@ class XGBoostClassifier(XGBoostModel):
    :type unlimited_queries: boolean, optional
    """
    def __init__(self, model: XGBClassifier, output_type: ModelOutputType, input_shape: Tuple[int, ...],
-                 nb_classes: int,black_box_access: Optional[bool] = True,
+                 nb_classes: int, black_box_access: Optional[bool] = True,
                 unlimited_queries: Optional[bool] = True, **kwargs):
        super().__init__(model, output_type, black_box_access, unlimited_queries, **kwargs)
        self._art_model = ArtXGBoostClassifier(model, nb_features=input_shape[0], nb_classes=nb_classes)