Commit f72cf3d9 authored by evanfeinberg's avatar evanfeinberg
Browse files

fixed multitask case

parent f19ebde9
Loading
Loading
Loading
Loading
+0 −7
Original line number Diff line number Diff line
@@ -324,15 +324,8 @@ def _df_to_numpy(df, feature_types, tasks):
  y[missing] = 0.
  w[missing] = 0.

  print("x")
  print(x)
  print("y")
  print(y)
  print("w")
  print(w)
  return sorted_ids, x.astype(float), y.astype(float), w.astype(float)


def compute_mean_and_std(df):
  """
  Compute means/stds of X/y from sums/sum_squares of tensors.
+3 −3
Original line number Diff line number Diff line
@@ -40,7 +40,7 @@ def undo_transform(y, y_means, y_stds, output_transforms):
  else:
    raise ValueError("Unsupported output transforms %s." % str(output_transforms))

def compute_roc_auc_scores(y, y_pred, w):
def compute_roc_auc_scores(y, y_pred):
  """Transforms the results dict into roc-auc-scores and prints scores.

  Parameters
@@ -51,7 +51,7 @@ def compute_roc_auc_scores(y, y_pred, w):
    "classification" or "regression".
  """
  try:
    score = roc_auc_score(y, y_pred, sample_weight=w)
    score = roc_auc_score(y, y_pred)
  except ValueError:
    warnings.warn("ROC AUC score calculation failed.")
    score = 0.5
@@ -103,7 +103,7 @@ class Evaluator(object):
        # Sometimes all samples have zero weight. In this case, continue.
        if not len(y):
          continue
        auc = compute_roc_auc_scores(y, y_pred, w)
        auc = compute_roc_auc_scores(y, y_pred)
        mcc = matthews_corrcoef(y, y_pred)
        recall = recall_score(y, y_pred)
        accuracy = accuracy_score(y, y_pred)
+0 −76
Original line number Diff line number Diff line
@@ -240,79 +240,3 @@ class TestAPI(unittest.TestCase):

    model = MultiTaskDNN(task_types, model_params)
    self._create_model(train_dataset, test_dataset, model)


'''
class TestMultitaskVectorAPI(unittest.TestCase):
  """
  Test top-level API for singletask vector models."
  """
  def setUp(self):
    current_dir = os.path.dirname(os.path.abspath(__file__))
    self.input_file = os.path.join(current_dir, "multitask_example.csv")
    self.tasks = ["task0", "task1", "task2", "task3", "task4", "task5", "task6",
                  "task7", "task8", "task9", "task10", "task11", "task12",
                  "task13", "task14", "task15", "task16"]
    self.smiles_field = "smiles"
    self.feature_dir = tempfile.mkdtemp()
    self.samples_dir = tempfile.mkdtemp()
    self.train_dir = tempfile.mkdtemp()
    self.test_dir = tempfile.mkdtemp()
    self.model_dir = tempfile.mkdtemp()

  def tearDown(self):
    shutil.rmtree(self.feature_dir)
    shutil.rmtree(self.samples_dir)
    shutil.rmtree(self.train_dir)
    shutil.rmtree(self.test_dir)
    shutil.rmtree(self.model_dir)

  def test_API(self):
    """Straightforward test of multitask deepchem classification API."""
    splittype = "scaffold"
    feature_types = ["ECFP"]
    output_transforms = []
    input_transforms = []
    task_type = "classification"
    # TODO(rbharath): There should be some automatic check to ensure that all
    # required model_params are specified.
    model_params = {"nb_hidden": 10, "activation": "relu",
                    "dropout": .5, "learning_rate": .01,
                    "momentum": .9, "nesterov": False,
                    "decay": 1e-4, "batch_size": 5,
                    "nb_epoch": 2}
    model_name = "multitask_deep_classifier"

    # Featurize input
    featurizer = DataFeaturizer(tasks=self.tasks,
                                smiles_field=self.smiles_field,
                                verbose=True)
    feature_files = featurizer.featurize(self.input_file, feature_types, self.feature_dir)

    # Transform data into arrays for ML
    samples = FeaturizedSamples(self.samples_dir, feature_files,
                                reload_data=False)

    # Split into train/test
    train_samples, test_samples = samples.train_test_split(
        splittype, self.train_dir, self.test_dir)
    train_dataset = Dataset(self.train_dir, train_samples, feature_types)
    test_dataset = Dataset(self.test_dir, test_samples, feature_types)

    # Transforming train/test data
    train_dataset.transform(input_transforms, output_transforms)
    test_dataset.transform(input_transforms, output_transforms)

    # Fit model
    task_types = {task: task_type for task in self.tasks}
    model_params["data_shape"] = train_dataset.get_data_shape()
    model = Model.model_builder(model_name, task_types, model_params)
    model.fit(train_dataset)
    model.save(self.model_dir)

    # Eval model on train
    evaluator = Evaluator(model, test_dataset, verbose=True)
    with tempfile.NamedTemporaryFile() as test_csv_out:
      with tempfile.NamedTemporaryFile() as test_stats_out:
        evaluator.compute_model_performance(test_csv_out, test_stats_out)
'''
 No newline at end of file