Commit 84ea2a9b authored by Bharath Ramsundar's avatar Bharath Ramsundar
Browse files

Handling merge conflicts

parents 4a784b76 7fa4cdce
Loading
Loading
Loading
Loading

datasets/gbd3k.pkl.gz

0 → 100644
+405 KiB

File added.

No diff preview for this file type.

+14 −10
Original line number Diff line number Diff line
@@ -114,7 +114,8 @@ class DataFeaturizer(object):
    self.verbose = verbose
    self.log_every_n = log_every_n

  def featurize(self, input_file, feature_dir, samples_dir, shard_size=128, worker_pool=None):
  def featurize(self, input_file, feature_dir, samples_dir,
                shard_size=128, worker_pool=None):
    """Featurize provided file and write to specified location."""
    input_type = _get_input_type(input_file)

@@ -141,15 +142,14 @@ class DataFeaturizer(object):
      raw_df_shard = raw_df.iloc[range(interval_points[j], interval_points[j+1])]
      
      df = self._standardize_df(raw_df_shard) 
      log("Aggregating User-Specified Features", self.verbose)

      for compound_featurizer in self.compound_featurizers:
        log("Currently feauturizing feature_type: %s"
        log("Currently featurizing feature_type: %s"
            % compound_featurizer.__class__.__name__, self.verbose)
        self._featurize_compounds(df, compound_featurizer, worker_pool=worker_pool)

      for complex_featurizer in self.complex_featurizers:
        log("Currently feauturizing feature_type: %s"
        log("Currently featurizing feature_type: %s"
            % complex_featurizer.__class__.__name__, self.verbose)
        self._featurize_complexes(df, complex_featurizer, worker_pool=worker_pool)

@@ -159,8 +159,7 @@ class DataFeaturizer(object):

    featurizers = self.compound_featurizers + self.complex_featurizers
    samples = FeaturizedSamples(samples_dir=samples_dir, featurizers=featurizers, 
                                dataset_files=shard_files,
                                reload_data=False)
                                dataset_files=shard_files, reload_data=False)

    return samples

@@ -190,6 +189,9 @@ class DataFeaturizer(object):
    df["smiles"] = ori_df[[self.smiles_field]]
    for task in self.tasks:
      df[task] = ori_df[[task]]
    if self.user_specified_features is not None:
      for feature in self.user_specified_features:
        df[feature] = ori_df[[feature]]
    if self.split_field is not None:
      df["split"] = ori_df[[self.split_field]]
    if self.protein_pdb_field is not None:
@@ -261,22 +263,24 @@ class DataFeaturizer(object):
      into final features dataframe
    """
    if self.user_specified_features is not None:
      log("Adding user-defined features.", self.verbose)
      log("Aggregating User-Specified Features", self.verbose)
      #log("Adding user-defined features.", self.verbose)
      features_data = []
      for ind, row in ori_df.iterrows():
        # pandas rows are tuples (row_num, row_data)
        feature_list = []
        for feature_name in self.user_specified_features:
          feature_list.append(row[feature_name])
        features_data.append({"user-specified-features": np.array(feature_list)})
      df["user-specified-features"] = pd.DataFrame(features_data)
        features_data.append(np.array(feature_list))
      df["user-specified-features"] = features_data


def map_function(data_tuple, featurizer):
  featurizer = NNScoreComplexFeaturizer()
  ind, ligand_pdb, protein_pdb = data_tuple
  print("Mapping on ind %d" % ind)
  print("ind, type(ligand_pdb), type(protein_pdb): %s " % str((ind, type(ligand_pdb), type(protein_pdb))))
  print("ind, type(ligand_pdb), type(protein_pdb): %s " %
        str((ind, type(ligand_pdb), type(protein_pdb))))
  return featurizer.featurize_complexes([ligand_pdb], [protein_pdb])

class FeaturizedSamples(object):
+2 −2
Original line number Diff line number Diff line
@@ -79,8 +79,8 @@ def hydrogenate_and_compute_partial_charges(input_file, input_format,
  hyd_conversion.SetInAndOutFormats(str(input_format), str("pdb"))
  mol = openbabel.OBMol()
  hyd_conversion.ReadFile(mol, str(input_file))
  # AddHydrogens(polaronly, correctForPH, pH)
  mol.AddHydrogens(True, True, 7.4)
  # AddHydrogens(not-polaronly, correctForPH, pH)
  mol.AddHydrogens(False, True, 7.4)
  hyd_conversion.WriteFile(mol, str(hyd_output))

  if verbose:
+2 −18
Original line number Diff line number Diff line
@@ -80,23 +80,6 @@ class Model(object):
    else:
      return "regression"

  #@staticmethod
  #def load(model_dir):
  #  """Dispatcher function for loading."""
  #  params = load_from_disk(Model.get_params_filename(model_dir))
  #  model_class = params["model_class"]
  #  if model_class in Model.registered_model_classes:
  #    model = Model.registered_model_classes[model_class](
  #        task_types=params["task_types"],
  #        model_params=params["model_params"])
  #    model.load(model_dir)
  #  else:
  #    model = Model.registered_model_classes["SklearnModel"](model_instance=model_class,
  #                         task_types=params["task_types"],
  #                         model_params=params["model_params"])
  #    model.load(model_dir)
  #  return model

  def save(self, out_dir):
    """Dispatcher function for saving."""
    params = {"model_params" : self.model_params,
@@ -149,7 +132,8 @@ class Model(object):
      y_preds = []
      for j in range(len(interval_points)-1):
        indices = range(interval_points[j], interval_points[j+1])
        y_pred_on_batch = self.predict_on_batch(X[indices, :]).reshape((len(indices),len(task_names)))
        y_pred_on_batch = self.predict_on_batch(X[indices, :]).reshape(
            (len(indices),len(task_names)))
        y_preds.append(y_pred_on_batch)

      y_pred = np.concatenate(y_preds)
+5 −1
Original line number Diff line number Diff line
@@ -30,9 +30,13 @@ class Dataset(object):
      os.makedirs(data_dir)
    self.data_dir = data_dir

    if featurizers is not None:
      feature_types = [featurizer.__class__.__name__ for featurizer in featurizers]
    else:
      feature_types = None

    if use_user_specified_features:
      feature_types += ["user-specified-features"]
      feature_types = ["user-specified-features"]

    if samples is not None and feature_types is not None:
      if not isinstance(feature_types, list):
Loading