Commit 4a68b00b authored by evanfeinberg's avatar evanfeinberg
Browse files

updated travis and coulomb_matrices

parent d7ea5f1c
Loading
Loading
Loading
Loading
+1 −0
Original line number Diff line number Diff line
@@ -20,6 +20,7 @@ install:
- conda install -c omnia keras
- conda install seaborn
- conda install six
- conda install -c https://conda.anaconda.org/groakat pathos
- python setup.py install
script:
- nosetests -v deepchem
+57 −0
Original line number Diff line number Diff line
@@ -155,3 +155,60 @@ class CoulombMatrix(Featurizer):
        else:
          continue
    return d

class CoulombMatrixEig(CoulombMatrix):
  """
  Calculate the eigenvales of Coulomb matrices for molecules.

  Parameters
  ----------
  max_atoms : int
      Maximum number of atoms for any molecule in the dataset. Used to
      pad the Coulomb matrix.
  remove_hydrogens : bool, optional (default False)
      Whether to remove hydrogens before constructing Coulomb matrix.
  randomize : bool, optional (default False)
      Whether to randomize Coulomb matrices to remove dependence on atom
      index order.
  n_samples : int, optional (default 1)
      Number of random Coulomb matrices to generate if randomize is True.
  seed : int, optional
      Random seed.
  """

  conformers = True
  name = 'coulomb_matrix'

  def __init__(self, max_atoms, remove_hydrogens=False, randomize=False,
               n_samples=1, seed=None):
    self.max_atoms = int(max_atoms)
    self.remove_hydrogens = remove_hydrogens
    self.randomize = randomize
    self.n_samples = n_samples
    if seed is not None:
      seed = int(seed)
    self.seed = seed

  def _featurize(self, mol):
    """
    Calculate eigenvalues of Coulomb matrix for molecules. Eigenvalues
    are returned sorted by absolute value in descending order and padded
    by max_atoms. 

    Parameters
    ----------
    mol : RDKit Mol
        Molecule.
    """
    cmat = self.coulomb_matrix(mol)
    features = []
    for f in cmat:
      w, v = np.linalg.eig(f)
      w_abs = np.abs(w)
      sortidx = np.argsort(w_abs)
      sortidx = sortidx[::-1]
      w = w[sortidx]
      f = pad_array(w, self.max_atoms)
      features.append(f)
    features = np.asarray(features)
    return features
+5 −3
Original line number Diff line number Diff line
@@ -2,9 +2,11 @@
#SBATCH --job-name=controller
#SBATCH --output=controller_%A_%a.out
#SBATCH --error=controller_%A_%a.err
#SBATCH --time=24:00:00
#SBATCH --partition=normal
#SBATCH -n 1
#SBATCH --time=12:00:00
##SBATCH --partition=normal
#SBATCH --partition=bigmem
#SBATCH --qos=bigmem
#SBATCH -n 4
ipcontroller --ip='*' --log-to-file=False &
for i in {0..90}; do
    echo $i;
+3 −1
Original line number Diff line number Diff line
@@ -5,8 +5,10 @@ from deepchem.featurizers.basic import RDKitDescriptors
from deepchem.featurizers.nnscore import NNScoreComplexFeaturizer
from deepchem.featurizers.grid_featurizer import GridFeaturizer

dataset_file = "../../datasets/pdbbind_core_df.pkl.gz"
dataset_file = "../../../datasets/pdbbind_full_df.pkl.gz"
print("About to load dataset form disk.")
dataset = load_from_disk(dataset_file)
print("Loaded dataset.")

grid_featurizer = GridFeaturizer(voxel_width=16.0, feature_types="voxel_combined", voxel_feature_types=["ecfp",
                                 "splif", "hbond", "pi_stack", "cation_pi", "salt_bridge"], ecfp_power=9, splif_power=9,