SciCodePile/SciCode-Domain-Code
DATA1: Domain-Specific Code Dataset Dataset Overview DATA1 is a large-scale domain-specific code dataset focusing on code samples from interdisciplinary fields such as biology, chemistry, materials science, and related areas. The dataset is collected and organized from GitHub repositories, covering 178 different domain topics with over 1.1 billion lines of code. Dataset Statistics Total Datasets: 178 CSV files Total Data Size: ~115 GB Total Lines… See the full description on the dataset page: https://huggingface.co/datasets/SciCodePile/SciCode-Domain-Code.
41k
1"keyword","repo_name","file_path","file_extension","file_size","line_count","content","language"
2"Hydrophilic","openvax/pepdata","setup.py",".py","2574","77","# Copyright (c) 2014-2018. Mount Sinai School of Medicine3#4# Licensed under the Apache License, Version 2.0 (the ""License"");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an ""AS IS"" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16 17from __future__ import print_function, division, absolute_import18import os19import re20 21from setuptools import setup, find_packages22 23readme_dir = os.path.dirname(__file__)24readme_path = os.path.join(readme_dir, 'README.md')25 26try:27 with open(readme_path, 'r') as f:28 readme_markdown = f.read()29except:30 print(""Failed to load README file"")31 readme_markdown = """"32 33try:34 import pypandoc35 readme_restructured = pypandoc.convert(readme_markdown, to='rst', format='md')36except:37 readme_restructured = readme_markdown38 print(""Conversion of long_description from markdown to reStructuredText failed, skipping..."")39 40with open('pepdata/__init__.py', 'r') as f:41 version = re.search(42 r'^__version__\s*=\s*[\'""]([^\'""]*)[\'""]',43 f.read(),44 re.MULTILINE).group(1)45 46if __name__ == '__main__':47 setup(48 name='pepdata',49 version=version,50 description=""Immunological peptide datasets and amino acid properties"",51 author=""Alex Rubinsteyn"",52 author_email=""alex.rubinsteyn@mssm.edu"",53 url=""https://github.com/openvax/pepdata"",54 license=""http://www.apache.org/licenses/LICENSE-2.0.html"",55 classifiers=[56 'Development Status :: 3 - Alpha',57 'Environment :: Console',58 'Operating System :: OS Independent',59 'Intended Audience :: Science/Research',60 'License :: OSI Approved :: Apache Software License',61 'Programming Language :: Python',62 'Topic :: Scientific/Engineering :: Bio-Informatics',63 ],64 install_requires=[65 'numpy>=1.7',66 'scipy>=0.9',67 'pandas>=0.17',68 'scikit-learn>=0.14.1',69 'progressbar33',70 'biopython>=1.65',71 'datacache>=0.4.4',72 'lxml',73 ],74 long_description=readme_restructured,75 packages=find_packages(exclude=""test""),76 include_package_data=True77 )78","Python"
79"Hydrophilic","openvax/pepdata","deploy.sh",".sh","273","10","./lint.sh && \80./test.sh && \81python3 -m pip install --upgrade build && \82python3 -m pip install --upgrade twine && \83rm -rf dist && \84python3 -m build && \85git --version && \86python3 -m twine upload dist/* && \87git tag ""$(python3 pepdata/version.py)"" && \88git push --tags","Shell"
89"Hydrophilic","openvax/pepdata","develop.sh",".sh","25","4","set -e90 91pip install -e .92","Shell"
93"Hydrophilic","openvax/pepdata","lint.sh",".sh","154","10","#!/bin/bash94set -o errexit95 96find pepdata test -name '*.py' \97 | xargs pylint \98 --errors-only \99 --disable=print-statement100 101echo 'Passes pylint check'102","Shell"
103"Hydrophilic","openvax/pepdata","test.sh",".sh","56","4","pytest --cov=pepdata/ --cov-report=term-missing tests104 105 106","Shell"
107"Hydrophilic","openvax/pepdata","pepdata/common.py",".py","967","28","# Copyright (c) 2014-2016. Mount Sinai School of Medicine108#109# Licensed under the Apache License, Version 2.0 (the ""License"");110# you may not use this file except in compliance with the License.111# You may obtain a copy of the License at112#113# http://www.apache.org/licenses/LICENSE-2.0114#115# Unless required by applicable law or agreed to in writing, software116# distributed under the License is distributed on an ""AS IS"" BASIS,117# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.118# See the License for the specific language governing permissions and119# limitations under the License.120 121 122from __future__ import print_function, division, absolute_import123 124import numpy as np125 126def transform_peptide(peptide, property_dict):127 return np.array([property_dict[amino_acid] for amino_acid in peptide])128 129def transform_peptides(peptides, property_dict):130 return np.array([131 [property_dict[aa] for aa in peptide]132 for peptide in peptides])133 134","Python"
135"Hydrophilic","openvax/pepdata","pepdata/static_data.py",".py","741","19","# Licensed under the Apache License, Version 2.0 (the ""License"");136# you may not use this file except in compliance with the License.137# You may obtain a copy of the License at138#139# http://www.apache.org/licenses/LICENSE-2.0140#141# Unless required by applicable law or agreed to in writing, software142# distributed under the License is distributed on an ""AS IS"" BASIS,143# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.144# See the License for the specific language governing permissions and145# limitations under the License.146 147 148from __future__ import print_function, division, absolute_import149from os.path import dirname, realpath, join150 151PACKAGE_DIR = dirname(realpath(__file__))152MATRIX_DIR = join(PACKAGE_DIR, 'matrices')153","Python"
154"Hydrophilic","openvax/pepdata","pepdata/version.py",".py","121","8","__version__ = ""1.2.0""155 156 157def print_version():158 print(f""v{__version__}"")159 160if __name__ == ""__main__"":161 print_version()","Python"
162"Hydrophilic","openvax/pepdata","pepdata/__init__.py",".py","597","27","from .amino_acid_alphabet import (163 AminoAcid,164 canonical_amino_acids,165 canonical_amino_acid_letters,166 extended_amino_acids,167 extended_amino_acid_letters,168 amino_acid_letter_indices,169 amino_acid_name_indices,170)171from .peptide_vectorizer import PeptideVectorizer172from .version import __version__173from . import iedb174 175 176 177__all__ = [178 ""iedb"",179 ""AminoAcid"",180 ""canonical_amino_acids"",181 ""canonical_amino_acid_letters"",182 ""extended_amino_acids"",183 ""extended_amino_acid_letters"",184 ""amino_acid_letter_indices"",185 ""amino_acid_name_indices"",186 ""PeptideVectorizer"",187]188","Python"
189"Hydrophilic","openvax/pepdata","pepdata/amino_acid.py",".py","1287","37","# Licensed under the Apache License, Version 2.0 (the ""License"");190# you may not use this file except in compliance with the License.191# You may obtain a copy of the License at192#193# http://www.apache.org/licenses/LICENSE-2.0194#195# Unless required by applicable law or agreed to in writing, software196# distributed under the License is distributed on an ""AS IS"" BASIS,197# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.198# See the License for the specific language governing permissions and199# limitations under the License.200 201 202from __future__ import print_function, division, absolute_import203 204class AminoAcid(object):205 def __init__(206 self, full_name, short_name, letter, contains=None):207 self.letter = letter208 self.full_name = full_name209 self.short_name = short_name210 if not contains:211 contains = [letter]212 self.contains = contains213 214 def __str__(self):215 return (216 (""AminoAcid(full_name='%s', short_name='%s', letter='%s', ""217 ""contains=%s)"") % (218 self.letter, self.full_name, self.short_name, self.contains))219 220 def __repr__(self):221 return str(self)222 223 def __eq__(self, other):224 return other.__class__ is AminoAcid and self.letter == other.letter225","Python"
226"Hydrophilic","openvax/pepdata","pepdata/amino_acid_alphabet.py",".py","4682","161","# Licensed under the Apache License, Version 2.0 (the ""License"");227# you may not use this file except in compliance with the License.228# You may obtain a copy of the License at229#230# http://www.apache.org/licenses/LICENSE-2.0231#232# Unless required by applicable law or agreed to in writing, software233# distributed under the License is distributed on an ""AS IS"" BASIS,234# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.235# See the License for the specific language governing permissions and236# limitations under the License.237 238 239""""""240Quantify amino acids by their physical/chemical properties241""""""242 243from __future__ import print_function, division, absolute_import244 245import numpy as np246 247from .amino_acid import AminoAcid248 249canonical_amino_acids = [250 AminoAcid(""Alanine"", ""Ala"", ""A""),251 AminoAcid(""Arginine"", ""Arg"", ""R""),252 AminoAcid(""Asparagine"",""Asn"", ""N""),253 AminoAcid(""Aspartic Acid"", ""Asp"", ""D""),254 AminoAcid(""Cysteine"", ""Cys"", ""C""),255 AminoAcid(""Glutamic Acid"", ""Glu"", ""E""),256 AminoAcid(""Glutamine"", ""Gln"", ""Q""),257 AminoAcid(""Glycine"", ""Gly"", ""G""),258 AminoAcid(""Histidine"", ""His"", ""H""),259 AminoAcid(""Isoleucine"", ""Ile"", ""I""),260 AminoAcid(""Leucine"", ""Leu"", ""L""),261 AminoAcid(""Lysine"", ""Lys"", ""K""),262 AminoAcid(""Methionine"", ""Met"", ""M""),263 AminoAcid(""Phenylalanine"", ""Phe"", ""F""),264 AminoAcid(""Proline"", ""Pro"", ""P""),265 AminoAcid(""Serine"", ""Ser"", ""S""),266 AminoAcid(""Threonine"", ""Thr"", ""T""),267 AminoAcid(""Tryptophan"", ""Trp"", ""W""),268 AminoAcid(""Tyrosine"", ""Tyr"", ""Y""),269 AminoAcid(""Valine"", ""Val"", ""V"")270]271 272canonical_amino_acid_letters = [aa.letter for aa in canonical_amino_acids]273 274###275# Post-translation modifications commonly detected by mass-spec276###277 278# TODO: figure out three letter codes for modified AAs279 280modified_amino_acids = [281 AminoAcid(""Phospho-Serine"", ""Sep"", ""s""),282 AminoAcid(""Phospho-Threonine"", ""???"", ""t""),283 AminoAcid(""Phospho-Tyrosine"", ""???"", ""y""),284 AminoAcid(""Cystine"", ""???"", ""c""),285 AminoAcid(""Methionine sulfoxide"", ""???"", ""m""),286 AminoAcid(""Pyroglutamate"", ""???"", ""q""),287 AminoAcid(""Pyroglutamic acid"", ""???"", ""n""),288]289 290###291# Amino acid tokens which represent multiple canonical amino acids292###293wildcard_amino_acids = [294 AminoAcid(""Unknown"", ""Xaa"", ""X"", contains=set(canonical_amino_acid_letters)),295 AminoAcid(""Asparagine-or-Aspartic-Acid"", ""Asx"", ""B"", contains={""D"", ""N""}),296 AminoAcid(""Glutamine-or-Glutamic-Acid"", ""Glx"", ""Z"", contains={""E"", ""Q""}),297 AminoAcid(""Leucine-or-Isoleucine"", ""Xle"", ""J"", contains={""I"", ""L""})298]299 300###301# Canonical amino acids + wilcard tokens302###303 304canonical_amino_acids_with_unknown = canonical_amino_acids + wildcard_amino_acids305 306 307###308# Rare amino acids which aren't considered part of the core 20 ""canonical""309###310 311rare_amino_acids = [312 AminoAcid(""Selenocysteine"", ""Sec"", ""U""),313 AminoAcid(""Pyrrolysine"", ""Pyl"", ""O""),314]315 316###317# Extended amino acids + wildcard tokens318###319 320extended_amino_acids = canonical_amino_acids + rare_amino_acids + wildcard_amino_acids321extended_amino_acid_letters = [322 aa.letter for aa in extended_amino_acids323]324extended_amino_acids_with_unknown_names = [325 aa.full_name for aa in extended_amino_acids326]327 328 329amino_acid_letter_indices = {330 c: i for (i, c) in331 enumerate(extended_amino_acid_letters)332}333 334 335amino_acid_letter_pairs = [336 ""%s%s"" % (x, y)337 for y in extended_amino_acids338 for x in extended_amino_acids339]340 341 342amino_acid_name_indices = {343 aa_name: i for (i, aa_name)344 in enumerate(extended_amino_acids_with_unknown_names)345}346 347amino_acid_pair_positions = {348 pair: i for (i, pair) in enumerate(amino_acid_letter_pairs)349}350 351def index_to_full_name(idx):352 return extended_amino_acids[idx].full_name353 354def index_to_short_name(idx):355 return extended_amino_acids[idx].short_name356 357def index_to_letter(idx):358 return extended_amino_acids[idx]359 360def letter_to_index(x):361 """"""362 Convert from an amino acid's letter code to its position index363 """"""364 assert x in amino_acid_letter_indices, ""Unknown amino acid: %s"" % x365 return amino_acid_letter_indices[x]366 367def peptide_to_indices(xs):368 return [amino_acid_letter_indices[x] for x in xs]369 370def letter_to_short_name(x):371 return index_to_short_name(letter_to_index(x))372 373def peptide_to_short_amino_acid_names(xs):374 return [amino_acid_letter_indices[x] for x in xs]375 376def dict_to_amino_acid_matrix(d, alphabet=canonical_amino_acids):377 n_aa = len(d)378 result_matrix = np.zeros((n_aa, n_aa), dtype=""float32"")379 for i, aa_row in enumerate(alphabet):380 d_row = d[aa_row.letter]381 for j, aa_col in enumerate(alphabet):382 value = d_row[aa_col.letter]383 result_matrix[i, j] = value384 return result_matrix385 386","Python"
387"Hydrophilic","openvax/pepdata","pepdata/amino_acid_properties.py",".py","6268","360","# Licensed under the Apache License, Version 2.0 (the ""License"");388# you may not use this file except in compliance with the License.389# You may obtain a copy of the License at390#391# http://www.apache.org/licenses/LICENSE-2.0392#393# Unless required by applicable law or agreed to in writing, software394# distributed under the License is distributed on an ""AS IS"" BASIS,395# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.396# See the License for the specific language governing permissions and397# limitations under the License.398 399from __future__ import print_function, division, absolute_import400 401from .amino_acid_alphabet import letter_to_index402 403""""""404Quantify amino acids by their physical/chemical properties405""""""406 407 408def aa_dict_to_positional_list(aa_property_dict):409 value_list = [None] * 20410 for letter, value in aa_property_dict.items():411 idx = letter_to_index(letter)412 assert idx >= 0413 assert idx < 20414 value_list[idx] = value415 assert all(elt is not None for elt in value_list), \416 ""Missing amino acids in:\n%s"" % aa_property_dict.keys()417 return value_list418 419def parse_property_table(table_string):420 value_dict = {}421 for line in table_string.splitlines():422 line = line.strip()423 if not line:424 continue425 fields = line.split("" "")426 fields = [f for f in fields if len(f.strip()) > 0]427 assert len(fields) >= 2428 value, letter = fields[:2]429 assert letter not in value_dict, ""Repeated amino acid "" + line430 value_dict[letter] = float(value)431 return value_dict432 433 434""""""435Amino acids property tables copied from CRASP website436""""""437 438hydropathy = parse_property_table(""""""4391.80000 A ALA440-4.5000 R ARG441-3.5000 N ASN442-3.5000 D ASP4432.50000 C CYS444-3.5000 Q GLN445-3.5000 E GLU446-0.4000 G GLY447-3.2000 H HIS4484.50000 I ILE4493.80000 L LEU450-3.9000 K LYS4511.90000 M MET4522.80000 F PHE453-1.6000 P PRO454-0.8000 S SER455-0.7000 T THR456-0.9000 W TRP457-1.3000 Y TYR4584.20000 V VAL459"""""")460 461volume = parse_property_table(""""""46291.5000 A ALA463202.0000 R ARG464135.2000 N ASN465124.5000 D ASP466118.0000 C CYS467161.1000 Q GLN468155.1000 E GLU46966.40000 G GLY470167.3000 H HIS471168.8000 I ILE472167.9000 L LEU473171.3000 K LYS474170.8000 M MET475203.4000 F PHE476129.3000 P PRO47799.10000 S SER478122.1000 T THR479237.6000 W TRP480203.6000 Y TYR481141.7000 V VAL482"""""")483 484polarity = parse_property_table(""""""4850.0000 A ALA48652.000 R ARG4873.3800 N ASN48840.700 D ASP4891.4800 C CYS4903.5300 Q GLN49149.910 E GLU4920.0000 G GLY49351.600 H HIS4940.1500 I ILE4950.4500 L LEU49649.500 K LYS4971.4300 M MET4980.3500 F PHE4991.5800 P PRO5001.6700 S SER5011.6600 T THR5022.1000 W TRP5031.6100 Y TYR5040.1300 V VAL505"""""")506 507pK_side_chain = parse_property_table(""""""5080.0000 A ALA50912.480 R ARG5100.0000 N ASN5113.6500 D ASP5128.1800 C CYS5130.0000 Q GLN5144.2500 E GLU5150.0000 G GLY5166.0000 H HIS5170.0000 I ILE5180.0000 L LEU51910.530 K LYS5200.0000 M MET5210.0000 F PHE5220.0000 P PRO5230.0000 S SER5240.0000 T THR5250.0000 W TRP52610.700 Y TYR5270.0000 V VAL528"""""")529 530prct_exposed_residues = parse_property_table(""""""53115.0000 A ALA53267.0000 R ARG53349.0000 N ASN53450.0000 D ASP5355.00000 C CYS53656.0000 Q GLN53755.0000 E GLU53810.0000 G GLY53934.0000 H HIS54013.0000 I ILE54116.0000 L LEU54285.0000 K LYS54320.0000 M MET54410.0000 F PHE54545.0000 P PRO54632.0000 S SER54732.0000 T THR54817.0000 W TRP54941.0000 Y TYR55014.0000 V VAL551"""""")552 553hydrophilicity = parse_property_table(""""""554-0.5000 A ALA5553.00000 R ARG5560.20000 N ASN5573.00000 D ASP558-1.0000 C CYS5590.20000 Q GLN5603.00000 E GLU5610.00000 G GLY562-0.5000 H HIS563-1.8000 I ILE564-1.8000 L LEU5653.00000 K LYS566-1.3000 M MET567-2.5000 F PHE5680.00000 P PRO5690.30000 S SER570-0.4000 T THR571-3.4000 W TRP572-2.3000 Y TYR573-1.5000 V VAL574"""""")575 576accessible_surface_area = parse_property_table(""""""57727.8000 A ALA57894.7000 R ARG57960.1000 N ASN58060.6000 D ASP58115.5000 C CYS58268.7000 Q GLN58368.2000 E GLU58424.5000 G GLY58550.7000 H HIS58622.8000 I ILE58727.6000 L LEU588103.000 K LYS58933.5000 M MET59025.5000 F PHE59151.5000 P PRO59242.0000 S SER59345.0000 T THR59434.7000 W TRP59555.2000 Y TYR59623.7000 V VAL597"""""")598 599local_flexibility = parse_property_table(""""""600705.42000 A ALA6011484.2800 R ARG602513.46010 N ASN60334.960000 D ASP6042412.5601 C CYS6051087.8300 Q GLN6061158.6600 E GLU60733.180000 G GLY6081637.1300 H HIS6095979.3701 I ILE6104985.7300 L LEU611699.69000 K LYS6124491.6602 M MET6135203.8599 F PHE614431.96000 P PRO615174.76000 S SER616601.88000 T THR6176374.0698 W TRP6184291.1001 Y TYR6194474.4199 V VAL620"""""")621 622accessible_surface_area_folded = parse_property_table(""""""62331.5000 A ALA62493.8000 R ARG62562.2000 N ASN62660.9000 D ASP62713.9000 C CYS62874.0000 Q GLN62972.3000 E GLU63025.2000 G GLY63146.7000 H HIS63223.0000 I ILE63329.0000 L LEU634110.300 K LYS63530.5000 M MET63628.7000 F PHE63753.7000 P PRO63844.2000 S SER63946.0000 T THR64041.7000 W TRP64159.1000 Y TYR64223.5000 V VAL643"""""")644 645refractivity = parse_property_table(""""""6464.34000 A ALA64726.6600 R ARG64813.2800 N ASN64912.0000 D ASP65035.7700 C CYS65117.5600 Q GLN65217.2600 E GLU6530.00000 G GLY65421.8100 H HIS65519.0600 I ILE65618.7800 L LEU65721.2900 K LYS65821.6400 M MET65929.4000 F PHE66010.9300 P PRO6616.35000 S SER66211.0100 T THR66342.5300 W TRP66431.5300 Y TYR66513.9200 V VAL666"""""")667 668 669mass = parse_property_table(""""""67070.079 A ALA671156.188 R ARG672114.104 N ASN673115.089 D ASP674103.144 C CYS675128.131 Q GLN676129.116 E GLU67757.052 G GLY678137.142 H HIS679113.160 I ILE680113.160 L LEU681128.174 K LYS682131.198 M MET683147.177 F PHE68497.177 P PRO68587.078 S SER686101.105 T THR687186.213 W TRP688163.170 Y TYR68999.133 V VAL690"""""")691 692###693# Values copied from:694# ""Solvent accessibility of AA in known protein structures""695# http://prowl.rockefeller.edu/aainfo/access.htm696###697""""""698Solvent accessibility of AA in known protein structures699 700Figure 1.701 702S 0.70 0.20 0.10703T 0.71 0.16 0.13704A 0.48 0.35 0.17705G 0.51 0.36 0.13706P 0.78 0.13 0.09707C 0.32 0.54 0.14708D 0.81 0.09 0.10709E 0.93 0.04 0.03710Q 0.81 0.10 0.09711N 0.82 0.10 0.08712L 0.41 0.49 0.10713I 0.39 0.47 0.14714V 0.40 0.50 0.10715M 0.44 0.20 0.36716F 0.42 0.42 0.16717Y 0.67 0.20 0.13718W 0.49 0.44 0.07719K 0.93 0.02 0.05720R 0.84 0.05 0.11721H 0.66 0.19 0.15722""""""723 724solvent_exposed_area = dict(725 S=0.70,726 T=0.71,727 A=0.48,728 G=0.51,729 P=0.78,730 C=0.32,731 D=0.81,732 E=0.93,733 Q=0.81,734 N=0.82,735 L=0.41,736 I=0.39,737 V=0.40,738 M=0.44,739 F=0.42,740 Y=0.67,741 W=0.49,742 K=0.93,743 R=0.84,744 H=0.66,745)746","Python"
747"Hydrophilic","openvax/pepdata","pepdata/reduced_alphabet.py",".py","1784","58","# Copyright (c) 2014-2018. Mount Sinai School of Medicine748#749# Licensed under the Apache License, Version 2.0 (the ""License"");750# you may not use this file except in compliance with the License.751# You may obtain a copy of the License at752#753# http://www.apache.org/licenses/LICENSE-2.0754#755# Unless required by applicable law or agreed to in writing, software756# distributed under the License is distributed on an ""AS IS"" BASIS,757# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.758# See the License for the specific language governing permissions and759# limitations under the License.760 761""""""762Amino acid groupings from763'Reduced amino acid alphabets improve the sensitivity...' by764Peterson, Kondev, et al.765http://www.rpgroup.caltech.edu/publications/Peterson2008.pdf766""""""767from __future__ import print_function, division, absolute_import768 769def dict_from_list(groups):770 aa_to_group = {}771 for i, group in enumerate(groups):772 for c in group:773 aa_to_group[c] = group[0]774 return aa_to_group775 776gbmr4 = dict_from_list([""ADKERNTSQ"", ""YFLIVMCWH"", ""G"", ""P""])777 778sdm12 = dict_from_list([779 ""A"", ""D"", ""KER"", ""N"", ""TSQ"", ""YF"", ""LIVM"", ""C"", ""W"", ""H"", ""G"", ""P""780])781 782hsdm17 = dict_from_list([783 ""A"", ""D"", ""KE"", ""R"", ""N"", ""T"", ""S"", ""Q"", ""Y"",784 ""F"", ""LIV"", ""M"", ""C"", ""W"", ""H"", ""G"", ""P""785])786 787""""""788Other alphabets from789http://bio.math-inf.uni-greifswald.de/viscose/html/alphabets.html790""""""791 792# hydrophilic vs. hydrophobic793hp2 = dict_from_list([""AGTSNQDEHRKP"", ""CMFILVWY""])794 795murphy10 = dict_from_list([796 ""LVIM"", ""C"", ""A"", ""G"", ""ST"", ""P"", ""FYW"", ""EDNQ"", ""KR"", ""H""797])798 799alex6 = dict_from_list([""C"", ""G"", ""P"", ""FYW"", ""AVILM"", ""STNQRHKDE""])800 801aromatic2 = dict_from_list([""FHWY"", ""ADKERNTSQLIVMCGP""])802 803hp_vs_aromatic = dict_from_list([""H"", ""CMILV"", ""FWY"", ""ADKERNTSQGP""])804","Python"
805"Hydrophilic","openvax/pepdata","pepdata/peptide_vectorizer.py",".py","2942","84","# Copyright (c) 2014-2016. Mount Sinai School of Medicine806#807# Licensed under the Apache License, Version 2.0 (the ""License"");808# you may not use this file except in compliance with the License.809# You may obtain a copy of the License at810#811# http://www.apache.org/licenses/LICENSE-2.0812#813# Unless required by applicable law or agreed to in writing, software814# distributed under the License is distributed on an ""AS IS"" BASIS,815# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.816# See the License for the specific language governing permissions and817# limitations under the License.818 819 820from __future__ import print_function, division, absolute_import821 822import numpy as np823from sklearn.feature_extraction.text import CountVectorizer824from sklearn.preprocessing import normalize825 826def make_count_vectorizer(reduced_alphabet, max_ngram):827 if reduced_alphabet is None:828 preprocessor = None829 else:830 preprocessor = lambda s: """".join([reduced_alphabet[si] for si in s])831 832 return CountVectorizer(833 analyzer='char',834 ngram_range=(1, max_ngram),835 dtype=np.float,836 preprocessor=preprocessor)837 838class PeptideVectorizer(object):839 """"""840 Make n-gram frequency vectors from peptide sequences841 """"""842 def __init__(843 self,844 max_ngram=1,845 normalize_row=True,846 reduced_alphabet=None,847 training_already_reduced=False):848 self.reduced_alphabet = reduced_alphabet849 self.max_ngram = max_ngram850 self.normalize_row = normalize_row851 self.training_already_reduced = training_already_reduced852 self.count_vectorizer = None853 854 def __getstate__(self):855 return {856 'reduced_alphabet': self.reduced_alphabet,857 'count_vectorizer': self.count_vectorizer,858 'training_already_reduced': self.training_already_reduced,859 'normalize_row': self.normalize_row,860 'max_ngram': self.max_ngram,861 }862 863 def fit_transform(self, amino_acid_strings):864 self.count_vectorizer = \865 make_count_vectorizer(self.reduced_alphabet, self.max_ngram)866 867 if self.training_already_reduced:868 c = make_count_vectorizer(None, self.max_ngram)869 X = c.fit_transform(amino_acid_strings).todense()870 self.count_vectorizer.vocabulary_ = c.vocabulary_871 else:872 c = self.count_vectorizer873 X = c.fit_transform(amino_acid_strings).todense()874 875 if self.normalize_row:876 X = normalize(X, norm='l1')877 return X878 879 def fit(self, amino_acid_strings):880 self.fit_transform(amino_acid_strings)881 882 def transform(self, amino_acid_strings):883 assert self.count_vectorizer, ""Must call 'fit' before 'transform'""884 X = self.count_vectorizer.transform(amino_acid_strings).todense()885 if self.normalize_row:886 X = normalize(X, norm='l1')887 return X888","Python"
889"Hydrophilic","openvax/pepdata","pepdata/chou_fasman.py",".py","3279","75","# Licensed under the Apache License, Version 2.0 (the ""License"");890# you may not use this file except in compliance with the License.891# You may obtain a copy of the License at892#893# http://www.apache.org/licenses/LICENSE-2.0894#895# Unless required by applicable law or agreed to in writing, software896# distributed under the License is distributed on an ""AS IS"" BASIS,897# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.898# See the License for the specific language governing permissions and899# limitations under the License.900 901from __future__ import print_function, division, absolute_import902 903from .amino_acid_alphabet import amino_acid_name_indices904 905# Chou-Fasman of structural properties from906# http://prowl.rockefeller.edu/aainfo/chou.htm907chou_fasman_table = """"""908Alanine 142 83 66 0.06 0.076 0.035 0.058909Arginine 98 93 95 0.070 0.106 0.099 0.085910Aspartic Acid 101 54 146 0.147 0.110 0.179 0.081911Asparagine 67 89 156 0.161 0.083 0.191 0.091912Cysteine 70 119 119 0.149 0.050 0.117 0.128913Glutamic Acid 151 037 74 0.056 0.060 0.077 0.064914Glutamine 111 110 98 0.074 0.098 0.037 0.098915Glycine 57 75 156 0.102 0.085 0.190 0.152916Histidine 100 87 95 0.140 0.047 0.093 0.054917Isoleucine 108 160 47 0.043 0.034 0.013 0.056918Leucine 121 130 59 0.061 0.025 0.036 0.070919Lysine 114 74 101 0.055 0.115 0.072 0.095920Methionine 145 105 60 0.068 0.082 0.014 0.055921Phenylalanine 113 138 60 0.059 0.041 0.065 0.065922Proline 57 55 152 0.102 0.301 0.034 0.068923Serine 77 75 143 0.120 0.139 0.125 0.106924Threonine 83 119 96 0.086 0.108 0.065 0.079925Tryptophan 108 137 96 0.077 0.013 0.064 0.167926Tyrosine 69 147 114 0.082 0.065 0.114 0.125927Valine 106 170 50 0.062 0.048 0.028 0.053928""""""929 930 931def parse_chou_fasman(table):932 alpha_helix_score_dict = {}933 beta_sheet_score_dict = {}934 turn_score_dict = {}935 936 for line in table.split(""\n""):937 fields = [field for field in line.split("" "") if len(field.strip()) > 0]938 if len(fields) == 0:939 continue940 941 if fields[1] == 'Acid':942 name = fields[0] + "" "" + fields[1]943 fields = fields[1:]944 else:945 name = fields[0]946 947 assert name in amino_acid_name_indices, ""Invalid amino acid name %s"" % name948 letter = amino_acid_name_indices[name]949 alpha = int(fields[1])950 beta = int(fields[2])951 turn = int(fields[3])952 alpha_helix_score_dict[letter] = alpha953 beta_sheet_score_dict[letter] = beta954 turn_score_dict[letter] = turn955 956 assert len(alpha_helix_score_dict) == 20957 assert len(beta_sheet_score_dict) == 20958 assert len(turn_score_dict) == 20959 return alpha_helix_score_dict, beta_sheet_score_dict, turn_score_dict960 961alpha_helix_score, beta_sheet_score, turn_score = \962 parse_chou_fasman(chou_fasman_table)963","Python"
964"Hydrophilic","openvax/pepdata","pepdata/pmbec.py",".py","3019","89","# Copyright (c) 2014-2016. Mount Sinai School of Medicine965#966# Licensed under the Apache License, Version 2.0 (the ""License"");967# you may not use this file except in compliance with the License.968# You may obtain a copy of the License at969#970# http://www.apache.org/licenses/LICENSE-2.0971#972# Unless required by applicable law or agreed to in writing, software973# distributed under the License is distributed on an ""AS IS"" BASIS,974# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.975# See the License for the specific language governing permissions and976# limitations under the License.977 978from __future__ import print_function, division, absolute_import979from os.path import join980 981from .static_data import MATRIX_DIR982 983from .amino_acid_alphabet import dict_to_amino_acid_matrix984 985def read_pmbec_coefficients(986 key_type='row',987 verbose=True,988 filename=join(MATRIX_DIR, 'pmbec.mat')):989 """"""990 Parameters991 ------------992 993 filename : str994 Location of PMBEC coefficient matrix995 996 key_type : str997 'row' : every key is a single amino acid,998 which maps to a dictionary for that row999 'pair' : every key is a tuple of amino acids1000 'pair_string' : every key is a string of two amino acid characters1001 1002 verbose : bool1003 Print rows of matrix as we read them1004 """"""1005 d = {}1006 if key_type == 'row':1007 def add_pair(row_letter, col_letter, value):1008 if row_letter not in d:1009 d[row_letter] = {}1010 d[row_letter][col_letter] = value1011 elif key_type == 'pair':1012 def add_pair(row_letter, col_letter, value):1013 d[(row_letter, col_letter)] = value1014 1015 else:1016 assert key_type == 'pair_string', \1017 ""Invalid dictionary key type: %s"" % key_type1018 1019 def add_pair(row_letter, col_letter, value):1020 d[""%s%s"" % (row_letter, col_letter)] = value1021 1022 with open(filename, 'r') as f:1023 lines = [line for line in f.read().split('\n') if len(line) > 0]1024 header = lines[0]1025 if verbose:1026 print(header)1027 residues = [1028 x for x in header.split()1029 if len(x) == 1 and x != ' ' and x != '\t'1030 ]1031 assert len(residues) == 201032 if verbose:1033 print(residues)1034 for line in lines[1:]:1035 cols = [1036 x1037 for x in line.split(' ')1038 if len(x) > 0 and x != ' ' and x != '\t'1039 ]1040 assert len(cols) == 21, ""Expected 20 values + letter, got %s"" % cols1041 row_letter = cols[0]1042 for i, col in enumerate(cols[1:]):1043 col_letter = residues[i]1044 assert col_letter != ' ' and col_letter != '\t'1045 value = float(col)1046 add_pair(row_letter, col_letter, value)1047 return d1048 1049# dictionary of PMBEC coefficient accessed like pmbec_dict[""V""][""R""]1050pmbec_dict = read_pmbec_coefficients(key_type=""row"")1051pmbec_matrix = dict_to_amino_acid_matrix(pmbec_dict)1052","Python"
1053"Hydrophilic","openvax/pepdata","pepdata/residue_contact_energies.py",".py","2937","77","# Licensed under the Apache License, Version 2.0 (the ""License"");1054# you may not use this file except in compliance with the License.1055# You may obtain a copy of the License at1056#1057# http://www.apache.org/licenses/LICENSE-2.01058#1059# Unless required by applicable law or agreed to in writing, software1060# distributed under the License is distributed on an ""AS IS"" BASIS,1061# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1062# See the License for the specific language governing permissions and1063# limitations under the License.1064 1065from __future__ import print_function, division, absolute_import1066 1067from os.path import join1068 1069from .amino_acid_alphabet import canonical_amino_acid_letters, dict_to_amino_acid_matrix1070from .static_data import MATRIX_DIR1071 1072 1073def parse_interaction_table(table, amino_acid_order=""ARNDCQEGHILKMFPSTWYV""):1074 table = table.strip()1075 while "" "" in table:1076 table = table.replace("" "", "" "")1077 1078 lines = [l.strip() for l in table.split(""\n"")]1079 lines = [l for l in lines if len(l) > 0 and not l.startswith(""#"")]1080 assert len(lines) == 20, ""Malformed amino acid interaction table""1081 d = {}1082 for i, line in enumerate(lines):1083 coeff_strings = line.split("" "")1084 assert len(coeff_strings) == 20, \1085 ""Malformed row in amino acid interaction table""1086 x = amino_acid_order[i]1087 d[x] = {}1088 for j, coeff_str in enumerate(coeff_strings):1089 value = float(coeff_str)1090 y = amino_acid_order[j]1091 d[x][y] = value1092 return d1093 1094def transpose_interaction_dict(d):1095 transposed = {}1096 for x in canonical_amino_acid_letters:1097 transposed[x] = {}1098 for y in canonical_amino_acid_letters:1099 transposed[x][y] = d[y][x]1100 return transposed1101 1102 1103with open(join(MATRIX_DIR, 'strand_vs_coil.txt'), 'r') as f:1104 # Strand vs. Coil1105 strand_vs_coil_dict = parse_interaction_table(f.read())1106 strand_vs_coil_array = dict_to_amino_acid_matrix(strand_vs_coil_dict)1107 1108 # Coil vs. Strand1109 coil_vs_strand_dict = transpose_interaction_dict(strand_vs_coil_dict)1110 coil_vs_strand_array = dict_to_amino_acid_matrix(coil_vs_strand_dict)1111 1112with open(join(MATRIX_DIR, 'helix_vs_strand.txt'), 'r') as f:1113 # Helix vs. Strand1114 helix_vs_strand_dict = parse_interaction_table(f.read())1115 helix_vs_strand_array = dict_to_amino_acid_matrix(helix_vs_strand_dict)1116 1117 # Strand vs. Helix1118 strand_vs_helix_dict = transpose_interaction_dict(helix_vs_strand_dict)1119 strand_vs_helix_array = dict_to_amino_acid_matrix(strand_vs_helix_dict)1120 1121with open(join(MATRIX_DIR, 'helix_vs_coil.txt'), 'r') as f:1122 # Helix vs. Coil1123 helix_vs_coil_dict = parse_interaction_table(f.read())1124 helix_vs_coil_array = dict_to_amino_acid_matrix(helix_vs_coil_dict)1125 1126 # Coil vs. Helix1127 coil_vs_helix_dict = transpose_interaction_dict(helix_vs_coil_dict)1128 coil_vs_helix_array = dict_to_amino_acid_matrix(coil_vs_helix_dict)1129","Python"
1130"Hydrophilic","openvax/pepdata","pepdata/blosum.py",".py","2600","76","# Licensed under the Apache License, Version 2.0 (the ""License"");1131# you may not use this file except in compliance with the License.1132# You may obtain a copy of the License at1133#1134# http://www.apache.org/licenses/LICENSE-2.01135#1136# Unless required by applicable law or agreed to in writing, software1137# distributed under the License is distributed on an ""AS IS"" BASIS,1138# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1139# See the License for the specific language governing permissions and1140# limitations under the License.1141 1142from __future__ import print_function, division, absolute_import1143 1144from os.path import join1145 1146from .static_data import MATRIX_DIR1147 1148from .amino_acid_alphabet import dict_to_amino_acid_matrix1149 1150def parse_blosum_table(table, coeff_type=int, key_type='row'):1151 """"""1152 Parse a table of pairwise amino acid coefficient (e.g. BLOSUM50)1153 """"""1154 1155 lines = table.split(""\n"")1156 # drop comments1157 lines = [line for line in lines if not line.startswith(""#"")]1158 # drop CR endline characters1159 lines = [line.replace(""\r"", """") for line in lines]1160 # skip empty lines1161 lines = [line for line in lines if line]1162 1163 labels = lines[0].split()1164 1165 if len(labels) < 20:1166 raise ValueError(1167 ""Expected 20+ amino acids but first line '%s' has %d fields"" % (1168 lines[0],1169 len(labels)))1170 coeffs = {}1171 for line in lines[1:]:1172 1173 fields = line.split()1174 assert len(fields) >= 21, \1175 ""Expected AA and 20+ coefficients but '%s' has %d fields"" % (1176 line, len(fields))1177 x = fields[0]1178 for i, coeff_str in enumerate(fields[1:]):1179 y = labels[i]1180 coeff = coeff_type(coeff_str)1181 if key_type == 'pair':1182 coeffs[(x, y)] = coeff1183 elif key_type == 'pair_string':1184 coeffs[x + y] = coeff1185 else:1186 assert key_type == 'row', ""Unknown key type: %s"" % key_type1187 if x not in coeffs:1188 coeffs[x] = {}1189 coeffs[x][y] = coeff1190 return coeffs1191 1192 1193with open(join(MATRIX_DIR, 'BLOSUM30'), 'r') as f:1194 blosum30_dict = parse_blosum_table(f.read())1195 blosum30_matrix = dict_to_amino_acid_matrix(blosum30_dict)1196 1197with open(join(MATRIX_DIR, 'BLOSUM50'), 'r') as f:1198 blosum50_dict = parse_blosum_table(f.read())1199 blosum50_matrix = dict_to_amino_acid_matrix(blosum50_dict)1200 