Team Ai
Datasetpublic

SciCodePile/SciCode-Domain-Code

DATA1: Domain-Specific Code Dataset Dataset Overview DATA1 is a large-scale domain-specific code dataset focusing on code samples from interdisciplinary fields such as biology, chemistry, materials science, and related areas. The dataset is collected and organized from GitHub repositories, covering 178 different domain topics with over 1.1 billion lines of code. Dataset Statistics Total Datasets: 178 CSV files Total Data Size: ~115 GB Total Lines… See the full description on the dataset page: https://huggingface.co/datasets/SciCodePile/SciCode-Domain-Code.

sourceHugging Faceapache-2.0updated 7mo agoView on Hugging Face
4likes1kdownloads
dataset_Hydrophilic.csv2178 linesDownload Raw Back to data
1"keyword","repo_name","file_path","file_extension","file_size","line_count","content","language"
2"Hydrophilic","openvax/pepdata","setup.py",".py","2574","77","# Copyright (c) 2014-2018. Mount Sinai School of Medicine3#4# Licensed under the Apache License, Version 2.0 (the ""License"");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an ""AS IS"" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16 17from __future__ import print_function, division, absolute_import18import os19import re20 21from setuptools import setup, find_packages22 23readme_dir = os.path.dirname(__file__)24readme_path = os.path.join(readme_dir, 'README.md')25 26try:27    with open(readme_path, 'r') as f:28        readme_markdown = f.read()29except:30    print(""Failed to load README file"")31    readme_markdown = """"32 33try:34    import pypandoc35    readme_restructured = pypandoc.convert(readme_markdown, to='rst', format='md')36except:37    readme_restructured = readme_markdown38    print(""Conversion of long_description from markdown to reStructuredText failed, skipping..."")39 40with open('pepdata/__init__.py', 'r') as f:41    version = re.search(42        r'^__version__\s*=\s*[\'""]([^\'""]*)[\'""]',43        f.read(),44        re.MULTILINE).group(1)45 46if __name__ == '__main__':47    setup(48        name='pepdata',49        version=version,50        description=""Immunological peptide datasets and amino acid properties"",51        author=""Alex Rubinsteyn"",52        author_email=""alex.rubinsteyn@mssm.edu"",53        url=""https://github.com/openvax/pepdata"",54        license=""http://www.apache.org/licenses/LICENSE-2.0.html"",55        classifiers=[56            'Development Status :: 3 - Alpha',57            'Environment :: Console',58            'Operating System :: OS Independent',59            'Intended Audience :: Science/Research',60            'License :: OSI Approved :: Apache Software License',61            'Programming Language :: Python',62            'Topic :: Scientific/Engineering :: Bio-Informatics',63        ],64        install_requires=[65            'numpy>=1.7',66            'scipy>=0.9',67            'pandas>=0.17',68            'scikit-learn>=0.14.1',69            'progressbar33',70            'biopython>=1.65',71            'datacache>=0.4.4',72            'lxml',73        ],74        long_description=readme_restructured,75        packages=find_packages(exclude=""test""),76        include_package_data=True77    )78","Python"
79"Hydrophilic","openvax/pepdata","deploy.sh",".sh","273","10","./lint.sh && \80./test.sh && \81python3 -m pip install --upgrade build && \82python3 -m pip install --upgrade twine && \83rm -rf dist && \84python3 -m build && \85git --version && \86python3 -m twine upload dist/* && \87git tag ""$(python3 pepdata/version.py)"" &&  \88git push --tags","Shell"
89"Hydrophilic","openvax/pepdata","develop.sh",".sh","25","4","set -e90 91pip install -e .92","Shell"
93"Hydrophilic","openvax/pepdata","lint.sh",".sh","154","10","#!/bin/bash94set -o errexit95 96find pepdata test -name '*.py' \97  | xargs pylint \98  --errors-only \99  --disable=print-statement100 101echo 'Passes pylint check'102","Shell"
103"Hydrophilic","openvax/pepdata","test.sh",".sh","56","4","pytest --cov=pepdata/ --cov-report=term-missing tests104 105 106","Shell"
107"Hydrophilic","openvax/pepdata","pepdata/common.py",".py","967","28","# Copyright (c) 2014-2016. Mount Sinai School of Medicine108#109# Licensed under the Apache License, Version 2.0 (the ""License"");110# you may not use this file except in compliance with the License.111# You may obtain a copy of the License at112#113#     http://www.apache.org/licenses/LICENSE-2.0114#115# Unless required by applicable law or agreed to in writing, software116# distributed under the License is distributed on an ""AS IS"" BASIS,117# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.118# See the License for the specific language governing permissions and119# limitations under the License.120 121 122from __future__ import print_function, division, absolute_import123 124import numpy as np125 126def transform_peptide(peptide, property_dict):127    return np.array([property_dict[amino_acid] for amino_acid in peptide])128 129def transform_peptides(peptides, property_dict):130    return np.array([131        [property_dict[aa] for aa in peptide]132        for peptide in peptides])133 134","Python"
135"Hydrophilic","openvax/pepdata","pepdata/static_data.py",".py","741","19","# Licensed under the Apache License, Version 2.0 (the ""License"");136# you may not use this file except in compliance with the License.137# You may obtain a copy of the License at138#139#     http://www.apache.org/licenses/LICENSE-2.0140#141# Unless required by applicable law or agreed to in writing, software142# distributed under the License is distributed on an ""AS IS"" BASIS,143# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.144# See the License for the specific language governing permissions and145# limitations under the License.146 147 148from __future__ import print_function, division, absolute_import149from os.path import dirname, realpath, join150 151PACKAGE_DIR = dirname(realpath(__file__))152MATRIX_DIR = join(PACKAGE_DIR, 'matrices')153","Python"
154"Hydrophilic","openvax/pepdata","pepdata/version.py",".py","121","8","__version__ = ""1.2.0""155 156 157def print_version():158    print(f""v{__version__}"")159 160if __name__ == ""__main__"":161    print_version()","Python"
162"Hydrophilic","openvax/pepdata","pepdata/__init__.py",".py","597","27","from .amino_acid_alphabet import (163    AminoAcid,164    canonical_amino_acids,165    canonical_amino_acid_letters,166    extended_amino_acids,167    extended_amino_acid_letters,168    amino_acid_letter_indices,169    amino_acid_name_indices,170)171from .peptide_vectorizer import PeptideVectorizer172from .version import __version__173from . import iedb174 175 176 177__all__ = [178    ""iedb"",179    ""AminoAcid"",180    ""canonical_amino_acids"",181    ""canonical_amino_acid_letters"",182    ""extended_amino_acids"",183    ""extended_amino_acid_letters"",184    ""amino_acid_letter_indices"",185    ""amino_acid_name_indices"",186    ""PeptideVectorizer"",187]188","Python"
189"Hydrophilic","openvax/pepdata","pepdata/amino_acid.py",".py","1287","37","# Licensed under the Apache License, Version 2.0 (the ""License"");190# you may not use this file except in compliance with the License.191# You may obtain a copy of the License at192#193#     http://www.apache.org/licenses/LICENSE-2.0194#195# Unless required by applicable law or agreed to in writing, software196# distributed under the License is distributed on an ""AS IS"" BASIS,197# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.198# See the License for the specific language governing permissions and199# limitations under the License.200 201 202from __future__ import print_function, division, absolute_import203 204class AminoAcid(object):205    def __init__(206            self, full_name, short_name, letter, contains=None):207        self.letter = letter208        self.full_name = full_name209        self.short_name = short_name210        if not contains:211            contains = [letter]212        self.contains = contains213 214    def __str__(self):215        return (216            (""AminoAcid(full_name='%s', short_name='%s', letter='%s', ""217             ""contains=%s)"") % (218            self.letter, self.full_name, self.short_name, self.contains))219 220    def __repr__(self):221        return str(self)222 223    def __eq__(self, other):224        return other.__class__ is AminoAcid and self.letter == other.letter225","Python"
226"Hydrophilic","openvax/pepdata","pepdata/amino_acid_alphabet.py",".py","4682","161","# Licensed under the Apache License, Version 2.0 (the ""License"");227# you may not use this file except in compliance with the License.228# You may obtain a copy of the License at229#230#     http://www.apache.org/licenses/LICENSE-2.0231#232# Unless required by applicable law or agreed to in writing, software233# distributed under the License is distributed on an ""AS IS"" BASIS,234# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.235# See the License for the specific language governing permissions and236# limitations under the License.237 238 239""""""240Quantify amino acids by their physical/chemical properties241""""""242 243from __future__ import print_function, division, absolute_import244 245import numpy as np246 247from .amino_acid import AminoAcid248 249canonical_amino_acids = [250    AminoAcid(""Alanine"", ""Ala"", ""A""),251    AminoAcid(""Arginine"", ""Arg"", ""R""),252    AminoAcid(""Asparagine"",""Asn"", ""N""),253    AminoAcid(""Aspartic Acid"", ""Asp"", ""D""),254    AminoAcid(""Cysteine"", ""Cys"", ""C""),255    AminoAcid(""Glutamic Acid"", ""Glu"", ""E""),256    AminoAcid(""Glutamine"", ""Gln"", ""Q""),257    AminoAcid(""Glycine"", ""Gly"", ""G""),258    AminoAcid(""Histidine"", ""His"", ""H""),259    AminoAcid(""Isoleucine"",  ""Ile"", ""I""),260    AminoAcid(""Leucine"", ""Leu"", ""L""),261    AminoAcid(""Lysine"", ""Lys"", ""K""),262    AminoAcid(""Methionine"",  ""Met"", ""M""),263    AminoAcid(""Phenylalanine"", ""Phe"", ""F""),264    AminoAcid(""Proline"", ""Pro"", ""P""),265    AminoAcid(""Serine"", ""Ser"", ""S""),266    AminoAcid(""Threonine"", ""Thr"", ""T""),267    AminoAcid(""Tryptophan"", ""Trp"", ""W""),268    AminoAcid(""Tyrosine"", ""Tyr"", ""Y""),269    AminoAcid(""Valine"", ""Val"", ""V"")270]271 272canonical_amino_acid_letters = [aa.letter for aa in canonical_amino_acids]273 274###275# Post-translation modifications commonly detected by mass-spec276###277 278# TODO: figure out three letter codes for modified AAs279 280modified_amino_acids = [281    AminoAcid(""Phospho-Serine"", ""Sep"", ""s""),282    AminoAcid(""Phospho-Threonine"", ""???"", ""t""),283    AminoAcid(""Phospho-Tyrosine"", ""???"", ""y""),284    AminoAcid(""Cystine"", ""???"", ""c""),285    AminoAcid(""Methionine sulfoxide"", ""???"", ""m""),286    AminoAcid(""Pyroglutamate"", ""???"", ""q""),287    AminoAcid(""Pyroglutamic acid"", ""???"", ""n""),288]289 290###291# Amino acid tokens which represent multiple canonical amino acids292###293wildcard_amino_acids = [294    AminoAcid(""Unknown"", ""Xaa"", ""X"", contains=set(canonical_amino_acid_letters)),295    AminoAcid(""Asparagine-or-Aspartic-Acid"", ""Asx"",  ""B"", contains={""D"", ""N""}),296    AminoAcid(""Glutamine-or-Glutamic-Acid"", ""Glx"", ""Z"", contains={""E"", ""Q""}),297    AminoAcid(""Leucine-or-Isoleucine"", ""Xle"", ""J"", contains={""I"", ""L""})298]299 300###301# Canonical amino acids + wilcard tokens302###303 304canonical_amino_acids_with_unknown = canonical_amino_acids + wildcard_amino_acids305 306 307###308# Rare amino acids which aren't considered part of the core 20 ""canonical""309###310 311rare_amino_acids = [312    AminoAcid(""Selenocysteine"", ""Sec"", ""U""),313    AminoAcid(""Pyrrolysine"", ""Pyl"", ""O""),314]315 316###317# Extended amino acids + wildcard tokens318###319 320extended_amino_acids = canonical_amino_acids + rare_amino_acids + wildcard_amino_acids321extended_amino_acid_letters = [322    aa.letter for aa in extended_amino_acids323]324extended_amino_acids_with_unknown_names = [325    aa.full_name for aa in extended_amino_acids326]327 328 329amino_acid_letter_indices = {330    c: i for (i, c) in331    enumerate(extended_amino_acid_letters)332}333 334 335amino_acid_letter_pairs = [336    ""%s%s"" % (x, y)337    for y in extended_amino_acids338    for x in extended_amino_acids339]340 341 342amino_acid_name_indices = {343    aa_name: i for (i, aa_name)344    in enumerate(extended_amino_acids_with_unknown_names)345}346 347amino_acid_pair_positions = {348    pair: i for (i, pair) in enumerate(amino_acid_letter_pairs)349}350 351def index_to_full_name(idx):352    return extended_amino_acids[idx].full_name353 354def index_to_short_name(idx):355    return extended_amino_acids[idx].short_name356 357def index_to_letter(idx):358    return extended_amino_acids[idx]359 360def letter_to_index(x):361    """"""362    Convert from an amino acid's letter code to its position index363    """"""364    assert x in amino_acid_letter_indices, ""Unknown amino acid: %s"" % x365    return amino_acid_letter_indices[x]366 367def peptide_to_indices(xs):368    return [amino_acid_letter_indices[x] for x in xs]369 370def letter_to_short_name(x):371    return index_to_short_name(letter_to_index(x))372 373def peptide_to_short_amino_acid_names(xs):374    return [amino_acid_letter_indices[x] for x in xs]375 376def dict_to_amino_acid_matrix(d, alphabet=canonical_amino_acids):377    n_aa = len(d)378    result_matrix = np.zeros((n_aa, n_aa), dtype=""float32"")379    for i, aa_row in enumerate(alphabet):380        d_row = d[aa_row.letter]381        for j, aa_col in enumerate(alphabet):382            value = d_row[aa_col.letter]383            result_matrix[i, j] = value384    return result_matrix385 386","Python"
387"Hydrophilic","openvax/pepdata","pepdata/amino_acid_properties.py",".py","6268","360","# Licensed under the Apache License, Version 2.0 (the ""License"");388# you may not use this file except in compliance with the License.389# You may obtain a copy of the License at390#391#     http://www.apache.org/licenses/LICENSE-2.0392#393# Unless required by applicable law or agreed to in writing, software394# distributed under the License is distributed on an ""AS IS"" BASIS,395# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.396# See the License for the specific language governing permissions and397# limitations under the License.398 399from __future__ import print_function, division, absolute_import400 401from .amino_acid_alphabet import letter_to_index402 403""""""404Quantify amino acids by their physical/chemical properties405""""""406 407 408def aa_dict_to_positional_list(aa_property_dict):409    value_list = [None] * 20410    for letter, value in aa_property_dict.items():411        idx = letter_to_index(letter)412        assert idx >= 0413        assert idx < 20414        value_list[idx] = value415    assert all(elt is not None for elt in value_list), \416        ""Missing amino acids in:\n%s"" % aa_property_dict.keys()417    return value_list418 419def parse_property_table(table_string):420    value_dict = {}421    for line in table_string.splitlines():422        line = line.strip()423        if not line:424            continue425        fields = line.split("" "")426        fields = [f for f in fields if len(f.strip()) > 0]427        assert len(fields) >= 2428        value, letter = fields[:2]429        assert letter not in value_dict, ""Repeated amino acid "" + line430        value_dict[letter] = float(value)431    return value_dict432 433 434""""""435Amino acids property tables copied from CRASP website436""""""437 438hydropathy = parse_property_table(""""""4391.80000 A ALA440-4.5000 R ARG441-3.5000 N ASN442-3.5000 D ASP4432.50000 C CYS444-3.5000 Q GLN445-3.5000 E GLU446-0.4000 G GLY447-3.2000 H HIS4484.50000 I ILE4493.80000 L LEU450-3.9000 K LYS4511.90000 M MET4522.80000 F PHE453-1.6000 P PRO454-0.8000 S SER455-0.7000 T THR456-0.9000 W TRP457-1.3000 Y TYR4584.20000 V VAL459"""""")460 461volume = parse_property_table(""""""46291.5000 A ALA463202.0000 R ARG464135.2000 N ASN465124.5000 D ASP466118.0000 C CYS467161.1000 Q GLN468155.1000 E GLU46966.40000 G GLY470167.3000 H HIS471168.8000 I ILE472167.9000 L LEU473171.3000 K LYS474170.8000 M MET475203.4000 F PHE476129.3000 P PRO47799.10000 S SER478122.1000 T THR479237.6000 W TRP480203.6000 Y TYR481141.7000 V VAL482"""""")483 484polarity = parse_property_table(""""""4850.0000 A ALA48652.000 R ARG4873.3800 N ASN48840.700 D ASP4891.4800 C CYS4903.5300 Q GLN49149.910 E GLU4920.0000 G GLY49351.600 H HIS4940.1500 I ILE4950.4500 L LEU49649.500 K LYS4971.4300 M MET4980.3500 F PHE4991.5800 P PRO5001.6700 S SER5011.6600 T THR5022.1000 W TRP5031.6100 Y TYR5040.1300 V VAL505"""""")506 507pK_side_chain = parse_property_table(""""""5080.0000 A ALA50912.480 R ARG5100.0000 N ASN5113.6500 D ASP5128.1800 C CYS5130.0000 Q GLN5144.2500 E GLU5150.0000 G GLY5166.0000 H HIS5170.0000 I ILE5180.0000 L LEU51910.530 K LYS5200.0000 M MET5210.0000 F PHE5220.0000 P PRO5230.0000 S SER5240.0000 T THR5250.0000 W TRP52610.700 Y TYR5270.0000 V VAL528"""""")529 530prct_exposed_residues = parse_property_table(""""""53115.0000 A ALA53267.0000 R ARG53349.0000 N ASN53450.0000 D ASP5355.00000 C CYS53656.0000 Q GLN53755.0000 E GLU53810.0000 G GLY53934.0000 H HIS54013.0000 I ILE54116.0000 L LEU54285.0000 K LYS54320.0000 M MET54410.0000 F PHE54545.0000 P PRO54632.0000 S SER54732.0000 T THR54817.0000 W TRP54941.0000 Y TYR55014.0000 V VAL551"""""")552 553hydrophilicity = parse_property_table(""""""554-0.5000 A ALA5553.00000 R ARG5560.20000 N ASN5573.00000 D ASP558-1.0000 C CYS5590.20000 Q GLN5603.00000 E GLU5610.00000 G GLY562-0.5000 H HIS563-1.8000 I ILE564-1.8000 L LEU5653.00000 K LYS566-1.3000 M MET567-2.5000 F PHE5680.00000 P PRO5690.30000 S SER570-0.4000 T THR571-3.4000 W TRP572-2.3000 Y TYR573-1.5000 V VAL574"""""")575 576accessible_surface_area = parse_property_table(""""""57727.8000 A ALA57894.7000 R ARG57960.1000 N ASN58060.6000 D ASP58115.5000 C CYS58268.7000 Q GLN58368.2000 E GLU58424.5000 G GLY58550.7000 H HIS58622.8000 I ILE58727.6000 L LEU588103.000 K LYS58933.5000 M MET59025.5000 F PHE59151.5000 P PRO59242.0000 S SER59345.0000 T THR59434.7000 W TRP59555.2000 Y TYR59623.7000 V VAL597"""""")598 599local_flexibility = parse_property_table(""""""600705.42000 A ALA6011484.2800 R ARG602513.46010 N ASN60334.960000 D ASP6042412.5601 C CYS6051087.8300 Q GLN6061158.6600 E GLU60733.180000 G GLY6081637.1300 H HIS6095979.3701 I ILE6104985.7300 L LEU611699.69000 K LYS6124491.6602 M MET6135203.8599 F PHE614431.96000 P PRO615174.76000 S SER616601.88000 T THR6176374.0698 W TRP6184291.1001 Y TYR6194474.4199 V VAL620"""""")621 622accessible_surface_area_folded = parse_property_table(""""""62331.5000 A ALA62493.8000 R ARG62562.2000 N ASN62660.9000 D ASP62713.9000 C CYS62874.0000 Q GLN62972.3000 E GLU63025.2000 G GLY63146.7000 H HIS63223.0000 I ILE63329.0000 L LEU634110.300 K LYS63530.5000 M MET63628.7000 F PHE63753.7000 P PRO63844.2000 S SER63946.0000 T THR64041.7000 W TRP64159.1000 Y TYR64223.5000 V VAL643"""""")644 645refractivity = parse_property_table(""""""6464.34000 A ALA64726.6600 R ARG64813.2800 N ASN64912.0000 D ASP65035.7700 C CYS65117.5600 Q GLN65217.2600 E GLU6530.00000 G GLY65421.8100 H HIS65519.0600 I ILE65618.7800 L LEU65721.2900 K LYS65821.6400 M MET65929.4000 F PHE66010.9300 P PRO6616.35000 S SER66211.0100 T THR66342.5300 W TRP66431.5300 Y TYR66513.9200 V VAL666"""""")667 668 669mass = parse_property_table(""""""67070.079 A ALA671156.188 R ARG672114.104 N ASN673115.089 D ASP674103.144 C CYS675128.131 Q GLN676129.116 E GLU67757.052 G GLY678137.142 H HIS679113.160 I ILE680113.160 L LEU681128.174 K LYS682131.198 M MET683147.177 F PHE68497.177 P PRO68587.078 S SER686101.105 T THR687186.213 W TRP688163.170 Y TYR68999.133 V VAL690"""""")691 692###693# Values copied from:694# ""Solvent accessibility of AA in known protein structures""695# http://prowl.rockefeller.edu/aainfo/access.htm696###697""""""698Solvent accessibility of AA in known protein structures699 700Figure 1.701 702S   0.70    0.20    0.10703T   0.71    0.16    0.13704A   0.48    0.35    0.17705G   0.51    0.36    0.13706P   0.78    0.13    0.09707C   0.32    0.54    0.14708D   0.81    0.09    0.10709E   0.93    0.04    0.03710Q   0.81    0.10    0.09711N   0.82    0.10    0.08712L   0.41    0.49    0.10713I   0.39    0.47    0.14714V   0.40    0.50    0.10715M   0.44    0.20    0.36716F   0.42    0.42    0.16717Y   0.67    0.20    0.13718W   0.49    0.44    0.07719K   0.93    0.02    0.05720R   0.84    0.05    0.11721H   0.66    0.19    0.15722""""""723 724solvent_exposed_area = dict(725    S=0.70,726    T=0.71,727    A=0.48,728    G=0.51,729    P=0.78,730    C=0.32,731    D=0.81,732    E=0.93,733    Q=0.81,734    N=0.82,735    L=0.41,736    I=0.39,737    V=0.40,738    M=0.44,739    F=0.42,740    Y=0.67,741    W=0.49,742    K=0.93,743    R=0.84,744    H=0.66,745)746","Python"
747"Hydrophilic","openvax/pepdata","pepdata/reduced_alphabet.py",".py","1784","58","# Copyright (c) 2014-2018. Mount Sinai School of Medicine748#749# Licensed under the Apache License, Version 2.0 (the ""License"");750# you may not use this file except in compliance with the License.751# You may obtain a copy of the License at752#753#     http://www.apache.org/licenses/LICENSE-2.0754#755# Unless required by applicable law or agreed to in writing, software756# distributed under the License is distributed on an ""AS IS"" BASIS,757# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.758# See the License for the specific language governing permissions and759# limitations under the License.760 761""""""762Amino acid groupings from763'Reduced amino acid alphabets improve the sensitivity...' by764Peterson, Kondev, et al.765http://www.rpgroup.caltech.edu/publications/Peterson2008.pdf766""""""767from __future__ import print_function, division, absolute_import768 769def dict_from_list(groups):770    aa_to_group = {}771    for i, group in enumerate(groups):772        for c in group:773            aa_to_group[c] = group[0]774    return aa_to_group775 776gbmr4 = dict_from_list([""ADKERNTSQ"", ""YFLIVMCWH"", ""G"", ""P""])777 778sdm12 = dict_from_list([779    ""A"", ""D"", ""KER"", ""N"", ""TSQ"", ""YF"", ""LIVM"", ""C"", ""W"", ""H"", ""G"", ""P""780])781 782hsdm17 = dict_from_list([783    ""A"", ""D"", ""KE"", ""R"", ""N"", ""T"", ""S"", ""Q"", ""Y"",784    ""F"", ""LIV"", ""M"", ""C"", ""W"", ""H"", ""G"", ""P""785])786 787""""""788Other alphabets from789http://bio.math-inf.uni-greifswald.de/viscose/html/alphabets.html790""""""791 792# hydrophilic vs. hydrophobic793hp2 = dict_from_list([""AGTSNQDEHRKP"", ""CMFILVWY""])794 795murphy10 = dict_from_list([796    ""LVIM"", ""C"", ""A"", ""G"", ""ST"", ""P"", ""FYW"", ""EDNQ"", ""KR"", ""H""797])798 799alex6 = dict_from_list([""C"", ""G"", ""P"", ""FYW"", ""AVILM"", ""STNQRHKDE""])800 801aromatic2 = dict_from_list([""FHWY"", ""ADKERNTSQLIVMCGP""])802 803hp_vs_aromatic = dict_from_list([""H"", ""CMILV"", ""FWY"", ""ADKERNTSQGP""])804","Python"
805"Hydrophilic","openvax/pepdata","pepdata/peptide_vectorizer.py",".py","2942","84","# Copyright (c) 2014-2016. Mount Sinai School of Medicine806#807# Licensed under the Apache License, Version 2.0 (the ""License"");808# you may not use this file except in compliance with the License.809# You may obtain a copy of the License at810#811#     http://www.apache.org/licenses/LICENSE-2.0812#813# Unless required by applicable law or agreed to in writing, software814# distributed under the License is distributed on an ""AS IS"" BASIS,815# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.816# See the License for the specific language governing permissions and817# limitations under the License.818 819 820from __future__ import print_function, division, absolute_import821 822import numpy as np823from sklearn.feature_extraction.text import CountVectorizer824from sklearn.preprocessing import normalize825 826def make_count_vectorizer(reduced_alphabet, max_ngram):827    if reduced_alphabet is None:828        preprocessor = None829    else:830        preprocessor = lambda s: """".join([reduced_alphabet[si] for si in s])831 832    return CountVectorizer(833        analyzer='char',834        ngram_range=(1, max_ngram),835        dtype=np.float,836        preprocessor=preprocessor)837 838class PeptideVectorizer(object):839    """"""840    Make n-gram frequency vectors from peptide sequences841    """"""842    def __init__(843            self,844            max_ngram=1,845            normalize_row=True,846            reduced_alphabet=None,847            training_already_reduced=False):848        self.reduced_alphabet = reduced_alphabet849        self.max_ngram = max_ngram850        self.normalize_row = normalize_row851        self.training_already_reduced = training_already_reduced852        self.count_vectorizer = None853 854    def __getstate__(self):855        return {856            'reduced_alphabet': self.reduced_alphabet,857            'count_vectorizer': self.count_vectorizer,858            'training_already_reduced': self.training_already_reduced,859            'normalize_row': self.normalize_row,860            'max_ngram': self.max_ngram,861        }862 863    def fit_transform(self, amino_acid_strings):864        self.count_vectorizer = \865            make_count_vectorizer(self.reduced_alphabet, self.max_ngram)866 867        if self.training_already_reduced:868            c = make_count_vectorizer(None, self.max_ngram)869            X = c.fit_transform(amino_acid_strings).todense()870            self.count_vectorizer.vocabulary_ = c.vocabulary_871        else:872            c = self.count_vectorizer873            X = c.fit_transform(amino_acid_strings).todense()874 875        if self.normalize_row:876            X = normalize(X, norm='l1')877        return X878 879    def fit(self, amino_acid_strings):880        self.fit_transform(amino_acid_strings)881 882    def transform(self, amino_acid_strings):883        assert self.count_vectorizer, ""Must call 'fit' before 'transform'""884        X = self.count_vectorizer.transform(amino_acid_strings).todense()885        if self.normalize_row:886            X = normalize(X, norm='l1')887        return X888","Python"
889"Hydrophilic","openvax/pepdata","pepdata/chou_fasman.py",".py","3279","75","# Licensed under the Apache License, Version 2.0 (the ""License"");890# you may not use this file except in compliance with the License.891# You may obtain a copy of the License at892#893#     http://www.apache.org/licenses/LICENSE-2.0894#895# Unless required by applicable law or agreed to in writing, software896# distributed under the License is distributed on an ""AS IS"" BASIS,897# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.898# See the License for the specific language governing permissions and899# limitations under the License.900 901from __future__ import print_function, division, absolute_import902 903from .amino_acid_alphabet import amino_acid_name_indices904 905# Chou-Fasman of structural properties from906# http://prowl.rockefeller.edu/aainfo/chou.htm907chou_fasman_table = """"""908Alanine        142     83       66      0.06    0.076   0.035   0.058909Arginine        98     93       95      0.070   0.106   0.099   0.085910Aspartic Acid  101     54      146      0.147   0.110   0.179   0.081911Asparagine      67     89      156      0.161   0.083   0.191   0.091912Cysteine        70    119      119      0.149   0.050   0.117   0.128913Glutamic Acid  151    037       74      0.056   0.060   0.077   0.064914Glutamine      111    110       98      0.074   0.098   0.037   0.098915Glycine         57     75      156      0.102   0.085   0.190   0.152916Histidine      100     87       95      0.140   0.047   0.093   0.054917Isoleucine     108    160       47      0.043   0.034   0.013   0.056918Leucine        121    130       59      0.061   0.025   0.036   0.070919Lysine         114     74      101      0.055   0.115   0.072   0.095920Methionine     145    105       60      0.068   0.082   0.014   0.055921Phenylalanine  113    138       60      0.059   0.041   0.065   0.065922Proline         57     55      152      0.102   0.301   0.034   0.068923Serine          77     75      143      0.120   0.139   0.125   0.106924Threonine       83    119       96      0.086   0.108   0.065   0.079925Tryptophan     108    137       96      0.077   0.013   0.064   0.167926Tyrosine        69    147      114      0.082   0.065   0.114   0.125927Valine         106    170       50      0.062   0.048   0.028   0.053928""""""929 930 931def parse_chou_fasman(table):932    alpha_helix_score_dict = {}933    beta_sheet_score_dict = {}934    turn_score_dict = {}935 936    for line in table.split(""\n""):937        fields = [field for field in line.split("" "") if len(field.strip()) > 0]938        if len(fields) == 0:939            continue940 941        if fields[1] == 'Acid':942            name = fields[0] + "" "" + fields[1]943            fields = fields[1:]944        else:945            name = fields[0]946 947        assert name in amino_acid_name_indices, ""Invalid amino acid name %s"" % name948        letter = amino_acid_name_indices[name]949        alpha = int(fields[1])950        beta = int(fields[2])951        turn = int(fields[3])952        alpha_helix_score_dict[letter] = alpha953        beta_sheet_score_dict[letter] = beta954        turn_score_dict[letter] = turn955 956    assert len(alpha_helix_score_dict) == 20957    assert len(beta_sheet_score_dict) == 20958    assert len(turn_score_dict) == 20959    return alpha_helix_score_dict, beta_sheet_score_dict, turn_score_dict960 961alpha_helix_score, beta_sheet_score, turn_score = \962    parse_chou_fasman(chou_fasman_table)963","Python"
964"Hydrophilic","openvax/pepdata","pepdata/pmbec.py",".py","3019","89","# Copyright (c) 2014-2016. Mount Sinai School of Medicine965#966# Licensed under the Apache License, Version 2.0 (the ""License"");967# you may not use this file except in compliance with the License.968# You may obtain a copy of the License at969#970#     http://www.apache.org/licenses/LICENSE-2.0971#972# Unless required by applicable law or agreed to in writing, software973# distributed under the License is distributed on an ""AS IS"" BASIS,974# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.975# See the License for the specific language governing permissions and976# limitations under the License.977 978from __future__ import print_function, division, absolute_import979from os.path import join980 981from .static_data import MATRIX_DIR982 983from .amino_acid_alphabet import dict_to_amino_acid_matrix984 985def read_pmbec_coefficients(986        key_type='row',987        verbose=True,988        filename=join(MATRIX_DIR, 'pmbec.mat')):989    """"""990    Parameters991    ------------992 993    filename : str994        Location of PMBEC coefficient matrix995 996    key_type : str997        'row' : every key is a single amino acid,998           which maps to a dictionary for that row999        'pair' : every key is a tuple of amino acids1000        'pair_string' : every key is a string of two amino acid characters1001 1002    verbose : bool1003        Print rows of matrix as we read them1004    """"""1005    d = {}1006    if key_type == 'row':1007        def add_pair(row_letter, col_letter, value):1008            if row_letter not in d:1009                d[row_letter] = {}1010            d[row_letter][col_letter] = value1011    elif key_type == 'pair':1012        def add_pair(row_letter, col_letter, value):1013            d[(row_letter, col_letter)] = value1014 1015    else:1016        assert key_type == 'pair_string', \1017            ""Invalid dictionary key type: %s"" % key_type1018 1019        def add_pair(row_letter, col_letter, value):1020            d[""%s%s"" % (row_letter, col_letter)] = value1021 1022    with open(filename, 'r') as f:1023        lines = [line for line in f.read().split('\n') if len(line) > 0]1024        header = lines[0]1025        if verbose:1026            print(header)1027        residues = [1028            x for x in header.split()1029            if len(x) == 1 and x != ' ' and x != '\t'1030        ]1031        assert len(residues) == 201032        if verbose:1033            print(residues)1034        for line in lines[1:]:1035            cols = [1036                x1037                for x in line.split(' ')1038                if len(x) > 0 and x != ' ' and x != '\t'1039            ]1040            assert len(cols) == 21, ""Expected 20 values + letter, got %s"" % cols1041            row_letter = cols[0]1042            for i, col in enumerate(cols[1:]):1043                col_letter = residues[i]1044                assert col_letter != ' ' and col_letter != '\t'1045                value = float(col)1046                add_pair(row_letter, col_letter, value)1047    return d1048 1049# dictionary of PMBEC coefficient accessed like pmbec_dict[""V""][""R""]1050pmbec_dict = read_pmbec_coefficients(key_type=""row"")1051pmbec_matrix = dict_to_amino_acid_matrix(pmbec_dict)1052","Python"
1053"Hydrophilic","openvax/pepdata","pepdata/residue_contact_energies.py",".py","2937","77","# Licensed under the Apache License, Version 2.0 (the ""License"");1054# you may not use this file except in compliance with the License.1055# You may obtain a copy of the License at1056#1057#     http://www.apache.org/licenses/LICENSE-2.01058#1059# Unless required by applicable law or agreed to in writing, software1060# distributed under the License is distributed on an ""AS IS"" BASIS,1061# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1062# See the License for the specific language governing permissions and1063# limitations under the License.1064 1065from __future__ import print_function, division, absolute_import1066 1067from os.path import join1068 1069from .amino_acid_alphabet import canonical_amino_acid_letters, dict_to_amino_acid_matrix1070from .static_data import MATRIX_DIR1071 1072 1073def parse_interaction_table(table, amino_acid_order=""ARNDCQEGHILKMFPSTWYV""):1074    table = table.strip()1075    while ""  "" in table:1076        table = table.replace(""  "", "" "")1077 1078    lines = [l.strip() for l in table.split(""\n"")]1079    lines = [l for l in lines if len(l) > 0 and not l.startswith(""#"")]1080    assert len(lines) == 20, ""Malformed amino acid interaction table""1081    d = {}1082    for i, line in enumerate(lines):1083        coeff_strings = line.split("" "")1084        assert len(coeff_strings) == 20, \1085            ""Malformed row in amino acid interaction table""1086        x = amino_acid_order[i]1087        d[x] = {}1088        for j, coeff_str in enumerate(coeff_strings):1089            value = float(coeff_str)1090            y = amino_acid_order[j]1091            d[x][y] = value1092    return d1093 1094def transpose_interaction_dict(d):1095    transposed = {}1096    for x in canonical_amino_acid_letters:1097        transposed[x] = {}1098        for y in canonical_amino_acid_letters:1099            transposed[x][y] = d[y][x]1100    return transposed1101 1102 1103with open(join(MATRIX_DIR, 'strand_vs_coil.txt'), 'r') as f:1104    # Strand vs. Coil1105    strand_vs_coil_dict = parse_interaction_table(f.read())1106    strand_vs_coil_array = dict_to_amino_acid_matrix(strand_vs_coil_dict)1107 1108    # Coil vs. Strand1109    coil_vs_strand_dict = transpose_interaction_dict(strand_vs_coil_dict)1110    coil_vs_strand_array = dict_to_amino_acid_matrix(coil_vs_strand_dict)1111 1112with open(join(MATRIX_DIR, 'helix_vs_strand.txt'), 'r') as f:1113    # Helix vs. Strand1114    helix_vs_strand_dict = parse_interaction_table(f.read())1115    helix_vs_strand_array = dict_to_amino_acid_matrix(helix_vs_strand_dict)1116 1117    # Strand vs. Helix1118    strand_vs_helix_dict = transpose_interaction_dict(helix_vs_strand_dict)1119    strand_vs_helix_array = dict_to_amino_acid_matrix(strand_vs_helix_dict)1120 1121with open(join(MATRIX_DIR, 'helix_vs_coil.txt'), 'r') as f:1122    # Helix vs. Coil1123    helix_vs_coil_dict = parse_interaction_table(f.read())1124    helix_vs_coil_array = dict_to_amino_acid_matrix(helix_vs_coil_dict)1125 1126    # Coil vs. Helix1127    coil_vs_helix_dict = transpose_interaction_dict(helix_vs_coil_dict)1128    coil_vs_helix_array = dict_to_amino_acid_matrix(coil_vs_helix_dict)1129","Python"
1130"Hydrophilic","openvax/pepdata","pepdata/blosum.py",".py","2600","76","# Licensed under the Apache License, Version 2.0 (the ""License"");1131# you may not use this file except in compliance with the License.1132# You may obtain a copy of the License at1133#1134#     http://www.apache.org/licenses/LICENSE-2.01135#1136# Unless required by applicable law or agreed to in writing, software1137# distributed under the License is distributed on an ""AS IS"" BASIS,1138# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1139# See the License for the specific language governing permissions and1140# limitations under the License.1141 1142from __future__ import print_function, division, absolute_import1143 1144from os.path import join1145 1146from .static_data import MATRIX_DIR1147 1148from .amino_acid_alphabet import dict_to_amino_acid_matrix1149 1150def parse_blosum_table(table, coeff_type=int, key_type='row'):1151    """"""1152    Parse a table of pairwise amino acid coefficient (e.g. BLOSUM50)1153    """"""1154 1155    lines = table.split(""\n"")1156    # drop comments1157    lines = [line for line in lines if not line.startswith(""#"")]1158    # drop CR endline characters1159    lines = [line.replace(""\r"", """") for line in lines]1160    # skip empty lines1161    lines = [line for line in lines if line]1162 1163    labels = lines[0].split()1164 1165    if len(labels) < 20:1166        raise ValueError(1167            ""Expected 20+ amino acids but first line '%s' has %d fields"" % (1168                lines[0],1169                len(labels)))1170    coeffs = {}1171    for line in lines[1:]:1172 1173        fields = line.split()1174        assert len(fields) >= 21, \1175            ""Expected AA and 20+ coefficients but '%s' has %d fields"" % (1176                line, len(fields))1177        x = fields[0]1178        for i, coeff_str in enumerate(fields[1:]):1179            y = labels[i]1180            coeff = coeff_type(coeff_str)1181            if key_type == 'pair':1182                coeffs[(x, y)] = coeff1183            elif key_type == 'pair_string':1184                coeffs[x + y] = coeff1185            else:1186                assert key_type == 'row', ""Unknown key type: %s"" % key_type1187                if x not in coeffs:1188                    coeffs[x] = {}1189                coeffs[x][y] = coeff1190    return coeffs1191 1192 1193with open(join(MATRIX_DIR, 'BLOSUM30'), 'r') as f:1194    blosum30_dict = parse_blosum_table(f.read())1195    blosum30_matrix = dict_to_amino_acid_matrix(blosum30_dict)1196 1197with open(join(MATRIX_DIR, 'BLOSUM50'), 'r') as f:1198    blosum50_dict = parse_blosum_table(f.read())1199    blosum50_matrix = dict_to_amino_acid_matrix(blosum50_dict)1200 

Showing the first 1,200 of 2178 lines. Download the file for the rest.