Team Ai
Datasetpublic

SciCodePile/SciCode-Domain-Code

DATA1: Domain-Specific Code Dataset Dataset Overview DATA1 is a large-scale domain-specific code dataset focusing on code samples from interdisciplinary fields such as biology, chemistry, materials science, and related areas. The dataset is collected and organized from GitHub repositories, covering 178 different domain topics with over 1.1 billion lines of code. Dataset Statistics Total Datasets: 178 CSV files Total Data Size: ~115 GB Total Lines… See the full description on the dataset page: https://huggingface.co/datasets/SciCodePile/SciCode-Domain-Code.

sourceHugging Faceapache-2.0updated 7mo agoView on Hugging Face
4likes1.3kdownloads
dataset_Codon.csv10591 linesDownload Raw Back to data
1"keyword","repo_name","file_path","file_extension","file_size","line_count","content","language"
2"Codon","LKremer/MSA_trimmer","alignment_trimmer.py",".py","9112","221","#!/usr/bin/env python33 4 5from __future__ import division, print_function6from Bio import SeqIO7from Bio.Seq import Seq8from Bio.SeqRecord import SeqRecord9from Bio.Alphabet import SingleLetterAlphabet10import argparse11import os12 13 14def gap_proportion(msa_column):15    gap_counter = 016    for aa_or_codon in msa_column:17        if aa_or_codon in ('-', '---'):18            gap_counter += 119    return gap_counter / len(msa_column)20 21 22class Alignment():23    def __init__(self, alignment_fasta, maxgap,24                 trim_positions, mask, is_cds=False):25        self._is_cds = is_cds26        self._letter_n = 3 if is_cds else 127        self._maxgap = maxgap28        self._fasta = []29        self._trimmed_fasta = []30        with open(alignment_fasta, 'r') as input_handle:31            for fasta_record in SeqIO.parse(input_handle, 'fasta'):32                self._fasta.append(fasta_record)33                self._trimmed_fasta.append([])34        self._first_seq = self._fasta[0]35        self._assert_fasta_is_aligned()36        self._set_mask_char(mask)37        self._parse_trim_positions(trim_positions)38        fname, fextension = os.path.splitext(alignment_fasta)39        self._output_filepath = '{}_{}{}'.format(40            fname,41            ""masked"" if self._mask_char else ""trimmed"",42            fextension,43        )44        return45 46    def _parse_trim_positions(self, trim_pos_list):47        self._trim_positions = set()48        if trim_pos_list:49            for trim_pos in trim_pos_list:50                if '-' in trim_pos:  # it's a range!51                    start_raw, stop_raw = trim_pos.split('-')52                    start = int(start_raw) - 153                    stop = int(stop_raw)54                    if start >= stop:55                        raise Exception('Invalid range:', trim_pos)56                    self._trim_positions |= set(range(start, stop))57                else:58                    pos = self._str_to_pos(trim_pos)59                    self._trim_positions.add(pos)60        return61 62    def _str_to_pos(self, pos_str):63        try:64            pos = int(pos_str) - 1  # account for zero-indexing65        except ValueError:66            raise Exception(pos_str, 'is not a valid position!')67        self[pos]  # raises IndexError when pos is out of range of the MSA68        return pos69 70    def _set_mask_char(self, mask_param):71        if mask_param:72            if len(mask_param) != 1:73                raise Exception('Invalid --mask character:', mask_param)74            self._mask_char = mask_param * self._letter_n75        else:76            self._mask_char = False77        return78 79    def _assert_fasta_is_aligned(self):80        for fa_entry in self._fasta[1:]:81            if len(fa_entry) != len(self._first_seq):82                raise Exception('The sequences {} and {} have different '83                                'lengths. Is your fasta really aligned'84                                '?'.format(fa_entry.id, self._first_seq.id))85        return86 87    def __getitem__(self, i):88        ''' returns the n-th column of the multifasta, works for89        CDS and protein fastas '''90        n = self._letter_n  # the number of characters per position91        # three for CDS sequences (codons), one for protein sequences92        start = i * n93        end = start + n94        if start < 0 or end > len(self._first_seq):95            raise IndexError('Cannot access MSA position {} - index out '96                             'of range'.format(i))97        return tuple(str(fa.seq[start:end]) for fa in self._fasta)98 99    def __iter__(self):100        ''' column-based iteration over MSA fasta, works for101        CDS and protein fastas '''102        for i in range(int(len(self._first_seq) / self._letter_n)):103            yield self[i]104 105    def _trim_column(self, msa_column, col_index):106        ''' trim an MSA column if the number of gaps exceeds the user-specified107        threshold, or if it is within the user-specified trim-range '''108        col_str = ' '.join(msa_column)109        gap_prop = gap_proportion(msa_column)110 111        if (col_index in self._trim_positions) or (gap_prop > self._maxgap):112            if self._mask_char:113                action = ' - masked'114            else:115                action = ' - trimmed'116            trim_column = True117        else:118            action = ''119            trim_column = False120 121        for row_i, aa_or_codon in enumerate(msa_column):122            if trim_column:  # trim (or mask) the column123                if self._mask_char:  # mask it124                    self._trimmed_fasta[row_i].append(self._mask_char)125                else:  # don't add the column at all126                    break127            else:  # keep the column as it was128                self._trimmed_fasta[row_i].append(aa_or_codon)129        print('{: >3} {} ({:.1%} gaps){}'.format(130              col_index + 1, col_str, gap_prop, action))131        return132 133    def _trim_fasta(self):134        for col_index, msa_column in enumerate(self):135            self._trim_column(msa_column, col_index)136        self._trimmed_seqs = []137        for i, trimmed_seq_list in enumerate(self._trimmed_fasta):138            trimmed_seq = ''.join(trimmed_seq_list)139            input_seq = self._fasta[i]140            out_record = SeqRecord(141                Seq(trimmed_seq, SingleLetterAlphabet),142                id=input_seq.id,143                description=input_seq.description144            )145            self._trimmed_seqs.append(out_record)146 147    def dump_trimmed_fasta(self):148        self._trim_fasta()149        SeqIO.write(self._trimmed_seqs, self._output_filepath, 'fasta')150        if self._mask_char:151            print('  Wrote ""{}""-masked fasta to {}\n'.format(152                self._mask_char, self._output_filepath))153        else:154            print('  Wrote trimmed fasta to {}\n'.format(self._output_filepath))155        return156 157    def __eq__(self, other):158        """""" defines an equality test with the == operator159        returns False when a position of one alignment is a gap160        when it isn't a gap in the other alignment """"""161        for col_i, self_column in enumerate(self):162            other_column = other[col_i]163            for row_i, aa_or_codon in enumerate(self_column):164                if aa_or_codon in ('-', '---'):165                    if other_column[row_i] not in ('-', '---'):166                        print('\nCodon mismatch in row {} column {}'167                              '\n""{}"" does not match ""{}""\n'.format(168                                row_i, col_i, aa_or_codon, other_column[row_i]))169                        return False170        return True171 172    def __ne__(self, other):173        """""" defines a non-equality test with the != operator """"""174        return not self.__eq__(other)175 176 177def main(pep_alignment_fasta, cds_alignment_fasta,178         trim_gappy, trim_positions, mask):179    if pep_alignment_fasta:180        pep_al = Alignment(pep_alignment_fasta, maxgap=trim_gappy, mask=mask,181                           trim_positions=trim_positions)182        pep_al.dump_trimmed_fasta()183    if cds_alignment_fasta:184        cds_al = Alignment(cds_alignment_fasta, maxgap=trim_gappy, mask=mask,185                           trim_positions=trim_positions, is_cds=True)186        if pep_alignment_fasta and cds_alignment_fasta:187            if pep_al != cds_al:188                raise Exception('The two alignments have a mismatch'189                                '- see printouts above.')190            else:191                print('Peptide and CDS alignment are in agreement!\n')192        cds_al.dump_trimmed_fasta()193    return194 195 196if __name__ == '__main__':197    parser = argparse.ArgumentParser(198        description='A program to remove (trim) or mask columns of fasta '199        'multiple sequence alignments. Columns are removed if they exceed a '200        'user-specified ""gappyness"" treshold, and/or if they are specified by '201        'column index. Also works for codon alignments. The trimmed alignment '202        'is written to [your_alignment]_trimmed.fa')203    parser.add_argument('-p', '--pep_alignment_fasta')204    parser.add_argument('-c', '--cds_alignment_fasta')205    parser.add_argument('--trim_gappy', type=float, default=9999,206                        help='the maximum allowed gap proportion for an MSA '207                        'column, all columns with a higher proportion of gaps '208                        'will be trimmed (default: off)')209    parser.add_argument('--trim_positions', help='specify columns to be '210                        'trimmed from the alignment, e.g. ""1-10 42"" to trim the '211                        'first ten columns and the 42th column', nargs='+')212    parser.add_argument('--mask', default=False, const='-',213                        action='store', nargs='?', help='mask the alignment '214                        'instead of trimming. if a character is specified after'215                        ' --mask, that character will be used for masking '216                        '(default char: ""-"")')217    args = parser.parse_args()218    if args.pep_alignment_fasta or args.cds_alignment_fasta:219        main(**vars(args))220    else:221        parser.print_help()222","Python"
223"Codon","adelq/dnds","setup.py",".py","1034","30","from setuptools import setup224 225package = 'dnds'226version = '2.1'227 228setup(229    name=package,230    version=version,231    description=""Calculate dN/dS ratio precisely (Ka/Ks) using a codon-by-codon counting method."",232    long_description=open(""README.rst"").read(),233    author=""Adel Qalieh"",234    author_email=""adelq@med.umich.edu"",235    url=""https://github.com/adelq/dnds"",236    license=""MIT"",237    py_modules=['dnds', 'codons'],238    classifiers=[239        'Development Status :: 5 - Production/Stable',240        'Intended Audience :: Science/Research',241        'License :: OSI Approved :: MIT License',242        'Natural Language :: English',243        'Operating System :: OS Independent',244        'Programming Language :: Python :: 2',245        'Programming Language :: Python :: 2.7',246        'Programming Language :: Python :: 3',247        'Programming Language :: Python :: 3.4',248        'Programming Language :: Python :: 3.5',249        'Programming Language :: Python :: 3.6',250        'Topic :: Scientific/Engineering :: Bio-Informatics',251    ])252","Python"
253"Codon","adelq/dnds","test_dnds.py",".py","3156","97","from __future__ import division254from fractions import Fraction255from nose.tools import assert_equal, assert_almost_equal256from dnds import dnds, pnps, substitutions, dnds_codon, dnds_codon_pair, syn_sum, translate257 258# From Canvas practice problem259TEST_SEQ1 = 'ACTCCGAACGGGGCGTTAGAGTTGAAACCCGTTAGA'260TEST_SEQ2 = 'ACGCCGATCGGCGCGATAGGGTTCAAGCTCGTACGA'261# From in-class problem set262TEST_SEQ3 = 'ATGCTTTTGAAATCGATCGTTCGTTCACATCGATGGATC'263TEST_SEQ4 = 'ATGCGTTCGAAGTCGATCGATCGCTCAGATCGATCGATC'264# From http://bioinformatics.cvr.ac.uk/blog/calculating-dnds-for-ngs-datasets/265TEST_SEQ5 = 'ATGAAACCCGGGTTTTAA'266TEST_SEQ6 = 'ATGAAACGCGGCTACTAA'267 268 269def test_translate():270    assert_equal(translate(TEST_SEQ1), 'TPNGALELKPVR')271    assert_equal(translate(TEST_SEQ2), 'TPIGAIGFKLVR')272    assert_equal(translate(TEST_SEQ3), 'MLLKSIVRSHRWI')273 274 275def test_dnds_codon_easy():276    assert_equal([0, 0, 1], dnds_codon('ACC'))277 278 279def test_dnds_codon_harder1():280    assert_equal([0, 0, Fraction(1, 3)], dnds_codon('AAC'))281    assert_equal([0, 0, Fraction(2, 3)], dnds_codon('ATC'))282 283 284def test_dnds_codon_harder2():285    assert_equal([Fraction(1, 3), 0, Fraction(1, 3)], dnds_codon('AGA'))286    assert_equal([Fraction(1, 3), 0, 1], dnds_codon('CGA'))287 288 289def test_dnds_codon_ngs():290    assert_equal([0, 0, 0], dnds_codon('TGG'))291    assert_equal([Fraction(1, 3), 0, 1], dnds_codon('CGG'))292 293 294def test_dnds_codon_ngs_sums():295    assert_equal(sum(dnds_codon('ATG')), 0)296    assert_equal(sum(dnds_codon('AAA')), Fraction(1, 3))297    assert_equal(sum(dnds_codon('CCC')), 1)298    assert_equal(sum(dnds_codon('GGG')), 1)299    assert_equal(sum(dnds_codon('TTT')), Fraction(1, 3))300    assert_equal(sum(dnds_codon('TAA')), Fraction(2, 3))301 302 303def test_dnds_codon_ps():304    assert_equal(dnds_codon('CTT'), [0, 0, 1])305    assert_equal(dnds_codon('CGT'), [0, 0, 1])306 307 308def test_dnds_codon_pair():309    assert_equal([Fraction(1, 3), 0, Fraction(2, 3)],310                 dnds_codon_pair('AGA', 'CGA'))311 312 313def test_dnds_codon_pair_harder1():314    assert_equal([Fraction(1, 6), 0, Fraction(1, 3)],315                 dnds_codon_pair('TTG', 'TTC'))316 317 318def test_dnds_codon_pair_harder2():319    assert_equal([Fraction(1, 6), 0, Fraction(1, 2)],320                 dnds_codon_pair('TTA', 'ATA'))321 322 323def test_syn_sum_ps():324    assert_almost_equal(syn_sum(TEST_SEQ3, TEST_SEQ4), 10, delta=1)325 326 327def test_syn_subs():328    assert_equal(substitutions(TEST_SEQ1, TEST_SEQ2), (5, 5))329    assert_equal(substitutions(TEST_SEQ3, TEST_SEQ4), (2, 5))330    assert_equal(substitutions(""ACCGGA"", ""ACAAGA""), (1, 1))331    assert_equal(substitutions(""TGC"", ""TAT""), (1, 1))332    assert_equal(substitutions(""ACC"", ""AAA""), (0.5, 1.5))333    assert_equal(substitutions(""CCC"", ""AAC""), (0, 2))334 335 336def test_pnps():337    assert_almost_equal(pnps(TEST_SEQ1, TEST_SEQ2), 0.269, delta=0.1)338    assert_almost_equal(pnps(TEST_SEQ3, TEST_SEQ4), 0.86, delta=0.1)339    assert_almost_equal(pnps(TEST_SEQ5, TEST_SEQ6), 0.1364 / 0.6001, delta=1e-4)340 341 342def test_dnds():343    assert_almost_equal(dnds(TEST_SEQ5, TEST_SEQ6), 0.1247, delta=1e-4)344 345# DNDS.pdf - wrong based on email conversation with Sean346#347# def test_syn_sum():348#     assert_close(syn_sum(TEST_SEQ1, TEST_SEQ2), 7.5833)349","Python"
350"Codon","adelq/dnds","dnds.py",".py","5340","169","""""""dnds351 352This module is a reference implementation of estimating nucleotide substitution353neutrality by estimating the percent of synonymous and nonsynonymous mutations.354""""""355from __future__ import print_function, division356from math import log357from fractions import Fraction358import logging359from codons import codons360 361BASES = {'A', 'G', 'T', 'C'}362 363 364def split_seq(seq, n=3):365    '''Returns sequence split into chunks of n characters, default is codons'''366    return [seq[i:i + n] for i in range(0, len(seq), n)]367 368 369def average_list(l1, l2):370    """"""Return the average of two lists""""""371    return [(i1 + i2) / 2 for i1, i2 in zip(l1, l2)]372 373 374def dna_to_protein(codon):375    '''Returns single letter amino acid code for given codon'''376    return codons[codon]377 378 379def translate(seq):380    """"""Translate a DNA sequence into the 1-letter amino acid sequence""""""381    return """".join([dna_to_protein(codon) for codon in split_seq(seq)])382 383 384def is_synonymous(codon1, codon2):385    '''Returns boolean whether given codons are synonymous'''386    return dna_to_protein(codon1) == dna_to_protein(codon2)387 388 389def dnds_codon(codon):390    '''Returns list of synonymous counts for a single codon.391    Calculations done per the methodology taught in class.392    http://sites.biology.duke.edu/rausher/DNDS.pdf393    '''394    syn_list = []395    for i in range(len(codon)):396        base = codon[i]397        other_bases = BASES - {base}398        syn = 0399        for new_base in other_bases:400            new_codon = codon[:i] + new_base + codon[i + 1:]401            syn += int(is_synonymous(codon, new_codon))402        syn_list.append(Fraction(syn, 3))403    return syn_list404 405 406def dnds_codon_pair(codon1, codon2):407    """"""Get the dN/dS for the given codon pair""""""408    return average_list(dnds_codon(codon1), dnds_codon(codon2))409 410 411def syn_sum(seq1, seq2):412    """"""Get the sum of synonymous sites from two DNA sequences""""""413    syn = 0414    codon_list1 = split_seq(seq1)415    codon_list2 = split_seq(seq2)416    for i in range(len(codon_list1)):417        codon1 = codon_list1[i]418        codon2 = codon_list2[i]419        dnds_list = dnds_codon_pair(codon1, codon2)420        syn += sum(dnds_list)421    return syn422 423 424def hamming(s1, s2):425    """"""Return the hamming distance between 2 DNA sequences""""""426    return sum(ch1 != ch2 for ch1, ch2 in zip(s1, s2)) + abs(len(s1) - len(s2))427 428 429def codon_subs(codon1, codon2):430    """"""Returns number of synonymous substitutions in provided codon pair431    Methodology for multiple substitutions from Dr. Swanson, UWashington432    https://faculty.washington.edu/wjs18/dnds.ppt433    """"""434    diff = hamming(codon1, codon2)435    if diff < 1:436        return 0437    elif diff == 1:438        return int(translate(codon1) == translate(codon2))439 440    syn = 0441    for i in range(len(codon1)):442        base1 = codon1[i]443        base2 = codon2[i]444        if base1 != base2:445            new_codon = codon1[:i] + base2 + codon1[i + 1:]446            syn += int(is_synonymous(codon1, new_codon))447            syn += int(is_synonymous(codon2, new_codon))448    return syn / diff449 450 451def substitutions(seq1, seq2):452    """"""Returns number of synonymous and nonsynonymous substitutions""""""453    dna_changes = hamming(seq1, seq2)454    codon_list1 = split_seq(seq1)455    codon_list2 = split_seq(seq2)456    syn = 0457    for i in range(len(codon_list1)):458        codon1 = codon_list1[i]459        codon2 = codon_list2[i]460        syn += codon_subs(codon1, codon2)461    return (syn, dna_changes - syn)462 463 464def clean_sequence(seq):465    """"""Clean up provided sequence by removing whitespace.""""""466    return seq.replace(' ', '')467 468 469def pnps(seq1, seq2):470    """"""Main function to calculate pN/pS between two DNA sequences.""""""471    # Strip any whitespace from both strings472    seq1 = clean_sequence(seq1)473    seq2 = clean_sequence(seq2)474    # Check that both sequences have the same length475    assert len(seq1) == len(seq2)476    # Check that sequences are codons477    assert len(seq1) % 3 == 0478 479    syn_sites = syn_sum(seq1, seq2)480    non_sites = len(seq1) - syn_sites481    logging.info('Sites (syn/nonsyn): {}, {}'.format(syn_sites, non_sites))482    syn_subs, non_subs = substitutions(seq1, seq2)483    logging.info('pN: {} / {}\t\tpS: {} / {}'484                 .format(non_subs, round(non_sites), syn_subs, round(syn_sites)))485    pn = non_subs / non_sites486    ps = syn_subs / syn_sites487    return pn / ps488 489 490def dnds(seq1, seq2):491    """"""Main function to calculate dN/dS between two DNA sequences per Nei &492    Gojobori 1986. This includes the per site conversion adapted from Jukes &493    Cantor 1967.494    """"""495    # Strip any whitespace from both strings496    seq1 = clean_sequence(seq1)497    seq2 = clean_sequence(seq2)498    # Check that both sequences have the same length499    assert len(seq1) == len(seq2)500    # Check that sequences are codons501    assert len(seq1) % 3 == 0502 503    syn_sites = syn_sum(seq1, seq2)504    non_sites = len(seq1) - syn_sites505    logging.info('Sites (syn/nonsyn): {}, {}'.format(syn_sites, non_sites))506    syn_subs, non_subs = substitutions(seq1, seq2)507    pn = non_subs / non_sites508    ps = syn_subs / syn_sites509    dn = -(3 / 4) * log(1 - (4 * pn / 3))510    ds = -(3 / 4) * log(1 - (4 * ps / 3))511    logging.info('dN: {}\t\tdS: {}'.format(round(dn, 3), round(ds, 3)))512    return dn / ds513 514 515if __name__ == '__main__':516    print(pnps('ACC GTG GGA TGC ACC GGT GTG CCC',517               'ACA GTG AGA TAT AAA GGA GAG AAC'))518","Python"
519"Codon","adelq/dnds","codons.py",".py","1045","67","codons = {520    ""TTT"": ""F"",521    ""TTC"": ""F"",522    ""TTA"": ""L"",523    ""TTG"": ""L"",524    ""TCT"": ""S"",525    ""TCC"": ""S"",526    ""TCA"": ""S"",527    ""TCG"": ""S"",528    ""TAT"": ""Y"",529    ""TAC"": ""Y"",530    ""TAA"": ""STOP"",531    ""TAG"": ""STOP"",532    ""TGT"": ""C"",533    ""TGC"": ""C"",534    ""TGA"": ""STOP"",535    ""TGG"": ""W"",536    ""CTT"": ""L"",537    ""CTC"": ""L"",538    ""CTA"": ""L"",539    ""CTG"": ""L"",540    ""CCT"": ""P"",541    ""CCC"": ""P"",542    ""CCA"": ""P"",543    ""CCG"": ""P"",544    ""CAT"": ""H"",545    ""CAC"": ""H"",546    ""CAA"": ""Q"",547    ""CAG"": ""Q"",548    ""CGT"": ""R"",549    ""CGC"": ""R"",550    ""CGA"": ""R"",551    ""CGG"": ""R"",552    ""ATT"": ""I"",553    ""ATC"": ""I"",554    ""ATA"": ""I"",555    ""ATG"": ""M"",556    ""ACT"": ""T"",557    ""ACC"": ""T"",558    ""ACA"": ""T"",559    ""ACG"": ""T"",560    ""AAT"": ""N"",561    ""AAC"": ""N"",562    ""AAA"": ""K"",563    ""AAG"": ""K"",564    ""AGT"": ""S"",565    ""AGC"": ""S"",566    ""AGA"": ""R"",567    ""AGG"": ""R"",568    ""GTT"": ""V"",569    ""GTC"": ""V"",570    ""GTA"": ""V"",571    ""GTG"": ""V"",572    ""GCT"": ""A"",573    ""GCC"": ""A"",574    ""GCA"": ""A"",575    ""GCG"": ""A"",576    ""GAT"": ""D"",577    ""GAC"": ""D"",578    ""GAA"": ""E"",579    ""GAG"": ""E"",580    ""GGT"": ""G"",581    ""GGC"": ""G"",582    ""GGA"": ""G"",583    ""GGG"": ""G""584}585","Python"
586"Codon","veg/hyphy-analyses","SCUEAL-WG/SCUEALfullG.sh",".sh","294","10","#!/bin/sh587 588#module load openmpi-x86_64589module load openmpi-1.10-x86_64590 591# Run MPI program through Ethernet eth0592 593mpirun -np 28 /localdisk/software/hyphy/hyphy-2.2.4/HYPHYMPI USEPATH=/dev/null BASEPATH=/localdisk/software/hyphy/lib/hyphy /localdisk/home/PATH_TO_FOLDER_/SCUEALfullG.bf594 595","Shell"
596"Codon","veg/hyphy-analyses","molerate/JSON.md",".md","4938","72","# Molerate JSON Schema Description597 598 599The top-level JSON object contains the following keys:600 601* **`analysis`** (Object): Contains metadata about the analysis performed.602    * `authors` (String): Name of the author(s).603    * `contact` (String): Contact email address.604    * `info` (String): Brief description of the analysis.605    * `labeling strategy` (String): The strategy used for labeling branches (e.g., ""all-descendants"").606    * `model` (String): The substitution model used (e.g., ""WAG"").607    * `rate variation` (String): The model used for rate variation across sites (e.g., ""GDD"").608    * `requirements` (String): Description of the input data required.609    * `version` (String): Version of the analysis tool.610 611* **`branch attributes`** (Object): Contains attributes for phylogenetic branches, grouped by partition (e.g., ""0"") and model.612    * `<partition_id>` (Object, e.g., ""0""): Represents a data partition.613        * `<branch_name>` (Object, e.g., ""HLacaPus1"", ""Node1""): Represents a specific branch in the tree.614            * `Proportional` (Number): Estimated branch length for the Proportional model.615            * `Proportional Partitioned` (Number): Estimated branch length for the Proportional Partitioned model.616            * `Reference` (Number): Branch length from the reference tree.617            * `Unconstrained Test` (Number): Estimated branch length for the Unconstrained Test model.618            * *( Potentially other model names as keys with numeric values )*619    * `attributes` (Object): Defines metadata for the models listed under branch names.620        * `<model_name>` (Object, e.g., ""Proportional""):621            * `attribute type` (String): Type of attribute (e.g., ""branch length"").622            * `display order` (Number): Suggested order for displaying this model's results.623 624* **`branch level analysis`** (Object): Contains detailed likelihood ratio test results for individual branches designated as 'test'.625    * `<branch_name>` (Object, e.g., ""HLamaAes1""): Represents a specific 'test' branch.626        * `LogL` (Number): Log-likelihood for the `Proportional+1` model at this branch.627        * `alternative` (Object): `Proportional+1` vs `Unconstrained Test`628            * `Corrected P-value` (Number): Corrected (Holm-Bonferroni) p-value.629            * `LRT` (Number): Likelihood Ratio Test statistic.630            * `p-value` (Number): Uncorrected p-value.631        * `null` (Object):  `Proportional` vs `Proportional+1`632            * `Corrected P-value` (Number): orrected (Holm-Bonferroni) p-value.633            * `LRT` (Number): Likelihood Ratio Test statistic.634            * `p-value` (Number): Uncorrected p-value.635 636* **`fits`** (Object): Contains details about the statistical fit of each evolutionary model tested.637    * `<model_name>` (Object, e.g., ""Proportional"", ""Unconstrained Test""):638        * `AIC-c` (Number): Corrected Akaike Information Criterion value.639        * `Log Likelihood` (Number): Log-likelihood value for the model fit.640        * `Rate Distributions` (Object): Parameters related to site rate variation.641            * *( Keys vary based on the rate variation model, e.g., GDD category rates, mixture weights, branch scalers )* (Number)642        * `display order` (Number): Suggested order for displaying this model's fit results.643        * `estimated parameters` (Number): Number of parameters estimated for this model.644 645* **`input`** (Object): Describes the input data used for the analysis.646    * `file name` (String): Path to the input alignment file.647    * `number of sequences` (Number): Count of sequences in the alignment.648    * `number of sites` (Number): Count of sites (columns) in the alignment.649    * `partition count` (Number): Number of data partitions (usually 0 or 1 if not partitioned).650    * `trees` (Object): Contains the input tree structure(s).651        * `<partition_id>` (String, e.g., ""0""): Tree structure in Newick format string.652 653* **`runtime`** (String):  Version of `HyPhy` used to execute the analysis.654 655* **`test results`** (Object): Contains results of Likelihood Ratio Tests comparing pairs of nested models.656    * `<model_comparison>` (Object, e.g., ""Proportional Partitioned:Unconstrained Test""): Represents a comparison between two models (Null:Alternative).657        * `Corrected P-value` (Number): Corrected (Holm-Bonferroni) p-value for the comparison.658        * `LRT` (Number): Likelihood Ratio Test statistic.659        * `Uncorrected P-value` (Number): Uncorrected p-value.660 661* **`tested`** (Object): Maps each branch name from the input tree to its classification in the analysis.662    * `<branch_name>` (String, e.g., ""HLacaPus1""): Value is either ""test"" or ""background"".663 664* **`timers`** (Object): Records the time duration for specific parts of the analysis.665    * `<component_name>` (Object, e.g., ""Overall"", ""Proportional""):666        * `order` (Number): Order index for the component.667        * `timer` (Number): Time duration in seconds.","Markdown"
668"Codon","Ensembl/ensembl-compara","setup.py",".py","776","22","# See the NOTICE file distributed with this work for additional information669# regarding copyright ownership.670#671# Licensed under the Apache License, Version 2.0 (the ""License"");672# you may not use this file except in compliance with the License.673# You may obtain a copy of the License at674#675#     http://www.apache.org/licenses/LICENSE-2.0676#677# Unless required by applicable law or agreed to in writing, software678# distributed under the License is distributed on an ""AS IS"" BASIS,679# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.680# See the License for the specific language governing permissions and681# limitations under the License.682""""""setuptools-based stub for editable installs""""""683 684from setuptools import setup685 686 687if __name__ == ""__main__"":688    setup()689","Python"
690"Codon","Ensembl/ensembl-compara","pull_request_template.md",".md","521","26","## Description691 692_Describe the problem you're addressing here._693 694**Related JIRA tickets:**695- ENSCOMPARASW-XXXX696 697## Overview of changes698_Give details of what changes were required to solve the problem. Break into sections if applicable._699 700#### Change 1701- _detail 1.1_702 703#### Change 2704- _detail 2.1_705 706## Testing707_How was this tested? Have new unit tests been included?_708 709## Notes710_Optional extra information._711 712---713 714For code reviewers: [code review SOP](https://www.ebi.ac.uk/seqdb/confluence/display/EnsCom/Code+review+SOP)715","Markdown"
716"Codon","Ensembl/ensembl-compara","run_all_tests.sh",".sh","1573","32","#!/bin/bash717 718# See the NOTICE file distributed with this work for additional information719# regarding copyright ownership.720# 721# Licensed under the Apache License, Version 2.0 (the ""License"");722# you may not use this file except in compliance with the License.723# You may obtain a copy of the License at724# 725#      http://www.apache.org/licenses/LICENSE-2.0726# 727# Unless required by applicable law or agreed to in writing, software728# distributed under the License is distributed on an ""AS IS"" BASIS,729# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.730# See the License for the specific language governing permissions and731# limitations under the License.732 733set -euxo pipefail734 735cd ""$ENSEMBL_ROOT_DIR""736prove -r ensembl-compara/travisci/all-housekeeping/737prove -r ensembl-compara/travisci/sql-unittest/738ensembl-test/scripts/runtests.pl ensembl-compara/modules/t739ensembl-test/scripts/runtests.pl ensembl/modules/t/compara.t740ensembl-test/scripts/runtests.pl ensembl-rest/t/genomic_alignment.t ensembl-rest/t/info.t ensembl-rest/t/taxonomy.t ensembl-rest/t/homology.t ensembl-rest/t/gene_tree.t ensembl-rest/t/cafe_tree.t ensembl-rest/t/family.t741cd ""$ENSEMBL_ROOT_DIR/ensembl-compara""742./travisci/perl-linter_harness.sh743find docs modules scripts sql travisci -iname '*.t' -print0 | xargs -0 -n 1 perl -c744find docs modules scripts sql travisci -iname '*.pl' -print0 | xargs -0 -n 1 perl -c745find docs modules scripts sql travisci -iname '*.pm' \! -name 'LoadSynonyms.pm' \! -name 'HALAdaptor.pm' \! -name 'HALXS.pm' -print0 | xargs -0 -n 1 perl -c746echo ""All good !""747","Shell"
748"Codon","Ensembl/ensembl-compara","DEPRECATED.md",".md","2457","87","> This file contains the list of methods deprecated in the Ensembl Compara749> API.  A method is deprecated when it is not functional any more750> (schema/data change) or has been replaced by a better one.  Backwards751> compatibility is provided whenever possible.  When a method is752> deprecated, a deprecation warning is thrown whenever the method is used.753> The warning also contains instructions on replacing the deprecated method754> and when it will be removed.755 756----757 758# Deprecated methods scheduled for deletion759 760## Ensembl 116761 762* `Bio::EnsEMBL::Compara::Utils::MasterDatabase::find_overlapping_genome_db_ids`763 764# Deprecated methods not yet scheduled for deletion765 766* `AlignedMember::get_cigar_array()`767* `AlignedMember::get_cigar_breakout()`768* `GenomicAlignTree::genomic_align_array()`769* `GenomicAlignTree::get_all_GenomicAligns()`770* `Homology::dn()`771* `Homology::dnds_ratio()`772* `Homology::ds()`773* `Homology::lnl()`774* `Homology::n()`775* `Homology::s()`776* `Homology::threshold_on_ds()`777* `HomologyAdaptor::update_genetic_distance()`778 779# Methods removed in previous versions of Ensembl780 781## Ensembl 110782 783* `DBSQL::'*MemberAdaptor::fetch_by_stable_id()`784 785## Ensembl 100786 787* `DBSQL::'*MemberAdaptor::get_source_taxon_count()`788 789## Ensembl 98790 791* `DBSQL::DnaFragAdaptor::fetch_all_by_GenomeDB_region()`792 793## Ensembl 96794 795* `AlignSlice::Slice::get_all_VariationFeatures_by_VariationSet`796* `AlignSlice::Slice::get_all_genotyped_VariationFeatures`797* `Taggable::get_value_for_XXX()`798* `Taggable::get_all_values_for_XXX()`799* `Taggable::get_XXX_value()`800 801## Ensembl 93802 803* `DnaFrag::isMT()`804* `DnaFrag::dna_type()`805* `MethodLinkSpeciesSet::species_set_obj()`806 807## Ensembl 92808 809* `Member::print_member()`810* `SyntenyRegion::regions()`811 812## Ensembl 91813 814* `Homology::print_homology()`815* `MethodLinkSpeciesSet::get_common_classification()`816* `NCBITaxon::binomial()`817* `NCBITaxon::ensembl_alias_name()`818* `NCBITaxon::common_name()`819* `DnaFragAdaptor::is_already_stored()`820* `GeneMemberAdaptor::fetch_all_by_source_Iterator()`821* `GeneMemberAdaptor::fetch_all_Iterator()`822* `GeneMemberAdaptor::load_all_from_seq_members()`823* `GenomeDBAdaptor::fetch_all_by_low_coverage()`824* `GenomeDBAdaptor::fetch_all_by_taxon_id_assembly()`825* `GenomeDBAdaptor::fetch_by_taxon_id()`826* `SeqMemberAdaptor::fetch_all_by_source_Iterator()`827* `SeqMemberAdaptor::fetch_all_Iterator()`828* `SeqMemberAdaptor::update_sequence()`829* `SequenceAdaptor::fetch_by_dbIDs()`830 831## Ensembl 89832 833* `SpeciesTree::species_tree()`834","Markdown"
835"Codon","Ensembl/ensembl-compara","conftest.py",".py","3680","88","# See the NOTICE file distributed with this work for additional information836# regarding copyright ownership.837#838# Licensed under the Apache License, Version 2.0 (the ""License"");839# you may not use this file except in compliance with the License.840# You may obtain a copy of the License at841#842#     http://www.apache.org/licenses/LICENSE-2.0843#844# Unless required by applicable law or agreed to in writing, software845# distributed under the License is distributed on an ""AS IS"" BASIS,846# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.847# See the License for the specific language governing permissions and848# limitations under the License.849""""""Local directory-specific hook implementations.850 851Since this file is located at the root of all ensembl-compara tests, every test in every subfolder will have852access to the plugins, hooks and fixtures defined here.853 854""""""855# Disable all the redefined-outer-name violations due to how pytest fixtures work856# pylint: disable=redefined-outer-name857 858import os859from pathlib import Path860import shutil861import time862 863import pytest864from _pytest.fixtures import FixtureRequest865 866from ensembl.compara.filesys import DirCmp867 868 869pytest_plugins = (""ensembl.utils.plugin"",)870 871 872def pytest_configure() -> None:873    """"""Adds global variables and configuration attributes required by Compara's unit tests.874 875    `Pytest initialisation hook876    <https://docs.pytest.org/en/latest/reference.html#_pytest.hookspec.pytest_configure>`_.877 878    """"""879    test_data_dir = Path(__file__).parent / 'src' / 'test_data'880    pytest.dbs_dir = test_data_dir / 'databases'  # type: ignore[attr-defined]881    pytest.files_dir = test_data_dir / 'flatfiles'  # type: ignore[attr-defined]882 883 884@pytest.fixture(scope=""session"")885def dir_cmp(request: FixtureRequest, tmp_path_factory: pytest.TempPathFactory) -> DirCmp:886    """"""Returns a directory tree comparison (:class:`DirCmp`) object.887 888    Requires a dictionary with the following keys:889 890        ref (:obj:`PathLike`): Reference root directory path.891        target (:obj:`PathLike`): Target root directory path.892 893    passed via `request.param`. In both cases, if a relative path is provided, the starting folder will be894    ``src/python/tests/flatfiles``. This fixture is intended to be used via indirect parametrization, for895    example::896 897        @pytest.mark.parametrize(""dir_cmp"", [{'ref': 'citest/reference', 'target': 'citest/target'}],898                                 indirect=True)899        def test_method(..., dir_cmp: DirCmp, ...):900 901    Args:902        request: Access to the requesting test context.903        tmp_path_factory: Temporary directory path factory fixture.904 905    """"""906    tmp_path = tmp_path_factory.mktemp(""dir_cmp_root"")907    # Get the source and temporary absolute paths for reference and target root directories908    ref = Path(request.param['ref'])  # type: ignore[attr-defined]909    ref_src = ref if ref.is_absolute() else pytest.files_dir / ref  # type: ignore910    ref_tmp = tmp_path / str(ref).replace(os.path.sep, '_')911    target = Path(request.param['target'])  # type: ignore[attr-defined]912    target_src = target if target.is_absolute() else pytest.files_dir / target  # type: ignore913    target_tmp = tmp_path / str(target).replace(os.path.sep, '_')914    # Copy directory trees (if they have not been copied already) ignoring file metadata915    if not ref_tmp.exists():916        shutil.copytree(ref_src, ref_tmp, copy_function=shutil.copy)917    # Sleep one second in between to ensure the timestamp differs between reference and target files918    time.sleep(1)919    if not target_tmp.exists():920        shutil.copytree(target_src, target_tmp, copy_function=shutil.copy)921    return DirCmp(ref_tmp, target_tmp)922","Python"
923"Codon","Ensembl/ensembl-compara","modules/Bio/EnsEMBL/Compara/HAL/HALXS/INLINE.h",".h","1732","44","/*924 * See the NOTICE file distributed with this work for additional information925 * regarding copyright ownership.926 * 927 * Licensed under the Apache License, Version 2.0 (the ""License"");928 * you may not use this file except in compliance with the License.929 * You may obtain a copy of the License at930 * 931 *      http://www.apache.org/licenses/LICENSE-2.0932 * 933 * Unless required by applicable law or agreed to in writing, software934 * distributed under the License is distributed on an ""AS IS"" BASIS,935 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.936 * See the License for the specific language governing permissions and937 * limitations under the License.938 */939 940#define Inline_Stack_Vars dXSARGS941#define Inline_Stack_Items items942#define Inline_Stack_Item(x) ST(x)943#define Inline_Stack_Reset sp = mark944#define Inline_Stack_Push(x) XPUSHs(x)945#define Inline_Stack_Done PUTBACK946#define Inline_Stack_Return(x) XSRETURN(x)947#define Inline_Stack_Void XSRETURN(0)948 949#define INLINE_STACK_VARS Inline_Stack_Vars950#define INLINE_STACK_ITEMS Inline_Stack_Items951#define INLINE_STACK_ITEM(x) Inline_Stack_Item(x)952#define INLINE_STACK_RESET Inline_Stack_Reset953#define INLINE_STACK_PUSH(x) Inline_Stack_Push(x)954#define INLINE_STACK_DONE Inline_Stack_Done955#define INLINE_STACK_RETURN(x) Inline_Stack_Return(x)956#define INLINE_STACK_VOID Inline_Stack_Void957 958#define inline_stack_vars Inline_Stack_Vars959#define inline_stack_items Inline_Stack_Items960#define inline_stack_item(x) Inline_Stack_Item(x)961#define inline_stack_reset Inline_Stack_Reset962#define inline_stack_push(x) Inline_Stack_Push(x)963#define inline_stack_done Inline_Stack_Done964#define inline_stack_return(x) Inline_Stack_Return(x)965#define inline_stack_void Inline_Stack_Void966","Unknown"
967"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/fix_leaf_names.py",".py","1811","58","#!/usr/bin/env python3968 969# See the NOTICE file distributed with this work for additional information970# regarding copyright ownership.971#972# Licensed under the Apache License, Version 2.0 (the ""License"");973# you may not use this file except in compliance with the License.974# You may obtain a copy of the License at975#976#      http://www.apache.org/licenses/LICENSE-2.0977#978# Unless required by applicable law or agreed to in writing, software979# distributed under the License is distributed on an ""AS IS"" BASIS,980# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.981# See the License for the specific language governing permissions and982# limitations under the License.983 984""""""985Replace leaf names in a species tree with production name.986Example:987    $ python fix_leaf_names.py -t astral_species_tree_neutral_bl.nwk -c input_genomes.csv -o test.nwk988""""""989 990import sys991import argparse992from os import path993from Bio import Phylo994import pandas as pd995 996# Parse command line arguments:997parser = argparse.ArgumentParser(998    description='Replace leafs of a species tree with production names.')999parser.add_argument(1000    '-t', metavar='tree', type=str, help=""Input tree."", required=True, default=None)1001parser.add_argument(1002    '-c', metavar='csv', type=str, help=""Input CSV."", required=True, default=None)1003parser.add_argument(1004    '-o', metavar='output', type=str, help=""Output tree."", required=True)1005 1006 1007if __name__ == '__main__':1008    args = parser.parse_args()1009 1010    df = pd.read_csv(args.c, delimiter=""\t"", header=None)1011    trans_map = {}1012    for r in df.itertuples():1013        trans_map[r[10]] = r[2]1014 1015    tree = Phylo.read(args.t, format=""newick"")1016    for leaf in tree.get_terminals():1017        leaf.name = trans_map[leaf.name]1018 1019    Phylo.write(1020        trees=tree,1021        file=args.o,1022        format=""newick""1023    )1024","Python"
1025"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/pick_third_site.py",".py","1845","59","#!/usr/bin/env python31026 1027# See the NOTICE file distributed with this work for additional information1028# regarding copyright ownership.1029#1030# Licensed under the Apache License, Version 2.0 (the ""License"");1031# you may not use this file except in compliance with the License.1032# You may obtain a copy of the License at1033#1034#     http://www.apache.org/licenses/LICENSE-2.01035#1036# Unless required by applicable law or agreed to in writing, software1037# distributed under the License is distributed on an ""AS IS"" BASIS,1038# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1039# See the License for the specific language governing permissions and1040# limitations under the License.1041""""""1042Pick out every third site from a codon alignment. The input sequences1043must have a length divisible by 3.1044 1045Example:1046    $ python  pick_third_site.py -i input.fas -o output.fas1047""""""1048 1049import sys1050import argparse1051from collections import OrderedDict1052 1053import pandas as pd1054from Bio import SeqIO1055 1056# Parse command line arguments:1057parser = argparse.ArgumentParser(1058    description='Pick out the third sites from a codon alignment.')1059parser.add_argument(1060    '-i', metavar='input', type=str, help=""Input fasta."", required=True)1061parser.add_argument(1062    '-o', metavar='output', type=str, help=""Output fasta."", required=True)1063 1064 1065if __name__ == '__main__':1066    args = parser.parse_args()1067 1068    fh = open(args.o, ""w"")1069 1070    for record in SeqIO.parse(args.i, ""fasta""):1071        # Check if sequence length is divisible by 3:1072        if len(record.seq) % 3 != 0:1073            sys.stderr.write(f""The length of sequence {record.id} is not divisible by 3!\n"")1074            sys.exit(1)1075        # Pick out every third site:1076        print(len(record.seq))1077        record.seq = record.seq[2::3]1078        # Write out record:1079        SeqIO.write(record, fh, ""fasta"")1080 1081    fh.flush()1082    fh.close()1083","Python"
1084"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/fetch_genomes_from_db.py",".py","4595","120","#!/usr/bin/env python31085 1086# See the NOTICE file distributed with this work for additional information1087# regarding copyright ownership.1088#1089# Licensed under the Apache License, Version 2.0 (the ""License"");1090# you may not use this file except in compliance with the License.1091# You may obtain a copy of the License at1092#1093#      http://www.apache.org/licenses/LICENSE-2.01094#1095# Unless required by applicable law or agreed to in writing, software1096# distributed under the License is distributed on an ""AS IS"" BASIS,1097# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1098# See the License for the specific language governing permissions and1099# limitations under the License.1100 1101""""""1102Locate genome dumps of a collection or a species set on disk.1103Example:1104    $ python fetch_genomes_from_db.py -u mysql://ensro@mysql-ens-compara-prod-1:4485/ensembl_compara_master -c pig_breeds -o test.tsv -d /hps/nobackup/flicek/ensembl/compara/shared/genome_dumps/vertebrates1105""""""1106 1107import sys1108import argparse1109import csv1110from os import path1111from sqlalchemy.engine.row import Row1112from sqlalchemy import create_engine, text1113 1114# Parse command line arguments:1115parser = argparse.ArgumentParser(1116    description='Locate genomes of a species set/collection in the dump directory.')1117parser.add_argument('-u', metavar='db_URL', type=str, help=""Compara master db URL."", required=True)1118parser.add_argument(1119    '-d', metavar='dump_dir', type=str, help=""Genome dump directory."", required=True)1120parser.add_argument(1121    '-s', metavar='ssid', type=str, help=""Species set id."", required=False, default=None)1122parser.add_argument(1123    '-c', metavar='cid', type=str, help=""Collection id."", required=False, default=None)1124parser.add_argument(1125    '-o', metavar='output', type=str, help=""Output CSV."", required=True)1126 1127 1128def _dir_revhash(gid: int) -> str:1129    """"""Build directory hash from genome db id.""""""1130    dir_hash = list(reversed(str(gid)))1131    dir_hash.pop()1132    return path.join(*dir_hash) if dir_hash else path.curdir1133 1134 1135def _build_dump_path(row: Row) -> str:1136    gcomp = """"1137    if row.genome_component is not None:1138        gcomp = f""comp{row.genome_component}.""1139    gpath = path.join(args.d, _dir_revhash(row.genome_db_id), f""{row.name}.{row.assembly}.{gcomp}soft.fa"")1140    return gpath1141 1142 1143if __name__ == '__main__':1144    args = parser.parse_args()1145 1146    db_url = args.u1147    ss_id = args.s1148    if ss_id == """":1149        ss_id = None1150    coll_id = args.c1151    if coll_id == """":1152        coll_id = None1153    engine = create_engine(db_url, future=True)1154 1155    if ss_id is None and coll_id is None:1156        sys.stderr.write(""Either a collection name or a species set id must be specified!\n"")1157        sys.exit(1)1158 1159    if ss_id is not None and coll_id is not None:1160        sys.stderr.write(""Specify either a collection name or a species set id!\n"")1161        sys.exit(1)1162 1163    ss_query = f""""""1164        SELECT genome_db_id, genome_db.name, assembly, genebuild,1165        strain_name, display_name, genome_component1166        FROM species_set JOIN genome_db USING(genome_db_id)1167        WHERE species_set_id = {ss_id};1168    """"""1169    c_query = f""""""1170        SELECT genome_db_id, genome_db.name, assembly, genebuild,1171        strain_name, display_name, genome_component1172        FROM species_set_header JOIN species_set USING(species_set_id)1173        JOIN genome_db USING(genome_db_id) WHERE species_set_header.name='{coll_id}'1174        AND species_set_header.last_release is NULL1175        AND species_set_header.first_release IS NOT NULL;1176    """"""1177    query = ss_query1178    if coll_id is not None:1179        query = c_query1180 1181    missing_fas = []1182    with engine.connect() as conn, open(args.o, ""w"") as ofh:1183        result = conn.execute(text(query))1184        writer = csv.writer(ofh, delimiter=""\t"", lineterminator=""\n"")1185        for row in result:1186            gpath = _build_dump_path(row)1187            fa_name = str(row.name) + ""_"" + str(row.assembly)1188            prod_name = row.name1189            if row.genome_component is not None:1190                fa_name += f""_{row.genome_component}""1191                prod_name += f""_{row.genome_component}""1192            dump_exists = path.exists(gpath)1193            if not dump_exists:1194                missing_fas.append(gpath)1195            writer.writerow([row.genome_db_id, prod_name, row.assembly, row.genebuild, row.strain_name,1196                             row.display_name, row.genome_component, gpath, dump_exists, fa_name])1197 1198    if len(missing_fas) > 0:1199        sys.stderr.write(""Fatal error! The following genome dumps are missing:\n"")1200        for p in missing_fas:

Showing the first 1,200 of 10591 lines. Download the file for the rest.