SciCodePile/SciCode-Domain-Code
DATA1: Domain-Specific Code Dataset Dataset Overview DATA1 is a large-scale domain-specific code dataset focusing on code samples from interdisciplinary fields such as biology, chemistry, materials science, and related areas. The dataset is collected and organized from GitHub repositories, covering 178 different domain topics with over 1.1 billion lines of code. Dataset Statistics Total Datasets: 178 CSV files Total Data Size: ~115 GB Total Lines… See the full description on the dataset page: https://huggingface.co/datasets/SciCodePile/SciCode-Domain-Code.
41.3k
1"keyword","repo_name","file_path","file_extension","file_size","line_count","content","language"
2"Codon","LKremer/MSA_trimmer","alignment_trimmer.py",".py","9112","221","#!/usr/bin/env python33 4 5from __future__ import division, print_function6from Bio import SeqIO7from Bio.Seq import Seq8from Bio.SeqRecord import SeqRecord9from Bio.Alphabet import SingleLetterAlphabet10import argparse11import os12 13 14def gap_proportion(msa_column):15 gap_counter = 016 for aa_or_codon in msa_column:17 if aa_or_codon in ('-', '---'):18 gap_counter += 119 return gap_counter / len(msa_column)20 21 22class Alignment():23 def __init__(self, alignment_fasta, maxgap,24 trim_positions, mask, is_cds=False):25 self._is_cds = is_cds26 self._letter_n = 3 if is_cds else 127 self._maxgap = maxgap28 self._fasta = []29 self._trimmed_fasta = []30 with open(alignment_fasta, 'r') as input_handle:31 for fasta_record in SeqIO.parse(input_handle, 'fasta'):32 self._fasta.append(fasta_record)33 self._trimmed_fasta.append([])34 self._first_seq = self._fasta[0]35 self._assert_fasta_is_aligned()36 self._set_mask_char(mask)37 self._parse_trim_positions(trim_positions)38 fname, fextension = os.path.splitext(alignment_fasta)39 self._output_filepath = '{}_{}{}'.format(40 fname,41 ""masked"" if self._mask_char else ""trimmed"",42 fextension,43 )44 return45 46 def _parse_trim_positions(self, trim_pos_list):47 self._trim_positions = set()48 if trim_pos_list:49 for trim_pos in trim_pos_list:50 if '-' in trim_pos: # it's a range!51 start_raw, stop_raw = trim_pos.split('-')52 start = int(start_raw) - 153 stop = int(stop_raw)54 if start >= stop:55 raise Exception('Invalid range:', trim_pos)56 self._trim_positions |= set(range(start, stop))57 else:58 pos = self._str_to_pos(trim_pos)59 self._trim_positions.add(pos)60 return61 62 def _str_to_pos(self, pos_str):63 try:64 pos = int(pos_str) - 1 # account for zero-indexing65 except ValueError:66 raise Exception(pos_str, 'is not a valid position!')67 self[pos] # raises IndexError when pos is out of range of the MSA68 return pos69 70 def _set_mask_char(self, mask_param):71 if mask_param:72 if len(mask_param) != 1:73 raise Exception('Invalid --mask character:', mask_param)74 self._mask_char = mask_param * self._letter_n75 else:76 self._mask_char = False77 return78 79 def _assert_fasta_is_aligned(self):80 for fa_entry in self._fasta[1:]:81 if len(fa_entry) != len(self._first_seq):82 raise Exception('The sequences {} and {} have different '83 'lengths. Is your fasta really aligned'84 '?'.format(fa_entry.id, self._first_seq.id))85 return86 87 def __getitem__(self, i):88 ''' returns the n-th column of the multifasta, works for89 CDS and protein fastas '''90 n = self._letter_n # the number of characters per position91 # three for CDS sequences (codons), one for protein sequences92 start = i * n93 end = start + n94 if start < 0 or end > len(self._first_seq):95 raise IndexError('Cannot access MSA position {} - index out '96 'of range'.format(i))97 return tuple(str(fa.seq[start:end]) for fa in self._fasta)98 99 def __iter__(self):100 ''' column-based iteration over MSA fasta, works for101 CDS and protein fastas '''102 for i in range(int(len(self._first_seq) / self._letter_n)):103 yield self[i]104 105 def _trim_column(self, msa_column, col_index):106 ''' trim an MSA column if the number of gaps exceeds the user-specified107 threshold, or if it is within the user-specified trim-range '''108 col_str = ' '.join(msa_column)109 gap_prop = gap_proportion(msa_column)110 111 if (col_index in self._trim_positions) or (gap_prop > self._maxgap):112 if self._mask_char:113 action = ' - masked'114 else:115 action = ' - trimmed'116 trim_column = True117 else:118 action = ''119 trim_column = False120 121 for row_i, aa_or_codon in enumerate(msa_column):122 if trim_column: # trim (or mask) the column123 if self._mask_char: # mask it124 self._trimmed_fasta[row_i].append(self._mask_char)125 else: # don't add the column at all126 break127 else: # keep the column as it was128 self._trimmed_fasta[row_i].append(aa_or_codon)129 print('{: >3} {} ({:.1%} gaps){}'.format(130 col_index + 1, col_str, gap_prop, action))131 return132 133 def _trim_fasta(self):134 for col_index, msa_column in enumerate(self):135 self._trim_column(msa_column, col_index)136 self._trimmed_seqs = []137 for i, trimmed_seq_list in enumerate(self._trimmed_fasta):138 trimmed_seq = ''.join(trimmed_seq_list)139 input_seq = self._fasta[i]140 out_record = SeqRecord(141 Seq(trimmed_seq, SingleLetterAlphabet),142 id=input_seq.id,143 description=input_seq.description144 )145 self._trimmed_seqs.append(out_record)146 147 def dump_trimmed_fasta(self):148 self._trim_fasta()149 SeqIO.write(self._trimmed_seqs, self._output_filepath, 'fasta')150 if self._mask_char:151 print(' Wrote ""{}""-masked fasta to {}\n'.format(152 self._mask_char, self._output_filepath))153 else:154 print(' Wrote trimmed fasta to {}\n'.format(self._output_filepath))155 return156 157 def __eq__(self, other):158 """""" defines an equality test with the == operator159 returns False when a position of one alignment is a gap160 when it isn't a gap in the other alignment """"""161 for col_i, self_column in enumerate(self):162 other_column = other[col_i]163 for row_i, aa_or_codon in enumerate(self_column):164 if aa_or_codon in ('-', '---'):165 if other_column[row_i] not in ('-', '---'):166 print('\nCodon mismatch in row {} column {}'167 '\n""{}"" does not match ""{}""\n'.format(168 row_i, col_i, aa_or_codon, other_column[row_i]))169 return False170 return True171 172 def __ne__(self, other):173 """""" defines a non-equality test with the != operator """"""174 return not self.__eq__(other)175 176 177def main(pep_alignment_fasta, cds_alignment_fasta,178 trim_gappy, trim_positions, mask):179 if pep_alignment_fasta:180 pep_al = Alignment(pep_alignment_fasta, maxgap=trim_gappy, mask=mask,181 trim_positions=trim_positions)182 pep_al.dump_trimmed_fasta()183 if cds_alignment_fasta:184 cds_al = Alignment(cds_alignment_fasta, maxgap=trim_gappy, mask=mask,185 trim_positions=trim_positions, is_cds=True)186 if pep_alignment_fasta and cds_alignment_fasta:187 if pep_al != cds_al:188 raise Exception('The two alignments have a mismatch'189 '- see printouts above.')190 else:191 print('Peptide and CDS alignment are in agreement!\n')192 cds_al.dump_trimmed_fasta()193 return194 195 196if __name__ == '__main__':197 parser = argparse.ArgumentParser(198 description='A program to remove (trim) or mask columns of fasta '199 'multiple sequence alignments. Columns are removed if they exceed a '200 'user-specified ""gappyness"" treshold, and/or if they are specified by '201 'column index. Also works for codon alignments. The trimmed alignment '202 'is written to [your_alignment]_trimmed.fa')203 parser.add_argument('-p', '--pep_alignment_fasta')204 parser.add_argument('-c', '--cds_alignment_fasta')205 parser.add_argument('--trim_gappy', type=float, default=9999,206 help='the maximum allowed gap proportion for an MSA '207 'column, all columns with a higher proportion of gaps '208 'will be trimmed (default: off)')209 parser.add_argument('--trim_positions', help='specify columns to be '210 'trimmed from the alignment, e.g. ""1-10 42"" to trim the '211 'first ten columns and the 42th column', nargs='+')212 parser.add_argument('--mask', default=False, const='-',213 action='store', nargs='?', help='mask the alignment '214 'instead of trimming. if a character is specified after'215 ' --mask, that character will be used for masking '216 '(default char: ""-"")')217 args = parser.parse_args()218 if args.pep_alignment_fasta or args.cds_alignment_fasta:219 main(**vars(args))220 else:221 parser.print_help()222","Python"
223"Codon","adelq/dnds","setup.py",".py","1034","30","from setuptools import setup224 225package = 'dnds'226version = '2.1'227 228setup(229 name=package,230 version=version,231 description=""Calculate dN/dS ratio precisely (Ka/Ks) using a codon-by-codon counting method."",232 long_description=open(""README.rst"").read(),233 author=""Adel Qalieh"",234 author_email=""adelq@med.umich.edu"",235 url=""https://github.com/adelq/dnds"",236 license=""MIT"",237 py_modules=['dnds', 'codons'],238 classifiers=[239 'Development Status :: 5 - Production/Stable',240 'Intended Audience :: Science/Research',241 'License :: OSI Approved :: MIT License',242 'Natural Language :: English',243 'Operating System :: OS Independent',244 'Programming Language :: Python :: 2',245 'Programming Language :: Python :: 2.7',246 'Programming Language :: Python :: 3',247 'Programming Language :: Python :: 3.4',248 'Programming Language :: Python :: 3.5',249 'Programming Language :: Python :: 3.6',250 'Topic :: Scientific/Engineering :: Bio-Informatics',251 ])252","Python"
253"Codon","adelq/dnds","test_dnds.py",".py","3156","97","from __future__ import division254from fractions import Fraction255from nose.tools import assert_equal, assert_almost_equal256from dnds import dnds, pnps, substitutions, dnds_codon, dnds_codon_pair, syn_sum, translate257 258# From Canvas practice problem259TEST_SEQ1 = 'ACTCCGAACGGGGCGTTAGAGTTGAAACCCGTTAGA'260TEST_SEQ2 = 'ACGCCGATCGGCGCGATAGGGTTCAAGCTCGTACGA'261# From in-class problem set262TEST_SEQ3 = 'ATGCTTTTGAAATCGATCGTTCGTTCACATCGATGGATC'263TEST_SEQ4 = 'ATGCGTTCGAAGTCGATCGATCGCTCAGATCGATCGATC'264# From http://bioinformatics.cvr.ac.uk/blog/calculating-dnds-for-ngs-datasets/265TEST_SEQ5 = 'ATGAAACCCGGGTTTTAA'266TEST_SEQ6 = 'ATGAAACGCGGCTACTAA'267 268 269def test_translate():270 assert_equal(translate(TEST_SEQ1), 'TPNGALELKPVR')271 assert_equal(translate(TEST_SEQ2), 'TPIGAIGFKLVR')272 assert_equal(translate(TEST_SEQ3), 'MLLKSIVRSHRWI')273 274 275def test_dnds_codon_easy():276 assert_equal([0, 0, 1], dnds_codon('ACC'))277 278 279def test_dnds_codon_harder1():280 assert_equal([0, 0, Fraction(1, 3)], dnds_codon('AAC'))281 assert_equal([0, 0, Fraction(2, 3)], dnds_codon('ATC'))282 283 284def test_dnds_codon_harder2():285 assert_equal([Fraction(1, 3), 0, Fraction(1, 3)], dnds_codon('AGA'))286 assert_equal([Fraction(1, 3), 0, 1], dnds_codon('CGA'))287 288 289def test_dnds_codon_ngs():290 assert_equal([0, 0, 0], dnds_codon('TGG'))291 assert_equal([Fraction(1, 3), 0, 1], dnds_codon('CGG'))292 293 294def test_dnds_codon_ngs_sums():295 assert_equal(sum(dnds_codon('ATG')), 0)296 assert_equal(sum(dnds_codon('AAA')), Fraction(1, 3))297 assert_equal(sum(dnds_codon('CCC')), 1)298 assert_equal(sum(dnds_codon('GGG')), 1)299 assert_equal(sum(dnds_codon('TTT')), Fraction(1, 3))300 assert_equal(sum(dnds_codon('TAA')), Fraction(2, 3))301 302 303def test_dnds_codon_ps():304 assert_equal(dnds_codon('CTT'), [0, 0, 1])305 assert_equal(dnds_codon('CGT'), [0, 0, 1])306 307 308def test_dnds_codon_pair():309 assert_equal([Fraction(1, 3), 0, Fraction(2, 3)],310 dnds_codon_pair('AGA', 'CGA'))311 312 313def test_dnds_codon_pair_harder1():314 assert_equal([Fraction(1, 6), 0, Fraction(1, 3)],315 dnds_codon_pair('TTG', 'TTC'))316 317 318def test_dnds_codon_pair_harder2():319 assert_equal([Fraction(1, 6), 0, Fraction(1, 2)],320 dnds_codon_pair('TTA', 'ATA'))321 322 323def test_syn_sum_ps():324 assert_almost_equal(syn_sum(TEST_SEQ3, TEST_SEQ4), 10, delta=1)325 326 327def test_syn_subs():328 assert_equal(substitutions(TEST_SEQ1, TEST_SEQ2), (5, 5))329 assert_equal(substitutions(TEST_SEQ3, TEST_SEQ4), (2, 5))330 assert_equal(substitutions(""ACCGGA"", ""ACAAGA""), (1, 1))331 assert_equal(substitutions(""TGC"", ""TAT""), (1, 1))332 assert_equal(substitutions(""ACC"", ""AAA""), (0.5, 1.5))333 assert_equal(substitutions(""CCC"", ""AAC""), (0, 2))334 335 336def test_pnps():337 assert_almost_equal(pnps(TEST_SEQ1, TEST_SEQ2), 0.269, delta=0.1)338 assert_almost_equal(pnps(TEST_SEQ3, TEST_SEQ4), 0.86, delta=0.1)339 assert_almost_equal(pnps(TEST_SEQ5, TEST_SEQ6), 0.1364 / 0.6001, delta=1e-4)340 341 342def test_dnds():343 assert_almost_equal(dnds(TEST_SEQ5, TEST_SEQ6), 0.1247, delta=1e-4)344 345# DNDS.pdf - wrong based on email conversation with Sean346#347# def test_syn_sum():348# assert_close(syn_sum(TEST_SEQ1, TEST_SEQ2), 7.5833)349","Python"
350"Codon","adelq/dnds","dnds.py",".py","5340","169","""""""dnds351 352This module is a reference implementation of estimating nucleotide substitution353neutrality by estimating the percent of synonymous and nonsynonymous mutations.354""""""355from __future__ import print_function, division356from math import log357from fractions import Fraction358import logging359from codons import codons360 361BASES = {'A', 'G', 'T', 'C'}362 363 364def split_seq(seq, n=3):365 '''Returns sequence split into chunks of n characters, default is codons'''366 return [seq[i:i + n] for i in range(0, len(seq), n)]367 368 369def average_list(l1, l2):370 """"""Return the average of two lists""""""371 return [(i1 + i2) / 2 for i1, i2 in zip(l1, l2)]372 373 374def dna_to_protein(codon):375 '''Returns single letter amino acid code for given codon'''376 return codons[codon]377 378 379def translate(seq):380 """"""Translate a DNA sequence into the 1-letter amino acid sequence""""""381 return """".join([dna_to_protein(codon) for codon in split_seq(seq)])382 383 384def is_synonymous(codon1, codon2):385 '''Returns boolean whether given codons are synonymous'''386 return dna_to_protein(codon1) == dna_to_protein(codon2)387 388 389def dnds_codon(codon):390 '''Returns list of synonymous counts for a single codon.391 Calculations done per the methodology taught in class.392 http://sites.biology.duke.edu/rausher/DNDS.pdf393 '''394 syn_list = []395 for i in range(len(codon)):396 base = codon[i]397 other_bases = BASES - {base}398 syn = 0399 for new_base in other_bases:400 new_codon = codon[:i] + new_base + codon[i + 1:]401 syn += int(is_synonymous(codon, new_codon))402 syn_list.append(Fraction(syn, 3))403 return syn_list404 405 406def dnds_codon_pair(codon1, codon2):407 """"""Get the dN/dS for the given codon pair""""""408 return average_list(dnds_codon(codon1), dnds_codon(codon2))409 410 411def syn_sum(seq1, seq2):412 """"""Get the sum of synonymous sites from two DNA sequences""""""413 syn = 0414 codon_list1 = split_seq(seq1)415 codon_list2 = split_seq(seq2)416 for i in range(len(codon_list1)):417 codon1 = codon_list1[i]418 codon2 = codon_list2[i]419 dnds_list = dnds_codon_pair(codon1, codon2)420 syn += sum(dnds_list)421 return syn422 423 424def hamming(s1, s2):425 """"""Return the hamming distance between 2 DNA sequences""""""426 return sum(ch1 != ch2 for ch1, ch2 in zip(s1, s2)) + abs(len(s1) - len(s2))427 428 429def codon_subs(codon1, codon2):430 """"""Returns number of synonymous substitutions in provided codon pair431 Methodology for multiple substitutions from Dr. Swanson, UWashington432 https://faculty.washington.edu/wjs18/dnds.ppt433 """"""434 diff = hamming(codon1, codon2)435 if diff < 1:436 return 0437 elif diff == 1:438 return int(translate(codon1) == translate(codon2))439 440 syn = 0441 for i in range(len(codon1)):442 base1 = codon1[i]443 base2 = codon2[i]444 if base1 != base2:445 new_codon = codon1[:i] + base2 + codon1[i + 1:]446 syn += int(is_synonymous(codon1, new_codon))447 syn += int(is_synonymous(codon2, new_codon))448 return syn / diff449 450 451def substitutions(seq1, seq2):452 """"""Returns number of synonymous and nonsynonymous substitutions""""""453 dna_changes = hamming(seq1, seq2)454 codon_list1 = split_seq(seq1)455 codon_list2 = split_seq(seq2)456 syn = 0457 for i in range(len(codon_list1)):458 codon1 = codon_list1[i]459 codon2 = codon_list2[i]460 syn += codon_subs(codon1, codon2)461 return (syn, dna_changes - syn)462 463 464def clean_sequence(seq):465 """"""Clean up provided sequence by removing whitespace.""""""466 return seq.replace(' ', '')467 468 469def pnps(seq1, seq2):470 """"""Main function to calculate pN/pS between two DNA sequences.""""""471 # Strip any whitespace from both strings472 seq1 = clean_sequence(seq1)473 seq2 = clean_sequence(seq2)474 # Check that both sequences have the same length475 assert len(seq1) == len(seq2)476 # Check that sequences are codons477 assert len(seq1) % 3 == 0478 479 syn_sites = syn_sum(seq1, seq2)480 non_sites = len(seq1) - syn_sites481 logging.info('Sites (syn/nonsyn): {}, {}'.format(syn_sites, non_sites))482 syn_subs, non_subs = substitutions(seq1, seq2)483 logging.info('pN: {} / {}\t\tpS: {} / {}'484 .format(non_subs, round(non_sites), syn_subs, round(syn_sites)))485 pn = non_subs / non_sites486 ps = syn_subs / syn_sites487 return pn / ps488 489 490def dnds(seq1, seq2):491 """"""Main function to calculate dN/dS between two DNA sequences per Nei &492 Gojobori 1986. This includes the per site conversion adapted from Jukes &493 Cantor 1967.494 """"""495 # Strip any whitespace from both strings496 seq1 = clean_sequence(seq1)497 seq2 = clean_sequence(seq2)498 # Check that both sequences have the same length499 assert len(seq1) == len(seq2)500 # Check that sequences are codons501 assert len(seq1) % 3 == 0502 503 syn_sites = syn_sum(seq1, seq2)504 non_sites = len(seq1) - syn_sites505 logging.info('Sites (syn/nonsyn): {}, {}'.format(syn_sites, non_sites))506 syn_subs, non_subs = substitutions(seq1, seq2)507 pn = non_subs / non_sites508 ps = syn_subs / syn_sites509 dn = -(3 / 4) * log(1 - (4 * pn / 3))510 ds = -(3 / 4) * log(1 - (4 * ps / 3))511 logging.info('dN: {}\t\tdS: {}'.format(round(dn, 3), round(ds, 3)))512 return dn / ds513 514 515if __name__ == '__main__':516 print(pnps('ACC GTG GGA TGC ACC GGT GTG CCC',517 'ACA GTG AGA TAT AAA GGA GAG AAC'))518","Python"
519"Codon","adelq/dnds","codons.py",".py","1045","67","codons = {520 ""TTT"": ""F"",521 ""TTC"": ""F"",522 ""TTA"": ""L"",523 ""TTG"": ""L"",524 ""TCT"": ""S"",525 ""TCC"": ""S"",526 ""TCA"": ""S"",527 ""TCG"": ""S"",528 ""TAT"": ""Y"",529 ""TAC"": ""Y"",530 ""TAA"": ""STOP"",531 ""TAG"": ""STOP"",532 ""TGT"": ""C"",533 ""TGC"": ""C"",534 ""TGA"": ""STOP"",535 ""TGG"": ""W"",536 ""CTT"": ""L"",537 ""CTC"": ""L"",538 ""CTA"": ""L"",539 ""CTG"": ""L"",540 ""CCT"": ""P"",541 ""CCC"": ""P"",542 ""CCA"": ""P"",543 ""CCG"": ""P"",544 ""CAT"": ""H"",545 ""CAC"": ""H"",546 ""CAA"": ""Q"",547 ""CAG"": ""Q"",548 ""CGT"": ""R"",549 ""CGC"": ""R"",550 ""CGA"": ""R"",551 ""CGG"": ""R"",552 ""ATT"": ""I"",553 ""ATC"": ""I"",554 ""ATA"": ""I"",555 ""ATG"": ""M"",556 ""ACT"": ""T"",557 ""ACC"": ""T"",558 ""ACA"": ""T"",559 ""ACG"": ""T"",560 ""AAT"": ""N"",561 ""AAC"": ""N"",562 ""AAA"": ""K"",563 ""AAG"": ""K"",564 ""AGT"": ""S"",565 ""AGC"": ""S"",566 ""AGA"": ""R"",567 ""AGG"": ""R"",568 ""GTT"": ""V"",569 ""GTC"": ""V"",570 ""GTA"": ""V"",571 ""GTG"": ""V"",572 ""GCT"": ""A"",573 ""GCC"": ""A"",574 ""GCA"": ""A"",575 ""GCG"": ""A"",576 ""GAT"": ""D"",577 ""GAC"": ""D"",578 ""GAA"": ""E"",579 ""GAG"": ""E"",580 ""GGT"": ""G"",581 ""GGC"": ""G"",582 ""GGA"": ""G"",583 ""GGG"": ""G""584}585","Python"
586"Codon","veg/hyphy-analyses","SCUEAL-WG/SCUEALfullG.sh",".sh","294","10","#!/bin/sh587 588#module load openmpi-x86_64589module load openmpi-1.10-x86_64590 591# Run MPI program through Ethernet eth0592 593mpirun -np 28 /localdisk/software/hyphy/hyphy-2.2.4/HYPHYMPI USEPATH=/dev/null BASEPATH=/localdisk/software/hyphy/lib/hyphy /localdisk/home/PATH_TO_FOLDER_/SCUEALfullG.bf594 595","Shell"
596"Codon","veg/hyphy-analyses","molerate/JSON.md",".md","4938","72","# Molerate JSON Schema Description597 598 599The top-level JSON object contains the following keys:600 601* **`analysis`** (Object): Contains metadata about the analysis performed.602 * `authors` (String): Name of the author(s).603 * `contact` (String): Contact email address.604 * `info` (String): Brief description of the analysis.605 * `labeling strategy` (String): The strategy used for labeling branches (e.g., ""all-descendants"").606 * `model` (String): The substitution model used (e.g., ""WAG"").607 * `rate variation` (String): The model used for rate variation across sites (e.g., ""GDD"").608 * `requirements` (String): Description of the input data required.609 * `version` (String): Version of the analysis tool.610 611* **`branch attributes`** (Object): Contains attributes for phylogenetic branches, grouped by partition (e.g., ""0"") and model.612 * `<partition_id>` (Object, e.g., ""0""): Represents a data partition.613 * `<branch_name>` (Object, e.g., ""HLacaPus1"", ""Node1""): Represents a specific branch in the tree.614 * `Proportional` (Number): Estimated branch length for the Proportional model.615 * `Proportional Partitioned` (Number): Estimated branch length for the Proportional Partitioned model.616 * `Reference` (Number): Branch length from the reference tree.617 * `Unconstrained Test` (Number): Estimated branch length for the Unconstrained Test model.618 * *( Potentially other model names as keys with numeric values )*619 * `attributes` (Object): Defines metadata for the models listed under branch names.620 * `<model_name>` (Object, e.g., ""Proportional""):621 * `attribute type` (String): Type of attribute (e.g., ""branch length"").622 * `display order` (Number): Suggested order for displaying this model's results.623 624* **`branch level analysis`** (Object): Contains detailed likelihood ratio test results for individual branches designated as 'test'.625 * `<branch_name>` (Object, e.g., ""HLamaAes1""): Represents a specific 'test' branch.626 * `LogL` (Number): Log-likelihood for the `Proportional+1` model at this branch.627 * `alternative` (Object): `Proportional+1` vs `Unconstrained Test`628 * `Corrected P-value` (Number): Corrected (Holm-Bonferroni) p-value.629 * `LRT` (Number): Likelihood Ratio Test statistic.630 * `p-value` (Number): Uncorrected p-value.631 * `null` (Object): `Proportional` vs `Proportional+1`632 * `Corrected P-value` (Number): orrected (Holm-Bonferroni) p-value.633 * `LRT` (Number): Likelihood Ratio Test statistic.634 * `p-value` (Number): Uncorrected p-value.635 636* **`fits`** (Object): Contains details about the statistical fit of each evolutionary model tested.637 * `<model_name>` (Object, e.g., ""Proportional"", ""Unconstrained Test""):638 * `AIC-c` (Number): Corrected Akaike Information Criterion value.639 * `Log Likelihood` (Number): Log-likelihood value for the model fit.640 * `Rate Distributions` (Object): Parameters related to site rate variation.641 * *( Keys vary based on the rate variation model, e.g., GDD category rates, mixture weights, branch scalers )* (Number)642 * `display order` (Number): Suggested order for displaying this model's fit results.643 * `estimated parameters` (Number): Number of parameters estimated for this model.644 645* **`input`** (Object): Describes the input data used for the analysis.646 * `file name` (String): Path to the input alignment file.647 * `number of sequences` (Number): Count of sequences in the alignment.648 * `number of sites` (Number): Count of sites (columns) in the alignment.649 * `partition count` (Number): Number of data partitions (usually 0 or 1 if not partitioned).650 * `trees` (Object): Contains the input tree structure(s).651 * `<partition_id>` (String, e.g., ""0""): Tree structure in Newick format string.652 653* **`runtime`** (String): Version of `HyPhy` used to execute the analysis.654 655* **`test results`** (Object): Contains results of Likelihood Ratio Tests comparing pairs of nested models.656 * `<model_comparison>` (Object, e.g., ""Proportional Partitioned:Unconstrained Test""): Represents a comparison between two models (Null:Alternative).657 * `Corrected P-value` (Number): Corrected (Holm-Bonferroni) p-value for the comparison.658 * `LRT` (Number): Likelihood Ratio Test statistic.659 * `Uncorrected P-value` (Number): Uncorrected p-value.660 661* **`tested`** (Object): Maps each branch name from the input tree to its classification in the analysis.662 * `<branch_name>` (String, e.g., ""HLacaPus1""): Value is either ""test"" or ""background"".663 664* **`timers`** (Object): Records the time duration for specific parts of the analysis.665 * `<component_name>` (Object, e.g., ""Overall"", ""Proportional""):666 * `order` (Number): Order index for the component.667 * `timer` (Number): Time duration in seconds.","Markdown"
668"Codon","Ensembl/ensembl-compara","setup.py",".py","776","22","# See the NOTICE file distributed with this work for additional information669# regarding copyright ownership.670#671# Licensed under the Apache License, Version 2.0 (the ""License"");672# you may not use this file except in compliance with the License.673# You may obtain a copy of the License at674#675# http://www.apache.org/licenses/LICENSE-2.0676#677# Unless required by applicable law or agreed to in writing, software678# distributed under the License is distributed on an ""AS IS"" BASIS,679# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.680# See the License for the specific language governing permissions and681# limitations under the License.682""""""setuptools-based stub for editable installs""""""683 684from setuptools import setup685 686 687if __name__ == ""__main__"":688 setup()689","Python"
690"Codon","Ensembl/ensembl-compara","pull_request_template.md",".md","521","26","## Description691 692_Describe the problem you're addressing here._693 694**Related JIRA tickets:**695- ENSCOMPARASW-XXXX696 697## Overview of changes698_Give details of what changes were required to solve the problem. Break into sections if applicable._699 700#### Change 1701- _detail 1.1_702 703#### Change 2704- _detail 2.1_705 706## Testing707_How was this tested? Have new unit tests been included?_708 709## Notes710_Optional extra information._711 712---713 714For code reviewers: [code review SOP](https://www.ebi.ac.uk/seqdb/confluence/display/EnsCom/Code+review+SOP)715","Markdown"
716"Codon","Ensembl/ensembl-compara","run_all_tests.sh",".sh","1573","32","#!/bin/bash717 718# See the NOTICE file distributed with this work for additional information719# regarding copyright ownership.720# 721# Licensed under the Apache License, Version 2.0 (the ""License"");722# you may not use this file except in compliance with the License.723# You may obtain a copy of the License at724# 725# http://www.apache.org/licenses/LICENSE-2.0726# 727# Unless required by applicable law or agreed to in writing, software728# distributed under the License is distributed on an ""AS IS"" BASIS,729# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.730# See the License for the specific language governing permissions and731# limitations under the License.732 733set -euxo pipefail734 735cd ""$ENSEMBL_ROOT_DIR""736prove -r ensembl-compara/travisci/all-housekeeping/737prove -r ensembl-compara/travisci/sql-unittest/738ensembl-test/scripts/runtests.pl ensembl-compara/modules/t739ensembl-test/scripts/runtests.pl ensembl/modules/t/compara.t740ensembl-test/scripts/runtests.pl ensembl-rest/t/genomic_alignment.t ensembl-rest/t/info.t ensembl-rest/t/taxonomy.t ensembl-rest/t/homology.t ensembl-rest/t/gene_tree.t ensembl-rest/t/cafe_tree.t ensembl-rest/t/family.t741cd ""$ENSEMBL_ROOT_DIR/ensembl-compara""742./travisci/perl-linter_harness.sh743find docs modules scripts sql travisci -iname '*.t' -print0 | xargs -0 -n 1 perl -c744find docs modules scripts sql travisci -iname '*.pl' -print0 | xargs -0 -n 1 perl -c745find docs modules scripts sql travisci -iname '*.pm' \! -name 'LoadSynonyms.pm' \! -name 'HALAdaptor.pm' \! -name 'HALXS.pm' -print0 | xargs -0 -n 1 perl -c746echo ""All good !""747","Shell"
748"Codon","Ensembl/ensembl-compara","DEPRECATED.md",".md","2457","87","> This file contains the list of methods deprecated in the Ensembl Compara749> API. A method is deprecated when it is not functional any more750> (schema/data change) or has been replaced by a better one. Backwards751> compatibility is provided whenever possible. When a method is752> deprecated, a deprecation warning is thrown whenever the method is used.753> The warning also contains instructions on replacing the deprecated method754> and when it will be removed.755 756----757 758# Deprecated methods scheduled for deletion759 760## Ensembl 116761 762* `Bio::EnsEMBL::Compara::Utils::MasterDatabase::find_overlapping_genome_db_ids`763 764# Deprecated methods not yet scheduled for deletion765 766* `AlignedMember::get_cigar_array()`767* `AlignedMember::get_cigar_breakout()`768* `GenomicAlignTree::genomic_align_array()`769* `GenomicAlignTree::get_all_GenomicAligns()`770* `Homology::dn()`771* `Homology::dnds_ratio()`772* `Homology::ds()`773* `Homology::lnl()`774* `Homology::n()`775* `Homology::s()`776* `Homology::threshold_on_ds()`777* `HomologyAdaptor::update_genetic_distance()`778 779# Methods removed in previous versions of Ensembl780 781## Ensembl 110782 783* `DBSQL::'*MemberAdaptor::fetch_by_stable_id()`784 785## Ensembl 100786 787* `DBSQL::'*MemberAdaptor::get_source_taxon_count()`788 789## Ensembl 98790 791* `DBSQL::DnaFragAdaptor::fetch_all_by_GenomeDB_region()`792 793## Ensembl 96794 795* `AlignSlice::Slice::get_all_VariationFeatures_by_VariationSet`796* `AlignSlice::Slice::get_all_genotyped_VariationFeatures`797* `Taggable::get_value_for_XXX()`798* `Taggable::get_all_values_for_XXX()`799* `Taggable::get_XXX_value()`800 801## Ensembl 93802 803* `DnaFrag::isMT()`804* `DnaFrag::dna_type()`805* `MethodLinkSpeciesSet::species_set_obj()`806 807## Ensembl 92808 809* `Member::print_member()`810* `SyntenyRegion::regions()`811 812## Ensembl 91813 814* `Homology::print_homology()`815* `MethodLinkSpeciesSet::get_common_classification()`816* `NCBITaxon::binomial()`817* `NCBITaxon::ensembl_alias_name()`818* `NCBITaxon::common_name()`819* `DnaFragAdaptor::is_already_stored()`820* `GeneMemberAdaptor::fetch_all_by_source_Iterator()`821* `GeneMemberAdaptor::fetch_all_Iterator()`822* `GeneMemberAdaptor::load_all_from_seq_members()`823* `GenomeDBAdaptor::fetch_all_by_low_coverage()`824* `GenomeDBAdaptor::fetch_all_by_taxon_id_assembly()`825* `GenomeDBAdaptor::fetch_by_taxon_id()`826* `SeqMemberAdaptor::fetch_all_by_source_Iterator()`827* `SeqMemberAdaptor::fetch_all_Iterator()`828* `SeqMemberAdaptor::update_sequence()`829* `SequenceAdaptor::fetch_by_dbIDs()`830 831## Ensembl 89832 833* `SpeciesTree::species_tree()`834","Markdown"
835"Codon","Ensembl/ensembl-compara","conftest.py",".py","3680","88","# See the NOTICE file distributed with this work for additional information836# regarding copyright ownership.837#838# Licensed under the Apache License, Version 2.0 (the ""License"");839# you may not use this file except in compliance with the License.840# You may obtain a copy of the License at841#842# http://www.apache.org/licenses/LICENSE-2.0843#844# Unless required by applicable law or agreed to in writing, software845# distributed under the License is distributed on an ""AS IS"" BASIS,846# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.847# See the License for the specific language governing permissions and848# limitations under the License.849""""""Local directory-specific hook implementations.850 851Since this file is located at the root of all ensembl-compara tests, every test in every subfolder will have852access to the plugins, hooks and fixtures defined here.853 854""""""855# Disable all the redefined-outer-name violations due to how pytest fixtures work856# pylint: disable=redefined-outer-name857 858import os859from pathlib import Path860import shutil861import time862 863import pytest864from _pytest.fixtures import FixtureRequest865 866from ensembl.compara.filesys import DirCmp867 868 869pytest_plugins = (""ensembl.utils.plugin"",)870 871 872def pytest_configure() -> None:873 """"""Adds global variables and configuration attributes required by Compara's unit tests.874 875 `Pytest initialisation hook876 <https://docs.pytest.org/en/latest/reference.html#_pytest.hookspec.pytest_configure>`_.877 878 """"""879 test_data_dir = Path(__file__).parent / 'src' / 'test_data'880 pytest.dbs_dir = test_data_dir / 'databases' # type: ignore[attr-defined]881 pytest.files_dir = test_data_dir / 'flatfiles' # type: ignore[attr-defined]882 883 884@pytest.fixture(scope=""session"")885def dir_cmp(request: FixtureRequest, tmp_path_factory: pytest.TempPathFactory) -> DirCmp:886 """"""Returns a directory tree comparison (:class:`DirCmp`) object.887 888 Requires a dictionary with the following keys:889 890 ref (:obj:`PathLike`): Reference root directory path.891 target (:obj:`PathLike`): Target root directory path.892 893 passed via `request.param`. In both cases, if a relative path is provided, the starting folder will be894 ``src/python/tests/flatfiles``. This fixture is intended to be used via indirect parametrization, for895 example::896 897 @pytest.mark.parametrize(""dir_cmp"", [{'ref': 'citest/reference', 'target': 'citest/target'}],898 indirect=True)899 def test_method(..., dir_cmp: DirCmp, ...):900 901 Args:902 request: Access to the requesting test context.903 tmp_path_factory: Temporary directory path factory fixture.904 905 """"""906 tmp_path = tmp_path_factory.mktemp(""dir_cmp_root"")907 # Get the source and temporary absolute paths for reference and target root directories908 ref = Path(request.param['ref']) # type: ignore[attr-defined]909 ref_src = ref if ref.is_absolute() else pytest.files_dir / ref # type: ignore910 ref_tmp = tmp_path / str(ref).replace(os.path.sep, '_')911 target = Path(request.param['target']) # type: ignore[attr-defined]912 target_src = target if target.is_absolute() else pytest.files_dir / target # type: ignore913 target_tmp = tmp_path / str(target).replace(os.path.sep, '_')914 # Copy directory trees (if they have not been copied already) ignoring file metadata915 if not ref_tmp.exists():916 shutil.copytree(ref_src, ref_tmp, copy_function=shutil.copy)917 # Sleep one second in between to ensure the timestamp differs between reference and target files918 time.sleep(1)919 if not target_tmp.exists():920 shutil.copytree(target_src, target_tmp, copy_function=shutil.copy)921 return DirCmp(ref_tmp, target_tmp)922","Python"
923"Codon","Ensembl/ensembl-compara","modules/Bio/EnsEMBL/Compara/HAL/HALXS/INLINE.h",".h","1732","44","/*924 * See the NOTICE file distributed with this work for additional information925 * regarding copyright ownership.926 * 927 * Licensed under the Apache License, Version 2.0 (the ""License"");928 * you may not use this file except in compliance with the License.929 * You may obtain a copy of the License at930 * 931 * http://www.apache.org/licenses/LICENSE-2.0932 * 933 * Unless required by applicable law or agreed to in writing, software934 * distributed under the License is distributed on an ""AS IS"" BASIS,935 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.936 * See the License for the specific language governing permissions and937 * limitations under the License.938 */939 940#define Inline_Stack_Vars dXSARGS941#define Inline_Stack_Items items942#define Inline_Stack_Item(x) ST(x)943#define Inline_Stack_Reset sp = mark944#define Inline_Stack_Push(x) XPUSHs(x)945#define Inline_Stack_Done PUTBACK946#define Inline_Stack_Return(x) XSRETURN(x)947#define Inline_Stack_Void XSRETURN(0)948 949#define INLINE_STACK_VARS Inline_Stack_Vars950#define INLINE_STACK_ITEMS Inline_Stack_Items951#define INLINE_STACK_ITEM(x) Inline_Stack_Item(x)952#define INLINE_STACK_RESET Inline_Stack_Reset953#define INLINE_STACK_PUSH(x) Inline_Stack_Push(x)954#define INLINE_STACK_DONE Inline_Stack_Done955#define INLINE_STACK_RETURN(x) Inline_Stack_Return(x)956#define INLINE_STACK_VOID Inline_Stack_Void957 958#define inline_stack_vars Inline_Stack_Vars959#define inline_stack_items Inline_Stack_Items960#define inline_stack_item(x) Inline_Stack_Item(x)961#define inline_stack_reset Inline_Stack_Reset962#define inline_stack_push(x) Inline_Stack_Push(x)963#define inline_stack_done Inline_Stack_Done964#define inline_stack_return(x) Inline_Stack_Return(x)965#define inline_stack_void Inline_Stack_Void966","Unknown"
967"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/fix_leaf_names.py",".py","1811","58","#!/usr/bin/env python3968 969# See the NOTICE file distributed with this work for additional information970# regarding copyright ownership.971#972# Licensed under the Apache License, Version 2.0 (the ""License"");973# you may not use this file except in compliance with the License.974# You may obtain a copy of the License at975#976# http://www.apache.org/licenses/LICENSE-2.0977#978# Unless required by applicable law or agreed to in writing, software979# distributed under the License is distributed on an ""AS IS"" BASIS,980# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.981# See the License for the specific language governing permissions and982# limitations under the License.983 984""""""985Replace leaf names in a species tree with production name.986Example:987 $ python fix_leaf_names.py -t astral_species_tree_neutral_bl.nwk -c input_genomes.csv -o test.nwk988""""""989 990import sys991import argparse992from os import path993from Bio import Phylo994import pandas as pd995 996# Parse command line arguments:997parser = argparse.ArgumentParser(998 description='Replace leafs of a species tree with production names.')999parser.add_argument(1000 '-t', metavar='tree', type=str, help=""Input tree."", required=True, default=None)1001parser.add_argument(1002 '-c', metavar='csv', type=str, help=""Input CSV."", required=True, default=None)1003parser.add_argument(1004 '-o', metavar='output', type=str, help=""Output tree."", required=True)1005 1006 1007if __name__ == '__main__':1008 args = parser.parse_args()1009 1010 df = pd.read_csv(args.c, delimiter=""\t"", header=None)1011 trans_map = {}1012 for r in df.itertuples():1013 trans_map[r[10]] = r[2]1014 1015 tree = Phylo.read(args.t, format=""newick"")1016 for leaf in tree.get_terminals():1017 leaf.name = trans_map[leaf.name]1018 1019 Phylo.write(1020 trees=tree,1021 file=args.o,1022 format=""newick""1023 )1024","Python"
1025"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/pick_third_site.py",".py","1845","59","#!/usr/bin/env python31026 1027# See the NOTICE file distributed with this work for additional information1028# regarding copyright ownership.1029#1030# Licensed under the Apache License, Version 2.0 (the ""License"");1031# you may not use this file except in compliance with the License.1032# You may obtain a copy of the License at1033#1034# http://www.apache.org/licenses/LICENSE-2.01035#1036# Unless required by applicable law or agreed to in writing, software1037# distributed under the License is distributed on an ""AS IS"" BASIS,1038# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1039# See the License for the specific language governing permissions and1040# limitations under the License.1041""""""1042Pick out every third site from a codon alignment. The input sequences1043must have a length divisible by 3.1044 1045Example:1046 $ python pick_third_site.py -i input.fas -o output.fas1047""""""1048 1049import sys1050import argparse1051from collections import OrderedDict1052 1053import pandas as pd1054from Bio import SeqIO1055 1056# Parse command line arguments:1057parser = argparse.ArgumentParser(1058 description='Pick out the third sites from a codon alignment.')1059parser.add_argument(1060 '-i', metavar='input', type=str, help=""Input fasta."", required=True)1061parser.add_argument(1062 '-o', metavar='output', type=str, help=""Output fasta."", required=True)1063 1064 1065if __name__ == '__main__':1066 args = parser.parse_args()1067 1068 fh = open(args.o, ""w"")1069 1070 for record in SeqIO.parse(args.i, ""fasta""):1071 # Check if sequence length is divisible by 3:1072 if len(record.seq) % 3 != 0:1073 sys.stderr.write(f""The length of sequence {record.id} is not divisible by 3!\n"")1074 sys.exit(1)1075 # Pick out every third site:1076 print(len(record.seq))1077 record.seq = record.seq[2::3]1078 # Write out record:1079 SeqIO.write(record, fh, ""fasta"")1080 1081 fh.flush()1082 fh.close()1083","Python"
1084"Codon","Ensembl/ensembl-compara","pipelines/SpeciesTreeFromBusco/scripts/fetch_genomes_from_db.py",".py","4595","120","#!/usr/bin/env python31085 1086# See the NOTICE file distributed with this work for additional information1087# regarding copyright ownership.1088#1089# Licensed under the Apache License, Version 2.0 (the ""License"");1090# you may not use this file except in compliance with the License.1091# You may obtain a copy of the License at1092#1093# http://www.apache.org/licenses/LICENSE-2.01094#1095# Unless required by applicable law or agreed to in writing, software1096# distributed under the License is distributed on an ""AS IS"" BASIS,1097# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.1098# See the License for the specific language governing permissions and1099# limitations under the License.1100 1101""""""1102Locate genome dumps of a collection or a species set on disk.1103Example:1104 $ python fetch_genomes_from_db.py -u mysql://ensro@mysql-ens-compara-prod-1:4485/ensembl_compara_master -c pig_breeds -o test.tsv -d /hps/nobackup/flicek/ensembl/compara/shared/genome_dumps/vertebrates1105""""""1106 1107import sys1108import argparse1109import csv1110from os import path1111from sqlalchemy.engine.row import Row1112from sqlalchemy import create_engine, text1113 1114# Parse command line arguments:1115parser = argparse.ArgumentParser(1116 description='Locate genomes of a species set/collection in the dump directory.')1117parser.add_argument('-u', metavar='db_URL', type=str, help=""Compara master db URL."", required=True)1118parser.add_argument(1119 '-d', metavar='dump_dir', type=str, help=""Genome dump directory."", required=True)1120parser.add_argument(1121 '-s', metavar='ssid', type=str, help=""Species set id."", required=False, default=None)1122parser.add_argument(1123 '-c', metavar='cid', type=str, help=""Collection id."", required=False, default=None)1124parser.add_argument(1125 '-o', metavar='output', type=str, help=""Output CSV."", required=True)1126 1127 1128def _dir_revhash(gid: int) -> str:1129 """"""Build directory hash from genome db id.""""""1130 dir_hash = list(reversed(str(gid)))1131 dir_hash.pop()1132 return path.join(*dir_hash) if dir_hash else path.curdir1133 1134 1135def _build_dump_path(row: Row) -> str:1136 gcomp = """"1137 if row.genome_component is not None:1138 gcomp = f""comp{row.genome_component}.""1139 gpath = path.join(args.d, _dir_revhash(row.genome_db_id), f""{row.name}.{row.assembly}.{gcomp}soft.fa"")1140 return gpath1141 1142 1143if __name__ == '__main__':1144 args = parser.parse_args()1145 1146 db_url = args.u1147 ss_id = args.s1148 if ss_id == """":1149 ss_id = None1150 coll_id = args.c1151 if coll_id == """":1152 coll_id = None1153 engine = create_engine(db_url, future=True)1154 1155 if ss_id is None and coll_id is None:1156 sys.stderr.write(""Either a collection name or a species set id must be specified!\n"")1157 sys.exit(1)1158 1159 if ss_id is not None and coll_id is not None:1160 sys.stderr.write(""Specify either a collection name or a species set id!\n"")1161 sys.exit(1)1162 1163 ss_query = f""""""1164 SELECT genome_db_id, genome_db.name, assembly, genebuild,1165 strain_name, display_name, genome_component1166 FROM species_set JOIN genome_db USING(genome_db_id)1167 WHERE species_set_id = {ss_id};1168 """"""1169 c_query = f""""""1170 SELECT genome_db_id, genome_db.name, assembly, genebuild,1171 strain_name, display_name, genome_component1172 FROM species_set_header JOIN species_set USING(species_set_id)1173 JOIN genome_db USING(genome_db_id) WHERE species_set_header.name='{coll_id}'1174 AND species_set_header.last_release is NULL1175 AND species_set_header.first_release IS NOT NULL;1176 """"""1177 query = ss_query1178 if coll_id is not None:1179 query = c_query1180 1181 missing_fas = []1182 with engine.connect() as conn, open(args.o, ""w"") as ofh:1183 result = conn.execute(text(query))1184 writer = csv.writer(ofh, delimiter=""\t"", lineterminator=""\n"")1185 for row in result:1186 gpath = _build_dump_path(row)1187 fa_name = str(row.name) + ""_"" + str(row.assembly)1188 prod_name = row.name1189 if row.genome_component is not None:1190 fa_name += f""_{row.genome_component}""1191 prod_name += f""_{row.genome_component}""1192 dump_exists = path.exists(gpath)1193 if not dump_exists:1194 missing_fas.append(gpath)1195 writer.writerow([row.genome_db_id, prod_name, row.assembly, row.genebuild, row.strain_name,1196 row.display_name, row.genome_component, gpath, dump_exists, fa_name])1197 1198 if len(missing_fas) > 0:1199 sys.stderr.write(""Fatal error! The following genome dumps are missing:\n"")1200 for p in missing_fas: