=item Daphnia pulex tests 2007 june 16, tandy v.5q melon:/bio/bio-grid/mb/EVidenceModeler/daphd scaffold_1 size=4193030 tandy HSP=876 ; match=202 ; region=114 scaffold_2 size=3740169 tandy HSP=1039 ; match=300 ; region=188 scaffold_3 size=3777634 tandy HSP=1255 ; match=320 ; region=170 scaffold_4 size=3075709 tandy HSP=1894 ; match=501 ; region=249 scaffold_5 size=2511979 tandy HSP=1012 ; match=180 ; region=81 scaffold_6 size=2406117 tandy HSP=1576 ; match=390 ; region=189 scaffold_7 size=2324446 tandy HSP=1016 ; match=260 ; region=124 scaffold_8 size=2335496 tandy HSP=705 ; match=167 ; region=92 scaffold_9 size=2251199 tandy HSP=1345 ; match=389 ; region=222 26 MB total Gene counts cat scaffold_?/dpulex1_predict.gff | grep mRNA | perl -ne'($r,$s,$t)=split; print "$s\n" if($t eq "mRNA");' | sort | uniq -c 11196 DGIL_SNO << this is 2x data stutter; 5598 is right 3315 JGI 4332 NCBI_GNO 2381 twinscan ?? drop these use 4332 as median gene count # genebest for Dpulex scaffold 1-9 # gene bestids scores/method (num is all ids x all gene matches, ~ 2/match) predictors num score lost tandem near equal DP_DGIL_SNO_ 4394 64.36 6.89 0.60 0.20 0.31 Dappu 1821 63.55 2.56 0.58 0.24 0.31 NCBI_GNO_ 3333 63.38 2.88 0.77 0.24 0.28 # geneinfo for Dpulex scaffold 1-9 # gene info, all-methods scores/method (count is all gene matches) predictor count eqfar eqnear eqsame fullgns gnids maxgns methods nexons tandkno tandnew tandems DP_DGIL_SNO_ 2296 1.00 1.22 3.32 1.51 3.89 2.11 1.81 4.33 371 103 0.44 Dappu 945 1.31 2.12 5.75 3.17 6.90 2.99 2.71 6.85 286 48 0.73 NCBI_GNO_ 1411 1.14 1.74 4.56 2.33 5.56 2.65 2.34 5.63 331 86 0.64 all 2709 0.96 1.11 2.95 1.34 3.52 1.99 1.72 3.96 383 121 0.42 # ............................................................. Using 501 for all known/new tandems, and 4332 gene predictions for these 9 scaffolds gives tandem rate of ~ 12% (is this calc right? known tandems are included in base count of 4000, but not 'new' tandem models)