diff --git a/.gitignore b/.gitignore index cc6f93b..a01cc10 100644 --- a/.gitignore +++ b/.gitignore @@ -12,3 +12,4 @@ mouse.protein.faa.psq petMar_lamp0.fasta petMar_lamp3.fasta petMar_mrna.lamp3.orig.fasta.gz +/.project diff --git a/annotate-seqs.py b/annotate-seqs.py index 848ff03..f7f6e28 100644 --- a/annotate-seqs.py +++ b/annotate-seqs.py @@ -37,7 +37,7 @@ def main(): o = ortho.get(name) if o: - annot = namedb.mouse_names.get(transform_name(o, args.ncbi)) + annot = namedb.data_names.get(transform_name(o, args.ncbi)) tr_dict[tr] = ('ortho', annot) else: if tr in tr_dict and tr_dict[tr][0] == 'ortho': @@ -52,7 +52,7 @@ def main(): h, score = h[0] score = round(float(score) / float(len(record.sequence)) * 100) - annot = namedb.mouse_names[transform_name(h, args.ncbi)] + annot = namedb.data_names[transform_name(h, args.ncbi)] if score > oldscore: tr_dict[tr] = (oldscore, annot) @@ -76,7 +76,7 @@ def main(): o = ortho.get(name) if o: - annot = namedb.mouse_names.get(transform_name(o, args.ncbi)) + annot = namedb.data_names.get(transform_name(o, args.ncbi)) annot = "ortho:" + annot annot_ortho_count += 1 else: @@ -85,7 +85,7 @@ def main(): if h: h, score = h[0] score = round(float(score) / float(len(record.sequence)) * 100) - annot = namedb.mouse_names[transform_name(h, args.ncbi)] + annot = namedb.data_names[transform_name(h, args.ncbi)] annot = "h=%d%% => " % score + annot annot += " " annot_homol_count += 1 diff --git a/make-namedb.py b/make-namedb.py index d6eb5f0..2f37254 100644 --- a/make-namedb.py +++ b/make-namedb.py @@ -4,11 +4,13 @@ import screed import sys -outfile = sys.argv[2] +# the name-db +outfile = "names.db" +seqFile = sys.argv[1] d = {} e = {} -for record in screed.open(sys.argv[1]): +for record in screed.open(seqFile): if record.name.startswith('gi|'): ident = record.name.split('|')[3] else: @@ -17,7 +19,8 @@ e[ident] = record.name fp = open(outfile, 'w') -dump(d, fp) +dump(seqFile, fp) +dump(d,fp) -fp = open(outfile + '.fullname', 'w') -dump(e, fp) +fp = open('fullnames.db', 'w') +dump(e, fp) \ No newline at end of file diff --git a/namedb.py b/namedb.py index 4b84d33..f5ec688 100644 --- a/namedb.py +++ b/namedb.py @@ -1,6 +1,6 @@ import cPickle import screed -mouse_names = cPickle.load(open('mouse.namedb')) -mouse_fullname = cPickle.load(open('mouse.namedb.fullname')) -mouse_seqs = screed.ScreedDB('mouse.protein.faa') +data_names = cPickle.load('names.db') +data_fullname = cPickle.load(open('fullnames.db')) +data_seqs = screed.ScreedDB(cPickle.load('names.db'))