2021-04-08 08:41:16 -07:00
|
|
|
import pandas as pd
|
2019-08-12 05:31:37 -07:00
|
|
|
import sys
|
|
|
|
|
import progressbar
|
|
|
|
|
import os
|
|
|
|
|
from neighbor_genes import read_genome_maps
|
|
|
|
|
from process_data import create_data_homology_ls
|
|
|
|
|
from read_get_gene_seq import read_gene_sequences
|
2021-04-08 08:41:16 -07:00
|
|
|
from access_data_rest import update_rest_protein
|
2019-08-12 05:31:37 -07:00
|
|
|
from prepare_synteny_matrix import write_fasta
|
2021-04-08 08:41:16 -07:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def read_database(fname, dirname):
|
|
|
|
|
df = pd.read_csv(dirname+"/"+fname, sep="\t", header=None)
|
|
|
|
|
label_dict = dict(ortholog_one2one=1,
|
|
|
|
|
other_paralog=0,
|
|
|
|
|
non_homolog=2,
|
|
|
|
|
ortholog_one2many=1,
|
|
|
|
|
ortholog_many2many=1,
|
|
|
|
|
within_species_paralog=0,
|
|
|
|
|
gene_split=4)
|
|
|
|
|
label = []
|
|
|
|
|
for _, row in df.iterrows():
|
2019-08-12 05:31:37 -07:00
|
|
|
label.append(label_dict[row[7]])
|
2021-04-08 08:41:16 -07:00
|
|
|
df = df.assign(label=label)
|
|
|
|
|
df = df.drop(7, axis=1)
|
|
|
|
|
df = df.drop(0, axis=1)
|
|
|
|
|
df.columns = ["gene_stable_id", "species", "homology_gene_stable_id",
|
|
|
|
|
"homology_species", "goc", "wga", "label"]
|
2019-08-12 05:31:37 -07:00
|
|
|
return df
|
|
|
|
|
|
2021-04-08 08:41:16 -07:00
|
|
|
|
2019-08-12 05:31:37 -07:00
|
|
|
def read_prediction_file_folder(dir_name):
|
2021-04-08 08:41:16 -07:00
|
|
|
lf = os.listdir(dir_name)
|
|
|
|
|
a_h = []
|
|
|
|
|
d_h = []
|
2019-08-12 05:31:37 -07:00
|
|
|
for x in progressbar.progressbar(lf):
|
2021-04-08 08:41:16 -07:00
|
|
|
df = read_database(x, dir_name)
|
2019-08-12 05:31:37 -07:00
|
|
|
a_h.append(df)
|
|
|
|
|
d_h.append(x.split(".")[0])
|
2021-04-08 08:41:16 -07:00
|
|
|
return a_h, d_h
|
|
|
|
|
|
2019-08-12 05:31:37 -07:00
|
|
|
|
2021-04-08 08:41:16 -07:00
|
|
|
def create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name):
|
|
|
|
|
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
|
|
|
|
protein_sequences = read_gene_sequences(
|
|
|
|
|
a_h, lsy, "pro_seq", "prediction_"+name)
|
|
|
|
|
protein_sequences = update_rest_protein(
|
|
|
|
|
protein_sequences, "prediction_"+name)
|
|
|
|
|
write_fasta(protein_sequences, "prediction_"+name)
|
2019-08-12 05:31:37 -07:00
|
|
|
print("Protein Sequences Loaded")
|
|
|
|
|
|
2021-04-08 08:41:16 -07:00
|
|
|
|
2019-08-12 05:31:37 -07:00
|
|
|
def main():
|
2021-04-08 08:41:16 -07:00
|
|
|
arg = sys.argv
|
|
|
|
|
dirname = arg[-1]
|
|
|
|
|
a_h, d_h = read_prediction_file_folder(dirname)
|
|
|
|
|
n = 3
|
2019-08-12 05:31:37 -07:00
|
|
|
|
2021-04-08 08:41:16 -07:00
|
|
|
a, d, ld, ldg, cmap, cimap = read_genome_maps() # read the genome maps
|
2019-08-12 05:31:37 -07:00
|
|
|
print("Genome Maps Loaded.")
|
|
|
|
|
|
2021-04-08 08:41:16 -07:00
|
|
|
create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, dirname)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
main()
|