compara-deep-learning/pfam_parser.py

54 lines
1.5 KiB
Python
Raw Permalink Normal View History

2021-04-08 08:41:16 -07:00
import pandas as pd
2019-08-29 08:45:54 -07:00
import progressbar
import gc
import sys
2021-04-08 08:41:16 -07:00
2019-08-29 08:45:54 -07:00
def pfam_parse(filename):
2021-04-08 08:41:16 -07:00
rlist = []
2019-08-29 08:45:54 -07:00
try:
with open(filename) as file:
2021-04-08 08:41:16 -07:00
file.seek(0, 0)
2019-08-29 08:45:54 -07:00
for line in progressbar.progressbar(file):
if line.startswith("#"):
continue
2021-04-08 08:41:16 -07:00
temp_dict = {}
x = [y for y in line.split(" ") if y != '']
if x[9] != "1":
2019-08-29 08:45:54 -07:00
continue
2021-04-08 08:41:16 -07:00
temp_dict["gene_stable_id"] = x[3]
temp_dict["accession"] = x[1]
temp_dict["tlen"] = x[2]
temp_dict["qlen"] = x[5]
temp_dict["domain"] = x[0]
temp_dict["hmm_from"] = x[15]
temp_dict["hmm_to"] = x[16]
temp_dict["ali_from"] = x[17]
temp_dict["ali_to"] = x[18]
temp_dict["env_from"] = x[19]
temp_dict["env_to"] = x[20]
2019-08-29 08:45:54 -07:00
rlist.append(temp_dict)
except Exception as e:
print(e)
return pd.DataFrame()
print(len(rlist))
2021-04-08 08:41:16 -07:00
tdf = pd.DataFrame(rlist)
2019-08-29 08:45:54 -07:00
return tdf
2021-04-08 08:41:16 -07:00
2019-08-29 08:45:54 -07:00
def main():
2021-04-08 08:41:16 -07:00
arg = sys.argv
fname_1 = arg[-2]
fname_2 = arg[-1]
df = pfam_parse(fname_1)
df.to_hdf("pfam_db_positive.h5", key="pfam_db_positive", mode="w")
df = ""
2019-08-29 08:45:54 -07:00
gc.collect()
2021-04-08 08:41:16 -07:00
df = pfam_parse(fname_2)
df.to_hdf("pfam_db_negative.h5", key="pfam_db_negative", mode="w")
2019-08-29 08:45:54 -07:00
print("Pfam Databases Written Successfully :)")
2021-04-08 08:41:16 -07:00
2019-08-29 08:45:54 -07:00
if __name__ == "__main__":
main()