compara-deep-learning/ftpg.py

85 lines
2.2 KiB
Python
Raw Normal View History

2019-08-11 23:10:45 -07:00
from ftplib import FTP
import progressbar
import os
import sys
import urllib.request as urllib
2021-04-08 08:41:16 -07:00
def download_data(x, dir_name):
fname = x.split("/")[-1]
path = os.path.join(dir_name, fname)
urllib.urlretrieve(x, path)
def get_data_file(file, dir):
2019-08-11 23:10:45 -07:00
if not os.path.isfile(file):
print("The specified file does not exist!!!")
sys.exit(1)
2021-04-08 08:41:16 -07:00
with open(file, "r")as f:
lf = f.read().splitlines()
2019-08-11 23:10:45 -07:00
if not os.path.exists(dir):
os.mkdir(dir)
for x in progressbar.progressbar(lf):
2021-04-08 08:41:16 -07:00
download_data(x, dir)
2019-08-11 23:10:45 -07:00
2021-04-08 08:41:16 -07:00
# This wil download all the fasta files for the coding sequences.
# To change the directory, change the argument in the get_data_file argument.
host = "ftp.ensembl.org"
2019-08-11 23:10:45 -07:00
user = "anonymous"
password = ""
print("Connecting to {}".format(host))
ftp = FTP(host)
ftp.login(user, password)
print("Connected to {}".format(host))
2021-04-08 08:41:16 -07:00
base_link = "ftp://ftp.ensembl.org"
# find sequences of all the cds files
list_of_files = ftp.nlst("/pub/release-96/fasta")
lt = []
for x in list_of_files:
y = ftp.nlst(x+"/cds")
2019-08-11 23:10:45 -07:00
for z in y:
if z.endswith(".cds.all.fa.gz"):
lt.append(z)
2021-04-08 08:41:16 -07:00
with open("seq_link.txt", "w") as file:
2019-08-11 23:10:45 -07:00
for x in lt:
file.write(base_link+x)
file.write("\n")
2021-04-08 08:41:16 -07:00
# find all the files with protein sequences
list_of_files = ftp.nlst("/pub/release-96/fasta")
lt = []
for x in progressbar.progressbar(list_of_files):
y = ftp.nlst(x+"/pep")
2019-08-11 23:10:45 -07:00
for z in y:
if z.endswith(".pep.all.fa.gz"):
lt.append(z)
2021-04-08 08:41:16 -07:00
with open("protein_seq.txt", "w") as file:
2019-08-11 23:10:45 -07:00
for x in lt:
file.write(base_link+x)
file.write("\n")
2021-04-08 08:41:16 -07:00
# get link of all the gtf files
list_of_files = ftp.nlst("/pub/release-96/gtf")
lt = []
for x in list_of_files:
y = ftp.nlst(x)
2019-08-11 23:10:45 -07:00
for z in y:
if z.endswith(".96.gtf.gz"):
lt.append(z)
2021-04-08 08:41:16 -07:00
with open("gtf_link.txt", "w") as file:
2019-08-11 23:10:45 -07:00
for x in lt:
file.write(base_link+x)
file.write("\n")
2019-08-14 08:08:02 -07:00
print("Downloading Data")
2021-04-08 08:41:16 -07:00
get_data_file("gtf_link.txt", "data")
get_data_file("seq_link.txt", "geneseq")
get_data_file("protein_seq.txt", "pro_seq")
2019-08-14 08:08:02 -07:00
print("Download Complete.................")