mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
86 lines
2.2 KiB
Python
86 lines
2.2 KiB
Python
|
|
from ftplib import FTP
|
||
|
|
import progressbar
|
||
|
|
import os
|
||
|
|
import sys
|
||
|
|
import urllib.request as urllib
|
||
|
|
import requests
|
||
|
|
|
||
|
|
def download_data(x,dir_name):
|
||
|
|
fname=x.split("/")[-1]
|
||
|
|
path=os.path.join(dir_name,fname)
|
||
|
|
urllib.urlretrieve(x,path)
|
||
|
|
|
||
|
|
def get_data_file(file,dir):
|
||
|
|
if not os.path.isfile(file):
|
||
|
|
print("The specified file does not exist!!!")
|
||
|
|
sys.exit(1)
|
||
|
|
|
||
|
|
with open(file,"r")as f:
|
||
|
|
lf=f.read().splitlines()
|
||
|
|
|
||
|
|
if not os.path.exists(dir):
|
||
|
|
os.mkdir(dir)
|
||
|
|
for x in progressbar.progressbar(lf):
|
||
|
|
download_data(x,dir)
|
||
|
|
|
||
|
|
|
||
|
|
|
||
|
|
#This wil download all the fasta files for the coding sequences. To change the directory, change the argument in the get_data_file argument.
|
||
|
|
host ="ftp.ensembl.org"
|
||
|
|
user = "anonymous"
|
||
|
|
password = ""
|
||
|
|
|
||
|
|
print("Connecting to {}".format(host))
|
||
|
|
ftp = FTP(host)
|
||
|
|
ftp.login(user, password)
|
||
|
|
print("Connected to {}".format(host))
|
||
|
|
base_link="ftp://ftp.ensembl.org"
|
||
|
|
#find sequences of all the cds files
|
||
|
|
l=ftp.nlst("/pub/release-96/fasta")
|
||
|
|
lt=[]
|
||
|
|
for x in l:
|
||
|
|
y=ftp.nlst(x+"/cds")
|
||
|
|
for z in y:
|
||
|
|
if z.endswith(".cds.all.fa.gz"):
|
||
|
|
lt.append(z)
|
||
|
|
|
||
|
|
with open("seq_link.txt","w") as file:
|
||
|
|
for x in lt:
|
||
|
|
file.write(base_link+x)
|
||
|
|
file.write("\n")
|
||
|
|
|
||
|
|
#find all the files with protein sequences
|
||
|
|
l=ftp.nlst("/pub/release-96/fasta")
|
||
|
|
lt=[]
|
||
|
|
for x in progressbar.progressbar(l):
|
||
|
|
y=ftp.nlst(x+"/pep")
|
||
|
|
for z in y:
|
||
|
|
if z.endswith(".pep.all.fa.gz"):
|
||
|
|
lt.append(z)
|
||
|
|
|
||
|
|
with open("protein_seq.txt","w") as file:
|
||
|
|
for x in lt:
|
||
|
|
file.write(base_link+x)
|
||
|
|
file.write("\n")
|
||
|
|
#get link of all the gtf files
|
||
|
|
l=ftp.nlst("/pub/release-96/gtf")
|
||
|
|
lt=[]
|
||
|
|
for x in l:
|
||
|
|
y=ftp.nlst(x)
|
||
|
|
for z in y:
|
||
|
|
if z.endswith(".96.gtf.gz"):
|
||
|
|
lt.append(z)
|
||
|
|
|
||
|
|
with open("gtf_link.txt","w") as file:
|
||
|
|
for x in lt:
|
||
|
|
file.write(base_link+x)
|
||
|
|
file.write("\n")
|
||
|
|
|
||
|
|
ch=input("Do you want to download the data?[y/n]")
|
||
|
|
if ch=='y':
|
||
|
|
print("Downloading Data.................")
|
||
|
|
get_data_file("gtf_link.txt","data")
|
||
|
|
get_data_file("seq_link.txt","geneseq")
|
||
|
|
get_data_file("protein_seq.txt","pro_seq")
|
||
|
|
print("Download Complete.................")
|