#This is a piece of sample code for iterating over the directory structure in
#  the top 2018 dataset
#You will have to specify correct paths for your machine.

import os

top_dir = 'top2018_cifs_mc_filtered_hom70'
#path to top level dir of cif storage

chain_list = open("top2018_chains_hom70_mcfilter_60pct_complete.txt")
#each line is one chain to look up
#19HC_A
#1A2Z_C
#1A4I_B

for line in chain_list:
  full = line.strip()
  pdb = line[0:4]
  short = line[0:2]
  pathlist = [top_dir,short,pdb,full+'_pruned_mc.cif']
  filepath = os.path.join(*pathlist)
  #e.g. 'top2018_cifs_mc_filtered_hom70/1a/1a2z/1a2z_C_pruned_mc.cif'
