"""
Ancient DNA tools - Library.
"""
from numpy.random import default_rng
from termcolor import colored
from datetime import date
from random import choice
from pathlib import Path
import itertools as it
import numpy as np
import warnings
import logging
import shutil
import math
import sys
import ast
import re
import os

__version__ = '2021.3.22'


def clean_snp(snp: str, div_snp="|") -> str:
    """
    Fix the format of the given SNP to the expected syntax. Remove unused information following the suffix ":" and
    replace the separator if necessary (to the standard considered within the code, i.e., "|"). This "cleaning" step
    has been added to be sure the input SNPs are conformed to the expected syntax, i.e., "0", "1", "0|0", "0|1", "1|0",
    or "1|1" (thus, be always careful to consider the type of VCF file you are using)
    :param str snp: The SNP to be checked
    :param str, optional div_snp: The expected separator
    :return: The cleaned SNP according to the expected syntax"""

    if div_snp == "|":
        alt_snp = "/"
    else:
        alt_snp = "|"

    if len(snp) > 1:
        if (div_snp not in snp) and (alt_snp not in snp):  # assumes single value with additional info, e.g., "1:blah"
            values = snp.split(":")
            c_snp = values[0]
        else:
            if div_snp not in snp:
                div_snp = alt_snp

            values = snp.split(":")
            values = values[0].split(div_snp)
            c_snp = "{}|{}".format(values[0], values[1])
    else:
        c_snp = snp

    return c_snp


def match_snp(snp: str, div_snp="|") -> int:
    """
    Convert the diallelic or haploidized SNP, with the genotype coding system of 0, 1, 2.
    Recognized values are the following: 0|0, 0|1, 1|0, 0|0 or 0 and 1 (converted respectively to 0|0 and 1|1).
    The separator could either be "|", i.e., phased, or "/", i.e., non-phased
    :param str snp: The SNP to be converted
    :param str, optional div_snp: Standard SNP divider (default, phased "|")
    :return: int: The genotype as coding system of 0, 1, 2"""

    # match the SNP value with its code
    if snp == "0" or snp == "1":
        code = match_snp("{0}{1}{0}".format(snp, div_snp), div_snp)
    elif snp == "0{}0".format(div_snp):
        code = 0
    elif snp == "1{}1".format(div_snp):
        code = 2
    else:
        code = 1

    return code


def update_asc(asc: list, ind_values: list, maf, is_haploid=False) -> list:
    """Update ASC calculation with haploid awareness"""
    upd_asc = asc.copy()
    f = float(maf)
    maf = min(f, 1-f)  # Ensure we're using minor allele frequency

    codes = []
    for v in ind_values:
        cleaned = clean_snp(v)
        if is_haploid:
            # Haploid: convert to 0/1 (0|0 becomes 0, 1|1 becomes 1)
            if cleaned in ["0", "0|0"]:
                codes.append(0)
            elif cleaned in ["1", "1|1"]:
                codes.append(1)
            else:
                codes.append(-1)  # Invalid for haploid
        else:
            # Diploid: standard 0/1/2 coding
            codes.append(match_snp(cleaned))

    n = len(codes)
    idx = 0
    for i in range(n - 1):
        for j in range(i + 1, n):
            if codes[i] == -1 or codes[j] == -1:
                continue  # Skip invalid genotypes

            if is_haploid:
                # Haploid formula: (x-f)(y-f)/[f(1-f)]
                coeff = ((codes[i] - f) * (codes[j] - f)) / (maf * (1-maf))
            else:
                # Diploid formula: (x-2f)(y-2f)/[2f(1-f)]
                coeff = ((codes[i] - 2*f) * (codes[j] - 2*f)) / (2 * maf * (1-maf))

            upd_asc[idx] += coeff
            idx += 1

    return upd_asc



def split_string(txt: str, length: int) -> [str]:
    """
    Distribute a given text string over multiple lines considering the specified line length
    :param str txt: The text to split
    :param int length: The length of the line
    :return: [str]: The list containing the text divided over multiple lines (if necessary)"""

    new_txt = [txt]
    if len(new_txt[0]) > length:
        tmp = txt.split()
        words = []
        for word in tmp:
            if len(word) <= length:
                words.append(word)
            else:
                w_start = 0
                w_end = length - 3
                words.append("{}...".format(word[w_start:w_end]))
                while len(word[w_end:]) > length - 3:
                    w_start = w_end
                    w_end = w_start + length - 6
                    words.append("...{}...".format(word[w_start:w_end]))
                words.append("...{}".format(word[w_end:]))
        idx = 1
        new_txt = []
        new_line = words[0]
        while idx < len(words):
            if len(new_line + " " + words[idx]) <= length:
                new_line = new_line + " " + words[idx]
            else:
                new_txt.append(new_line)
                new_line = words[idx]
            idx = idx + 1
        if len(new_line) > 0:
            new_txt.append(new_line)

    return new_txt




def print_progress_bar(n, max_n, txt="", prefix="", size=10):
    """
    Print a progress bar
    :param Number n: Current position
    :param Number max_n: Maximum value for position
    :param str, optional txt: Additional text to print (on the right side). By default none, i.e., empty string
    :param str, optional prefix: Additional text to print prior to the progress bar (on the left side). By default none,
    i.e., empty string
    :param int, optional size: Size in characters of the progress bar (default 10)"""

    prg = n / max_n
    sys.stdout.write('\r')
    sys.stdout.write(prefix + f"[{'=' * int(size * prg):{size}s}] {int(100 * prg)}%  {txt}")
    sys.stdout.flush()


def extract_freq(info: str, tag: str) -> float:
    """
    Given a string containing a series of tag, it extracts the frequency tag returning its floating value
    :param str info: The string containing one or more tags (each separated by ';')
    :param str tag: The TAG that defines the frequency value
    :return: float: The frequency value found in the info string"""

    idx = info.find(tag)
    val = float(-1)
    if not idx == -1:  # if TAG not present idx is -1
        start = idx + len(tag) + 1
        end = info[start:].find(";")

        # Extract the relevant string for the value
        if end == -1:
            val = info[start:]
        else:
            val = info[start: start + end]

        if "," in val:
            print("Possible multiple assignment for 'AFngsrelate': {}. Selecting the first value.".format(val))
            val = val.split(",")[0]

        try:
            val = float(val)  # convert the string to number
        except ValueError:
            msg = "Unrecognized freq. value: {}Assigning -1.".format(val)
            raise ValueError(msg)

    return val


# noinspection PyUnboundLocalVariable,PyTypeChecker
def vcf2asc(vcf_file: str, tag="AFngsrelate", x_chr="X", chr_idx=0):
    """ASC-only version"""
    """
    Compute the allele sharing coefficient between samples of a given VCF file.
    Results are stored in an ASC file (text file) having the same name of the input VCF file
    :param str vcf_file: Reference VCF or minor allele frequency file
    :param str, optional tag: MAFs are stored  within  the reference  VCF file,  consider the given TAG  (default
    'AFngsrelate')
    :param optional x_chr: Label associated with the X chromosome (default 'X')
    :param int, optional chr_idx: Column  index for chromosome  details (default 0)"""

    maf_idx = 7  # TAG column
    asc_file = vcf_file.replace(".vcf", ".asc")  # output file
    vcf_in = open(vcf_file, "r")  # input file
    for line in vcf_in:
        if line.startswith("#CHROM"):
            values = line.split()
            ids = values[9:]

            # initialize allele sharing coefficient lists and SNPs count
            n = len(ids)
            asc_autosome = [0] * int(n * (n - 1) / 2)
            n_autosome = 0
            asc_x = [0] * int(n * (n - 1) / 2)
            n_x = 0
        elif not line.startswith("#"):
            values = line.split()
            maf = extract_freq(info=values[maf_idx], tag=tag)
            ind_values = values[9:]
            if 0 < float(maf) < 1:
                if str(values[chr_idx]) == str(x_chr):
                    asc_x = update_asc(asc_x, ind_values, maf)
                    n_x = n_x + 1
                else:
                    asc_autosome = update_asc(asc_autosome, ind_values, maf)
                    n_autosome = n_autosome + 1
    vcf_in.close()

    for idx in range(len(asc_autosome)):
        if n_autosome > 0:
            asc_autosome[idx] = asc_autosome[idx] / n_autosome
        else:
            asc_autosome[idx] = '-'
        if n_x > 0:
            asc_x[idx] = asc_x[idx] / n_x
        else:
            asc_x[idx] = '-'

    n_ids = len(ids)
    asc_idx = 0
    asc_out = open(asc_file, "w")
    asc_out.write("ID 1\tID 2\tASC autosome\tASC X chromosome\n")
    for idx_1 in range(n_ids - 1):
        for idx_2 in range(idx_1 + 1, n):
            asc_out.write("{}\t{}\t{}\t{}\n".format(ids[idx_1], ids[idx_2], asc_autosome[asc_idx], asc_x[asc_idx]))
            asc_idx = asc_idx + 1
    asc_out.close()


# noinspection PyUnboundLocalVariable,PyTypeChecker
def kinship_windowed(g_data: list, win_len: int, win_shift: int, phased=False, is_haploid=False) -> list:
    """Modified to auto-detect haploid data and remove MSM"""
    """
    Estimate the allele sharing coefficient and mismatch coefficient over a shifting window for a given set of SNPs
    :param list g_data: The data block to be analyzed (in the form of [ [CHR, POS, MAF, Value 1, ..., Value n], ... ])
    With CHR as a string, POS as an integer, MAF as a floating value and Value 1, ..., Value n as string in the form of
    "0|0", "0|1", "1|0", "1|1", "1" or "0"
    :param int win_len: Size of the window within which the allele sharing coefficient is estimated (in terms of number
    of positions)
    :param int win_shift: Shift size (in terms of number of positions)
    :param bool, optional phased: If set it assumes the data to be phased. Default behavior unphased (affects only the
    mismatch coefficient)
    :return list: A list of two items is returned. The first item contains the data structure (as described above) for
    the allele sharing coefficient, whereas the second item contains the data structure (as described above) for the
    mismatch coefficient"""

    # Auto-detect haploid status from first sample's genotype
    first_gt = clean_snp(g_data[0][3])  # First sample's genotype
    is_haploid = len(first_gt) == 1  # Single allele indicates haploid

    # sanity check for the input g_data
    check_chr = None
    for dt in g_data:
        if check_chr is None:
            check_chr = dt[0]
        else:
            if not check_chr == dt[0]:
                msg = "g_data not consistent, multiple chromosome found: {} and {}.".format(check_chr, dt[0])
                raise ValueError(msg)

    asc_win = []  # returned data structure for asc
    if win_shift > win_len:
        print("Shifting of the windows cannot be bigger that the window size.")
        print("Setting window shift equal to window size.")
        win_shift = win_len

    asc_next_start = 0  # used to determine the starting index for the next window
    asc_start_idx = asc_next_start  # starting index of the window
    stop = False  # flag to stop the loop
    asc_stop = False

    while not stop:
        # initialize allele sharing coefficient lists and SNPs count for the current segment
        n = len(g_data[0][3:])  # number of samples/individuals
        asc_seg = [0] * int(n * (n - 1) / 2)  # number of combinations among individuals

        asc_idx = asc_start_idx  # first value to be considered is at the starting index of the window
        asc_n_seg = 0
        done = False
        asc_done = False

        while not done:  # continue until the length of the SNP values does not reach the window length
            try:
                if not asc_stop and asc_n_seg < win_len:
                    asc_ind_values = g_data[asc_idx][3:]  # sample values from individuals
                    maf = g_data[asc_idx][2]  # alternate allele frequency
                    if 0 < float(maf) < 1:  # a value we can use
                        asc_seg = update_asc(asc_seg, asc_ind_values, maf, is_haploid)
                        asc_n_seg += 1
                        if asc_n_seg == win_shift:  # if necessary save the index where the next window will start
                            asc_next_start = asc_idx + 1
                else:
                    asc_done = True
            except IndexError:  # last window may not have all the SNPs required in a window, finalize it as it is
                asc_stop = True
                asc_finalized = False

            done = asc_done
            asc_idx += 1

        # segment completed, finalize asc value
        if (not asc_stop) or (asc_stop and not asc_finalized):
            for idx in range(len(asc_seg)):
                if asc_n_seg > 0:
                    asc_seg[idx] = asc_seg[idx] / asc_n_seg
                else:
                    asc_seg[idx] = '-'
            if len(asc_seg) == 1:
                asc_win.append(tuple([asc_n_seg, asc_seg[0]]))
            else:
                asc_win.append(tuple([asc_n_seg, tuple(asc_seg)]))
            if asc_stop and not asc_finalized:
                asc_finalized = True

        asc_start_idx = asc_next_start  # update starting index for the new window
        stop = asc_stop

    return asc_win


# noinspection PyUnboundLocalVariable
def dynamic_kinship(vcf_file: str, win_len: int, win_shift: int, tag="AFngsrelate",
                   chr_idx=0, pos_idx=1, phased=None, is_haploid=False):
    """MSM-free version"""
    """
    Compute the dynamic allele sharing coefficient (ASC) and the mismatch (MSM) coefficient between samples of a given
    VCF file. Dynamic refers to the ASC estimated within a given window size throughout the available SNPs
    :param str vcf_file: Input VCF file
    :param int win_len: Size of the window within which the allele sharing coefficient is estimated (in terms of number
    of SNPs)
    :param int win_shift: Shift size (in terms of number of SNPs)
    :param str, optional tag: MAFs are stored  within  the reference  VCF file,  consider the given TAG  (default
    'AFngsrelate')
    :param int, optional chr_idx: Column  index for chromosome  details (default 0)
    :param int, optional pos_idx: Column  index for SNP position (default 1)
    :param bool, optional phased: If set it assumes the data to be phased. Default behavior unphased (affects only the
    mismatch coefficient)"""

    idx_ref = {}
    idx = 0
    chrom = "-"
    vcf_in = open(vcf_file, "r")
    found_maf = False
    for line in vcf_in:
        line = line.strip('\n')
        if line.startswith("#CHROM"):
            values = line.split("\t")
            ids = values[9:]
            samples_data = []  # each entry is a list: [CHR, POS, MAF, V1, V2, ..., Vn]
        elif not line.startswith("#"):
            values = line.split("\t")
            maf = extract_freq(info=values[7], tag=tag)
            if not maf == -1:  # if there is at least one frequency value we retain all ASC computations
                found_maf = True
            ind_values = values[9:]
            samples_data.append([values[chr_idx], int(values[pos_idx]), maf] + ind_values)
            if chrom != values[chr_idx]:
                if chrom != "-":
                    idx_ref[chrom] = (idx_ref[chrom], idx - 1)
                idx_ref[values[chr_idx]] = idx
                chrom = values[chr_idx]
            idx += 1
    idx_ref[values[chr_idx]] = (idx_ref[values[chr_idx]], idx - 1)
    vcf_in.close()

    idxs = []
    for idx1 in range(len(ids)):
        for idx2 in range(idx1 + 1, len(ids)):
            idxs.append((idx1, idx2))
    asc_chr = {"id": ids,
               "idx": idxs
               }

    for key, val in idx_ref.items():
            asc_chr[key] = kinship_windowed(
                samples_data[val[0]: val[1] + 1],
                win_len, win_shift,
                phased=phased,
                is_haploid=is_haploid
            )

    if not found_maf:
        asc_chr = ["DISCARD", asc_chr]
        warn_msg = "No frequency values were found in the VCF file under the tag '{}'. Therefore, the ASC analysis " \
                   "has been discarded.".format(tag)
        warnings.warn(warn_msg)

    return asc_chr


def maf2vcf(frq_file: str, vcf_file: str):
    """
    Insert the frequency values stored in a file into the VCF file
    :param str frq_file : The file that contains the frequency values
    :param str vcf_file : The VCF file where the frequency values will be added"""

    info_header = "##INFO=<ID=AFngsrelate,Number=A,Type=Float,Description=\"Allele Frequency\">\n"  # ngsRelate TAG
    af_defined = False  # flag to avoid to insert twice the ngsRelate TAG if already there

    old_file = vcf_file + ".old"  # backup/old file name
    shutil.copy(vcf_file, old_file)  # copy the original file (so it is not lost)
    frq = open(frq_file, "r")  # frequency file pointer

    vcf_in = open(old_file, "r")  # input VCF file pointer
    vcf_out = open(vcf_file, "w")  # output VCF file pointer

    info_idx = -1  # will contain the index for the INFO column
    entry_line = 0
    for line in vcf_in:  # for each line in the file
        if line.startswith("##"):  # header: copy it to the new file
            if line.startswith("##INFO") and "ID=AFngsrelate" in line:  # TAG was already in the header
                af_defined = True
            vcf_out.write(line)
        elif line.startswith("#CHROM"):  # column header: we need to find the index for the INFO column
            if not af_defined:  # before adding the column names, if necessary add the TAG line
                vcf_out.write(info_header)
            header = line.split()  # decompose the line

            # Identify the index of the INFO column:
            #
            info_idx = 0
            found = False
            while not found and info_idx < len(header):
                if header[info_idx] == "INFO":
                    found = True
                else:
                    info_idx = info_idx + 1

            if not found:  # the INFO column is not in the file, throw an error
                # Close the opened files:
                #
                vcf_in.close()
                vcf_out.close()
                frq.close()

                __restore_vcf(vcf_file)  # restore the original file

                msg = "INFO column not found in the VCF file '{}'".format(vcf_file)
                raise ValueError(msg)

            vcf_out.write(line)  # write on the output file
        else:  # data row: we need to select the INFO field and add the frequency value
            entry_line = entry_line + 1
            # Read the VCF line details:
            #
            entry = line.split()  # decompose the line
            prv_value = entry[info_idx]  # previous INFO value

            # Read the frequency details:
            #
            frq_entry = frq.readline()
            while frq_entry.startswith(
                    "#"):  # or frq_entry == "":  # header or empty line, discard and read the next line
                frq_entry = frq.readline()  # read (and discard) the header of the frequency file which won't be used

            if not frq_entry:
                vcf_in.close()
                vcf_out.close()
                frq.close()

                __restore_vcf(vcf_file)  # restore the original file

                msg = "Unexpected end of frequency file: The frequency file has less entries than the VCF file."
                raise ValueError(msg)

            frq_entry = frq_entry.strip('\n')
            try:
                _ = float(frq_entry)  # check if it is really a data line
            except ValueError:  # a header without the #, read and split the next line
                vcf_in.close()
                vcf_out.close()
                frq.close()

                __restore_vcf(vcf_file)  # restore the original file

                msg = "Unexpected frequency value: '{}'".format(frq_entry)
                raise ValueError(msg)

            if prv_value == ".":  # if empty just insert the TAG value
                entry[info_idx] = "AFngsrelate={}".format(frq_entry)
            elif prv_value[-1] == ";":  # if not empty attach the TAG at the end
                entry[info_idx] = entry[info_idx] + "AFngsrelate={}".format(frq_entry)
            else:  # if not empty attach the TAG at the end
                entry[info_idx] = entry[info_idx] + ";AFngsrelate={}".format(frq_entry)

            # Create the new line and write it to the output file:
            #
            new_line = "\t".join(entry)
            new_line = new_line + "\n"
            vcf_out.write(new_line)  # write on the output file

    frq_entry = frq.readline()
    while frq_entry.startswith("#"):  # or frq_entry == "":  # header or empty line, discard and read the next line
        frq_entry = frq.readline()  # read (and discard) the header of the frequency file which won't be used

    if not frq_entry:
        # Close the opened files:
        #
        vcf_in.close()
        vcf_out.close()
        frq.close()
    else:
        vcf_in.close()
        vcf_out.close()
        frq.close()

        __restore_vcf(vcf_file)  # restore the original file

        msg = "Unexpected values from frequency file: The frequency file has more entries than the VCF file."
        raise ValueError(msg)


def __restore_vcf(vcf_file: str, suffix=".old"):
    """
    Restore the previous version of the VCF file (if any)
    :param str vcf_file :  The original name (i.e., without the .old suffix) of the file that needs to be restored
    :param str, optional suffix : The suffix extension attached to the file"""

    old_file = vcf_file + suffix  # name of the backup/old file
    old_path = Path(old_file)
    if old_path.is_file():  # there is a file to be restored
        vcf_path = Path(vcf_file)
        if vcf_path.is_file():  # there is a file that need to be removed
            os.unlink(vcf_file)  # remove the file
        shutil.move(old_file, vcf_file)  # restore the previous file
    else:
        msg = "The VCF file '{}' has no previous version.".format(vcf_file)
        raise ValueError(msg)


def fix_vcf(vcf_in: str, vcf_out: str, missing_values=0, feedback=True):
    """
    Given a VCF file generates a new VCF file where all misplaced entries are rearranged and duplicate entries
    as well as unsupported SNP lines are discarded. Among unsupported SNP checks there are also missing reference
    values. Checks also for alternate missing values coupled with samples referring to it, i.e., having 1s (note that
    if an alternate missing values is referenced in a sample, it is automatically discarded without considering the
    general rule about missing values, i.e., even if missing values are accepted and should be retained, the ones
    associated to the alternate value are still discarded. This is simply done for simplicity.).

    :param str vcf_in: The input VCF file (to be fixed).
    :param str vcf_out: The name of the (fixed) output VCF file.
    :param int, optional missing_values: Filter the data in presence of missing data. By default, it filters all
    SNPs having any missing values, i.e., a given SNP is discarded if there is at least one individual with a missing
    value in that position (missing_values=0 as zero missing values are retained). Alternatively, it can be set to 1,
    i.e., the SNP is retained if there is at most a single missing value (e.g., ./1), otherwise it is discarded (e.g.,
    ./.). Finally, if it is set to 2 it retains also SNPs with two missing values, e.g., ./., thus not applying any
    filter. Briefly, the values allowed for the parameter missing_values are either 0, 1, or 2.
    :param bool, optional feedback: If set the function provides a more detailed feedback,
    otherwise it will just output minimum information."""

    # examine the VCF file
    print("Scanning the file...")
    vcf_summary = summary(vcf_file=vcf_in, missing_values=missing_values, feedback=feedback)
    unsupported = vcf_summary["unsupported_entry"]
    duplicates = vcf_summary["duplicate_entry"]
    misplaced = vcf_summary["misplaced_entry"]

    msg_warning = (len(unsupported) > 0) or (len(duplicates) > 0) or (len(misplaced) > 0)
    if msg_warning:
        print("Sorting through the reported lines...")

    skip = None
    if len(unsupported) > 0:
        unsupported = entries2tab(unsupported)  # these lines are discarded as containing unsupported values
        skip = unsupported["line"]
    if len(duplicates) > 0:
        duplicates = entries2tab(duplicates)  # these lines are discarded as containing duplicates
        if skip is None:
            skip = duplicates["line"]
        else:
            skip = np.unique(np.concatenate((skip, duplicates)))  # if necessary merge and remove duplicates
    if len(misplaced) > 0:
        misplaced = entries2tab(misplaced)
        if skip is not None:  # there are lines to skip let's filter out those who are also in misplaced
            ignore_misplaced = np.intersect1d(misplaced["line"], skip)
            if len(ignore_misplaced) > 0:  # we have misplaced to ignore
                next_idx = 0
                tmp = np.zeros(len(misplaced) - len(ignore_misplaced), dtype=misplaced.dtype)
                for m_el in misplaced:
                    if m_el["line"] not in ignore_misplaced:
                        tmp[next_idx] = m_el
                        next_idx += 1
                misplaced = tmp
    start_idx = 0
    if msg_warning:
        print("Writing the VCF file...")
        last_chromosome = -1
        vcf_raw = open(vcf_in, "r")
        vcf_fix = open(vcf_out, "w")
        idx_line = -1
        ignored = 0  # used for debugging purposes
        written = 0  # used for debugging purposes
        for line in vcf_raw:
            idx_line = idx_line + 1
            if line.startswith("#"):  # comment line, rewrite it
                vcf_fix.write(line)
            else:
                [found, start_idx] = if_present(v=idx_line, v_list=skip, start_idx=start_idx)
                if not found:  # the line needs to be written, but we may add some misplaced lines before
                    # select the chromosome and position of the current line
                    values = line.split()
                    values[0] = values[0].lower().replace("x", "23")
                    line_chromosome = int(values[0])
                    line_position = int(values[1])

                    # if the line_chromosome is different from the previous written chromosome we need to be sure that
                    # we add all the remaining misplaced entries of the previous chromosome (last_chromosome)
                    if len(misplaced) > 0:
                        if not (line_chromosome == last_chromosome):
                            condition = (misplaced["chr"] == last_chromosome)
                            misplaced_chromosome = np.extract(condition, misplaced)
                            if np.size(misplaced_chromosome):
                                for row in np.nditer(misplaced_chromosome):
                                    misplaced_line = vcf_summary["misplaced_entry"][row["idx"]][1]
                                    vcf_fix.write(misplaced_line)  # write it on file

                                    # add the line to the skip list not to write it again
                                    tmp = skip
                                    skip = np.zeros(len(skip) + 1)
                                    added = 0
                                    for idx_skip in range(len(skip)):
                                        if idx_skip - added < len(tmp):
                                            if added == 0 and row["line"] < tmp[idx_skip - added]:
                                                skip[idx_skip] = row["line"]
                                                added = 1
                                                ignored -= 1  # it'll be later on ignored, but it's been already written
                                            else:
                                                skip[idx_skip] = tmp[idx_skip - added]
                                    if added == 0:
                                        skip[-1] = row["line"]
                                        ignored -= 1  # it will be later on ignored, but it has been already written

                                # drop all the entries of last_chromosome from the misplaced array
                                condition = (misplaced["chr"] != last_chromosome)
                                misplaced = np.extract(condition, misplaced)

                            last_chromosome = line_chromosome

                        # select all the misplaced entries related to the line_chromosome
                        condition = (misplaced["chr"] == line_chromosome)
                        misplaced_chromosome = np.extract(condition, misplaced)

                        # if the line_position, i.e., the position of the current file entry is smaller than
                        # the first misplaced then write the current file entry and continue to the next line,
                        # otherwise loop through the misplaced entries and insert them in the output file until
                        # the line_position can be inserted
                        while len(misplaced_chromosome) > 0 and line_position > misplaced_chromosome[0]["pos"]:
                            idx_row = misplaced_chromosome[0]["idx"]
                            misplaced_line = vcf_summary["misplaced_entry"][idx_row][1]  # misplaced line to be written
                            vcf_fix.write(misplaced_line)  # write it on file

                            tmp = skip
                            skip = np.zeros(len(skip) + 1)
                            added = 0
                            for idx_skip in range(len(skip)):
                                if idx_skip - added < len(tmp):
                                    if added == 0 and misplaced_chromosome[0]["line"] < tmp[idx_skip - added]:
                                        skip[idx_skip] = misplaced_chromosome[0]["line"]
                                        added = 1
                                        ignored -= 1  # it will be later on ignored, but it has been already written
                                    else:
                                        skip[idx_skip] = tmp[idx_skip - added]
                            if added == 0:
                                skip[-1] = misplaced_chromosome[0]["line"]
                                ignored -= 1  # it will be later on ignored, but it has been already written

                            # drop the inserted entry from both misplaced (the whole set)
                            # and misplaced_chromosome (the current one)
                            curr_line = misplaced_chromosome[0]["line"]
                            condition = (misplaced_chromosome["line"] != curr_line)
                            misplaced_chromosome = np.extract(condition, misplaced_chromosome)
                            condition = (misplaced["line"] != curr_line)
                            misplaced = np.extract(condition, misplaced)

                    vcf_fix.write(line)  # write the current line
                    written += 1
                else:
                    ignored += 1

        vcf_raw.close()
        vcf_fix.close()
        print("Output file created: {}".format(vcf_out))
        print("(Of the initial {:,} SNPs a total of {:,} SNPs were retained while the remaining {:,} SNPs were "
              "discarded. See '{}' for more details)".format(written + ignored,
                                                             written, ignored, "summary_{}.log".format(vcf_in)))
    else:
        msg = "The VCF file '{}' does not contain misplaced, duplicate or unsupported entries.".format(vcf_in)
        logging.basicConfig(format='%(asctime)s - %(message)s', level=logging.INFO)
        logging.info(msg)


def entries2tab(entries: [int, str]):
    """
    Given the misplaced_entry or duplicate_entry of the summary dictionary,
    returns a numpy array with the chromosomes and positions sorted
    :param [int, str] entries : The key of the summary dictionary where the first column identify the line
        number in the VCF file from which it was generated and str is the line itself
    :return: numpy array : The table defined as follows:
        idx : The reference index of the misplaced_entry or duplicate_entry list
        line : The line number associated with the relative misplaced/duplicate entry
        chr : The chromosome number pertinent to the text line of the misplaced/duplicate entry
        pos : The position of the chromosome pertinent to the text line of the misplaced/duplicate entry"""

    d = np.zeros(len(entries), dtype=[("idx", int), ("line", int), ("chr", int), ("pos", int)])
    for idx in range(len(entries)):
        d[idx][0] = idx
        d[idx][1] = entries[idx][0]
        values = entries[idx][1].split()
        values[0] = values[0].lower().replace("x", "23")
        d[idx][2] = int(values[0])
        d[idx][3] = int(values[1])

    d.sort(order=["chr", "pos"])

    return d


def if_present(v: int, v_list, start_idx: int) -> [bool, int]:
    done = False
    found = False
    idx = start_idx
    while not done:
        if idx >= len(v_list):
            done = True
        elif v == v_list[idx]:
            done = True
            found = True
            idx += 1
        elif (idx > 0 and v_list[idx - 1] < v < v_list[idx]) or (idx == 0 and v < v_list[idx]):
            done = True
        else:
            idx += 1

    return [found, idx]
