#!/usr/bin/python3

import regex as re
import sys
import argparse

"""

Written by DJ3CE, Cedric.

Mimics a previous shell script, hopefully speeding things up and learning a
bit of Python coding...
"""

class NullWriter:
    def write(self, s): pass


"""
Precompiled regular expression tuples, together with their
replacement.
To be used like `a = re_char_replacements[i][0].sub(re_char_replacements[i][1], input)`
"""
re_char_replacements = [
    (re.compile("\p{Zs}")," "),
    (re.compile("\p{Cf}"),""),
    #(re.compile("\N{U+FFFD}"),""), # !? Python no likes dis.
    (re.compile("Ã"),"A"),
    (re.compile("Ã¨"),""), # Liege/Belgium
    (re.compile("Ã¦"),"ae"), # Denmark
    (re.compile("|"),""), # La Mancha?!
    (re.compile("Å"),""),
    (re.compile("\$"),"s"),
    #
    # Transformations
    (re.compile("[öő]|Ã¶"),"oe"),
    (re.compile("ä"),"ae"),
    (re.compile("[üű]|Ã¼"),"ue"),
    (re.compile("Ö"),"Oe"),
    (re.compile("Ö"),"Oe"),
    (re.compile("Ü"),"Ue"),
    (re.compile("ß|Ã\x9f"),"ss"),
    #
    # Reduction/Simplification
    (re.compile("[ÉÊË]"),"E"),
    (re.compile("[èéëěêėę]"),"e"),
    (re.compile("[ÁÂÅÃÄ]"),"A"),
    (re.compile("[åáàâãąāª]"),"a"),
    (re.compile("[ÓØÔÕ]"),"O"),
    (re.compile("[øóôòõ]"),"o"),
    (re.compile("[ÎÍİ]"),"I"),
    (re.compile("[īïîıìíi̇]"),"i"),
    (re.compile("Ñ"),"N"),
    (re.compile("[ňñń]"),"n"),
    (re.compile("Ú"),"U"),
    (re.compile("[ūůú]"),"u"),
    (re.compile("[ÇČĆĆ]"),"C"),
    (re.compile("[çčć]"),"c"),
    (re.compile("[S̄ȘŠŚŞ]"),"S"),
    (re.compile("[șšśş]"),"s"),
    (re.compile("æ"),"ae"),
    (re.compile("Æ"),"AE"),
    (re.compile("[țť]"),"t"),
    (re.compile("°"),"o"),
    (re.compile("k̄"),"k"),
    (re.compile("đ"),"d"),
    (re.compile("Đ"),"D"),
    (re.compile("[žźż]"),"z"),
    (re.compile("[ŻŽ]"),"Z"),
    (re.compile("ł"),"l"),
    (re.compile("Ł"),"L"),
    (re.compile("ř"),"r"),
    (re.compile("ğ"),"g"),
    (re.compile("Ğ"),"G"),
    (re.compile("ý"),"y"),
    (re.compile("º"),"o"),
    (re.compile("A¤"),"ae"),
    (re.compile("A\x92"),"'"),
    (re.compile("A\x83"),"e"),
    (re.compile("[\`\´’”“]"),"'"),
    (re.compile("\x0a+"),"\x0a"),
    (re.compile("[­​]")," "),
    (re.compile("[\*\{\}=°♥©!¥\x7f]"),"") # tr/*{}=º//d;
]
# AA
# ♥
# 
# o S
# 
# 



# Global precompiled regular expressions for various things
re_sanity = re.compile("^[-_.# \/a-zA-Z0-9@,; :&?\(\)\[\]\"']*[\x0d\x0a]?$")

# Original formulation:
#re_callsign = re.compile(',\"?([A-Z]{1,2}[A-Z0-9]|[0-9][A-Z]{1,2})[0-9]{0,2}[A-Z]{1,5}([-\/][MPLJ])?\"?,')
# RegExp based on (length of) ITU prefixes
#                             Single      | Three    |HB0 vs | W   WD | D      W         |3...
re_callsign = re.compile(',\"?([2WRNMKIGF]|S[ST][A-Z]|HB[0-9]|[A-Z][0-9A-Z]|[0-14-9][A-Z]|3[A-CE-Z]|3D[A-Z])[A-Z0-9]{1,5}([-\/][MPLJ])?\"?,')
# Based on formula 
# NN9A, NN9NA, NN9NNA, NN9NNNA, A9A, A9NA, A9NNA, A9NNNA
# from Wikipedia + suffix
# Lacks prefix detection
#re_callsign = re.compile(',\"?([0-9A-Z][0-9A-Z][0-9][A-Z]|[0-9A-Z][0-9A-Z][0-9][0-9A-Z][A-Z]|[0-9A-Z][0-9A-Z][0-9][0-9A-Z][0-9A-Z][A-Z]|[0-9A-Z][0-9A-Z][0-9][0-9A-Z][0-9A-Z][0-9A-Z][A-Z]|[A-Z][0-9][A-Z]|[A-Z][0-9][0-9A-Z][A-Z]|[A-Z][0-9][0-9A-Z][0-9A-Z][A-Z]|[A-Z][0-9][0-9A-Z][0-9A-Z][0-9A-Z][A-Z])([-\/][MPLJ])?\"?,')
re_fixno = re.compile("[0-9]+")

re_nonone = (
    (re.compile(' None","'),'","'),
    (re.compile(' None,'),','),
) # ! Die Quotes müssen nicht da sein!

# Global variables for ITU statistics
stat_callsign_prefix = {}
stat_callsign_itu = {}

# Global variables for counting
no_valid = 0
no_discarded = 0
no_no_callsign = 0

# Global argument variables
args = None

def printf(format, *args):
    sys.stdout.write(format % args)

def printef(format, *args):
    sys.stderr.write(format % args)

def fprintf(fh, format, *args):
    fh.write(format % args)

def print_log(level, format, *largs):
    global args
    if args.verbose >= level:
        sys.stderr.write(format % largs)

def retuple(re,text):
    return re[0].sub(re[1],ln)

def convertNonAscii(ln):
    # Lines 34 to 93 of sh-script...
    # Convert characters:
    for r in re_char_replacements:
        ln = r[0].sub(r[1],ln)

    return ln

def checkSanity(i):
    return re_sanity.search(i) != None

def itu(call):
    s = re_callsign.search(call) # Search: *somewhere* in the string
    if s != None :
        return s.groups()
    else :
        return None

def findItu(s, itu_list):
    for i in itu_list:
        if (i[0].match(s)):
            return (i[1], i[2])

"""
Functions for handing over to main-worker function
"""

def option_convertNonAscii(ln, valid=True):
    ln = convertNonAscii(ln)
    return True, ln

def option_checksanity(ln, valid=True):
    if not valid:
        return valid, None
    san = checkSanity(ln)
    if not san:
        print_log(1,"No sanity: %s", ln)
    return san, None


def option_callsign_prefix_stat(ln, valid=True):
    if not valid: 
        print_log(4,"Skipping invalid entry.\n")
        return True, None

    global stat_callsign_prefix
    global no_no_callsign
    ln_itu = itu(ln)
    if ln_itu is not None : 
        if not ln_itu[0] in stat_callsign_prefix :
            stat_callsign_prefix[ln_itu[0]] = 0

        stat_callsign_prefix[ln_itu[0]] += 1
    else :
        no_no_callsign += 1
        print_log(2,"No callsign detected: %s", ln)

    return True, None

def option_filter_gen(regexp, invert=False):
    reg = re.compile(regexp)

    def fn(ln, valid=True):
        ln_itu = itu(ln)
        if ln_itu is None :
            return invert, None
        match = reg.match(ln_itu[0])
        return (match is not None)^invert, None

    return fn

def option_drop_no_callsign(ln, valid=True):
    ln_itu = itu(ln)
    return (ln_itu is not None), None

def option_drop_no_itu(ln, valid=True):
    ln_itu = itu(ln)
    if (ln_itu is None):
        return False, None

    return (findItu(ln_itu, itulist) is not None), None # WONT WORK (no itulist)

def option_nonone(ln, valid=True):
    for re in re_nonone:
        ln = re[0].sub(re[1],ln)
    #ln = retuple(re_nonone, ln)
    return valid, ln

def post_callsign_prefix_stat(itulist, output=sys.stdout,
        missingPrefix=NullWriter()):
    global stat_callsign_prefix, no_no_callsign
    total = 0

    for i in stat_callsign_prefix.keys():
        # Iterate over prefixes, match them with itu and sum up:
        if itulist is None:
            total += stat_callsign_prefix[i]
            fprintf(output, "%s %s\n", i, stat_callsign_prefix[i])
            continue

        r = findItu(i,itulist)

        if r is None:
            no_no_callsign += stat_callsign_prefix[i]
            missingPrefix.write("%s\n"%(i))
            print_log(1,"Missing prefix %s (n=%i)\n",i,stat_callsign_prefix[i])
            continue

        if r[0] not in stat_callsign_itu:
            stat_callsign_itu[r[0]] = 0
        stat_callsign_itu[r[0]] += stat_callsign_prefix[i]

    for k in stat_callsign_itu.keys():
        fprintf(output, "%s %i\n", k, stat_callsign_itu[k])
        total += stat_callsign_itu[k]

    print_log(1,"Total stat: %i\n", total)
    print_log(0,"Invalid/no country: %i\n", no_no_callsign)

    return True

"""
Main worker function
"""
def parsecontacts(f, procfn=(option_convertNonAscii, option_checksanity,
    option_callsign_prefix_stat ), outfile=sys.stdout, dropped=NullWriter()):
    global args

    no_valid = 0
    no_discarded = 0

    for ln in f.readlines():

        isValid = True
        for fn in procfn:
            isValid_fn, nln = fn(ln, isValid)
            isValid = isValid and isValid_fn
            if nln is not None:
                ln = nln

        if isValid:
            no_valid += 1
            if args.modify_index:
                ln = re_fixno.sub("%i"%(no_valid), ln, 1)
            fprintf(outfile,"%s",ln)
        else:
            no_discarded += 1
            fprintf(dropped,"%s",ln)


    # Output statistics:
    printef("Valid:     %d\n",no_valid)
    printef("Discarded: %d\n",no_discarded)

def readItuList(fn, sep="\t"):
    itu_list = []
    r = re.compile("^(.*)%s(.*)$"%(sep))

    with open(fn, "r", encoding="utf-8") as f:
        for ln in f.readlines():
            i = r.match(ln)
            if ( i != None ):
                g = i.groups()
                itu_list.append((re.compile(g[0]),g[0],g[1]))

    return itu_list

class FileOpen(object):
    def __init__(self, **kwargs):
        self.kwargs = kwargs

    def __call__(self, values):

        if values == '-' :
            if 'r' in self.kwargs.mode:
                return sys.stdin
            elif any(c in self.kwargs.mode for c in 'wax'):
                return sys.stdout
            else :
                raise ValueError("argument '-' with mode %r" % self.kwargs.mode)
        try :
            return open(values, **self.kwargs)
        except OSError as e:
            raise ArgumentTypeError("Can't open '%s': %s"% (values, e))
        

class FilterAction(argparse.Action):
    def __init__(self, option_strings, dest, nargs=None, **kwargs):
        if nargs is None:
            raise ValueError("No nargs!")
        super().__init__(option_strings, dest, **kwargs)

    def __call__(self, parser, namespace, values, option_string=None):
        invert = False
        if option_string != '-f' :
            invert = True

        fn = option_filter_gen(values, invert)

        opt = getattr(namespace, self.dest)
        if opt is None :
            opt = [fn]
        else :
            opt.append(fn)

        setattr(namespace, self.dest, opt)

def convert_non_ASCII():
    global args
    
    # Parse arguments
    argp = argparse.ArgumentParser()
    # Input has header? --no-header turns off
    argp.add_argument('--header', action='store_const', default=True,
        const=True, 
        help="Copy input file header line to output (default)")
    argp.add_argument('--modify-index', action='store_const', default=False,
        const=True,
        help="Modify first column to be continuous index column")
    argp.add_argument('--no-header', action='store_const',
        dest='header', const=False, 
        help="Turn off copying header line to output")
    # ITU List, input file
    argp.add_argument('--itu-prefix', type=readItuList,
        metavar='itu-prefixlist.csv',
        help="File with ITU callsign prefixes, for grouping callsigns")
    # Generate ITU stat, output file
    argp.add_argument('--itu', type=argparse.FileType('w',
        encoding='utf-8'), default=None, metavar='statfile.csv',
        help="Write Callsign-Prefix Statistics to file")
    # Convert non-ascii
    argp.add_argument('-c','--convert', dest='action',
        action='append_const', const=option_convertNonAscii,
        help="Convert non ASCII symbols to ASCII ones")
    # Sanitize (drop invalid lines)
    argp.add_argument('-s','--sanitize', dest='action',
        action='append_const', const=option_checksanity,
        help="Drop lines with dubious characters")
    # No-callsign (drop invalid lines)
    argp.add_argument('-d','--drop-no-callsign', dest='action',
        action='append_const', const=option_drop_no_callsign,
        help="Drop lines where no callsign is detected")
    argp.add_argument('-w','--drop-no-itu', dest='action',
        action='append_const', const=option_drop_no_itu,
        help="Drop lines, where the callsign prefix does not match ITU list")
    argp.add_argument('--no-none', dest='action',
        action='append_const', const=option_nonone,
        help="Remove ' None' from names")
    # Create itu statistics 
    #argp.add_argument('-i','--itu-stat', dest='action',
    #    action='append_const', const=option_callsign_prefix_stat)
    # Filter by regexp
    argp.add_argument('-f','--filter', dest='action', nargs=1,
        action=FilterAction, metavar='filterRegExp',
        help="Only keep lines with callsign matching given RegExp")
    # Exclude by regexp
    argp.add_argument('-x','--exclude', dest='action', nargs=1,
        action=FilterAction, metavar='excludeRegExp',
        help="Drop lines with callsign matching given RegExp")
    # Verbose output
    argp.add_argument('-v','--verbose', action='count', default=0)
    # Output non existing ITU Prefixes
    argp.add_argument('--missingPrefixes', default=NullWriter(),
        type=argparse.FileType('w', encoding='utf-8'),
        help="Write callsign-prefixes not found in ITU list to file")
    # Output dropped lines
    argp.add_argument('--dropped', default=NullWriter(),
        type=argparse.FileType('w', encoding='utf-8'),
        help="Write dropped lines to file")

    # Input and Output file for callsigns
    #argp.add_argument('input', metavar='input-file', type=open)
    argp.add_argument('input', metavar='input-file', 
        type=FileOpen(encoding='utf-8',errors='ignore'))
        #type=FileOpen(encoding='utf-8'))
    argp.add_argument('output', metavar='output-file',
        type=FileOpen(mode='w', encoding='utf-8', newline='\r\n'))

    args = argp.parse_args()

    if args.action is None :
        args.action = []

    args.action.append(option_callsign_prefix_stat)

    # Copy header:
    if args.header :
        head = args.input.readline()
        args.output.write(head)
        args.dropped.write(head)

    parsecontacts(args.input, outfile=args.output, procfn=args.action,
            dropped=args.dropped)

    if args.itu is not None :
        post_callsign_prefix_stat(args.itu_prefix, output=args.itu,
                missingPrefix=args.missingPrefixes)


if __name__ == "__main__":
    convert_non_ASCII()
