#!/usr/bin/env python3
# alasm 1.0: ALASM sources (TR-DOS files of type H) as plain text, for Midnight Commander and the command line.
# Copyright (c) 2026 Spectre (Optical Brothers), https://www.zxby.org. MIT License (see LICENSE).
#
#   alasm [-cp866] [-f] INPUT [OUTPUT]
#
# INPUT is an 'H' file as it is on the disk, or the same in a hobeta file ($H). OUTPUT is INPUT with the extension
# .asm by default (an existing one is kept unless -f), - is stdout. The text is what QC's alasm2asm gives: UTF-8 with
# LF line ends, no spaces at the line ends, the bytes #11-#1F as their pictures (U+2411-U+241F); with -cp866, ALASM's
# own text export (cp866, CR line ends). A file that is no ALASM source goes to stdout as it is (mc's F3 on another .H
# file shows it), to a file not at all. The problems (unknown tokens, a broken file) go to stderr, but not with
# stdout as OUTPUT: there ?XX in the text shows them. See README.md.
#
# An 'H' file: a 64-byte header (#28-#2F: SIGNATURE), then lines; each line is <len><len-1 bytes>, len = 0 ends the
# file. Inside a line, as ALASM's str2txt (al0_44) shows it and its export (exptxt in alImpExp) writes it:
#
#   #01-#0F  a run of that many spaces (#00: 256)
#   #10      the rest of the line is plain text (the byte itself not shown)
#   #11-#1F  written as they are
#   #FF      not shown; a token right after it (or after a run of spaces) is not moved to the mnemonic column
#   >= #80   a token, padded to the mnemonic column when it would start before it; the first one is the mnemonic
#            or directive (#80 + its index in MNEMONICS), the rest are operands (OPERANDS)
#   ;        starts the comment: the rest of the line is plain text
#   "        starts a string: plain text up to the next " or the end of the line
#
# Plain text is raw cp866 without the bytes under #20 and #FF, and ends at column 64. The tokens are ALASM's own tables
# (mnemtkn, regstkn in al0_44; the versions: at MNEMONICS). The spaces trimmed at the line ends are what Python's
# str.isspace takes, as in alasm2asm.

import os
import struct
import sys

# ALASM 4.44-4.46's up to ENDIF (#E0), then the ones later versions added: EXD, JNZ, JZ, JNC, JC (4.47, 5.0), RUN
# (5.0f2); none after it up to 5.09, the last. A token a later version renamed keeps 4.4x's name (#D3 IF: IF0 since
# 5.03; #D2 UNTIL: UNTIL0 since 5.07): an 'H' file doesn't tell its version.
MNEMONICS = '''INCLUDE INCBIN MACRO LOCAL RLCA RRCA HALT CALL PUSH RETN RETI DJNZ OUTI OUTD LDIR CPIR INIR OTIR LDDR
    CPDR INDR OTDR DD DEFB DEFW DEFS DISP ENDM EDUP ENDL MAIN ELSE DISPLAY EXA DB DW DS NOP INC DEC RLA RRA DAA CPL SCF
    CCF ADD ADC SUB SBC AND XOR RET POP RST EXX RLC RRC SLA SRA SLI SRL BIT RES SET OUT NEG RRD RLD LDI CPI INI LDD CPD
    IND ORG EQU ENT INF DUP IFN REPEAT UNTIL IF LD JR JP OR CP EX DI EI IN RL RR IM ENDIF
    EXD JNZ JZ JNC JC RUN'''.split()

OPERANDS = {0x9F: '(BC)', 0xA0: '(DE)', 0xA1: '(HL)', 0xA2: '(SP)', 0xA3: '(IX)', 0xA4: '(IY)',
            0xD0: '(C)', 0xD1: '(IX', 0xD2: '(IY', 0xD3: "AF'"}
OPERANDS.update((0xE0 + i, r) for i, r in enumerate(
    'BC DE HL AF IX IY SP NZ NC PO PE HX LX HY LY B C D E H L A P M Z R I'.split()))

MNEMONIC_COLUMN = 8  # where ALASM's export puts the mnemonic
HEADER = 0x40
SIGNATURE = b'\xf3\x76\xc7\xdd\xfd\xed\xb0\xd9'  # at #28 of the header: ALASM writes it into every source
HOBETA_HEADER = 17
SECTOR = 256
TEXT_COLUMNS = 64  # where ALASM ends a string, a comment or a line of plain text
MAX_PROBLEMS = 10  # told on stderr; ALASM's help texts saved as H files have thousands

CP866 = bytes(range(256)).decode('cp866')  # a byte's character
CONTROL_PICTURES = {c: 0x2400 + c for c in range(0x11, 0x20)}  # #11-#1F in UTF-8: U+2411-U+241F


class Error(Exception):
    pass


def render_line(line, where, problems):
    """A line of an 'H' file as text, as ALASM's str2txt (al0_44) makes it for its screen and its export. Its buffer
    is a 256-byte page whose low address byte is the column: past 255 the column wraps, and the export writes the
    line up to the wrapped column."""
    out = ''
    have_mnem = gap = False  # gap: the byte before was #FF or a run of spaces, a token stays where it is
    i, n = 0, len(line)
    while i < n:
        b = line[i]
        i += 1
        if b == 0xFF:
            gap = True
        elif b < 0x10:
            out += ' ' * (b or 256)
            gap = True
        elif b in (0x22, 0x3B, 0x10):
            # A string, a comment, or plain text (#10, not shown): raw bytes, those under #20 and #FF dropped, up
            # to TEXT_COLUMNS; a string's closing quote goes back to the tokens.
            if b != 0x10:
                out += chr(b)
            while len(out) % 256 != TEXT_COLUMNS and i < n:
                c = line[i]
                i += 1
                if c >= 0x20 and c != 0xFF:
                    out += CP866[c]
                    if b == 0x22 and c == 0x22:
                        break
            else:
                break
            gap = False
        elif b >= 0x80:
            if len(out) % 256 < MNEMONIC_COLUMN and not gap:
                out += ' ' * (MNEMONIC_COLUMN - len(out) % 256)
            if have_mnem:
                op = OPERANDS.get(b)
                if op is None:
                    problems.append('%s: unknown operand token #%02X' % (where, b))
                    op = '?%02X' % b
                out += op
            else:
                k = b - 0x80
                if k >= len(MNEMONICS):
                    problems.append('%s: unknown mnemonic token #%02X' % (where, b))
                out += (MNEMONICS[k] if k < len(MNEMONICS) else '?%02X' % b) + ' '
                have_mnem = True
            gap = False
        else:  # #11-#1F too, as they are
            out += CP866[b]
            gap = False
    return out[:len(out) % 256]


def convert(data, name):
    """The lines of an 'H' file as text, and the problems met (name is the file's, for them)."""
    lines, problems = [], []
    p = HEADER
    while True:
        if p >= len(data):
            problems.append('%s: no end marker' % name)
            break
        n = data[p]
        if n == 0:
            break
        end = min(p + n, len(data))
        lines.append(render_line(data[min(p + 1, end):end], '%s:%d' % (name, len(lines) + 1), problems))
        p += n
    if p + 1 != len(data):
        problems.append('%s: end marker at %d, file length %d' % (name, p, len(data)))
    return lines, problems


def text(lines, cp866):
    if cp866:
        return ''.join(line + '\r' for line in lines).encode('cp866')
    return ''.join(line.translate(CONTROL_PICTURES).rstrip() + '\n' for line in lines).encode('utf-8')


def hobeta_body(data):
    """A hobeta file's body, as long as its header says; None if the header's checksum is wrong."""
    if len(data) < HOBETA_HEADER or struct.unpack('<H', data[15:17])[0] != (105 + 257 * sum(data[:15])) & 0xFFFF:
        return None
    # Byte 14 is the sectors (13-14: their length in bytes); some old writers put the sectors in byte 13.
    if data[13] == 0:
        sectors = data[14]
    elif data[14] == 0:
        sectors = data[13]
    else:
        sectors = (data[13] + 256 * data[14] + SECTOR - 1) // SECTOR
    body = data[HOBETA_HEADER:HOBETA_HEADER + sectors * SECTOR] if sectors else data[HOBETA_HEADER:]
    length = struct.unpack('<H', data[11:13])[0]  # the file's length, as the catalog has it
    if (length + SECTOR - 1) // SECTOR == sectors:
        body = body[:length]
    return body


def source(data):
    """The 'H' file in data (as it is on the disk, or in a hobeta file), or None if it is no ALASM source."""
    body = hobeta_body(data)
    if body is not None and body[0x28:0x30] == SIGNATURE:
        return body
    if data[0x28:0x30] == SIGNATURE:
        return data
    return None


def write(path, data, force):
    try:
        with open(path, 'wb' if force else 'xb') as f:
            f.write(data)
    except FileExistsError:
        raise Error('%s exists (-f overwrites it)' % path)


def usage(out):
    out.write('usage: alasm [-cp866] [-f] INPUT [OUTPUT]\n'
              '  INPUT   an ALASM source: a TR-DOS file of type H, raw or hobeta ($H)\n'
              '  OUTPUT  the text; INPUT.asm by default, - for stdout\n'
              '  -cp866  cp866 with CR line ends, as ALASM exports (UTF-8 with LF by default)\n'
              '  -f      overwrite OUTPUT if it exists\n')


def main(argv):
    args = argv[1:]
    cp866 = force = False
    while args and args[0].startswith('-') and args[0] != '-':
        a = args.pop(0)
        if a in ('-cp866', '--cp866'):
            cp866 = True
        elif a == '-f':
            force = True
        elif a in ('-h', '--help'):
            usage(sys.stdout)
            return 0
        elif a == '--':
            break
        else:
            sys.stderr.write('alasm: unknown option %s\n' % a)
            usage(sys.stderr)
            return 2
    if len(args) not in (1, 2):
        usage(sys.stderr)
        return 2
    src = args[0]
    dest = args[1] if len(args) == 2 else os.path.splitext(src)[0] + '.asm'
    try:
        with open(src, 'rb') as f:
            data = f.read()
        h = source(data)
        if dest == '-':
            # Another file goes as it is. No problems told: mc would show them in a box over the text (where ?XX
            # shows them).
            sys.stdout.buffer.write(data if h is None else text(convert(h, src)[0], cp866))
            sys.stdout.flush()
            return 0
        if h is None:
            raise Error('%s: not an ALASM source (no ALASM header)' % src)
        lines, problems = convert(h, src)
        if os.path.exists(dest) and os.path.samefile(src, dest):
            raise Error('%s: the output would overwrite the input' % dest)
        write(dest, text(lines, cp866), force)
        for p in problems[:MAX_PROBLEMS]:
            sys.stderr.write('alasm: %s\n' % p)
        if len(problems) > MAX_PROBLEMS:
            sys.stderr.write('alasm: %s: %d problems in all\n' % (src, len(problems)))
    except BrokenPipeError:  # the reader of stdout has gone (mc's viewer closed early)
        os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
    except (Error, OSError) as e:
        if isinstance(e, OSError) and e.filename and e.strerror:
            e = '%s: %s' % (e.filename, e.strerror)
        sys.stderr.write('alasm: %s\n' % e)
        return 1
    return 0


if __name__ == '__main__':
    sys.exit(main(sys.argv))
