#!/usr/bin/env bash

# Check if a file was provided, otherwise default to a dash for stdin
if [ -n "$1" ]; then
    INPUT_SRC="$1"
else
    INPUT_SRC="-"
fi

# Use Python 3 to process and print everything safely as UTF-8 bytes
python3 -c '
import sys
import unicodedata

# Explicitly grab our argument target
target = sys.argv[1] if len(sys.argv) > 1 else "-"

# Read input from file or stdin with explicit UTF-8 handling
if target == "-":
    content = sys.stdin.buffer.read().decode("utf-8", errors="replace")
else:
    with open(target, "r", encoding="utf-8", errors="replace") as f:
        content = f.read()

# Loop through every character and extract Unicode properties
for char in content:
    code_point = "U+{:04X}".format(ord(char))
    
    try:
        name = unicodedata.name(char)
    except ValueError:
        category = unicodedata.category(char)
        if category == "Cc":
            name = "CONTROL CHARACTER (repr: {})".format(repr(char))
        else:
            name = "UNKNOWN CATEGORY: {}".format(category)

    # Format the display output line
    display_char = char if char not in "\n\r\t" else " "
    output_line = "{:<3} | {:<7} | {}\n".format(display_char, code_point, name)
    
    # CRITICAL FIX FOR ASCII TERMINALS:
    # Instead of print(), we convert the text line into raw UTF-8 bytes 
    # and force it straight into the stdout buffer. This completely bypasses the ASCII check.
    sys.stdout.buffer.write(output_line.encode("utf-8"))
' "$INPUT_SRC"
