import re
from typing import List

MACHINE_CODE_RE = re.compile(r'MHA-[0-9a-fA-F]{8}')
KEYWORDS = [
    'computer_name', 'machine_code', 'processor', 'ram', 'storage',
    'backup_status', 'assigned_to', 'machine_location'
]


def strip_html_tags(html: str) -> str:
    """Remove HTML tags and collapse whitespace."""
    # remove scripts/styles first
    html = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', html, flags=re.S|re.I)
    text = re.sub(r'<[^>]+>', '', html)
    text = re.sub(r'\s+', ' ', text).strip()
    return text


def extract_relevant_lines(text: str) -> List[str]:
    """Return lines that look like machine info or contain keywords."""
    lines = [l.strip() for l in re.split(r'[\n\r]+', text) if l.strip()]
    relevant = []
    for line in lines:
        if MACHINE_CODE_RE.search(line):
            relevant.append(line)
            continue
        low = line.lower()
        if any(k in low for k in KEYWORDS):
            relevant.append(line)
            continue
        # quick heuristics for specs
        if re.search(r"\b\d+GB\b", line, flags=re.I) or re.search(r"nvme|ssd|hdd", line, flags=re.I):
            relevant.append(line)
            continue
    return relevant


def extract_from_tables(html: str) -> List[str]:
    """Parse HTML tables and return rows as pipe-separated cell text.

    This is a heuristic parser (no external deps) intended to pull machine
    rows from Jinja-rendered tables in templates/machines.html and index.html.
    """
    rows = []
    # find table blocks
    tables = re.findall(r'<table[^>]*>(.*?)</table>', html, flags=re.S|re.I)
    for table in tables:
        # find tr blocks
        trs = re.findall(r'<tr[^>]*>(.*?)</tr>', table, flags=re.S|re.I)
        for tr in trs:
            # extract th/td cells
            cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]>', tr, flags=re.S|re.I)
            if not cells:
                continue
            clean_cells = []
            for c in cells:
                # strip inner tags
                t = re.sub(r'<[^>]+>', '', c)
                t = re.sub(r'\s+', ' ', t).strip()
                clean_cells.append(t)
            # join with separator for easy scanning
            row = ' | '.join(clean_cells)
            rows.append(row)
    return rows


def sanitize_for_ai(html: str, max_chars: int = 2000) -> str:
    """Produce a sanitized, minimal text payload from HTML that only contains
    likely machine names, codes and specs for passing to an AI.

    If no clear machine lines are found, return a truncated plain-text preview.
    """
    # First try to extract table rows from the raw HTML (table rows are
    # the most likely place machine codes and specs appear in the templates).
    table_lines = extract_from_tables(html)
    if table_lines:
        # Filter table lines for relevance
        relevant_table = [l for l in table_lines if MACHINE_CODE_RE.search(l) or any(k in l.lower() for k in KEYWORDS) or re.search(r"\b\d+GB\b", l, flags=re.I) or re.search(r"nvme|ssd|hdd", l, flags=re.I)]
        if relevant_table:
            out = '\n'.join(relevant_table)
            return out[:max_chars]

    text = strip_html_tags(html)
    relevant = extract_relevant_lines(text)
    if relevant:
        out = '\n'.join(relevant)
    else:
        out = text[:max_chars]
    return out


def find_machine_code(text: str) -> str:
    """Return first machine code found in text or empty string."""
    m = MACHINE_CODE_RE.search(text)
    return m.group(0) if m else ''
