"""Decide which S3 folder (key prefix) a local file belongs in.

The decision uses three signals, cheapest first:
  1. file extension  -> narrows to document / image / raw media
  2. filename tokens -> weighted heavily, they are usually deliberate
  3. sniffed content -> weighted lightly, for files whose name says nothing

Every result carries a `confident` flag. Callers must confirm with a human
before uploading anything that is not confident.
"""

from __future__ import annotations

import mimetypes
import re
from dataclasses import dataclass, field
from pathlib import Path

from s3_config import (
    CONTENT_SNIFF_BYTES,
    CONTENT_TYPES,
    DOCUMENT_EXTENSIONS,
    FALLBACK_CONTENT_TYPE,
    IMAGES_PREFIX,
    IMAGE_EXTENSIONS,
    MEDIA_EXTENSIONS,
    OTHER_PREFIX,
    PDF_SNIFF_PAGES,
    PERMITS_PREFIX,
    PERMIT_KEYWORDS,
    RAW_MEDIA_PREFIX,
    REQUIREMENTS_PREFIX,
    REQUIREMENTS_KEYWORDS,
    SCREENSHOTS_PREFIX,
    SCREENSHOT_TOKENS,
    TEXT_EXTENSIONS,
    TRANSCRIPTS_PREFIX,
    TRANSCRIPT_KEYWORDS,
)

FILENAME_WEIGHT = 3
CONTENT_WEIGHT = 1

# A result is confident when the winner clears this score and beats the
# runner-up by this margin. Otherwise we ask.
MIN_WINNING_SCORE = 3
MIN_MARGIN = 2

# "10:04:37" or "00:12" at the start of a line — the shape of a transcript.
_TIMESTAMP_LINE = re.compile(r"^\s*\[?\d{1,2}:\d{2}(:\d{2})?\]?\s", re.MULTILINE)


@dataclass(frozen=True)
class Classification:
    """Where a file should go, and how sure we are."""

    path: Path
    prefix: str
    content_type: str
    confident: bool
    reason: str
    scores: dict[str, int] = field(default_factory=dict)

    @property
    def key(self) -> str:
        """Full S3 key: prefix + filename."""
        return f"{self.prefix}{self.path.name}"

    @property
    def alternatives(self) -> list[str]:
        """Other plausible prefixes, best first, excluding the chosen one."""
        ranked = sorted(self.scores.items(), key=lambda kv: kv[1], reverse=True)
        return [p for p, score in ranked if p != self.prefix and score > 0]


# --- Content type -----------------------------------------------------------


def content_type_for(path: Path) -> str:
    """Correct Content-Type for the upload, from the extension."""
    explicit = CONTENT_TYPES.get(path.suffix.lower())
    if explicit:
        return explicit
    guessed, _ = mimetypes.guess_type(path.name)
    return guessed or FALLBACK_CONTENT_TYPE


# --- Content sniffing -------------------------------------------------------


def _sniff_text(path: Path) -> str:
    """Lowercased sample of the file's text, or '' when it cannot be read as text."""
    suffix = path.suffix.lower()

    if suffix == ".pdf":
        return _sniff_pdf(path).lower()

    if suffix in TEXT_EXTENSIONS:
        try:
            with path.open("rb") as handle:
                return handle.read(CONTENT_SNIFF_BYTES).decode("utf-8", "ignore").lower()
        except OSError:
            return ""

    return ""


def _sniff_pdf(path: Path) -> str:
    """Text from the first few PDF pages.

    Uses pypdf when installed. Without it, falls back to scanning raw bytes —
    that only catches text stored uncompressed, so classification of PDFs
    leans on the filename unless pypdf is available.
    """
    try:
        from pypdf import PdfReader  # type: ignore[import-not-found]
    except ImportError:
        try:
            with path.open("rb") as handle:
                return handle.read(CONTENT_SNIFF_BYTES).decode("latin-1", "ignore")
        except OSError:
            return ""

    try:
        reader = PdfReader(str(path))
        pages = reader.pages[:PDF_SNIFF_PAGES]
        return "\n".join(page.extract_text() or "" for page in pages)
    except Exception:
        # A malformed or encrypted PDF must not break the upload path; the
        # filename signal still applies.
        return ""


# --- Scoring ----------------------------------------------------------------


def _score(keywords: tuple[str, ...], haystack_name: str, haystack_content: str) -> int:
    total = 0
    for keyword in keywords:
        if keyword in haystack_name:
            total += FILENAME_WEIGHT
        elif keyword in haystack_content:
            total += CONTENT_WEIGHT
    return total


def _classify_document(path: Path, name: str) -> tuple[str, dict[str, int], str]:
    content = _sniff_text(path)

    scores = {
        PERMITS_PREFIX: _score(PERMIT_KEYWORDS, name, content),
        TRANSCRIPTS_PREFIX: _score(TRANSCRIPT_KEYWORDS, name, content),
        REQUIREMENTS_PREFIX: _score(REQUIREMENTS_KEYWORDS, name, content),
    }

    # Timestamped lines are a strong structural tell for a transcript.
    if content and len(_TIMESTAMP_LINE.findall(content)) >= 5:
        scores[TRANSCRIPTS_PREFIX] += FILENAME_WEIGHT

    winner = max(scores, key=lambda prefix: scores[prefix])
    sniffed = "filename and content" if content else "filename only"
    reason = f"{path.suffix.lower() or 'file'} scored on {sniffed}: " + ", ".join(
        f"{prefix.rstrip('/').split('/')[-1]}={scores[prefix]}" for prefix in scores
    )
    return winner, scores, reason


def _classify_image(path: Path, name: str) -> tuple[str, dict[str, int], str, bool]:
    hits = [token for token in SCREENSHOT_TOKENS if token in name]
    if hits:
        return (
            SCREENSHOTS_PREFIX,
            {SCREENSHOTS_PREFIX: FILENAME_WEIGHT * len(hits)},
            f"filename contains {hits[0]!r}",
            True,
        )

    if path.suffix.lower() == ".png":
        # PNG with no naming signal is genuinely ambiguous — screenshots and
        # exported reference visuals both land here. Ask.
        return (
            IMAGES_PREFIX,
            {IMAGES_PREFIX: 1, SCREENSHOTS_PREFIX: 1},
            "PNG with no screenshot marker in the filename — could be either",
            False,
        )

    return (
        IMAGES_PREFIX,
        {IMAGES_PREFIX: FILENAME_WEIGHT},
        "image with no screenshot marker in the filename",
        True,
    )


# --- Entry point ------------------------------------------------------------


def classify_file(path: str | Path) -> Classification:
    """Work out the S3 prefix and Content-Type for a local file."""
    path = Path(path)
    if not path.is_file():
        raise FileNotFoundError(f"Not a file: {path}")

    name = path.name.lower()
    suffix = path.suffix.lower()
    content_type = content_type_for(path)

    if suffix in MEDIA_EXTENSIONS:
        return Classification(
            path=path,
            prefix=RAW_MEDIA_PREFIX,
            content_type=content_type,
            confident=True,
            reason=f"raw {content_type.split('/')[0]} source file",
            scores={RAW_MEDIA_PREFIX: FILENAME_WEIGHT},
        )

    if suffix in IMAGE_EXTENSIONS:
        prefix, scores, reason, confident = _classify_image(path, name)
        return Classification(
            path=path,
            prefix=prefix,
            content_type=content_type,
            confident=confident,
            reason=reason,
            scores=scores,
        )

    if suffix in DOCUMENT_EXTENSIONS:
        prefix, scores, reason = _classify_document(path, name)
        ranked = sorted(scores.values(), reverse=True)
        runner_up = ranked[1] if len(ranked) > 1 else 0
        confident = ranked[0] >= MIN_WINNING_SCORE and (ranked[0] - runner_up) >= MIN_MARGIN
        return Classification(
            path=path,
            prefix=prefix,
            content_type=content_type,
            confident=confident,
            reason=reason,
            scores=scores,
        )

    return Classification(
        path=path,
        prefix=OTHER_PREFIX,
        content_type=content_type,
        confident=False,
        reason=f"unrecognised type {suffix or '(no extension)'} — no folder rule covers it",
        scores={OTHER_PREFIX: 0},
    )
