"""Static configuration for the heavy-haul-agent S3 integration.

Nothing secret lives here. Credentials are read at runtime from the
environment (AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY / AWS_DEFAULT_REGION)
or from ~/.aws/credentials by boto3's normal provider chain.
"""

from __future__ import annotations

import os

# --- Bucket -----------------------------------------------------------------

DEFAULT_BUCKET = "heavy-haul-agent"
DEFAULT_REGION = "us-east-1"


def bucket_name() -> str:
    """Bucket to operate on; override with S3_BUCKET_NAME."""
    return os.environ.get("S3_BUCKET_NAME") or DEFAULT_BUCKET


def region_name() -> str:
    """Region to operate in; override with AWS_DEFAULT_REGION."""
    return os.environ.get("AWS_DEFAULT_REGION") or DEFAULT_REGION


# --- Key prefixes (the "folders") -------------------------------------------

REQUIREMENTS_PREFIX = "documents/requirements/"
TRANSCRIPTS_PREFIX = "documents/transcripts/"
PERMITS_PREFIX = "documents/permits/"
SCREENSHOTS_PREFIX = "images/screenshots/"
IMAGES_PREFIX = "images/general/"
RAW_MEDIA_PREFIX = "raw-media/"
OTHER_PREFIX = "other/"

KNOWN_PREFIXES = (
    REQUIREMENTS_PREFIX,
    TRANSCRIPTS_PREFIX,
    PERMITS_PREFIX,
    SCREENSHOTS_PREFIX,
    IMAGES_PREFIX,
    RAW_MEDIA_PREFIX,
    OTHER_PREFIX,
)

# --- Raw media lifecycle ----------------------------------------------------

RAW_MEDIA_EXPIRY_DAYS = 90

RAW_MEDIA_LIFECYCLE_NOTE = (
    f"{RAW_MEDIA_PREFIX} holds raw audio/video source files. These are large and "
    f"are meant to be transient: put a lifecycle rule on the bucket that expires "
    f"objects under {RAW_MEDIA_PREFIX} after {RAW_MEDIA_EXPIRY_DAYS} days. "
    f"This code never applies that rule for you — run `python cli.py lifecycle` "
    f"to print it."
)


def raw_media_lifecycle_rule() -> dict:
    """The lifecycle rule that should be applied to the bucket (never auto-applied)."""
    return {
        "Rules": [
            {
                "ID": "expire-raw-media-after-90-days",
                "Status": "Enabled",
                "Filter": {"Prefix": RAW_MEDIA_PREFIX},
                "Expiration": {"Days": RAW_MEDIA_EXPIRY_DAYS},
                "AbortIncompleteMultipartUpload": {"DaysAfterInitiation": 7},
            }
        ]
    }


# --- Content types ----------------------------------------------------------

CONTENT_TYPES = {
    # documents
    ".pdf": "application/pdf",
    ".txt": "text/plain; charset=utf-8",
    ".md": "text/markdown; charset=utf-8",
    ".csv": "text/csv; charset=utf-8",
    ".json": "application/json",
    ".sql": "application/sql",
    ".vtt": "text/vtt",
    ".srt": "application/x-subrip",
    ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
    # images
    ".jpg": "image/jpeg",
    ".jpeg": "image/jpeg",
    ".png": "image/png",
    ".gif": "image/gif",
    ".webp": "image/webp",
    ".svg": "image/svg+xml",
    ".heic": "image/heic",
    ".tiff": "image/tiff",
    ".tif": "image/tiff",
    # audio
    ".mp3": "audio/mpeg",
    ".m4a": "audio/mp4",
    ".wav": "audio/wav",
    ".aac": "audio/aac",
    ".flac": "audio/flac",
    ".ogg": "audio/ogg",
    # video
    ".mp4": "video/mp4",
    ".mov": "video/quicktime",
    ".m4v": "video/x-m4v",
    ".avi": "video/x-msvideo",
    ".mkv": "video/x-matroska",
    ".webm": "video/webm",
}

FALLBACK_CONTENT_TYPE = "application/octet-stream"

IMAGE_EXTENSIONS = frozenset(
    ext for ext, ct in CONTENT_TYPES.items() if ct.startswith("image/")
)
MEDIA_EXTENSIONS = frozenset(
    ext
    for ext, ct in CONTENT_TYPES.items()
    if ct.startswith("audio/") or ct.startswith("video/")
)
TEXT_EXTENSIONS = frozenset({".txt", ".md", ".csv", ".json", ".vtt", ".srt"})
DOCUMENT_EXTENSIONS = TEXT_EXTENSIONS | {".pdf", ".docx"}

# --- Classification keywords ------------------------------------------------
# Matched case-insensitively against the filename (weighted heavily) and
# against sniffed file content (weighted lightly).

PERMIT_KEYWORDS = (
    "permit",
    "license",
    "licence",
    "compliance",
    "oversize",
    "overweight",
    "escort",
    "certificate of insurance",
    "certificate",
    "authorization",
    "authorisation",
    "bond",
    "dot number",
    "mc number",
    "regulation",
    "inspection",
    "order details",
)

TRANSCRIPT_KEYWORDS = (
    "transcript",
    "meeting room",
    "meeting notes",
    "recording",
    "zoom",
    "webinar",
    "call notes",
    "speaker 1",
    "speaker 2",
    "participants:",
)

REQUIREMENTS_KEYWORDS = (
    "requirement",
    "spec",
    "specification",
    "architecture",
    "vision",
    "implementation",
    "task",
    "scope",
    "user story",
    "acceptance criteria",
    "design doc",
    "proposal",
    "rfc",
    "roadmap",
    "milestone",
    "deliverable",
)

SCREENSHOT_TOKENS = (
    "screenshot",
    "screen shot",
    "screen_shot",
    "screencapture",
    "screen capture",
    "screen-capture",
    "capture",
    "snip",
    "grab",
    "mockup",
    "wireframe",
    "preview",
    "ui-",
    "test-",
)

# How much of a file to sniff when looking for content keywords.
CONTENT_SNIFF_BYTES = 200_000
PDF_SNIFF_PAGES = 5
