# Depending on your decision:
# * Keep markdown files → use content_store.py directly, remove the Article SQLAlchemy model
# * Use the database → repurpose content_store.py as a seeder script that reads markdown files and imports them into the DB, then it moves to scripts/ or app/automation/


import os
import re
import time
from dataclasses import dataclass
from typing import Any, Optional

import yaml
import markdown as md
import bleach

from .models import ArticleMeta, Article, SectionId
from pathlib import Path

_IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp")

FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n(.*)$", re.S)

ALLOWED_TAGS = bleach.sanitizer.ALLOWED_TAGS.union({
    "p","h1","h2","h3","h4","h5","h6","pre","code","blockquote",
    "ul","ol","li","strong","em","hr","br","img","a",
    "table","thead","tbody","tr","th","td",
})
ALLOWED_ATTRS = {
    **bleach.sanitizer.ALLOWED_ATTRIBUTES,
    "a": ["href", "title", "rel", "target"],
    "img": ["src", "alt", "title", "loading"],
    "th": ["colspan", "rowspan"],
    "td": ["colspan", "rowspan"],
}
ALLOWED_PROTOCOLS = ["http", "https", "mailto"]

def _safe_cover_url(v) -> str | None:
    """Evita cosas raras tipo 'javascript:'; permitimos /... o http(s)://..."""
    if not v:
        return None
    s = str(v).strip()
    if s.startswith("/"):
        return s
    if s.startswith("https://") or s.startswith("http://"):
        return s
    return None

def _find_inline_image(content_dir:str, section:str, slug:str) -> str | None:
    """
    Busca:content/_images/<section>/<slug>-2.<ext>
    Devuelve URL pública: /api/images/<section>/<slug>-2.<ext>
    """
    base = Path(content_dir).resolve() / "_images" / section
    for ext in _IMAGE_EXTS:
        p = base / f"{slug}-2{ext}"
        if p.is_file():
            return f"/api/images/{section}/{slug}-2{ext}"
    return None

def _find_cover_image(content_dir:str, section:str, slug:str) -> str | None:
    """
    Busca:content/_images/<section>/<slug>.<ext>
    Devuelve URL pública: /api/images/<section>/<slug>.<ext>
    """
    base = Path(content_dir).resolve() / "_images" / section
    for ext in _IMAGE_EXTS:
        p = base / f"{slug}{ext}"
        if p.is_file():
            return f"/api/images/{section}/{slug}{ext}"
    return None


def _sanitize_html(html:str) -> str:
    cleaned = bleach.clean(
        html,
        tags=ALLOWED_TAGS,
        attributes=ALLOWED_ATTRS,
        protocols=ALLOWED_PROTOCOLS,
        strip=True,
    )
    # Link hardening
    cleaned = bleach.linkify(cleaned, callbacks=[bleach.callbacks.nofollow, bleach.callbacks.target_blank])
    return cleaned

def _parse_frontmatter(raw:str) -> tuple[dict[str, Any], str]:
    m = FRONTMATTER_RE.match(raw)
    if not m:
        return {}, raw
    fm_text, body = m.group(1), m.group(2)
    data = yaml.safe_load(fm_text) or {}
    if not isinstance(data, dict):
        data = {}
    return data, body

def _excerpt_from_text(text:str, limit:int = 180) -> str:
    t = re.sub(r"\s+", " ", text).strip()
    return (t[:limit] + "…") if len(t) > limit else t

def _safe_section(section:str) -> Optional[SectionId]:
    allowed = {
        "nacionales","internacionales","opiniones","ciencias","criptos","trading",
        "futurologia","filosofia","entretenimiento","curiosidades","salud"
    }
    return section if section in allowed else None

@dataclass
class CacheState:
    stamp:float
    max_mtime:float
    articles:list[Article]

class ContentStore:
    def __init__(self, content_dir:str):
        self.content_dir = os.path.abspath(content_dir)
        self._cache: Optional[CacheState] = None

    def _compute_max_mtime(self) -> float:
        max_m = 0.0
        for root, _, files in os.walk(self.content_dir):
            for fn in files:
                low = fn.lower()
                if low.endswith(".md") or low.endswith(_IMAGE_EXTS):
                    p = os.path.join(root, fn)
                    try:
                        max_m = max(max_m, os.path.getmtime(p))
                    except FileNotFoundError:
                        pass
        return max_m

    def _load_all(self) -> list[Article]:
        out:list[Article] = []

        for section in os.listdir(self.content_dir):
            sec_path = os.path.join(self.content_dir, section)
            if not os.path.isdir(sec_path):
                continue

            sec_id = _safe_section(section)
            if not sec_id:
                continue

            for fn in os.listdir(sec_path):
                if not fn.lower().endswith(".md"):
                    continue
                path = os.path.join(sec_path, fn)
                slug = os.path.splitext(fn)[0]

                with open(path, "r", encoding="utf-8") as f:
                    raw = f.read()

                fm, body = _parse_frontmatter(raw)

                title = str(fm.get("title") or slug.replace("-", " ").title())
                subtitle = fm.get("subtitle")
                author = fm.get("author")
                date = str(fm.get("date") or time.strftime("%Y-%m-%d", time.gmtime(os.path.getmtime(path))))
                tags = fm.get("tags") or []
                if not isinstance(tags, list):
                    tags = []
                tags = [str(t) for t in tags][:20]

                raw_cover = fm.get("cover_image")
                cover_image = _safe_cover_url(raw_cover) or _find_cover_image(self.content_dir, section, slug)
                raw_inline = fm.get("inline_image")
                inline_image = _safe_cover_url(raw_inline) or _find_inline_image(self.content_dir, section, slug)
                excerpt = fm.get("excerpt")
                if not excerpt:
                    excerpt = _excerpt_from_text(re.sub(r"[#*_>`~\[\]\(\)]", " ", body))

                html = md.markdown(
                    body,
                    extensions=["extra", "codehilite", "tables", "fenced_code", "toc"],
                    output_format="html5",
                )
                html = _sanitize_html(html)

                meta = ArticleMeta(
                    slug=slug,
                    section=sec_id,
                    title=title,
                    subtitle=str(subtitle) if subtitle else None,
                    author=str(author) if author else None,
                    date=date,
                    tags=tags,
                    excerpt=str(excerpt) if excerpt else None,
                    cover_image=str(cover_image) if cover_image else None,
                    inline_image=str(inline_image) if inline_image else None,   # <---
                )
                out.append(Article(**meta.model_dump(), body_html=html, markdown_path=path))

        # Orden descendente por date (ISO strings funcionan ok si usás YYYY-MM-DD o ISO datetime)
        out.sort(key=lambda a: a.date, reverse=True)
        return out

    def _refresh_if_needed(self):
        max_mtime = self._compute_max_mtime()
        if self._cache and self._cache.max_mtime == max_mtime:
            return
        articles = self._load_all()
        self._cache = CacheState(stamp=time.time(), max_mtime=max_mtime, articles=articles)

    def list_articles(self, section:Optional[str], q:Optional[str], limit:int, offset:int) -> list[ArticleMeta]:
        self._refresh_if_needed()
        assert self._cache

        items = self._cache.articles

        if section:
            sec = _safe_section(section)
            if not sec:
                return []
            items = [a for a in items if a.section == sec]

        if q:
            qq = q.lower().strip()
            items = [
                a for a in items
                if qq in a.title.lower()
                or (a.subtitle or "").lower().find(qq) >= 0
                or " ".join(a.tags).lower().find(qq) >= 0
                or (a.excerpt or "").lower().find(qq) >= 0
            ]

        sliced = items[offset: offset + limit]
        return [ArticleMeta(**a.model_dump(exclude={"body_html","markdown_path"})) for a in sliced]

    def get_article(self, slug:str) -> Optional[Article]:
        self._refresh_if_needed()
        assert self._cache
        for a in self._cache.articles:
            if a.slug == slug:
                return a
        return None
