"""
ExploitDB client — downloads and parses exploit descriptions from CSV.

Source: GitLab mirror of files_exploits.csv
Only embeds descriptions (title, platform, type, CVE mapping) — not source code.
"""

import csv
import io
import logging
import re
from pathlib import Path

from knowledge_base.chunking import ChunkStrategy
from knowledge_base.curation.base_client import BaseClient
from knowledge_base.curation.pins import PinMismatchError, get_feed_ref, verify_sha256
from knowledge_base.curation.safe_http import MAX_EXPLOITDB_CSV_BYTES, safe_get

logger = logging.getLogger(__name__)

# T15: pinned to an immutable commit (see pins.py) instead of the mutable
# ``main`` branch head. GitLab serves a raw blob at a fixed commit.
EXPLOITDB_CSV_URL_TEMPLATE = (
    "https://gitlab.com/exploit-database/exploitdb/-/raw/{ref}/files_exploits.csv"
)


class ExploitDBClient(BaseClient):
    """Fetches and parses ExploitDB exploit descriptions."""

    SOURCE = "exploitdb"
    NODE_LABEL = "ExploitDBChunk"

    def __init__(self, cache_dir: str = None):
        self.cache_dir = Path(cache_dir) if cache_dir else (
            Path(__file__).parent.parent / "data" / "cache" / "exploitdb"
        )

    def fetch(self, **kwargs) -> list[dict]:
        """Download files_exploits.csv from GitLab mirror.

        Returns list of dicts: [{id, description, date_published, platform, type, port}]
        """
        self.cache_dir.mkdir(parents=True, exist_ok=True)
        cache_file = self.cache_dir / "files_exploits.csv"

        # Download CSV
        # Sec #3/#6: hostname-allowlisted + size-capped via safe_get.
        # CSV is ~50MB today; the 200MB cap leaves ample headroom.
        url = EXPLOITDB_CSV_URL_TEMPLATE.format(ref=get_feed_ref(self.SOURCE))
        try:
            resp = safe_get(
                url, timeout=60, max_bytes=MAX_EXPLOITDB_CSV_BYTES
            )
            resp.raise_for_status()
            # T15: integrity check before we trust the bytes. A mismatch is a
            # security event — abort the feed, never fall back to a cache.
            verify_sha256(self.SOURCE, resp.content)
            csv_text = resp.text
            cache_file.write_text(csv_text, encoding="utf-8")
            logger.info(f"Downloaded ExploitDB CSV ({len(csv_text)} chars)")
        except PinMismatchError:
            raise
        except Exception as e:
            logger.warning(f"Failed to download ExploitDB CSV: {e}")
            if cache_file.exists():
                csv_text = cache_file.read_text(encoding="utf-8")
                logger.info("Using cached ExploitDB CSV")
            else:
                return []

        # Sanitize unusual line terminators (U+2028 Line Separator, U+2029 Paragraph Separator)
        csv_text = csv_text.replace("\u2028", " ").replace("\u2029", " ")

        # Parse CSV
        results = []
        reader = csv.DictReader(io.StringIO(csv_text))
        for row in reader:
            edb_id = row.get("id", "").strip()
            description = row.get("description", "").strip()
            if not edb_id or not description:
                continue

            results.append({
                "id": edb_id,
                "description": description,
                "date_published": row.get("date_published", "").strip(),
                "platform": row.get("platform", "").strip(),
                "type": row.get("type", "").strip(),
                "port": row.get("port", "").strip(),
                # `codes` is a semicolon-delimited list of external IDs
                # for this exploit (CVE-*, OSVDB-*, BID-*, etc.). The CVE
                # entries here are the authoritative cross-reference;
                # parsing CVEs from `description` is at best a fallback.
                "codes": row.get("codes", "").strip(),
            })

        logger.info(f"Parsed {len(results)} exploits from ExploitDB CSV")
        return results

    def to_chunks(self, raw_data: list[dict]) -> list[dict]:
        """Convert ExploitDB entries to chunks. One chunk per exploit."""
        chunks = []
        for entry in raw_data:
            edb_id = entry["id"]
            description = entry["description"]
            platform = entry.get("platform", "")
            exploit_type = entry.get("type", "")
            date_published = entry.get("date_published", "")

            # Parse the `codes` column from the CSV. Format is a
            # semicolon-delimited list of external identifiers, e.g.
            # "CVE-2003-0087;OSVDB-7996;BID-6840". Mixed prefixes are
            # preserved as-is — CVE-*, OSVDB-*, BID-*, etc.
            codes_raw = entry.get("codes", "")
            codes: list[str] = []
            for token in codes_raw.split(";"):
                token = token.strip()
                if token and token not in codes:
                    codes.append(token)

            # Pull out CVE entries for the indexed `cve_id` property.
            # Falls back to scanning the description if `codes` had none —
            # rare but does happen.
            cve_ids = [c.upper() for c in codes if c.upper().startswith("CVE-")]
            if not cve_ids:
                for m in re.findall(r"CVE-\d{4}-\d+", description, re.IGNORECASE):
                    m_up = m.upper()
                    if m_up not in cve_ids:
                        cve_ids.append(m_up)
                # If we found CVEs in the description but not codes, fold
                # them into the codes list too so the field stays complete.
                for cve in cve_ids:
                    if cve not in codes:
                        codes.append(cve)

            cve_id = cve_ids[0] if cve_ids else None

            # edb_id is already in `title` and as a top-level node property
            # — no need to repeat it as a content prefix.
            content = description
            if platform:
                content += f" | Platform: {platform}"
            if exploit_type:
                content += f" | Type: {exploit_type}"
            if date_published:
                content += f" | Published: {date_published}"
            if codes:
                content += f" | Codes: {', '.join(codes)}"

            chunk_id = ChunkStrategy.generate_chunk_id(self.SOURCE, edb_id)
            chunks.append({
                "chunk_id": chunk_id,
                "content": content,
                "title": f"EDB-{edb_id}: {description[:80]}",
                "source": self.SOURCE,
                "edb_id": edb_id,
                "cve_id": cve_id,        # primary CVE, indexed (edb_cve)
                "codes": codes,          # full list of external IDs (CVE/OSVDB/BID/...)
                "platform": platform,
                "exploit_type": exploit_type,
                "published_date": date_published or None,
                # All ExploitDB chunks share a single CSV file. Loading
                # the full document is heavy (~25 MB) and rarely useful
                # since the chunk content is already a verbatim row, but
                # the field is set for schema uniformity across sources.
                "source_path": "knowledge_base/data/cache/exploitdb/files_exploits.csv",
            })

        logger.info(f"Created {len(chunks)} ExploitDB chunks")
        return chunks
