from __future__ import annotations

import io
import re
import zipfile
from datetime import UTC, date, datetime
from pathlib import Path

import pandas as pd

from psx_signal.data.artifacts import MarketDataArtifact
from psx_signal.data.schema import CANONICAL_EQUITY_COLUMNS, CANONICAL_INDEX_COLUMNS, NUMERIC_COLUMNS


ALIASES = {
    "symbol": {"symbol", "scrip", "ticker", "security", "securitycode"},
    "date": {"date", "tradingdate", "trade_date"},
    "open": {"open", "openprice"},
    "high": {"high", "highprice"},
    "low": {"low", "lowprice"},
    "close": {"close", "closingprice", "current", "price"},
    "previous_close": {"previousclose", "prevclose", "ldcp"},
    "volume": {"volume", "turnover", "shares"},
    "trades": {"trades", "nooftrades", "tradecount"},
    "value_traded": {"valuetraded", "value", "turnovervalue"},
    "market_cap": {"marketcap", "marketcapitalization"},
    "sector": {"sector", "sectorname"},
    "company_name": {"companyname", "company", "name"},
    "index": {"index", "indexname"},
}


def _token(value: object) -> str:
    return re.sub(r"[^a-z0-9_]", "", str(value).strip().lower())


def _column_map(columns: pd.Index) -> dict[object, str]:
    result: dict[object, str] = {}
    for column in columns:
        token = _token(column)
        for canonical, aliases in ALIASES.items():
            if token in aliases:
                result[column] = canonical
                break
    return result


def read_artifact(artifact: MarketDataArtifact) -> pd.DataFrame:
    suffix = artifact.path.suffix.lower()
    if suffix == ".zip":
        with zipfile.ZipFile(artifact.path) as archive:
            members = [name for name in archive.namelist() if Path(name).suffix.lower() in {".csv", ".txt"}]
            if len(members) != 1:
                raise ValueError(f"Expected one CSV/TXT in {artifact.path.name}; found {len(members)}")
            with archive.open(members[0]) as handle:
                content = handle.read()
        return pd.read_csv(io.BytesIO(content), sep=None, engine="python")
    return pd.read_csv(artifact.path, sep=None, engine="python")


def normalize_equities(artifact: MarketDataArtifact, raw_reference: str) -> pd.DataFrame:
    raw = read_artifact(artifact)
    frame = raw.rename(columns=_column_map(raw.columns))
    required_without_date = {"symbol", "open", "high", "low", "close", "volume"}
    missing = required_without_date - set(frame.columns)
    if missing:
        raise ValueError(f"Missing equity columns in {artifact.path.name}: {', '.join(sorted(missing))}")
    if "date" not in frame:
        if artifact.trading_date is None:
            raise ValueError(f"No date column and no date in filename: {artifact.path.name}")
        frame["date"] = artifact.trading_date
    frame["symbol"] = frame["symbol"].astype("string").str.strip().str.upper()
    frame["date"] = pd.to_datetime(frame["date"], errors="raise").dt.normalize()
    for column in set(NUMERIC_COLUMNS).intersection(frame.columns):
        frame[column] = pd.to_numeric(frame[column].astype("string").str.replace(",", "", regex=False), errors="coerce")
    for column in ("adjusted_open", "adjusted_high", "adjusted_low", "adjusted_close", "adjusted_volume", "adjustment_factor"):
        if column not in frame:
            frame[column] = pd.NA
    frame["is_adjusted"] = False
    frame["provider"] = artifact.provider
    frame["fetched_at"] = datetime.now(UTC).isoformat()
    frame["source_reference"] = artifact.source_reference
    frame["raw_artifact"] = raw_reference
    for column in CANONICAL_EQUITY_COLUMNS:
        if column not in frame:
            frame[column] = pd.NA
    return frame[list(CANONICAL_EQUITY_COLUMNS)].sort_values(["date", "symbol"]).reset_index(drop=True)


def normalize_indices(artifact: MarketDataArtifact, raw_reference: str, default_index: str = "KSE100") -> pd.DataFrame:
    raw = read_artifact(artifact)
    frame = raw.rename(columns=_column_map(raw.columns))
    if "date" not in frame:
        if artifact.trading_date is None:
            raise ValueError(f"No date column and no date in filename: {artifact.path.name}")
        frame["date"] = artifact.trading_date
    if "index" not in frame:
        if "symbol" in frame:
            frame["index"] = frame["symbol"]
        else:
            frame["index"] = default_index
    if "close" not in frame:
        raise ValueError(f"Missing index close in {artifact.path.name}")
    frame["date"] = pd.to_datetime(frame["date"], errors="raise").dt.normalize()
    frame["index"] = frame["index"].astype("string").str.strip().str.upper()
    for column in ("open", "high", "low", "close", "volume"):
        if column not in frame:
            frame[column] = pd.NA
        frame[column] = pd.to_numeric(frame[column].astype("string").str.replace(",", "", regex=False), errors="coerce")
    frame["provider"] = artifact.provider
    frame["fetched_at"] = datetime.now(UTC).isoformat()
    frame["source_reference"] = artifact.source_reference
    frame["raw_artifact"] = raw_reference
    return frame[list(CANONICAL_INDEX_COLUMNS)].sort_values(["date", "index"]).reset_index(drop=True)
