from __future__ import annotations

from collections import defaultdict
from pathlib import Path
import csv
import os
import re
from typing import Any

from .cache import get_or_fetch, read_cache
from .config import CACHE_DIR, CACHE_TTL_DAYS
from .csv_filter import download_filter_csv
from .datagouv import get_dataset, resources_matching
from .geo_api import agency_epci_codes, agency_scope
from .http_client import ExternalApiError, get_json

DATASET_SLUG = os.environ.get("CONSO_ENAF_DATASET_SLUG", "consommation-despaces-naturels-agricoles-et-forestiers")
LEGACY_DATASET_SLUG = "consommation-despaces-naturels-agricoles-et-forestiers-du-1er-janvier-2009-au-1er-janvier-2024"

# API Donnees foncieres / Cerema. La documentation publique expose notamment
# /indicateurs/conso_espace/communes/{code_insee}. On conserve la base en variable
# car l'URL publique peut evoluer.
APIFONCIER_BASE_URL = os.environ.get("APIFONCIER_BASE_URL", "https://apidf-preprod.cerema.fr").rstrip("/")

# Couche ArcGIS utilisee par certains visualiseurs du portail. Elle est facultative :
# le connecteur fonctionne aussi avec les CSV data.gouv mis en cache.
CEREMA_FEATURESERVER_LAYER_URL = os.environ.get(
    "CEREMA_ARTIF_FEATURESERVER_LAYER_URL",
    "https://services.arcgis.com/d3voDfTFbHOCRwVR/ArcGIS/rest/services/Consommation_d_espaces_naturels__agricoles_et_forestiers/FeatureServer/0",
).rstrip("/")

CONSO_YEAR_MIN = int(os.environ.get("CONSO_ENAF_YEAR_MIN", "2009"))
CONSO_YEAR_MAX = int(os.environ.get("CONSO_ENAF_YEAR_MAX", "2024"))
CEREMA_TIMEOUT = float(os.environ.get("CEREMA_ARTIF_TIMEOUT", "20"))
CEREMA_PAGE_SIZE = int(os.environ.get("CEREMA_ARTIF_PAGE_SIZE", "2000"))
TTL_SECONDS = CACHE_TTL_DAYS * 24 * 3600

CODE_COLUMNS_COMMUNE = ["idcom", "code_commune", "codgeo", "code_geo", "code", "insee", "Code commune", "CODGEO"]
CODE_COLUMNS_EPCI = ["epci24", "code_epci", "siren_epci", "epci", "code", "siren", "Code EPCI", "EPCI"]

# Champs de la couche FeatureServer si elle est disponible.
CEREMA_OUT_FIELDS = ",".join([
    "idcom",
    "idcomtxt",
    "iddep",
    "iddeptxt",
    "epci24",
    "epci24txt",
    "flux_secteur",
    "flux_periode",
    "flux_artificialisation",
])

ANNUAL_TOTAL_RE = re.compile(r"^naf(\d{2})art(\d{2})$", re.IGNORECASE)
ANNUAL_DEST_RE = re.compile(r"^art(\d{2})(hab|act|mix|rou|fer|inc)(\d{2})$", re.IGNORECASE)
TOTAL_TOTAL_RE = re.compile(r"^nafart(\d{2})(\d{2})$", re.IGNORECASE)
TOTAL_DEST_RE = re.compile(r"^art(hab|act|mix|rou|fer|inc)(\d{2})(\d{2})$", re.IGNORECASE)
YEAR_RE = re.compile(r"20\d{2}")

DEST_LABELS = {
    "hab": "Habitat",
    "act": "Activités",
    "mix": "Mixte",
    "rou": "Routes",
    "fer": "Ferroviaire",
    "inc": "Destination inconnue",
}


def normalize_level(level: str) -> str:
    return "commune" if str(level).lower() == "commune" else "epci"


def filtered_cache_path(level: str) -> Path:
    return CACHE_DIR / f"artificialisation_{normalize_level(level)}_aula.csv"


def metadata(*, force: bool = False) -> dict[str, Any]:
    dataset = get_dataset(DATASET_SLUG, force=force)
    resources = {
        "commune": resources_matching(dataset, terms=["commune"], formats=["csv", "zip"]),
        "epci": resources_matching(dataset, terms=["epci"], formats=["csv", "zip"]),
        "departement": resources_matching(dataset, terms=["departement"], formats=["csv", "zip"]),
        "region": resources_matching(dataset, terms=["region"], formats=["csv", "zip"]),
        "documentation": resources_matching(dataset, terms=["description"], formats=["csv", "pdf"]),
    }
    return {
        "dataset": {
            "id": dataset.get("id"),
            "title": dataset.get("title"),
            "slug": dataset.get("slug"),
            "page": dataset.get("page"),
            "last_update": dataset.get("last_update"),
            "frequency": dataset.get("frequency"),
            "license": (dataset.get("license") or {}).get("title") if isinstance(dataset.get("license"), dict) else dataset.get("license"),
        },
        "resources": resources,
        "api_donnees_foncieres": {
            "base_url": APIFONCIER_BASE_URL,
            "commune_endpoint": f"{APIFONCIER_BASE_URL}/indicateurs/conso_espace/communes/{{code_insee}}",
            "mode": "optionnel, surtout pour lecture directe commune",
        },
        "cerema_feature_server": {
            "layer_url": CEREMA_FEATURESERVER_LAYER_URL,
            "query_url": f"{CEREMA_FEATURESERVER_LAYER_URL}/query",
            "mode": "optionnel, cache JSON par territoire si la couche est accessible",
            "fields_used": CEREMA_OUT_FIELDS.split(","),
        },
        "local_csv_cache": csv_cache_status(),
    }


def _best_resource(level: str, *, force: bool = False) -> dict[str, Any] | None:
    candidates = metadata(force=force)["resources"].get(normalize_level(level), [])
    csv_candidates = [r for r in candidates if str(r.get("format") or "").lower() == "csv"]
    return (csv_candidates or candidates or [None])[0]


def refresh_filtered_cache(level: str = "epci", *, force: bool = False) -> dict[str, Any]:
    level = normalize_level(level)
    resource = _best_resource(level, force=force)
    if not resource:
        return {"ok": False, "error": f"Aucune ressource {level} trouvée", "level": level}
    url = resource.get("url") or resource.get("latest")
    if not url:
        return {"ok": False, "error": "Ressource sans URL", "resource": resource}

    if level == "epci":
        codes = set(agency_epci_codes())
        columns = CODE_COLUMNS_EPCI
    else:
        scope = agency_scope(live=True, force=force)
        codes = {str(c.get("code")) for c in scope.get("communes", []) if c.get("code")}
        columns = CODE_COLUMNS_COMMUNE

    output = filtered_cache_path(level)
    result = download_filter_csv(url, output, codes, columns)
    return {
        "ok": True,
        "level": level,
        "resource": {"title": resource.get("title"), "url": url, "format": resource.get("format")},
        "filter": result,
        "warning": "Le CSV communal peut être volumineux ; commencer par le niveau EPCI pour les tests.",
    }


def csv_cache_status() -> list[dict[str, Any]]:
    out: list[dict[str, Any]] = []
    for level in ["epci", "commune"]:
        path = filtered_cache_path(level)
        out.append({
            "level": level,
            "exists": path.exists(),
            "path": str(path),
            "size_bytes": path.stat().st_size if path.exists() else 0,
        })
    return out


def local_status() -> dict[str, Any]:
    return {
        "source": "Cerema / Portail national de l'artificialisation",
        "dataset_slug": DATASET_SLUG,
        "csv_caches": csv_cache_status(),
        "api_donnees_foncieres_base_url": APIFONCIER_BASE_URL,
        "cerema_feature_server_layer_url": CEREMA_FEATURESERVER_LAYER_URL,
    }


def _normalise_header(value: str) -> str:
    return "".join(ch.lower() for ch in value if ch.isalnum())


def _choose_column(headers: list[str], candidates: list[str]) -> str | None:
    mapping = {_normalise_header(h): h for h in headers}
    for candidate in candidates:
        key = _normalise_header(candidate)
        if key in mapping:
            return mapping[key]
    return None


def _delimiter_from_header(header: str) -> str:
    return ";" if header.count(";") >= header.count(",") else ","


def _read_csv_rows(path: Path) -> list[dict[str, str]]:
    if not path.exists():
        return []
    with path.open("r", encoding="utf-8-sig", newline="") as f:
        sample = f.readline()
        if not sample:
            return []
        delimiter = _delimiter_from_header(sample)
        f.seek(0)
        return list(csv.DictReader(f, delimiter=delimiter))


def _rows_from_filtered_cache(code: str, level: str) -> list[dict[str, str]]:
    level = normalize_level(level)
    rows = _read_csv_rows(filtered_cache_path(level))
    if not rows:
        return []
    headers = list(rows[0].keys())
    code_col = _choose_column(headers, CODE_COLUMNS_COMMUNE if level == "commune" else CODE_COLUMNS_EPCI)
    if not code_col:
        return []
    return [row for row in rows if str(row.get(code_col, "")).strip() == str(code).strip()]


def _float(value: Any) -> float | None:
    if value is None or value == "":
        return None
    try:
        return float(str(value).replace("\u202f", "").replace(" ", "").replace(",", "."))
    except Exception:
        return None


def _m2_to_ha(value: Any) -> float:
    parsed = _float(value)
    return 0.0 if parsed is None else parsed / 10000.0


def _name_from_rows(rows: list[dict[str, Any]], level: str) -> str | None:
    if not rows:
        return None
    candidates = ["idcomtxt", "nom_commune", "libelle_commune", "commune"] if level == "commune" else ["epci24txt", "nom_epci", "libelle_epci", "epci"]
    first = rows[0]
    for key in candidates:
        for header in first.keys():
            if _normalise_header(header) == _normalise_header(key) and first.get(header):
                return str(first.get(header))
    return None


def summarize_csv_rows(rows: list[dict[str, Any]], *, code: str, level: str) -> dict[str, Any]:
    level = normalize_level(level)
    if rows:
        normalised_headers = {_normalise_header(h) for h in rows[0].keys()}
        # Certains CSV/mirroirs Cerema sont en format long, identique à la couche FeatureServer
        # avec flux_secteur / flux_periode / flux_artificialisation. On les traite avant
        # le format large des Fichiers fonciers nafXXartYY.
        if "fluxartificialisation" in normalised_headers:
            summary = summarize_feature_rows(rows, code=code, level=level)
            summary["status"] = "cerema_csv_cache"
            summary["source"] = "Cerema — Portail national de l’artificialisation / data.gouv"
            summary["source_detail"] = "Flux ENAF vers artificialisé issus des Fichiers Fonciers, diffusés en open data."
            return summary
    annual_by_year: dict[int, float] = defaultdict(float)
    destinations: dict[str, float] = defaultdict(float)
    total_period_values: list[float] = []

    for row in rows:
        for key, value in row.items():
            if value in (None, ""):
                continue
            k = str(key).strip().lower()
            annual_total = ANNUAL_TOTAL_RE.match(k)
            if annual_total:
                start_year = 2000 + int(annual_total.group(1))
                annual_by_year[start_year] += _m2_to_ha(value)
                continue
            annual_dest = ANNUAL_DEST_RE.match(k)
            if annual_dest:
                dest = annual_dest.group(2).lower()
                destinations[DEST_LABELS.get(dest, dest)] += _m2_to_ha(value)
                continue
            total_total = TOTAL_TOTAL_RE.match(k)
            if total_total:
                total_period_values.append(_m2_to_ha(value))
                continue
            total_dest = TOTAL_DEST_RE.match(k)
            if total_dest:
                dest = total_dest.group(1).lower()
                destinations[DEST_LABELS.get(dest, dest)] += _m2_to_ha(value)

    series = [
        {"year": year, "value": round(value, 3)}
        for year, value in sorted(annual_by_year.items())
        if CONSO_YEAR_MIN <= year <= CONSO_YEAR_MAX and value
    ]
    total_ha = round(sum(item["value"] for item in series), 3) if series else round(sum(total_period_values), 3)
    annual_average_ha = round(total_ha / len(series), 3) if series else 0.0
    recent = round(sum(item["value"] for item in series[-5:]), 3) if series else 0.0
    destinations_list = [
        {"label": label, "value_ha": round(value, 3)}
        for label, value in sorted(destinations.items(), key=lambda item: item[1], reverse=True)
        if value
    ]
    period = f"{series[0]['year']}-{series[-1]['year'] + 1}" if series else f"{CONSO_YEAR_MIN}-{CONSO_YEAR_MAX}"

    return {
        "available": bool(series or total_ha),
        "ready": bool(series or total_ha),
        "status": "cerema_csv_cache" if rows else "cache_missing",
        "source": "Cerema — Portail national de l’artificialisation / data.gouv",
        "source_detail": "Indicateurs de consommation d'espaces ENAF produits pour le portail national à partir des Fichiers Fonciers.",
        "producer": "Cerema",
        "provider": "Cerema",
        "source_id": "cerema_portail_artificialisation",
        "official": bool(series or total_ha),
        "portal": "Portail national de l’artificialisation des sols",
        "dataset_slug": DATASET_SLUG,
        "period": period,
        "unit": "ha",
        "total_ha": total_ha,
        "annual_average_ha": annual_average_ha,
        "recent_ha": recent,
        "recent_period": "5 dernières années disponibles" if series else None,
        "series": series,
        "destinations": destinations_list,
        "by_sector": destinations_list,
        "territory": {"code": str(code), "name": _name_from_rows(rows, level), "level": level},
        "methodology": methodology(),
        "raw_count": len(rows),
    }


def methodology() -> dict[str, Any]:
    return {
        "producer": "Cerema",
        "portal": "Portail national de l'artificialisation des sols",
        "base": "Fichiers Fonciers retraités pour le suivi de la consommation d'espaces naturels, agricoles et forestiers.",
        "definition": "La consommation ENAF correspond aux flux d'espaces naturels, agricoles ou forestiers vers des espaces urbanisés/artificialises selon la méthodologie nationale.",
        "not_same_as_building_footprint": True,
        "aggregation": "Les valeurs annuelles des champs nafXXartYY sont additionnées et converties de m² en hectares ; les destinations habitat, activité, mixte et infrastructures sont lues dans les champs dédiés si disponibles.",
        "warning": "À ne pas confondre avec l'emprise bâtie de la carte bâtiments : un bâtiment peut consommer plus ou moins d'ENAF selon le contexte parcellaire et l'état initial du sol.",
        "limits": [
            "Donnée millésimée : afficher le millésime et la date de rafraîchissement du cache.",
            "Pour les bilans territoriaux, privilégier les données communales ou EPCI du portail plutôt que des calculs cartographiques approximatifs.",
        ],
    }


def _sql_quote(value: str) -> str:
    return str(value).replace("'", "''")


def _scope_check(code: str, level: str) -> dict[str, Any]:
    level = normalize_level(level)
    code = str(code)
    if level == "epci":
        ok = code in set(agency_epci_codes())
        return {"checked": True, "in_scope": ok, "level": level, "message": None if ok else "EPCI hors périmètre AULA configuré"}
    cached = agency_scope(live=False)
    communes = cached.get("communes") or []
    if communes:
        ok = code in {str(c.get("code")) for c in communes if c.get("code")}
        return {"checked": True, "in_scope": ok, "level": level, "message": None if ok else "Commune hors périmètre AULA d'après le cache API Géo"}
    return {
        "checked": False,
        "in_scope": True,
        "level": level,
        "message": "Cache des communes AULA absent : lancer le rafraîchissement API Géo pour verrouiller le filtre communal.",
    }


def cerema_layer_info(*, force: bool = False) -> dict[str, Any]:
    item = get_or_fetch(
        "cerema_artificialisation_layer_info",
        TTL_SECONDS,
        lambda: get_json(CEREMA_FEATURESERVER_LAYER_URL, params={"f": "json"}, timeout=CEREMA_TIMEOUT),
        force=force,
        source=CEREMA_FEATURESERVER_LAYER_URL,
    )
    payload = item["payload"]
    fields = payload.get("fields", []) or []
    return {
        "name": payload.get("name"),
        "description": payload.get("description"),
        "copyrightText": payload.get("copyrightText"),
        "maxRecordCount": payload.get("maxRecordCount"),
        "supportedQueryFormats": payload.get("supportedQueryFormats"),
        "fields": [{"name": f.get("name"), "alias": f.get("alias"), "type": f.get("type")} for f in fields],
        "cache_status": item.get("cache_status"),
        "cached_at": item.get("cached_at"),
    }


def _fetch_feature_server_rows(code: str, level: str) -> list[dict[str, Any]]:
    field = "idcom" if normalize_level(level) == "commune" else "epci24"
    where = f"{field} = '{_sql_quote(code)}'"
    rows: list[dict[str, Any]] = []
    offset = 0
    while True:
        payload = get_json(
            f"{CEREMA_FEATURESERVER_LAYER_URL}/query",
            params={
                "where": where,
                "outFields": CEREMA_OUT_FIELDS,
                "returnGeometry": "false",
                "f": "json",
                "orderByFields": "flux_periode,flux_secteur,idcom",
                "resultOffset": offset,
                "resultRecordCount": CEREMA_PAGE_SIZE,
            },
            timeout=CEREMA_TIMEOUT,
        )
        if "error" in payload:
            raise ExternalApiError(str(payload["error"]))
        features = payload.get("features", []) or []
        rows.extend([(feature.get("attributes") or {}) for feature in features])
        if not payload.get("exceededTransferLimit") and len(features) < CEREMA_PAGE_SIZE:
            break
        if not features:
            break
        offset += CEREMA_PAGE_SIZE
    return rows


def cerema_rows_for_territory(
    code: str,
    level: str = "commune",
    *,
    force: bool = False,
    enforce_agency_scope: bool = True,
) -> dict[str, Any]:
    level = normalize_level(level)
    scope = _scope_check(code, level)
    if enforce_agency_scope and scope.get("checked") and not scope.get("in_scope"):
        return {"ok": False, "code": code, "level": level, "scope": scope, "rows": [], "error": scope.get("message")}
    key = f"cerema_artificialisation_{level}_{code}"
    item = get_or_fetch(
        key,
        TTL_SECONDS,
        lambda: _fetch_feature_server_rows(str(code), level),
        force=force,
        source=f"{CEREMA_FEATURESERVER_LAYER_URL}/query?{level}={code}",
    )
    return {
        "ok": True,
        "code": str(code),
        "level": level,
        "scope": scope,
        "rows": item.get("payload") or [],
        "count": len(item.get("payload") or []),
        "cache_status": item.get("cache_status"),
        "cached_at": item.get("cached_at"),
        "source_url": f"{CEREMA_FEATURESERVER_LAYER_URL}/query",
    }


def _is_percent_or_ratio_sector(label: str) -> bool:
    lower = label.lower()
    return "%" in lower or "rapport" in lower or "surface communale" in lower or "pourcentage" in lower or "taux" in lower


def _is_total_sector(label: str) -> bool:
    lower = label.lower()
    if _is_percent_or_ratio_sector(label):
        return False
    return "total" in lower or "ensemble" in lower or "tous secteurs" in lower or "flux d'artificialisation" in lower or "flux d’artificialisation" in lower


def _sector_label(label: str) -> str:
    lower = (label or "").lower()
    if "habitat" in lower:
        return "Habitat"
    if "activité" in lower or "activite" in lower:
        return "Activités"
    if "rout" in lower:
        return "Routes"
    if "ferro" in lower:
        return "Ferroviaire"
    if "mixte" in lower:
        return "Mixte"
    if "inconn" in lower:
        return "Destination inconnue"
    if "infrastructure" in lower:
        return "Infrastructures"
    return (label or "Destination inconnue").strip()


def _period_years(period: Any) -> tuple[int | None, int | None, bool]:
    years = [int(y) for y in YEAR_RE.findall(str(period or ""))]
    if len(years) >= 2:
        start, end = years[0], years[1]
        return start, end, (end - start == 1)
    if len(years) == 1:
        return years[0], years[0] + 1, True
    return None, None, False


def summarize_feature_rows(rows: list[dict[str, Any]], *, code: str, level: str) -> dict[str, Any]:
    level = normalize_level(level)
    annual_by_year: dict[int, float] = defaultdict(float)
    sector_totals: dict[str, float] = defaultdict(float)
    total_rows_seen = False
    for row in rows:
        value = _float(row.get("flux_artificialisation"))
        if value is None:
            continue
        sector_raw = str(row.get("flux_secteur") or "")
        if _is_percent_or_ratio_sector(sector_raw):
            continue
        start, _end, annual = _period_years(row.get("flux_periode"))
        if start is None:
            continue
        if _is_total_sector(sector_raw):
            total_rows_seen = True
            if annual:
                annual_by_year[int(start)] += float(value)
        elif not total_rows_seen:
            if annual:
                annual_by_year[int(start)] += float(value)
            sector_totals[_sector_label(sector_raw)] += float(value)

    series = [{"year": y, "value": round(v, 3)} for y, v in sorted(annual_by_year.items()) if v]
    total_ha = round(sum(item["value"] for item in series), 3)
    return {
        "available": bool(series or rows),
        "ready": bool(series or rows),
        "status": "cerema_feature_server",
        "source": "Cerema — Portail national de l’artificialisation",
        "source_detail": "Lecture directe d'une couche diffusée pour le portail national de l'artificialisation.",
        "producer": "Cerema",
        "provider": "Cerema",
        "source_id": "cerema_portail_artificialisation",
        "official": bool(series or total_ha),
        "portal": "Portail national de l’artificialisation des sols",
        "period": f"{series[0]['year']}-{series[-1]['year'] + 1}" if series else f"{CONSO_YEAR_MIN}-{CONSO_YEAR_MAX}",
        "unit": "ha",
        "total_ha": total_ha,
        "annual_average_ha": round(total_ha / len(series), 3) if series else 0,
        "recent_ha": round(sum(item["value"] for item in series[-5:]), 3) if series else 0,
        "recent_period": "5 dernières années disponibles" if series else None,
        "series": series,
        "destinations": [{"label": k, "value_ha": round(v, 3)} for k, v in sorted(sector_totals.items(), key=lambda i: i[1], reverse=True)],
        "by_sector": [{"label": k, "value_ha": round(v, 3)} for k, v in sorted(sector_totals.items(), key=lambda i: i[1], reverse=True)],
        "territory": {"code": str(code), "name": _name_from_rows(rows, level), "level": level},
        "methodology": methodology(),
        "raw_count": len(rows),
    }


def consumption_summary_for_territory(
    code: str,
    level: str = "commune",
    *,
    force: bool = False,
    enforce_agency_scope: bool = True,
) -> dict[str, Any]:
    rows_payload = cerema_rows_for_territory(code, level, force=force, enforce_agency_scope=enforce_agency_scope)
    if not rows_payload.get("ok"):
        return {
            "available": False,
            "status": "not_available",
            "source": "Cerema — Portail national de l’artificialisation",
            "error": rows_payload.get("error"),
            "scope": rows_payload.get("scope"),
            "series": [],
            "total_ha": 0,
            "annual_average_ha": 0,
            "recent_ha": 0,
            "methodology": methodology(),
        }
    summary = summarize_feature_rows(rows_payload.get("rows") or [], code=str(code), level=level)
    summary["cache_status"] = rows_payload.get("cache_status")
    summary["cached_at"] = rows_payload.get("cached_at")
    summary["scope"] = rows_payload.get("scope")
    summary["source_url"] = rows_payload.get("source_url")
    return summary


def cached_consumption_summary(code: str, level: str = "commune") -> dict[str, Any] | None:
    level = normalize_level(level)
    item = read_cache(f"cerema_artificialisation_{level}_{code}")
    if not item:
        return None
    return {
        **summarize_feature_rows(item.get("payload") or [], code=str(code), level=level),
        "cache_status": "hit",
        "cached_at": item.get("cached_at"),
    }


def _api_donnees_foncieres_commune(code: str, *, force: bool = False) -> dict[str, Any] | None:
    """Tentative de lecture directe de l'API Donnees foncieres pour une commune.

    L'API peut evoluer ; on garde donc une normalisation prudente et non bloquante.
    """
    key = f"apifoncier_conso_commune_{code}"
    item = get_or_fetch(
        key,
        TTL_SECONDS,
        lambda: get_json(f"{APIFONCIER_BASE_URL}/indicateurs/conso_espace/communes/{code}", timeout=CEREMA_TIMEOUT),
        force=force,
        source=f"{APIFONCIER_BASE_URL}/indicateurs/conso_espace/communes/{code}",
    )
    payload = item.get("payload")
    # Normalisation volontairement limitee : on expose le brut si le schema ne correspond pas.
    return {
        "available": True,
        "ready": True,
        "status": "api_donnees_foncieres_raw",
        "source": "Cerema — API Données foncières",
        "producer": "Cerema",
        "provider": "Cerema",
        "source_id": "cerema_api_donnees_foncieres",
        "official": False,
        "territory": {"code": str(code), "level": "commune"},
        "series": [],
        "total_ha": 0,
        "annual_average_ha": 0,
        "recent_ha": 0,
        "raw": payload,
        "cache_status": item.get("cache_status"),
        "cached_at": item.get("cached_at"),
        "methodology": methodology(),
        "warnings": ["Schéma API à confirmer avant affichage direct des valeurs dans le tableau de bord."],
    }


def consumption_for_territory(
    territoire: str,
    niveau: str = "commune",
    *,
    live: bool = False,
    force: bool = False,
    aggregate_communes: bool = True,
) -> dict[str, Any]:
    level = normalize_level(niveau)
    code = str(territoire)

    # 1. Priorité au CSV officiel filtré sur le périmètre agence, s'il existe.
    rows = _rows_from_filtered_cache(code, level)
    if rows:
        summary = summarize_csv_rows(rows, code=code, level=level)
        summary["mode"] = "csv_cache"
        summary["cache_path"] = str(filtered_cache_path(level))
        return summary

    # 2. Puis au cache JSON FeatureServer existant.
    cached = cached_consumption_summary(code, level)
    if cached and cached.get("available"):
        cached["mode"] = "feature_server_cache"
        return cached

    # 3. Si demande live, on tente la lecture directe FeatureServer. Pour commune, on expose aussi
    #    un fallback brut de l'API Donnees foncieres si la couche est indisponible.
    if live or force:
        try:
            direct = consumption_summary_for_territory(code, level, force=force, enforce_agency_scope=True)
            if direct.get("available"):
                direct["mode"] = "feature_server_live"
                return direct
        except Exception as exc:  # noqa: BLE001
            if level == "commune":
                try:
                    raw = _api_donnees_foncieres_commune(code, force=force)
                    if raw:
                        raw["mode"] = "api_donnees_foncieres_live_raw"
                        raw["feature_server_error"] = str(exc)
                        return raw
                except Exception:
                    pass
            return {
                "available": False,
                "ready": False,
                "status": "cerema_live_error",
                "source": "Cerema — Portail national de l’artificialisation",
                "territory": {"code": code, "level": level},
                "series": [],
                "total_ha": 0,
                "annual_average_ha": 0,
                "recent_ha": 0,
                "error": str(exc),
                "methodology": methodology(),
                "refresh_command": f"python scripts/refresh_external_cache.py --source artificialisation --level {level} --force",
            }

    return {
        "available": False,
        "ready": False,
        "status": "cache_missing",
        "source": "Cerema — Portail national de l’artificialisation",
        "territory": {"code": code, "level": level},
        "series": [],
        "total_ha": 0,
        "annual_average_ha": 0,
        "recent_ha": 0,
        "methodology": methodology(),
        "refresh_command": f"python scripts/refresh_external_cache.py --source artificialisation --level {level} --force",
        "message": "Cache Cerema absent pour ce territoire. Rafraîchir le cache ou appeler avec live=true.",
    }


def consumption_for_scope(
    territoire: str,
    niveau: str = "commune",
    *,
    live: bool = False,
    force: bool = False,
    aggregate_communes: bool = True,
) -> dict[str, Any]:
    return consumption_for_territory(territoire, niveau, live=live, force=force, aggregate_communes=aggregate_communes)


def available_for(territoire: str, niveau: str = "commune") -> dict[str, Any]:
    data = consumption_for_territory(territoire, niveau, live=False, force=False)
    return {
        "territoire": str(territoire),
        "niveau": normalize_level(niveau),
        "available": bool(data.get("available")),
        "status": data.get("status"),
        "mode": data.get("mode"),
        "rows": data.get("raw_count", 0),
        "source": data.get("source"),
        "refresh_command": data.get("refresh_command") or f"python scripts/refresh_external_cache.py --source artificialisation --level {normalize_level(niveau)} --force",
    }
