"""Atelier local : deux spécifications de partition, aucun fichier ancien réécrit."""
from datetime import datetime
from hashlib import sha256
from pathlib import Path
from tempfile import TemporaryDirectory
from urllib.parse import unquote, urlparse
import json
import platform
import pyarrow as pa
import pyiceberg
from pyiceberg.catalog.sql import SqlCatalog
from pyiceberg.partitioning import PartitionField, PartitionSpec
from pyiceberg.schema import Schema
from pyiceberg.transforms import DayTransform, MonthTransform
from pyiceberg.types import LongType, NestedField, TimestampType

def fingerprint(uri):
    return sha256(Path(unquote(urlparse(uri).path)).read_bytes()).hexdigest()

def files(table):
    return table.inspect.files().to_pylist()

arrow_schema = pa.schema([pa.field("id", pa.int64(), nullable=False),
                          pa.field("event_ts", pa.timestamp("us"), nullable=False)])

def batch(rows):
    return pa.Table.from_pylist([
        {"id": ident, "event_ts": datetime.fromisoformat(ts)} for ident, ts in rows
    ], schema=arrow_schema)

with TemporaryDirectory(prefix="iceberg-partitions-") as tmp:
    root = Path(tmp)
    catalog = SqlCatalog("atelier", uri=f"sqlite:///{root}/catalog.db",
                         warehouse=(root / "warehouse").as_uri())
    catalog.create_namespace("demo")
    table = catalog.create_table(
        "demo.events",
        schema=Schema(NestedField(1, "id", LongType(), required=True),
                      NestedField(2, "event_ts", TimestampType(), required=True)),
        partition_spec=PartitionSpec(PartitionField(
            source_id=2, field_id=1000, transform=MonthTransform(), name="event_month"
        )),
        properties={"format-version": "2"},
    )
    table.append(batch([(1, "2026-09-01T10:00:00"), (2, "2026-09-02T10:00:00")]))
    old_hashes = {f["file_path"]: fingerprint(f["file_path"]) for f in files(table)}
    before_spec = table.spec().spec_id
    with table.update_spec() as update:
        update.remove_field("event_month")
        update.add_field("event_ts", DayTransform(), "event_day")
    # La seule modification de spécification ne crée ni ne réécrit de fichier de données.
    assert {f["file_path"] for f in files(table)} == set(old_hashes)
    assert all(fingerprint(p) == h for p, h in old_hashes.items())
    # Une arrivée tardive de septembre utilise aussi la nouvelle spécification.
    table.append(batch([(3, "2026-10-01T10:00:00"), (4, "2026-10-02T10:00:00"),
                        (5, "2026-09-02T12:00:00")]))
    current_files = files(table)
    assert set(old_hashes) <= {f["file_path"] for f in current_files}
    assert all(fingerprint(p) == h for p, h in old_hashes.items())
    assert table.scan().to_arrow().num_rows == 5
    september = table.scan(row_filter="event_ts >= '2026-09-01T00:00:00' and event_ts < '2026-10-01T00:00:00'").to_arrow()
    assert sorted(september["id"].to_pylist()) == [1, 2, 5]
    by_spec = {}
    for f in current_files:
        entry = by_spec.setdefault(str(f["spec_id"]), {"files": 0, "rows": 0})
        entry["files"] += 1
        entry["rows"] += f["record_count"]
    print(json.dumps({
        "python": platform.python_version(), "pyiceberg": pyiceberg.__version__,
        "pyarrow": pa.__version__, "format_version": table.metadata.format_version,
        "old_spec_id": before_spec, "current_spec_id": table.spec().spec_id,
        "files_by_spec": by_spec, "old_files_unchanged": True,
        "total_rows": 5, "september_ids": sorted(september["id"].to_pylist()),
        "assertions_passed": True,
        "scope": "Local SQLite/PyArrow only; not a Snowflake benchmark"
    }, ensure_ascii=False, indent=2))
