# One-shot Iceberg sync for the ERA5 lake: registers new hive tiles from the
# era5-thermograph bucket into the iceberg/era5_daily table (see
# sync_iceberg.py). Everything ships as manylinux wheels — no toolchain.
FROM python:3.12-slim

# The pyiceberg extras pull sqlalchemy (sqlite catalog), pyarrow (FileIO +
# parquet footers) and s3fs (bucket listing/manifest reads).
COPY requirements.txt /tmp/requirements.txt
RUN pip install --no-cache-dir -r /tmp/requirements.txt

WORKDIR /app
COPY sync_iceberg.py .

# The sqlite catalog lives on a volume so runs share it; losing it is fine
# (the next run re-registers the table from its latest metadata JSON).
ENV THERMOGRAPH_ICEBERG_CATALOG_DB=/state/iceberg-catalog.db \
    PYTHONUNBUFFERED=1
VOLUME /state

ENTRYPOINT ["python", "sync_iceberg.py"]
