"""
CD Report Generator API
========================
Single-file FastAPI service that:
  1. Downloads a CD-entries CSV from a given URL
  2. Generates one styled landscape-A3 PDF per drug (register name group)
  3. Zips all the PDFs together
  4. Uploads the zip to a target S3 "report_files/" location (built from a
     template in .env, using values parsed out of the CSV URL)
  5. Returns immediately, then POSTs the result to the correct Laravel
     endpoint (stage or live) once background processing finishes.

No database, no ORM models — everything happens in-memory / temp-dir per request.

-----------------------------------------------------------------------------
CONFIG (env vars)
-----------------------------------------------------------------------------
AWS_ACCESS_KEY_ID       - AWS access key (or use an attached IAM role / profile)
AWS_SECRET_ACCESS_KEY   - AWS secret key
AWS_REGION              - e.g. "eu-west-2"           (default: eu-west-2)
S3_BUCKET               - bucket name that backs the pharm.amazonaws.com domain
S3_PUBLIC_BASE_URL      - optional. Public base URL to build the returned link,
                           e.g. "https://pharm.amazonaws.com". If not set, a
                           presigned URL is generated instead.
ZIP_KEY_TEMPLATE        - template used to build the destination S3 key for
                           the zip, using placeholders parsed from csv_url:
                             {receive_form} - "local" / "stage" / "live" segment
                             {pharmacy_id}  - the folder segment right before
                                              "report_files" in csv_url
                             {report_stem}  - the CSV filename without ".csv"
                           Default:
                             "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.zip"
STORAGE_BACKEND         - "s3" (default) or "local".
                           Set to "local" for local development/testing with
                           NO AWS account or credentials needed at all —
                           /generate-report will save the zip to
                           LOCAL_OUTPUT_DIR on disk instead of uploading to S3.
LOCAL_OUTPUT_DIR        - folder to save zips into when STORAGE_BACKEND=local
                           (default: "./local_zip_output")
MAX_ROWS_PER_PDF        - max rows rendered into ONE ReportLab document before
                           generate_pdfs()/generate_single_pdf() split off
                           another "_PartN" PDF. Exists to bound peak memory
                           on huge CSVs — see the comment above the constant
                           for the full explanation. Default: 20000

-----------------------------------------------------------------------------
CALLBACK ROUTING (Laravel)
-----------------------------------------------------------------------------
csv_url's first path segment tells us which environment the request came
from ("local", "stage", or "live"), which decides which Laravel endpoint
gets the completion callback:

    csv_url segment    -> callback goes to
    ----------------------------------------
    local              -> STAGE_LARAVEL_URL   (no separate local Laravel API)
    stage              -> STAGE_LARAVEL_URL
    live               -> LIVE_LARAVEL_URL

STAGE_LARAVEL_URL  - default "https://devstage.pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php"
LIVE_LARAVEL_URL   - default "https://pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php"
LARAVEL_SECRET_KEY - shared secret used to HMAC-sign the callback (Bearer token
                      sent is HMAC-SHA256(pharmacy_id|csv_filename), hex digest).
CALLBACK_TIMEOUT_SECONDS - request timeout for the callback POST (default: 30)

-----------------------------------------------------------------------------
LOGGING
-----------------------------------------------------------------------------
Every call to /generate-report, /generate-csv-to-pdf-report, and their
/download variants (request received, validation result, background job
progress, storage result, callback attempt/response, and any errors) is
written to a rotating log file, in addition to stdout (so it still shows up
under uvicorn/systemd/docker logs as before).

LOG_FILE_PATH   - path to the log file (default: "logs/api.log"). Parent
                   directories are created automatically if missing.
LOG_LEVEL       - default "INFO"
LOG_MAX_BYTES   - size in bytes before the log file rotates (default: 5MB)
LOG_BACKUP_COUNT- number of rotated backups to keep (default: 5)

-----------------------------------------------------------------------------
RUN
-----------------------------------------------------------------------------
pip install fastapi uvicorn pandas reportlab boto3 requests python-multipart python-dotenv
Create a ".env" file in the same directory (see .env.example) then run:
uvicorn main:app --host 0.0.0.0 --port 8000

-----------------------------------------------------------------------------
USAGE
-----------------------------------------------------------------------------
POST /generate-report
{
  "csv_url": "https://pharmsmart-pharmacy.s3.eu-west-2.amazonaws.com/local/pharmacies/N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09/report_files/6a38ce059932a-cd-entries.csv"
}

Returns IMMEDIATELY:
{
  "status": "success"
}

The heavy CSV -> PDFs -> zip -> upload work runs in a background task. Once
it finishes, this JSON is POSTed to the Laravel endpoint chosen by the
csv_url's environment segment (see CALLBACK ROUTING above):

{
  "request_type": "csv-to-pdf",
  "timestamp": 1784021199,
  "pharmacy_id": "N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09",
  "result_filename": "6a38ce059932a-cd-entries.zip",
  "csv_filename": "6a38ce059932a-cd-entries.csv"
}

with header:
  Authorization: Bearer <HMAC-SHA256(pharmacy_id|csv_filename, LARAVEL_SECRET_KEY)>

On failure, "status": "failed" and "error": "<message>" are added to that
same payload (Laravel isn't required to look at these, but they're there
for visibility / debugging).

There is also POST /generate-report/download which streams the zip straight
back in the HTTP response instead of uploading anywhere (handy for testing
without S3 credentials). This one is still synchronous.

POST /generate-csv-to-pdf-report and /generate-csv-to-pdf-report/download
behave the same way but produce a single combined PDF (or, for CSVs larger
than MAX_ROWS_PER_PDF rows, several PDF parts bundled into a zip — see
MAX_ROWS_PER_PDF above) instead of one PDF per drug.
"""

import gc
import hashlib
import hmac
import logging
import math
import os
import re
import resource
import shutil
import tempfile
import time
import zipfile
from html import escape
from logging.handlers import RotatingFileHandler
from pathlib import Path, PurePosixPath
from typing import Optional
from urllib.parse import urlparse

import boto3
import pandas as pd
import requests
from botocore.exceptions import BotoCoreError, ClientError
from dotenv import load_dotenv
from fastapi import BackgroundTasks, FastAPI, HTTPException
from fastapi.responses import FileResponse, JSONResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel, HttpUrl

load_dotenv()  # reads .env file in the working directory, if present

from reportlab.lib import colors
from reportlab.lib.enums import TA_LEFT
from reportlab.lib.pagesizes import A2, A3, landscape
from reportlab.lib.styles import ParagraphStyle
from reportlab.platypus import LongTable, Paragraph, SimpleDocTemplate, Table, TableStyle
from fastapi import FastAPI, HTTPException, Security, Depends, Request
from fastapi.security import APIKeyHeader

# --------------------------------------------------------------------------
# CONFIG
# --------------------------------------------------------------------------
AWS_REGION = os.environ.get("AWS_REGION", "eu-west-2")
S3_BUCKET = os.environ.get("S3_BUCKET", "")
S3_PUBLIC_BASE_URL = os.environ.get("S3_PUBLIC_BASE_URL", "")  # e.g. https://pharm.amazonaws.com
ZIP_KEY_TEMPLATE = os.environ.get(
    "ZIP_KEY_TEMPLATE",
    "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.zip",
)
# Used only by /generate-csv-to-pdf-report (single combined PDF, no zip —
# unless the CSV is large enough to need splitting, see MAX_ROWS_PER_PDF).
PDF_KEY_TEMPLATE = os.environ.get(
    "PDF_KEY_TEMPLATE",
    "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.pdf",
)
PRESIGNED_URL_EXPIRY_SECONDS = int(os.environ.get("PRESIGNED_URL_EXPIRY_SECONDS", 3600))  # 1 hour default

# Max rows rendered into ONE ReportLab document before generate_pdfs()/
# generate_single_pdf() split off another "_PartN" PDF. This exists because
# every cell becomes a Paragraph object held in memory before doc.build()
# runs — for a CSV with hundreds of thousands of rows that's millions of
# Paragraph objects and multiple GB of RAM, which gets the whole worker
# process OOM-killed by the OS (a SIGKILL that bypasses Python's exception
# handling entirely, so nothing gets logged — the process just dies).
# Splitting bounds peak memory to roughly one chunk's worth regardless of
# total CSV size. Tune down if you still see OOM kills; tune up if you'd
# rather have fewer, bigger PDF parts and have RAM to spare.
MAX_ROWS_PER_PDF = int(os.environ.get("MAX_ROWS_PER_PDF", 20000))

# Hard cap on the number of distinct register-name groups /generate-report
# will turn into separate PDFs. MAX_ROWS_PER_PDF only bounds rows WITHIN one
# group — it does nothing if the "register name" column turns out to have
# very high cardinality (e.g. the column-matching logic picked up a
# near-unique field instead of a true low-cardinality drug/register name),
# which can silently balloon into thousands of tiny PDF files. Each one is
# individually cheap, but ReportLab's font/style registries and general
# object churn accumulate across that many doc.build() calls, and a CSV
# that "should" be small can still exhaust memory this way. If this cap
# trips, it's almost always a sign the wrong column got picked as the
# grouping column — check the CSV's headers.
MAX_REGISTER_GROUPS = int(os.environ.get("MAX_REGISTER_GROUPS", 3000))

# Optional hard ceiling (MB) on process RSS during PDF generation. 0 disables
# it (default). If set, generate_pdfs()/generate_single_pdf() check RSS
# periodically and raise a normal, catchable HTTPException once the ceiling
# is crossed — turning what would otherwise be a silent, uncatchable OOM
# SIGKILL into a logged failure with a proper callback to Laravel. Set this
# comfortably below the server's actual available RAM (leave headroom for
# the OS, other workers, etc.).
MAX_MEMORY_MB = int(os.environ.get("MAX_MEMORY_MB", 0))


def _current_rss_mb() -> float:
    """Current process peak RSS in MB. ru_maxrss is KB on Linux (bytes on
    macOS) — this service is assumed to run on Linux servers."""
    return resource.getrusage(resource.RUSAGE_SELF).ru_maxrss / 1024


def _check_memory_guard(context: str) -> None:
    """Raise a controlled, catchable error if RSS has crossed MAX_MEMORY_MB.
    No-op if MAX_MEMORY_MB is 0 (default/disabled)."""
    if MAX_MEMORY_MB <= 0:
        return
    rss = _current_rss_mb()
    if rss > MAX_MEMORY_MB:
        raise HTTPException(
            status_code=507,
            detail=(
                f"Aborting {context}: process RSS {rss:.0f}MB exceeded "
                f"MAX_MEMORY_MB={MAX_MEMORY_MB}MB safety limit."
            ),
        )

# STORAGE_BACKEND controls where /generate-report saves the zip:
#   "s3"    (default) - uploads to real S3 via boto3, requires valid AWS creds
#   "local" - saves to LOCAL_OUTPUT_DIR on disk instead, no AWS needed at all.
STORAGE_BACKEND = os.environ.get("STORAGE_BACKEND", "s3").strip().lower()
LOCAL_OUTPUT_DIR = os.environ.get("LOCAL_OUTPUT_DIR", "./local_zip_output")
# Base URL used to build the downloadable link when STORAGE_BACKEND=local,
# e.g. "http://localhost:8000" locally, or "https://your-domain.com" in prod.
# Files are served at {LOCAL_PUBLIC_BASE_URL}/files/{key}
LOCAL_PUBLIC_BASE_URL = os.environ.get("LOCAL_PUBLIC_BASE_URL", "http://localhost:8000").rstrip("/")

# --------------------------------------------------------------------------
# LOGGING — rotating file handler, one line per event, plus stdout so
# existing uvicorn/systemd/docker log collection keeps working unchanged.
# --------------------------------------------------------------------------
LOG_FILE_PATH = os.environ.get("LOG_FILE_PATH", "logs/api.log")
LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO").upper()
LOG_MAX_BYTES = int(os.environ.get("LOG_MAX_BYTES", 5 * 1024 * 1024))  # 5MB
LOG_BACKUP_COUNT = int(os.environ.get("LOG_BACKUP_COUNT", 5))

Path(LOG_FILE_PATH).parent.mkdir(parents=True, exist_ok=True)

logger = logging.getLogger("cd_report_api")
logger.setLevel(LOG_LEVEL)
logger.propagate = False  # avoid double-logging via uvicorn's root logger

_log_formatter = logging.Formatter(
    fmt="%(asctime)s | %(levelname)-8s | %(message)s",
    datefmt="%Y-%m-%d %H:%M:%S",
)

if not logger.handlers:  # guard against duplicate handlers on --reload
    _file_handler = RotatingFileHandler(
        LOG_FILE_PATH, maxBytes=LOG_MAX_BYTES, backupCount=LOG_BACKUP_COUNT, encoding="utf-8"
    )
    _file_handler.setFormatter(_log_formatter)
    logger.addHandler(_file_handler)

    _console_handler = logging.StreamHandler()
    _console_handler.setFormatter(_log_formatter)
    logger.addHandler(_console_handler)

# --------------------------------------------------------------------------
# CALLBACK ROUTING (Laravel) — which URL to POST the completion result to,
# chosen by the "local"/"stage"/"live" segment parsed out of csv_url.
# "local" is intentionally routed to the STAGE url — there is no separate
# local Laravel API to call back into.
# --------------------------------------------------------------------------
STAGE_LARAVEL_URL = os.environ.get(
    "STAGE_LARAVEL_URL",
    "https://devstage.pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php",
)
LIVE_LARAVEL_URL = os.environ.get(
    "LIVE_LARAVEL_URL",
    "https://pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php",
)
AWS_FILE_CONVERT_API_KEY = os.environ.get("LARAVEL_SECRET_KEY", "")
CALLBACK_TIMEOUT_SECONDS = int(os.environ.get("CALLBACK_TIMEOUT_SECONDS", 30))

CALLBACK_URL_BY_ENV = {
    "local": STAGE_LARAVEL_URL,
    "stage": STAGE_LARAVEL_URL,
    "live": LIVE_LARAVEL_URL,
}


def resolve_callback_url(receive_form: str) -> str:
    url = CALLBACK_URL_BY_ENV.get(receive_form.lower())
    if not url:
        raise HTTPException(
            status_code=422,
            detail=f"Unrecognised environment '{receive_form}' in csv_url "
                    f"(expected one of: {', '.join(CALLBACK_URL_BY_ENV)}).",
        )
    return url


EXCLUDE_COLUMNS = ["ID", "Amended Entry ID"]

app = FastAPI(title="Pharmsmart CD Report Generator", version="1.0.0")

# When using local storage, serve the saved zips back over plain HTTP so
# zip_url in the response is an actual downloadable link — no S3 needed.
if STORAGE_BACKEND == "local":
    Path(LOCAL_OUTPUT_DIR).mkdir(parents=True, exist_ok=True)
    app.mount("/files", StaticFiles(directory=LOCAL_OUTPUT_DIR), name="files")

# --------------------------------------------------------------------------
# CORS / ORIGIN CHECK
# --------------------------------------------------------------------------

from fastapi import Request

ALLOWED_ORIGINS = [
    o.strip().rstrip("/")
    for o in os.environ.get("ALLOWED_ORIGINS", "").split(",")
    if o.strip()
]


def verify_domain(request: Request) -> None:
    if not ALLOWED_ORIGINS:
        # No allowlist configured -> skip this check entirely
        return

    origin = (request.headers.get("origin") or request.headers.get("referer") or "").rstrip("/")

    if not origin:
        # No Origin/Referer header at all — this is a normal, expected case
        # for server-to-server calls, curl, Postman, or same-origin Swagger UI.
        return

    if not any(origin == allowed or origin.startswith(allowed + "/") for allowed in ALLOWED_ORIGINS):
        logger.warning(f"Blocked request: origin '{origin}' not in ALLOWED_ORIGINS")
        raise HTTPException(status_code=403, detail=f"Origin '{origin}' not allowed.")

# --------------------------------------------------------------------------
# API KEY AUTH
# --------------------------------------------------------------------------
API_KEY = os.environ.get("API_KEY", "")

api_key_header = APIKeyHeader(name="X-API-Key", auto_error=False)


def verify_api_key(api_key: str = Security(api_key_header)) -> None:
    if not API_KEY:
        raise HTTPException(status_code=500, detail="Server API key not configured.")
    if not api_key or api_key != API_KEY:
        logger.warning("Rejected request: invalid or missing X-API-Key")
        raise HTTPException(status_code=401, detail="Invalid or missing API key.")

# --------------------------------------------------------------------------
# REQUEST / RESPONSE MODELS
# --------------------------------------------------------------------------

class GenerateReportRequest(BaseModel):
    csv_url: HttpUrl


class GenerateReportResponse(BaseModel):
    status: str

# --------------------------------------------------------------------------
# PDF GENERATION LOGIC (same behaviour as the original script)
# --------------------------------------------------------------------------
def clean_filename(name: str) -> str:
    name = str(name).strip()
    name = re.sub(r"[^\w\s.-]", "", name)
    name = re.sub(r"\s+", "_", name)
    return name[:95] or "Unknown_Drug"


def make_paragraph(value, style: ParagraphStyle) -> Paragraph:
    text = escape(str(value)).replace("\n", "<br/>")
    return Paragraph(text, style)


def get_column_widths(columns):
    weights = []

    for c in columns:
        lc = str(c).strip().lower()

        if "register" in lc and "name" in lc:
            weights.append(2.55)
        elif "timestamp" in lc or "date" in lc:
            weights.append(1.25)
        elif "action" in lc:
            weights.append(1.20)
        elif ("name" in lc or "address" in lc) and "register" not in lc:
            weights.append(2.35)
        elif "prescriber" in lc:
            weights.append(2.05)
        elif "collect" in lc:
            weights.append(1.65)
        elif "id" in lc or "request" in lc or "provided" in lc:
            weights.append(1.25)
        elif "quantity" in lc:
            weights.append(0.92)
        elif "balance" in lc:
            weights.append(0.95)
        elif "entry" in lc:
            weights.append(1.00)
        elif "sign" in lc:
            weights.append(0.65)
        else:
            weights.append(1.05)

    page_width, _ = landscape(A3)
    usable_width = page_width - 24
    total_weight = sum(weights)

    return [usable_width * w / total_weight for w in weights]


def read_csv_with_fallback(csv_path: Path) -> pd.DataFrame:
    """Read a CSV trying multiple encodings, since pharmacy exports are often
    not UTF-8 (commonly Windows-1252 / Latin-1 from Excel/Windows systems).
    Also falls back to a more lenient parser if some rows have a ragged
    number of fields (e.g. an unescaped comma inside a text field).

    Reads every column as dtype=str: this data ends up rendered as strings
    anyway (callers already do .astype(str)), and reading it that way from
    the start avoids pandas' per-chunk numeric-type inference/conversion —
    which is what throws the 'Columns (...) have mixed types' DtypeWarning
    and roughly doubles peak memory during parsing on large files."""
    encodings_to_try = ["utf-8-sig", "utf-8", "cp1252", "latin-1"]
    last_error: Optional[Exception] = None

    for encoding in encodings_to_try:
        try:
            return pd.read_csv(csv_path, encoding=encoding, dtype=str)
        except (UnicodeDecodeError, UnicodeError) as e:
            last_error = e
            continue
        except pd.errors.ParserError as e:
            try:
                bad_rows: list[int] = []

                def _log_bad_line(bad_line: list[str]) -> None:
                    bad_rows.append(len(bad_rows))
                    return None  # drop the row

                df = pd.read_csv(
                    csv_path,
                    encoding=encoding,
                    engine="python",
                    dtype=str,
                    on_bad_lines=_log_bad_line,
                )
                if bad_rows:
                    logger.warning(
                        f"Skipped {len(bad_rows)} malformed row(s) in {csv_path.name} "
                        f"(ragged field count, e.g. an unescaped comma in a text field)."
                    )
                return df
            except Exception as e2:
                last_error = e2
                continue

    raise HTTPException(status_code=422, detail=f"Could not parse CSV file: {last_error}")

def generate_pdfs(csv_path: Path, output_dir: Path) -> list[Path]:
    """Generate one drug-wise styled PDF per group, returns list of PDF paths.

    Any register-name group larger than MAX_ROWS_PER_PDF rows is split into
    multiple "_PartN" PDFs, built and flushed to disk one at a time. This is
    what keeps peak memory bounded on very large CSVs (previously, a single
    huge group meant building millions of ReportLab Paragraph objects in
    memory at once before doc.build() ran — enough to get the whole worker
    process OOM-killed silently, with no exception ever logged)."""
    df = read_csv_with_fallback(csv_path).fillna("").astype(str)

    df = df.drop(columns=[col for col in EXCLUDE_COLUMNS if col in df.columns])

    register_col = None
    for col in df.columns:
        if "register" in col.lower() and "name" in col.lower():
            register_col = col
            break
    if register_col is None:
        register_col = df.columns[0]

    total_rows_all = len(df)
    unique_groups = df[register_col].nunique(dropna=False)
    logger.info(
        f"[generate_pdfs] CSV has {total_rows_all} row(s), grouping by column "
        f"'{register_col}' -> {unique_groups} unique group(s) | RSS={_current_rss_mb():.0f}MB"
    )

    if unique_groups > MAX_REGISTER_GROUPS:
        logger.error(
            f"[generate_pdfs] Column '{register_col}' has {unique_groups} unique values, "
            f"exceeding MAX_REGISTER_GROUPS={MAX_REGISTER_GROUPS}. This almost always means "
            f"the wrong column was picked as the drug/register-name grouping column (it should "
            f"have low cardinality — a handful to a few hundred distinct drug names, not "
            f"thousands of near-unique values). Refusing to generate {unique_groups} separate "
            f"PDFs — check the CSV's column headers."
        )
        raise HTTPException(
            status_code=422,
            detail=(
                f"Column '{register_col}' has {unique_groups} unique values "
                f"(> MAX_REGISTER_GROUPS={MAX_REGISTER_GROUPS}); refusing to generate that "
                f"many per-drug PDFs. Check that the CSV's register/drug-name column was "
                f"identified correctly."
            ),
        )

    title_style = ParagraphStyle(
        "ReportTitle",
        fontName="Times-Bold",
        fontSize=12,
        leading=15,
        alignment=1,
        spaceAfter=8,
    )
    header_style = ParagraphStyle(
        "Header",
        fontName="Times-Bold",
        fontSize=7.2,
        leading=8.2,
        alignment=1,
        wordWrap="CJK",
    )
    body_style = ParagraphStyle(
        "Body",
        fontName="Times-Roman",
        fontSize=7.0,
        leading=8.2,
        wordWrap="CJK",
    )

    columns = list(df.columns)
    col_widths = get_column_widths(columns)  # same for every group, compute once
    pdf_paths: list[Path] = []
    groups_processed = 0

    for drug_name, group in df.groupby(register_col, dropna=False, sort=True):
        drug_label = str(drug_name).strip() or "Unknown Drug"
        total_rows = len(group)
        num_parts = max(1, math.ceil(total_rows / MAX_ROWS_PER_PDF))

        if num_parts > 1:
            logger.info(
                f"[generate_pdfs] Register '{drug_label}' has {total_rows} rows -> "
                f"splitting into {num_parts} PDF part(s) of up to {MAX_ROWS_PER_PDF} rows each"
            )

        for part_idx in range(num_parts):
            start = part_idx * MAX_ROWS_PER_PDF
            end = min(start + MAX_ROWS_PER_PDF, total_rows)
            chunk = group.iloc[start:end]

            suffix = f"_Part{part_idx + 1}" if num_parts > 1 else ""
            pdf_path = output_dir / f"CD_Report_{clean_filename(drug_label)}{suffix}.pdf"

            table_data = [[make_paragraph(c, header_style) for c in columns]]
            for _, row in chunk.iterrows():
                table_data.append([make_paragraph(value, body_style) for value in row.tolist()])

            title_text = f"CD Reports - {escape(drug_label)}"
            if num_parts > 1:
                title_text += f" (Part {part_idx + 1} of {num_parts})"

            doc = SimpleDocTemplate(
                str(pdf_path),
                pagesize=landscape(A3),
                leftMargin=12,
                rightMargin=12,
                topMargin=14,
                bottomMargin=12,
            )

            table = Table(
                table_data,
                colWidths=col_widths,
                repeatRows=1,
                splitByRow=True,
            )
            table.setStyle(
                TableStyle(
                    [
                        ("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#D0D0D0")),
                        ("TEXTCOLOR", (0, 0), (-1, 0), colors.black),
                        ("GRID", (0, 0), (-1, -1), 0.45, colors.black),
                        ("VALIGN", (0, 0), (-1, -1), "TOP"),
                        ("LEFTPADDING", (0, 0), (-1, -1), 2),
                        ("RIGHTPADDING", (0, 0), (-1, -1), 2),
                        ("TOPPADDING", (0, 0), (-1, -1), 3),
                        ("BOTTOMPADDING", (0, 0), (-1, -1), 3),
                    ]
                )
            )

            story = [Paragraph(title_text, title_style), table]
            doc.build(story)
            pdf_paths.append(pdf_path)

            # Release this chunk's Paragraph objects before building the next
            # one — this is what actually keeps peak memory bounded on huge
            # CSVs, rather than accumulating every chunk's data in memory.
            del table_data, table, story, doc, chunk
            gc.collect()

        del group

        groups_processed += 1
        if groups_processed % 100 == 0 or groups_processed == unique_groups:
            logger.info(
                f"[generate_pdfs] Progress: {groups_processed}/{unique_groups} group(s), "
                f"{len(pdf_paths)} PDF(s) so far | RSS={_current_rss_mb():.0f}MB"
            )
        _check_memory_guard(f"generate_pdfs (after group {groups_processed}/{unique_groups})")

    logger.info(f"[generate_pdfs] Done: {len(pdf_paths)} PDF file(s) from {len(df)} total row(s)")
    return pdf_paths


def zip_pdfs(pdf_paths: list[Path], zip_path: Path) -> Path:
    with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zipf:
        for pdf in pdf_paths:
            zipf.write(pdf, arcname=pdf.name)
    return zip_path


def generate_single_pdf(csv_path: Path, output_dir: Path, base_filename: str) -> list[Path]:
    """Convert an entire CSV into one or more compact landscape PDFs (no
    drug/register grouping — that's what generate_pdfs() does for
    /generate-report). Used by /generate-csv-to-pdf-report.

    Returns a list of PDF paths: a single-element list for CSVs up to
    MAX_ROWS_PER_PDF rows, or multiple "{base_filename}_PartN.pdf" files for
    larger ones — callers should zip the result together when more than one
    path comes back. Splitting (and freeing each chunk's memory before
    starting the next) is what keeps peak RAM bounded on huge CSVs instead
    of holding millions of ReportLab Paragraph objects in memory at once,
    which is what silently OOM-kills the worker process on very large files.

    Page size auto-scales from A3 to A2 for wide CSVs, and column widths are
    sized from actual content length so text wraps sensibly instead of
    overflowing."""
    df = read_csv_with_fallback(csv_path).fillna("").astype(str)

    num_cols = len(df.columns)
    page_size = landscape(A3) if num_cols <= 10 else landscape(A2)
    page_width = page_size[0]

    left_margin = 20
    right_margin = 20
    usable_width = page_width - left_margin - right_margin

    body_style = ParagraphStyle(
        "SinglePdfBody",
        fontName="Helvetica",
        fontSize=5,
        leading=6,
        alignment=TA_LEFT,
        wordWrap="CJK",
    )
    header_style = ParagraphStyle(
        "SinglePdfHeader",
        parent=body_style,
        fontName="Helvetica-Bold",
    )

    # Dynamic column widths from actual content length, capped so no single
    # column can hog the whole page, then scaled to fit usable_width exactly.
    lengths = []
    for col in df.columns:
        max_len = max(len(col), df[col].str.len().max() if len(df) else len(col))
        lengths.append(min(max_len, 40))
    total_len = sum(lengths) or 1
    col_widths = [max(35, usable_width * l / total_len) for l in lengths]
    total_width = sum(col_widths)
    if total_width > usable_width:
        scale = usable_width / total_width
        col_widths = [w * scale for w in col_widths]

    columns = list(df.columns)
    total_rows = len(df)
    num_parts = max(1, math.ceil(total_rows / MAX_ROWS_PER_PDF))

    if num_parts > 1:
        logger.info(
            f"[generate_single_pdf] {total_rows} rows -> splitting into {num_parts} "
            f"PDF part(s) of up to {MAX_ROWS_PER_PDF} rows each"
        )

    pdf_paths: list[Path] = []

    for part_idx in range(num_parts):
        start = part_idx * MAX_ROWS_PER_PDF
        end = min(start + MAX_ROWS_PER_PDF, total_rows)
        chunk = df.iloc[start:end]

        suffix = f"_Part{part_idx + 1}" if num_parts > 1 else ""
        pdf_path = output_dir / f"{base_filename}{suffix}.pdf"

        table_data = [[make_paragraph(c, header_style) for c in columns]]
        for _, row in chunk.iterrows():
            table_data.append([make_paragraph(v, body_style) for v in row])

        table = LongTable(table_data, repeatRows=1, colWidths=col_widths, splitByRow=True)
        table.setStyle(
            TableStyle(
                [
                    ("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#D9EAF7")),
                    ("GRID", (0, 0), (-1, -1), 0.25, colors.grey),
                    ("FONTNAME", (0, 0), (-1, 0), "Helvetica-Bold"),
                    ("VALIGN", (0, 0), (-1, -1), "TOP"),
                    ("LEFTPADDING", (0, 0), (-1, -1), 2),
                    ("RIGHTPADDING", (0, 0), (-1, -1), 2),
                    ("TOPPADDING", (0, 0), (-1, -1), 1),
                    ("BOTTOMPADDING", (0, 0), (-1, -1), 1),
                ]
            )
        )

        doc = SimpleDocTemplate(
            str(pdf_path),
            pagesize=page_size,
            leftMargin=left_margin,
            rightMargin=right_margin,
            topMargin=20,
            bottomMargin=20,
        )
        doc.build([table])
        pdf_paths.append(pdf_path)

        # Release this chunk's Paragraph objects before building the next one.
        del table_data, table, doc, chunk
        gc.collect()

        logger.info(
            f"[generate_single_pdf] Progress: part {part_idx + 1}/{num_parts} done "
            f"| RSS={_current_rss_mb():.0f}MB"
        )
        _check_memory_guard(f"generate_single_pdf (after part {part_idx + 1}/{num_parts})")

    logger.info(f"[generate_single_pdf] Done: {len(pdf_paths)} PDF file(s) from {total_rows} total row(s)")
    return pdf_paths


# --------------------------------------------------------------------------
# HELPERS: download CSV / upload zip / send callback
# --------------------------------------------------------------------------
def build_zip_key_from_csv_url(csv_url: str) -> str:
    """
    Extract:
        receive_form : local/stage/live
        pharmacy_id
        report_stem

    Example:
    https://.../local/pharmacies/<pharmacy_id>/report_files/<report>.csv
    """

    parts = PurePosixPath(urlparse(csv_url).path).parts

    # Expected:
    # ('/', 'local', 'pharmacies', '<pharmacy_id>', 'report_files', '<report>.csv')

    if len(parts) < 6:
        raise HTTPException(status_code=422, detail="Invalid csv_url format.")

    receive_form = parts[1]

    if parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    pharmacy_id = parts[3]
    report_stem = PurePosixPath(parts[5]).stem

    return ZIP_KEY_TEMPLATE.format(
        receive_form=receive_form,
        pharmacy_id=pharmacy_id,
        report_stem=report_stem,
    )


def build_pdf_key_from_csv_url(csv_url: str) -> str:
    """Same as build_zip_key_from_csv_url but for the single-PDF endpoint —
    stores the PDF alongside the CSV in the same report_files/ folder."""
    parts = PurePosixPath(urlparse(csv_url).path).parts

    if len(parts) < 6:
        raise HTTPException(status_code=422, detail="Invalid csv_url format.")

    receive_form = parts[1]

    if parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    pharmacy_id = parts[3]
    report_stem = PurePosixPath(parts[5]).stem

    return PDF_KEY_TEMPLATE.format(
        receive_form=receive_form,
        pharmacy_id=pharmacy_id,
        report_stem=report_stem,
    )


def parse_csv_url_parts(csv_url: str) -> tuple[str, str, str, str]:
    """Cheap, synchronous parse of csv_url ->
    (receive_form, pharmacy_id, csv_filename, zip_filename).

    receive_form is "local" / "stage" / "live" and decides which Laravel
    endpoint the completion callback goes to. Used so /generate-report can
    validate + build the response instantly, before the actual heavy work is
    handed off to the background task."""
    parts = PurePosixPath(urlparse(csv_url).path).parts

    if len(parts) < 6 or parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    receive_form = parts[1]
    pharmacy_id = parts[3]
    csv_filename = os.path.basename(urlparse(csv_url).path)
    zip_filename = f"{os.path.splitext(csv_filename)[0]}.zip"

    # Fail fast if the env segment isn't one we know how to route — better to
    # 422 immediately than to do all the work and have nowhere to report it.
    resolve_callback_url(receive_form)

    return receive_form, pharmacy_id, csv_filename, zip_filename


def download_csv(csv_url: str, dest_path: Path) -> None:
    try:
        with requests.get(csv_url, stream=True, timeout=60) as r:
            r.raise_for_status()
            with open(dest_path, "wb") as f:
                for chunk in r.iter_content(chunk_size=1024 * 256):
                    f.write(chunk)
    except requests.RequestException as e:
        raise HTTPException(status_code=400, detail=f"Failed to download CSV: {e}")


def upload_to_s3(local_zip_path: Path, bucket: str, key: str, key_flag: str) -> str:
    if not bucket:
        raise HTTPException(
            status_code=500,
            detail="No S3 bucket configured. Set S3_BUCKET env var or pass 'bucket' in the request.",
        )

    s3 = boto3.client("s3", region_name=AWS_REGION)
    try:
        if key_flag == "zip_file":
            s3.upload_file(str(local_zip_path), bucket, key, ExtraArgs={"ContentType": "application/zip"})
        elif key_flag == "pdf_file":
            s3.upload_file(str(local_zip_path), bucket, key, ExtraArgs={"ContentType": "application/pdf"})
        else:
            s3.upload_file(str(local_zip_path), bucket, key)
    except (BotoCoreError, ClientError) as e:
        raise HTTPException(status_code=500, detail=f"Failed to upload {key_flag} to S3: {e}")

    if S3_PUBLIC_BASE_URL:
        return f"{S3_PUBLIC_BASE_URL.rstrip('/')}/{key.lstrip('/')}"

    try:
        return s3.generate_presigned_url(
            "get_object",
            Params={"Bucket": bucket, "Key": key},
            ExpiresIn=PRESIGNED_URL_EXPIRY_SECONDS,
        )
    except (BotoCoreError, ClientError) as e:
        raise HTTPException(status_code=500, detail=f"Zip uploaded but failed to build a URL: {e}")


def save_to_local(local_zip_path: Path, key: str) -> str:
    """Local-disk equivalent of upload_to_s3, used when STORAGE_BACKEND=local."""
    dest_path = Path(LOCAL_OUTPUT_DIR) / key
    dest_path.parent.mkdir(parents=True, exist_ok=True)
    try:
        shutil.copyfile(local_zip_path, dest_path)
    except OSError as e:
        raise HTTPException(status_code=500, detail=f"Failed to save zip locally: {e}")
    return f"{LOCAL_PUBLIC_BASE_URL}/files/{key.lstrip('/')}"


def send_callback(callback_url: str, payload: dict) -> None:
    """POST the final result to the resolved Laravel endpoint, with an
    HMAC-signed Bearer token. Best-effort: logs a warning on failure instead
    of raising, since we're already inside a fire-and-forget background task
    at this point. The token itself is never logged."""

    pharmacy_id = payload.get("pharmacy_id")
    filename = payload.get("csv_filename")

    if not pharmacy_id or not filename:
        logger.warning("Cannot send callback: missing pharmacy_id or csv_filename in payload.")
        return

    data = f"{pharmacy_id}|{filename}"
    token = hmac.new(
        AWS_FILE_CONVERT_API_KEY.encode("utf-8"),
        data.encode("utf-8"),
        hashlib.sha256,
    ).hexdigest()
    headers = {
        "Content-Type": "application/json",
        "Authorization": f"Bearer {token}",
    }

    logger.info(
        f"Sending callback -> {callback_url} | pharmacy_id={pharmacy_id} "
        f"csv_filename={filename} status={payload.get('status', 'success')}"
    )

    try:
        resp = requests.post(
            callback_url,
            json=payload,
            headers=headers,
            timeout=CALLBACK_TIMEOUT_SECONDS,
        )

        logger.info(f"Callback response <- {callback_url} | http_status={resp.status_code}")

        if not resp.ok:
            logger.error(f"Callback non-OK response from {callback_url}: {resp.text[:500]}")

    except requests.RequestException as e:
        logger.error(f"Callback request FAILED to {callback_url}: {e}")


# --------------------------------------------------------------------------
# BACKGROUND WORKER
# --------------------------------------------------------------------------
def process_report_background(
    csv_url: str,
    receive_form: str,
    pharmacy_id: str,
    csv_filename: str,
    zip_filename: str,
) -> None:
    """Runs the actual CSV -> PDFs -> zip -> upload pipeline. Called via
    BackgroundTasks so /generate-report can return immediately. Always ends
    by POSTing a result (success or failure) to the Laravel URL resolved
    from receive_form (local/stage -> STAGE_LARAVEL_URL, live -> LIVE_LARAVEL_URL)."""
    callback_url = resolve_callback_url(receive_form)
    work_dir = Path(tempfile.mkdtemp(prefix="cd_report_"))

    log_ctx = f"pharmacy_id={pharmacy_id} csv_filename={csv_filename} env={receive_form}"
    logger.info(f"[generate-report] BACKGROUND JOB STARTED | {log_ctx}")

    base_payload = {
        "request_type": "csv-to-pdf",
        "timestamp": int(time.time()),
        "pharmacy_id": pharmacy_id,
        "result_filename": zip_filename,
        "csv_filename": csv_filename,
    }

    try:
        csv_path = work_dir / "input.csv"
        pdf_dir = work_dir / "pdfs"
        pdf_dir.mkdir()

        download_csv(csv_url, csv_path)
        logger.info(f"[generate-report] CSV downloaded ({csv_path.stat().st_size} bytes) | {log_ctx}")

        pdf_paths = generate_pdfs(csv_path, pdf_dir)
        if not pdf_paths:
            logger.error(f"[generate-report] No PDFs generated (empty CSV?) | {log_ctx}")
            send_callback(callback_url, {
                **base_payload,
                "status": "failed",
                "error": "No PDFs were generated from the CSV (empty file?).",
            })
            return

        logger.info(f"[generate-report] Generated {len(pdf_paths)} PDF(s) | {log_ctx}")

        zip_key = build_zip_key_from_csv_url(csv_url)
        zip_path = work_dir / Path(zip_key).name
        zip_pdfs(pdf_paths, zip_path)
        logger.info(f"[generate-report] Zipped {len(pdf_paths)} PDF(s) -> {zip_path.name} | {log_ctx}")

        if STORAGE_BACKEND == "local":
            zip_url = save_to_local(zip_path, zip_key)
        else:
            if S3_BUCKET and STORAGE_BACKEND == "s3":
                key_flag = 'zip_file'
                zip_url = upload_to_s3(zip_path, S3_BUCKET, zip_key, key_flag)
            else:
                zip_url = save_to_local(zip_path, zip_key)

        logger.info(f"[generate-report] Zip saved -> {zip_url} | {log_ctx}")

        send_callback(callback_url, base_payload)
        logger.info(f"[generate-report] BACKGROUND JOB COMPLETED (success) | {log_ctx}")

    except HTTPException as e:
        logger.error(f"[generate-report] BACKGROUND JOB FAILED (HTTPException {e.status_code}: {e.detail}) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e.detail),
        })
    except Exception as e:
        logger.exception(f"[generate-report] BACKGROUND JOB FAILED (unexpected error) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e),
        })
    finally:
        shutil.rmtree(work_dir, ignore_errors=True)


def process_csv_to_pdf_background(
    csv_url: str,
    receive_form: str,
    pharmacy_id: str,
    csv_filename: str,
    pdf_filename: str,
) -> None:
    """Runs the CSV -> PDF(s) -> upload pipeline for
    /generate-csv-to-pdf-report. Same background-task / callback-routing
    pattern as process_report_background(), but produces one combined PDF
    via generate_single_pdf() — UNLESS the CSV has more than
    MAX_ROWS_PER_PDF rows, in which case generate_single_pdf() returns
    multiple "_PartN" PDFs and this function bundles them into a zip instead
    (same OOM-avoidance reasoning as process_report_background). Either way,
    the callback payload's 'result_filename' points at whichever file
    actually got uploaded (.pdf or .zip)."""
    callback_url = resolve_callback_url(receive_form)
    work_dir = Path(tempfile.mkdtemp(prefix="csv_to_pdf_"))

    log_ctx = f"pharmacy_id={pharmacy_id} csv_filename={csv_filename} env={receive_form}"
    logger.info(f"[generate-csv-to-pdf-report] BACKGROUND JOB STARTED | {log_ctx}")

    base_payload = {
        "request_type": "csv-to-pdf",
        "timestamp": int(time.time()),
        "pharmacy_id": pharmacy_id,
        "result_filename": pdf_filename,  # overwritten below if it ends up zipped
        "csv_filename": csv_filename,
    }

    try:
        csv_path = work_dir / "input.csv"
        pdf_dir = work_dir / "pdfs"
        pdf_dir.mkdir()

        download_csv(csv_url, csv_path)
        logger.info(f"[generate-csv-to-pdf-report] CSV downloaded ({csv_path.stat().st_size} bytes) | {log_ctx}")

        base_name = os.path.splitext(pdf_filename)[0]
        pdf_paths = generate_single_pdf(csv_path, pdf_dir, base_name)

        if not pdf_paths:
            logger.error(f"[generate-csv-to-pdf-report] No PDF generated (empty CSV?) | {log_ctx}")
            send_callback(callback_url, {
                **base_payload,
                "status": "failed",
                "error": "No PDF was generated from the CSV (empty file?).",
            })
            return

        logger.info(f"[generate-csv-to-pdf-report] PDF generated -> {len(pdf_paths)} part(s) | {log_ctx}")

        if len(pdf_paths) == 1:
            result_path = pdf_paths[0]
            result_key = build_pdf_key_from_csv_url(csv_url)
            result_filename = pdf_filename
            key_flag = "pdf_file"
        else:
            # CSV needed splitting -> bundle the parts into a zip instead of a
            # single PDF, same as /generate-report always does.
            result_filename = f"{base_name}.zip"
            result_path = work_dir / result_filename
            zip_pdfs(pdf_paths, result_path)
            result_key = build_zip_key_from_csv_url(csv_url)
            key_flag = "zip_file"
            logger.info(
                f"[generate-csv-to-pdf-report] CSV too large for one PDF -> bundled "
                f"{len(pdf_paths)} parts into {result_filename} | {log_ctx}"
            )

        base_payload["result_filename"] = result_filename

        if STORAGE_BACKEND == "local":
            result_url = save_to_local(result_path, result_key)
        else:
            if S3_BUCKET and STORAGE_BACKEND == "s3":
                result_url = upload_to_s3(result_path, S3_BUCKET, result_key, key_flag)
            else:
                result_url = save_to_local(result_path, result_key)

        logger.info(f"[generate-csv-to-pdf-report] Result saved -> {result_url} | {log_ctx}")

        send_callback(callback_url, base_payload)
        logger.info(f"[generate-csv-to-pdf-report] BACKGROUND JOB COMPLETED (success) | {log_ctx}")

    except HTTPException as e:
        logger.error(f"[generate-csv-to-pdf-report] BACKGROUND JOB FAILED (HTTPException {e.status_code}: {e.detail}) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e.detail),
        })
    except Exception as e:
        logger.exception(f"[generate-csv-to-pdf-report] BACKGROUND JOB FAILED (unexpected error) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e),
        })
    finally:
        shutil.rmtree(work_dir, ignore_errors=True)


# --------------------------------------------------------------------------
# ENDPOINTS
# --------------------------------------------------------------------------
@app.post("/generate-report", response_model=GenerateReportResponse, dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_report(req: GenerateReportRequest, background_tasks: BackgroundTasks, request: Request):
    """Validates csv_url, kicks off the CSV -> PDF -> zip -> upload pipeline
    in the background, and returns immediately. Once the background job
    finishes (success or failure), the result is POSTed to whichever Laravel
    endpoint matches the csv_url's environment segment (local/stage -> stage,
    live -> live)."""
    csv_url = str(req.csv_url)
    client_ip = request.client.host if request.client else "unknown"

    logger.info(f"[generate-report] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    try:
        # Cheap synchronous parse + validation only — no download/PDF/zip work
        # happens here, and this also confirms we know where to send the callback.
        receive_form, pharmacy_id, csv_filename, zip_filename = parse_csv_url_parts(csv_url)
    except HTTPException as e:
        logger.error(f"[generate-report] REQUEST REJECTED from {client_ip} | csv_url={csv_url} | "
                     f"{e.status_code}: {e.detail}")
        raise

    logger.info(
        f"[generate-report] REQUEST VALIDATED | env={receive_form} pharmacy_id={pharmacy_id} "
        f"csv_filename={csv_filename} zip_filename={zip_filename}"
    )

    background_tasks.add_task(
        process_report_background,
        csv_url,
        receive_form,
        pharmacy_id,
        csv_filename,
        zip_filename,
    )

    logger.info(f"[generate-report] Background job queued | pharmacy_id={pharmacy_id} csv_filename={csv_filename}")

    return GenerateReportResponse(status="success")


@app.post("/generate-csv-to-pdf-report", response_model=GenerateReportResponse, dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_csv_to_pdf_report(req: GenerateReportRequest, background_tasks: BackgroundTasks, request: Request):
    """Same request/response/callback-routing contract as /generate-report,
    but converts the whole CSV into ONE combined landscape PDF (via
    generate_single_pdf()) instead of one PDF per drug + a zip — unless the
    CSV is large enough to need splitting (see MAX_ROWS_PER_PDF), in which
    case the result is bundled into a zip instead, same as /generate-report.
    The result is stored alongside the CSV in the same report_files/ S3 (or
    local) folder.

    Request body is identical to /generate-report:
      {"csv_url": "https://.../local/pharmacies/<id>/report_files/<name>.csv"}

    Returns IMMEDIATELY: {"status": "success"}

    Once the background job finishes, the result is POSTed to whichever
    Laravel endpoint matches the csv_url's environment segment (same
    local/stage -> stage, live -> live routing as /generate-report):

      {
        "request_type": "csv-to-pdf",
        "timestamp": 1784021199,
        "pharmacy_id": "N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09",
        "result_filename": "6a38ce059932a-cd-entries.pdf",
        "csv_filename": "6a38ce059932a-cd-entries.csv"
      }
    """
    csv_url = str(req.csv_url)
    client_ip = request.client.host if request.client else "unknown"

    logger.info(f"[generate-csv-to-pdf-report] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    try:
        # Reuses the same parser/validation as /generate-report — csv_url
        # structure and environment routing are identical, only the output
        # file extension differs (.pdf instead of .zip).
        receive_form, pharmacy_id, csv_filename, _zip_filename = parse_csv_url_parts(csv_url)
    except HTTPException as e:
        logger.error(f"[generate-csv-to-pdf-report] REQUEST REJECTED from {client_ip} | csv_url={csv_url} | "
                     f"{e.status_code}: {e.detail}")
        raise

    pdf_filename = f"{os.path.splitext(csv_filename)[0]}.pdf"

    logger.info(
        f"[generate-csv-to-pdf-report] REQUEST VALIDATED | env={receive_form} pharmacy_id={pharmacy_id} "
        f"csv_filename={csv_filename} pdf_filename={pdf_filename}"
    )

    background_tasks.add_task(
        process_csv_to_pdf_background,
        csv_url,
        receive_form,
        pharmacy_id,
        csv_filename,
        pdf_filename,
    )

    logger.info(f"[generate-csv-to-pdf-report] Background job queued | pharmacy_id={pharmacy_id} csv_filename={csv_filename}")

    return GenerateReportResponse(status="success")


@app.post("/generate-report/download", dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_report_download(csv_url: HttpUrl, request: Request, zip_filename: Optional[str] = None):
    """Same pipeline, but streams the zip back directly instead of uploading to S3.
    Useful for local testing without AWS credentials. This endpoint remains
    synchronous (no background task, no callback) since the point is to get
    the file back in the same request.

    zip_filename is OPTIONAL — if omitted, it's derived automatically from
    csv_url, e.g. ".../2906266-cd-entries.csv" -> "2906266-cd-entries.zip".
    """
    client_ip = request.client.host if request.client else "unknown"
    logger.info(f"[generate-report/download] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    work_dir = Path(tempfile.mkdtemp(prefix="cd_report_dl_"))
    csv_path = work_dir / "input.csv"
    pdf_dir = work_dir / "pdfs"
    pdf_dir.mkdir()

    try:
        download_csv(str(csv_url), csv_path)
        logger.info(f"[generate-report/download] CSV downloaded ({csv_path.stat().st_size} bytes) | csv_url={csv_url}")

        pdf_paths = generate_pdfs(csv_path, pdf_dir)
        if not pdf_paths:
            logger.error(f"[generate-report/download] No PDFs generated (empty CSV?) | csv_url={csv_url}")
            raise HTTPException(status_code=422, detail="No PDFs were generated from the CSV (empty file?).")

        logger.info(f"[generate-report/download] Generated {len(pdf_paths)} PDF(s) | csv_url={csv_url}")

        if not zip_filename:
            csv_filename = os.path.basename(urlparse(str(csv_url)).path)
            zip_filename = f"{os.path.splitext(csv_filename)[0]}.zip"
        elif not zip_filename.lower().endswith(".zip"):
            zip_filename = f"{zip_filename}.zip"

        zip_path = work_dir / zip_filename
        zip_pdfs(pdf_paths, zip_path)
        logger.info(f"[generate-report/download] Streaming zip back to client -> {zip_filename} | csv_url={csv_url}")

    except HTTPException:
        shutil.rmtree(work_dir, ignore_errors=True)
        raise
    except Exception as e:
        logger.exception(f"[generate-report/download] REQUEST FAILED (unexpected error) | csv_url={csv_url}")
        shutil.rmtree(work_dir, ignore_errors=True)
        raise HTTPException(status_code=500, detail=f"Failed to generate report: {e}")

    # NOTE: work_dir is intentionally not deleted before the file is streamed;
    # FileResponse streams from disk after this function returns. In production
    # add a BackgroundTask to clean up after the response is sent.
    return FileResponse(
        path=str(zip_path),
        filename=zip_filename,
        media_type="application/zip",
    )


@app.post("/generate-csv-to-pdf-report/download", dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_csv_to_pdf_report_download(csv_url: HttpUrl, request: Request, pdf_filename: Optional[str] = None):
    """Same pipeline as /generate-csv-to-pdf-report, but streams the result
    back directly instead of uploading to S3 — the PDF equivalent of
    /generate-report/download. Returns a single .pdf for CSVs up to
    MAX_ROWS_PER_PDF rows, or a .zip of "_PartN" PDFs for larger ones (same
    fallback as the background version). Synchronous (no background task,
    no callback) since the point is to get the file back in the same request.

    pdf_filename is OPTIONAL — if omitted, it's derived automatically from
    csv_url, e.g. ".../2906266-cd-entries.csv" -> "2906266-cd-entries.pdf".
    """
    client_ip = request.client.host if request.client else "unknown"
    logger.info(f"[generate-csv-to-pdf-report/download] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    work_dir = Path(tempfile.mkdtemp(prefix="csv_to_pdf_dl_"))
    csv_path = work_dir / "input.csv"
    pdf_dir = work_dir / "pdfs"
    pdf_dir.mkdir()

    try:
        download_csv(str(csv_url), csv_path)
        logger.info(f"[generate-csv-to-pdf-report/download] CSV downloaded ({csv_path.stat().st_size} bytes) | csv_url={csv_url}")

        if not pdf_filename:
            csv_filename = os.path.basename(urlparse(str(csv_url)).path)
            pdf_filename = f"{os.path.splitext(csv_filename)[0]}.pdf"
        elif not pdf_filename.lower().endswith(".pdf"):
            pdf_filename = f"{pdf_filename}.pdf"

        base_name = os.path.splitext(pdf_filename)[0]
        pdf_paths = generate_single_pdf(csv_path, pdf_dir, base_name)

        if not pdf_paths:
            logger.error(f"[generate-csv-to-pdf-report/download] No PDF generated (empty CSV?) | csv_url={csv_url}")
            raise HTTPException(status_code=422, detail="No PDF was generated from the CSV (empty file?).")

        if len(pdf_paths) == 1:
            result_path = pdf_paths[0]
            result_filename = pdf_filename
            media_type = "application/pdf"
        else:
            result_filename = f"{base_name}.zip"
            result_path = work_dir / result_filename
            zip_pdfs(pdf_paths, result_path)
            media_type = "application/zip"
            logger.info(
                f"[generate-csv-to-pdf-report/download] CSV too large for one PDF -> bundled "
                f"{len(pdf_paths)} parts into {result_filename} | csv_url={csv_url}"
            )

        logger.info(f"[generate-csv-to-pdf-report/download] Streaming result back to client -> {result_filename} | csv_url={csv_url}")

    except HTTPException:
        shutil.rmtree(work_dir, ignore_errors=True)
        raise
    except Exception as e:
        logger.exception(f"[generate-csv-to-pdf-report/download] REQUEST FAILED (unexpected error) | csv_url={csv_url}")
        shutil.rmtree(work_dir, ignore_errors=True)
        raise HTTPException(status_code=500, detail=f"Failed to generate report: {e}")

    # NOTE: work_dir is intentionally not deleted before the file is streamed;
    # FileResponse streams from disk after this function returns. In production
    # add a BackgroundTask to clean up after the response is sent.
    return FileResponse(
        path=str(result_path),
        filename=result_filename,
        media_type=media_type,
    )


@app.get("/health")
def health():
    return JSONResponse({"status": "ok"})