"""
CD Report Generator API
========================
Single-file FastAPI service that:
  1. Downloads a CD-entries CSV from a given URL
  2. Generates one styled landscape-A3 PDF per drug (register name group)
  3. Zips all the PDFs together
  4. Uploads the zip to a target S3 "report_files/" location (built from a
     template in .env, using values parsed out of the CSV URL)
  5. Returns immediately, then POSTs the result to the correct Laravel
     endpoint (stage or live) once background processing finishes.

No database, no ORM models — everything happens in-memory / temp-dir per request.

-----------------------------------------------------------------------------
CONFIG (env vars)
-----------------------------------------------------------------------------
AWS_ACCESS_KEY_ID       - AWS access key (or use an attached IAM role / profile)
AWS_SECRET_ACCESS_KEY   - AWS secret key
AWS_REGION              - e.g. "eu-west-2"           (default: eu-west-2)
S3_BUCKET               - bucket name that backs the pharm.amazonaws.com domain
S3_PUBLIC_BASE_URL      - optional. Public base URL to build the returned link,
                           e.g. "https://pharm.amazonaws.com". If not set, a
                           presigned URL is generated instead.
ZIP_KEY_TEMPLATE        - template used to build the destination S3 key for
                           the zip, using placeholders parsed from csv_url:
                             {receive_form} - "local" / "stage" / "live" segment
                             {pharmacy_id}  - the folder segment right before
                                              "report_files" in csv_url
                             {report_stem}  - the CSV filename without ".csv"
                           Default:
                             "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.zip"
STORAGE_BACKEND         - "s3" (default) or "local".
                           Set to "local" for local development/testing with
                           NO AWS account or credentials needed at all —
                           /generate-report will save the zip to
                           LOCAL_OUTPUT_DIR on disk instead of uploading to S3.
LOCAL_OUTPUT_DIR        - folder to save zips into when STORAGE_BACKEND=local
                           (default: "./local_zip_output")

-----------------------------------------------------------------------------
CALLBACK ROUTING (Laravel)
-----------------------------------------------------------------------------
csv_url's first path segment tells us which environment the request came
from ("local", "stage", or "live"), which decides which Laravel endpoint
gets the completion callback:

    csv_url segment    -> callback goes to
    ----------------------------------------
    local              -> STAGE_LARAVEL_URL   (no separate local Laravel API)
    stage              -> STAGE_LARAVEL_URL
    live               -> LIVE_LARAVEL_URL

STAGE_LARAVEL_URL  - default "https://devstage.pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php"
LIVE_LARAVEL_URL   - default "https://pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php"
LARAVEL_SECRET_KEY - shared secret used to HMAC-sign the callback (Bearer token
                      sent is HMAC-SHA256(pharmacy_id|csv_filename), hex digest).
CALLBACK_TIMEOUT_SECONDS - request timeout for the callback POST (default: 30)

-----------------------------------------------------------------------------
LOGGING
-----------------------------------------------------------------------------
Every call to /generate-report and /generate-report/download (request
received, validation result, background job progress, storage result,
callback attempt/response, and any errors) is written to a rotating log
file, in addition to stdout (so it still shows up under uvicorn/systemd/
docker logs as before).

LOG_FILE_PATH   - path to the log file (default: "logs/api.log"). Parent
                   directories are created automatically if missing.
LOG_LEVEL       - default "INFO"
LOG_MAX_BYTES   - size in bytes before the log file rotates (default: 5MB)
LOG_BACKUP_COUNT- number of rotated backups to keep (default: 5)

-----------------------------------------------------------------------------
RUN
-----------------------------------------------------------------------------
pip install fastapi uvicorn pandas reportlab boto3 requests python-multipart python-dotenv
Create a ".env" file in the same directory (see .env.example) then run:
uvicorn main:app --host 0.0.0.0 --port 8000

-----------------------------------------------------------------------------
USAGE
-----------------------------------------------------------------------------
POST /generate-report
{
  "csv_url": "https://pharmsmart-pharmacy.s3.eu-west-2.amazonaws.com/local/pharmacies/N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09/report_files/6a38ce059932a-cd-entries.csv"
}

Returns IMMEDIATELY:
{
  "status": "success"
}

The heavy CSV -> PDFs -> zip -> upload work runs in a background task. Once
it finishes, this JSON is POSTed to the Laravel endpoint chosen by the
csv_url's environment segment (see CALLBACK ROUTING above):

{
  "request_type": "csv-to-pdf",
  "timestamp": 1784021199,
  "pharmacy_id": "N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09",
  "zip_filename": "6a38ce059932a-cd-entries.zip",
  "csv_filename": "6a38ce059932a-cd-entries.csv"
}

with header:
  Authorization: Bearer <HMAC-SHA256(pharmacy_id|csv_filename, LARAVEL_SECRET_KEY)>

On failure, "status": "failed" and "error": "<message>" are added to that
same payload (Laravel isn't required to look at these, but they're there
for visibility / debugging).

There is also POST /generate-report/download which streams the zip straight
back in the HTTP response instead of uploading anywhere (handy for testing
without S3 credentials). This one is still synchronous.
"""

import hashlib
import hmac
import logging
import os
import re
import shutil
import tempfile
import time
import zipfile
from html import escape
from logging.handlers import RotatingFileHandler
from pathlib import Path, PurePosixPath
from typing import Optional
from urllib.parse import urlparse

import boto3
import pandas as pd
import requests
from botocore.exceptions import BotoCoreError, ClientError
from dotenv import load_dotenv
from fastapi import BackgroundTasks, FastAPI, HTTPException
from fastapi.responses import FileResponse, JSONResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel, HttpUrl

load_dotenv()  # reads .env file in the working directory, if present

from reportlab.lib import colors
from reportlab.lib.enums import TA_LEFT
from reportlab.lib.pagesizes import A2, A3, landscape
from reportlab.lib.styles import ParagraphStyle
from reportlab.platypus import LongTable, Paragraph, SimpleDocTemplate, Table, TableStyle
from fastapi import FastAPI, HTTPException, Security, Depends, Request
from fastapi.security import APIKeyHeader

# --------------------------------------------------------------------------
# CONFIG
# --------------------------------------------------------------------------
AWS_REGION = os.environ.get("AWS_REGION", "eu-west-2")
S3_BUCKET = os.environ.get("S3_BUCKET", "")
S3_PUBLIC_BASE_URL = os.environ.get("S3_PUBLIC_BASE_URL", "")  # e.g. https://pharm.amazonaws.com
ZIP_KEY_TEMPLATE = os.environ.get(
    "ZIP_KEY_TEMPLATE",
    "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.zip",
)
# Used only by /generate-csv-to-pdf-report (single combined PDF, no zip).
PDF_KEY_TEMPLATE = os.environ.get(
    "PDF_KEY_TEMPLATE",
    "{receive_form}/pharmacies/{pharmacy_id}/report_files/{report_stem}.pdf",
)
PRESIGNED_URL_EXPIRY_SECONDS = int(os.environ.get("PRESIGNED_URL_EXPIRY_SECONDS", 3600))  # 1 hour default

# STORAGE_BACKEND controls where /generate-report saves the zip:
#   "s3"    (default) - uploads to real S3 via boto3, requires valid AWS creds
#   "local" - saves to LOCAL_OUTPUT_DIR on disk instead, no AWS needed at all.
STORAGE_BACKEND = os.environ.get("STORAGE_BACKEND", "s3").strip().lower()
LOCAL_OUTPUT_DIR = os.environ.get("LOCAL_OUTPUT_DIR", "./local_zip_output")
# Base URL used to build the downloadable link when STORAGE_BACKEND=local,
# e.g. "http://localhost:8000" locally, or "https://your-domain.com" in prod.
# Files are served at {LOCAL_PUBLIC_BASE_URL}/files/{key}
LOCAL_PUBLIC_BASE_URL = os.environ.get("LOCAL_PUBLIC_BASE_URL", "http://localhost:8000").rstrip("/")

# --------------------------------------------------------------------------
# LOGGING — rotating file handler, one line per event, plus stdout so
# existing uvicorn/systemd/docker log collection keeps working unchanged.
# --------------------------------------------------------------------------
LOG_FILE_PATH = os.environ.get("LOG_FILE_PATH", "logs/api.log")
LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO").upper()
LOG_MAX_BYTES = int(os.environ.get("LOG_MAX_BYTES", 5 * 1024 * 1024))  # 5MB
LOG_BACKUP_COUNT = int(os.environ.get("LOG_BACKUP_COUNT", 5))

Path(LOG_FILE_PATH).parent.mkdir(parents=True, exist_ok=True)

logger = logging.getLogger("cd_report_api")
logger.setLevel(LOG_LEVEL)
logger.propagate = False  # avoid double-logging via uvicorn's root logger

_log_formatter = logging.Formatter(
    fmt="%(asctime)s | %(levelname)-8s | %(message)s",
    datefmt="%Y-%m-%d %H:%M:%S",
)

if not logger.handlers:  # guard against duplicate handlers on --reload
    _file_handler = RotatingFileHandler(
        LOG_FILE_PATH, maxBytes=LOG_MAX_BYTES, backupCount=LOG_BACKUP_COUNT, encoding="utf-8"
    )
    _file_handler.setFormatter(_log_formatter)
    logger.addHandler(_file_handler)

    _console_handler = logging.StreamHandler()
    _console_handler.setFormatter(_log_formatter)
    logger.addHandler(_console_handler)

# --------------------------------------------------------------------------
# CALLBACK ROUTING (Laravel) — which URL to POST the completion result to,
# chosen by the "local"/"stage"/"live" segment parsed out of csv_url.
# "local" is intentionally routed to the STAGE url — there is no separate
# local Laravel API to call back into.
# --------------------------------------------------------------------------
STAGE_LARAVEL_URL = os.environ.get(
    "STAGE_LARAVEL_URL",
    "https://devstage.pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php",
)
LIVE_LARAVEL_URL = os.environ.get(
    "LIVE_LARAVEL_URL",
    "https://pharmsmart.co.uk/api/ext/v2/pharmacy/aws-csv-pdf-converter/index.php",
)
AWS_FILE_CONVERT_API_KEY = os.environ.get("LARAVEL_SECRET_KEY", "")
CALLBACK_TIMEOUT_SECONDS = int(os.environ.get("CALLBACK_TIMEOUT_SECONDS", 30))

CALLBACK_URL_BY_ENV = {
    "local": STAGE_LARAVEL_URL,
    "stage": STAGE_LARAVEL_URL,
    "live": LIVE_LARAVEL_URL,
}


def resolve_callback_url(receive_form: str) -> str:
    url = CALLBACK_URL_BY_ENV.get(receive_form.lower())
    if not url:
        raise HTTPException(
            status_code=422,
            detail=f"Unrecognised environment '{receive_form}' in csv_url "
                    f"(expected one of: {', '.join(CALLBACK_URL_BY_ENV)}).",
        )
    return url


EXCLUDE_COLUMNS = ["ID", "Amended Entry ID"]

app = FastAPI(title="Pharmsmart CD Report Generator", version="1.0.0")

# When using local storage, serve the saved zips back over plain HTTP so
# zip_url in the response is an actual downloadable link — no S3 needed.
if STORAGE_BACKEND == "local":
    Path(LOCAL_OUTPUT_DIR).mkdir(parents=True, exist_ok=True)
    app.mount("/files", StaticFiles(directory=LOCAL_OUTPUT_DIR), name="files")

# --------------------------------------------------------------------------
# CORS / ORIGIN CHECK
# --------------------------------------------------------------------------

from fastapi import Request

ALLOWED_ORIGINS = [
    o.strip().rstrip("/")
    for o in os.environ.get("ALLOWED_ORIGINS", "").split(",")
    if o.strip()
]


def verify_domain(request: Request) -> None:
    if not ALLOWED_ORIGINS:
        # No allowlist configured -> skip this check entirely
        return

    origin = (request.headers.get("origin") or request.headers.get("referer") or "").rstrip("/")

    if not origin:
        # No Origin/Referer header at all — this is a normal, expected case
        # for server-to-server calls, curl, Postman, or same-origin Swagger UI.
        return

    if not any(origin == allowed or origin.startswith(allowed + "/") for allowed in ALLOWED_ORIGINS):
        logger.warning(f"Blocked request: origin '{origin}' not in ALLOWED_ORIGINS")
        raise HTTPException(status_code=403, detail=f"Origin '{origin}' not allowed.")

# --------------------------------------------------------------------------
# API KEY AUTH
# --------------------------------------------------------------------------
API_KEY = os.environ.get("API_KEY", "")

api_key_header = APIKeyHeader(name="X-API-Key", auto_error=False)


def verify_api_key(api_key: str = Security(api_key_header)) -> None:
    if not API_KEY:
        raise HTTPException(status_code=500, detail="Server API key not configured.")
    if not api_key or api_key != API_KEY:
        logger.warning("Rejected request: invalid or missing X-API-Key")
        raise HTTPException(status_code=401, detail="Invalid or missing API key.")

# --------------------------------------------------------------------------
# REQUEST / RESPONSE MODELS
# --------------------------------------------------------------------------

class GenerateReportRequest(BaseModel):
    csv_url: HttpUrl


class GenerateReportResponse(BaseModel):
    status: str

# --------------------------------------------------------------------------
# PDF GENERATION LOGIC (same behaviour as the original script)
# --------------------------------------------------------------------------
def clean_filename(name: str) -> str:
    name = str(name).strip()
    name = re.sub(r"[^\w\s.-]", "", name)
    name = re.sub(r"\s+", "_", name)
    return name[:95] or "Unknown_Drug"


def make_paragraph(value, style: ParagraphStyle) -> Paragraph:
    text = escape(str(value)).replace("\n", "<br/>")
    return Paragraph(text, style)


def get_column_widths(columns):
    weights = []

    for c in columns:
        lc = str(c).strip().lower()

        if "register" in lc and "name" in lc:
            weights.append(2.55)
        elif "timestamp" in lc or "date" in lc:
            weights.append(1.25)
        elif "action" in lc:
            weights.append(1.20)
        elif ("name" in lc or "address" in lc) and "register" not in lc:
            weights.append(2.35)
        elif "prescriber" in lc:
            weights.append(2.05)
        elif "collect" in lc:
            weights.append(1.65)
        elif "id" in lc or "request" in lc or "provided" in lc:
            weights.append(1.25)
        elif "quantity" in lc:
            weights.append(0.92)
        elif "balance" in lc:
            weights.append(0.95)
        elif "entry" in lc:
            weights.append(1.00)
        elif "sign" in lc:
            weights.append(0.65)
        else:
            weights.append(1.05)

    page_width, _ = landscape(A3)
    usable_width = page_width - 24
    total_weight = sum(weights)

    return [usable_width * w / total_weight for w in weights]


def read_csv_with_fallback(csv_path: Path) -> pd.DataFrame:
    """Read a CSV trying multiple encodings, since pharmacy exports are often
    not UTF-8 (commonly Windows-1252 / Latin-1 from Excel/Windows systems).
    Also falls back to a more lenient parser if some rows have a ragged
    number of fields (e.g. an unescaped comma inside a text field)."""
    encodings_to_try = ["utf-8-sig", "utf-8", "cp1252", "latin-1"]
    last_error: Optional[Exception] = None

    for encoding in encodings_to_try:
        try:
            return pd.read_csv(csv_path, encoding=encoding)
        except (UnicodeDecodeError, UnicodeError) as e:
            last_error = e
            continue
        except pd.errors.ParserError as e:
            try:
                bad_rows: list[int] = []

                def _log_bad_line(bad_line: list[str]) -> None:
                    bad_rows.append(len(bad_rows))
                    return None  # drop the row

                df = pd.read_csv(
                    csv_path,
                    encoding=encoding,
                    engine="python",
                    on_bad_lines=_log_bad_line,
                )
                if bad_rows:
                    logger.warning(
                        f"Skipped {len(bad_rows)} malformed row(s) in {csv_path.name} "
                        f"(ragged field count, e.g. an unescaped comma in a text field)."
                    )
                return df
            except Exception as e2:
                last_error = e2
                continue

    raise HTTPException(status_code=422, detail=f"Could not parse CSV file: {last_error}")

def generate_pdfs(csv_path: Path, output_dir: Path) -> list[Path]:
    """Generate one drug-wise styled PDF per group, returns list of PDF paths."""
    df = read_csv_with_fallback(csv_path).fillna("").astype(str)

    df = df.drop(columns=[col for col in EXCLUDE_COLUMNS if col in df.columns])

    register_col = None
    for col in df.columns:
        if "register" in col.lower() and "name" in col.lower():
            register_col = col
            break
    if register_col is None:
        register_col = df.columns[0]

    title_style = ParagraphStyle(
        "ReportTitle",
        fontName="Times-Bold",
        fontSize=12,
        leading=15,
        alignment=1,
        spaceAfter=8,
    )
    header_style = ParagraphStyle(
        "Header",
        fontName="Times-Bold",
        fontSize=7.2,
        leading=8.2,
        alignment=1,
        wordWrap="CJK",
    )
    body_style = ParagraphStyle(
        "Body",
        fontName="Times-Roman",
        fontSize=7.0,
        leading=8.2,
        wordWrap="CJK",
    )

    pdf_paths: list[Path] = []

    for drug_name, group in df.groupby(register_col, dropna=False, sort=True):
        drug_label = str(drug_name).strip() or "Unknown Drug"
        pdf_path = output_dir / f"CD_Report_{clean_filename(drug_label)}.pdf"

        columns = list(group.columns)
        table_data = [[make_paragraph(c, header_style) for c in columns]]
        for _, row in group.iterrows():
            table_data.append([make_paragraph(value, body_style) for value in row.tolist()])

        doc = SimpleDocTemplate(
            str(pdf_path),
            pagesize=landscape(A3),
            leftMargin=12,
            rightMargin=12,
            topMargin=14,
            bottomMargin=12,
        )

        table = Table(
            table_data,
            colWidths=get_column_widths(columns),
            repeatRows=1,
            splitByRow=True,
        )
        table.setStyle(
            TableStyle(
                [
                    ("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#D0D0D0")),
                    ("TEXTCOLOR", (0, 0), (-1, 0), colors.black),
                    ("GRID", (0, 0), (-1, -1), 0.45, colors.black),
                    ("VALIGN", (0, 0), (-1, -1), "TOP"),
                    ("LEFTPADDING", (0, 0), (-1, -1), 2),
                    ("RIGHTPADDING", (0, 0), (-1, -1), 2),
                    ("TOPPADDING", (0, 0), (-1, -1), 3),
                    ("BOTTOMPADDING", (0, 0), (-1, -1), 3),
                ]
            )
        )

        story = [Paragraph(f"CD Reports - {escape(drug_label)}", title_style), table]
        doc.build(story)
        pdf_paths.append(pdf_path)

    return pdf_paths


def zip_pdfs(pdf_paths: list[Path], zip_path: Path) -> Path:
    with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zipf:
        for pdf in pdf_paths:
            zipf.write(pdf, arcname=pdf.name)
    return zip_path


def generate_single_pdf(csv_path: Path, pdf_path: Path) -> Path:
    """Convert an entire CSV into ONE compact landscape PDF (no drug/register
    grouping — that's what generate_pdfs() above does for /generate-report).
    Used by /generate-csv-to-pdf-report. Page size auto-scales from A3 to A2
    for wide CSVs, and column widths are sized from actual content length so
    text wraps sensibly instead of overflowing."""
    df = read_csv_with_fallback(csv_path).fillna("").astype(str)

    num_cols = len(df.columns)
    page_size = landscape(A3) if num_cols <= 10 else landscape(A2)
    page_width = page_size[0]

    left_margin = 20
    right_margin = 20
    usable_width = page_width - left_margin - right_margin

    body_style = ParagraphStyle(
        "SinglePdfBody",
        fontName="Helvetica",
        fontSize=5,
        leading=6,
        alignment=TA_LEFT,
        wordWrap="CJK",
    )
    header_style = ParagraphStyle(
        "SinglePdfHeader",
        parent=body_style,
        fontName="Helvetica-Bold",
    )

    # Dynamic column widths from actual content length, capped so no single
    # column can hog the whole page, then scaled to fit usable_width exactly.
    lengths = []
    for col in df.columns:
        max_len = max(len(col), df[col].str.len().max() if len(df) else len(col))
        lengths.append(min(max_len, 40))
    total_len = sum(lengths) or 1
    col_widths = [max(35, usable_width * l / total_len) for l in lengths]
    total_width = sum(col_widths)
    if total_width > usable_width:
        scale = usable_width / total_width
        col_widths = [w * scale for w in col_widths]

    table_data = [[Paragraph(str(c), header_style) for c in df.columns]]
    for _, row in df.iterrows():
        table_data.append([Paragraph(v, body_style) for v in row])

    table = LongTable(table_data, repeatRows=1, colWidths=col_widths, splitByRow=True)
    table.setStyle(
        TableStyle(
            [
                ("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#D9EAF7")),
                ("GRID", (0, 0), (-1, -1), 0.25, colors.grey),
                ("FONTNAME", (0, 0), (-1, 0), "Helvetica-Bold"),
                ("VALIGN", (0, 0), (-1, -1), "TOP"),
                ("LEFTPADDING", (0, 0), (-1, -1), 2),
                ("RIGHTPADDING", (0, 0), (-1, -1), 2),
                ("TOPPADDING", (0, 0), (-1, -1), 1),
                ("BOTTOMPADDING", (0, 0), (-1, -1), 1),
            ]
        )
    )

    doc = SimpleDocTemplate(
        str(pdf_path),
        pagesize=page_size,
        leftMargin=left_margin,
        rightMargin=right_margin,
        topMargin=20,
        bottomMargin=20,
    )
    doc.build([table])
    return pdf_path


# --------------------------------------------------------------------------
# HELPERS: download CSV / upload zip / send callback
# --------------------------------------------------------------------------
def build_zip_key_from_csv_url(csv_url: str) -> str:
    """
    Extract:
        receive_form : local/stage/live
        pharmacy_id
        report_stem

    Example:
    https://.../local/pharmacies/<pharmacy_id>/report_files/<report>.csv
    """

    parts = PurePosixPath(urlparse(csv_url).path).parts

    # Expected:
    # ('/', 'local', 'pharmacies', '<pharmacy_id>', 'report_files', '<report>.csv')

    if len(parts) < 6:
        raise HTTPException(status_code=422, detail="Invalid csv_url format.")

    receive_form = parts[1]

    if parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    pharmacy_id = parts[3]
    report_stem = PurePosixPath(parts[5]).stem

    return ZIP_KEY_TEMPLATE.format(
        receive_form=receive_form,
        pharmacy_id=pharmacy_id,
        report_stem=report_stem,
    )


def build_pdf_key_from_csv_url(csv_url: str) -> str:
    """Same as build_zip_key_from_csv_url but for the single-PDF endpoint —
    stores the PDF alongside the CSV in the same report_files/ folder."""
    parts = PurePosixPath(urlparse(csv_url).path).parts

    if len(parts) < 6:
        raise HTTPException(status_code=422, detail="Invalid csv_url format.")

    receive_form = parts[1]

    if parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    pharmacy_id = parts[3]
    report_stem = PurePosixPath(parts[5]).stem

    return PDF_KEY_TEMPLATE.format(
        receive_form=receive_form,
        pharmacy_id=pharmacy_id,
        report_stem=report_stem,
    )


def parse_csv_url_parts(csv_url: str) -> tuple[str, str, str, str]:
    """Cheap, synchronous parse of csv_url ->
    (receive_form, pharmacy_id, csv_filename, zip_filename).

    receive_form is "local" / "stage" / "live" and decides which Laravel
    endpoint the completion callback goes to. Used so /generate-report can
    validate + build the response instantly, before the actual heavy work is
    handed off to the background task."""
    parts = PurePosixPath(urlparse(csv_url).path).parts

    if len(parts) < 6 or parts[2] != "pharmacies" or parts[4] != "report_files":
        raise HTTPException(status_code=422, detail="Unexpected csv_url structure.")

    receive_form = parts[1]
    pharmacy_id = parts[3]
    csv_filename = os.path.basename(urlparse(csv_url).path)
    zip_filename = f"{os.path.splitext(csv_filename)[0]}.zip"

    # Fail fast if the env segment isn't one we know how to route — better to
    # 422 immediately than to do all the work and have nowhere to report it.
    resolve_callback_url(receive_form)

    return receive_form, pharmacy_id, csv_filename, zip_filename


def download_csv(csv_url: str, dest_path: Path) -> None:
    try:
        with requests.get(csv_url, stream=True, timeout=60) as r:
            r.raise_for_status()
            with open(dest_path, "wb") as f:
                for chunk in r.iter_content(chunk_size=1024 * 256):
                    f.write(chunk)
    except requests.RequestException as e:
        raise HTTPException(status_code=400, detail=f"Failed to download CSV: {e}")


def upload_to_s3(local_zip_path: Path, bucket: str, key: str, key_flag: str) -> str:
    if not bucket:
        raise HTTPException(
            status_code=500,
            detail="No S3 bucket configured. Set S3_BUCKET env var or pass 'bucket' in the request.",
        )

    s3 = boto3.client("s3", region_name=AWS_REGION)
    try:
        if key_flag == "zip_file":
            s3.upload_file(str(local_zip_path), bucket, key, ExtraArgs={"ContentType": "application/zip"})
        elif key_flag == "pdf_file":
            s3.upload_file(str(local_zip_path), bucket, key, ExtraArgs={"ContentType": "application/pdf"})
    except (BotoCoreError, ClientError) as e:
        raise HTTPException(status_code=500, detail=f"Failed to upload {key_flag} to S3: {e}")

    if S3_PUBLIC_BASE_URL:
        return f"{S3_PUBLIC_BASE_URL.rstrip('/')}/{key.lstrip('/')}"

    try:
        return s3.generate_presigned_url(
            "get_object",
            Params={"Bucket": bucket, "Key": key},
            ExpiresIn=PRESIGNED_URL_EXPIRY_SECONDS,
        )
    except (BotoCoreError, ClientError) as e:
        raise HTTPException(status_code=500, detail=f"Zip uploaded but failed to build a URL: {e}")


def save_to_local(local_zip_path: Path, key: str) -> str:
    """Local-disk equivalent of upload_to_s3, used when STORAGE_BACKEND=local."""
    dest_path = Path(LOCAL_OUTPUT_DIR) / key
    dest_path.parent.mkdir(parents=True, exist_ok=True)
    try:
        shutil.copyfile(local_zip_path, dest_path)
    except OSError as e:
        raise HTTPException(status_code=500, detail=f"Failed to save zip locally: {e}")
    return f"{LOCAL_PUBLIC_BASE_URL}/files/{key.lstrip('/')}"


def send_callback(callback_url: str, payload: dict) -> None:
    """POST the final result to the resolved Laravel endpoint, with an
    HMAC-signed Bearer token. Best-effort: logs a warning on failure instead
    of raising, since we're already inside a fire-and-forget background task
    at this point. The token itself is never logged."""

    pharmacy_id = payload.get("pharmacy_id")
    filename = payload.get("csv_filename")

    if not pharmacy_id or not filename:
        logger.warning("Cannot send callback: missing pharmacy_id or csv_filename in payload.")
        return

    data = f"{pharmacy_id}|{filename}"
    token = hmac.new(
        AWS_FILE_CONVERT_API_KEY.encode("utf-8"),
        data.encode("utf-8"),
        hashlib.sha256,
    ).hexdigest()
    headers = {
        "Content-Type": "application/json",
        "Authorization": f"Bearer {token}",
    }

    logger.info(
        f"Sending callback -> {callback_url} | pharmacy_id={pharmacy_id} "
        f"csv_filename={filename} status={payload.get('status', 'success')}"
    )

    try:
        resp = requests.post(
            callback_url,
            json=payload,
            headers=headers,
            timeout=CALLBACK_TIMEOUT_SECONDS,
        )

        logger.info(f"Callback response <- {callback_url} | http_status={resp.status_code}")

        if not resp.ok:
            logger.error(f"Callback non-OK response from {callback_url}: {resp.text[:500]}")

    except requests.RequestException as e:
        logger.error(f"Callback request FAILED to {callback_url}: {e}")


# --------------------------------------------------------------------------
# BACKGROUND WORKER
# --------------------------------------------------------------------------
def process_report_background(
    csv_url: str,
    receive_form: str,
    pharmacy_id: str,
    csv_filename: str,
    zip_filename: str,
) -> None:
    """Runs the actual CSV -> PDFs -> zip -> upload pipeline. Called via
    BackgroundTasks so /generate-report can return immediately. Always ends
    by POSTing a result (success or failure) to the Laravel URL resolved
    from receive_form (local/stage -> STAGE_LARAVEL_URL, live -> LIVE_LARAVEL_URL)."""
    callback_url = resolve_callback_url(receive_form)
    work_dir = Path(tempfile.mkdtemp(prefix="cd_report_"))

    log_ctx = f"pharmacy_id={pharmacy_id} csv_filename={csv_filename} env={receive_form}"
    logger.info(f"[generate-report] BACKGROUND JOB STARTED | {log_ctx}")

    base_payload = {
        "request_type": "csv-to-pdf",
        "timestamp": int(time.time()),
        "pharmacy_id": pharmacy_id,
        "result_filename": zip_filename,
        "csv_filename": csv_filename,
    }

    try:
        csv_path = work_dir / "input.csv"
        pdf_dir = work_dir / "pdfs"
        pdf_dir.mkdir()

        download_csv(csv_url, csv_path)
        logger.info(f"[generate-report] CSV downloaded ({csv_path.stat().st_size} bytes) | {log_ctx}")

        pdf_paths = generate_pdfs(csv_path, pdf_dir)
        if not pdf_paths:
            logger.error(f"[generate-report] No PDFs generated (empty CSV?) | {log_ctx}")
            send_callback(callback_url, {
                **base_payload,
                "status": "failed",
                "error": "No PDFs were generated from the CSV (empty file?).",
            })
            return

        logger.info(f"[generate-report] Generated {len(pdf_paths)} PDF(s) | {log_ctx}")

        zip_key = build_zip_key_from_csv_url(csv_url)
        zip_path = work_dir / Path(zip_key).name
        zip_pdfs(pdf_paths, zip_path)
        logger.info(f"[generate-report] Zipped {len(pdf_paths)} PDF(s) -> {zip_path.name} | {log_ctx}")

        if STORAGE_BACKEND == "local":
            zip_url = save_to_local(zip_path, zip_key)
        else:
            if S3_BUCKET and STORAGE_BACKEND == "s3":
                key_flag = 'zip_file'
                zip_url = upload_to_s3(zip_path, S3_BUCKET, zip_key, key_flag)
            else:
                zip_url = save_to_local(zip_path, zip_key)

        logger.info(f"[generate-report] Zip saved -> {zip_url} | {log_ctx}")

        send_callback(callback_url, base_payload)
        logger.info(f"[generate-report] BACKGROUND JOB COMPLETED (success) | {log_ctx}")

    except HTTPException as e:
        logger.error(f"[generate-report] BACKGROUND JOB FAILED (HTTPException {e.status_code}: {e.detail}) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e.detail),
        })
    except Exception as e:
        logger.exception(f"[generate-report] BACKGROUND JOB FAILED (unexpected error) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e),
        })
    finally:
        shutil.rmtree(work_dir, ignore_errors=True)


def process_csv_to_pdf_background(
    csv_url: str,
    receive_form: str,
    pharmacy_id: str,
    csv_filename: str,
    pdf_filename: str,
) -> None:
    """Runs the CSV -> single PDF -> upload pipeline for
    /generate-csv-to-pdf-report. Same background-task / callback-routing
    pattern as process_report_background(), but produces one combined PDF
    (via generate_single_pdf()) instead of one PDF per drug + a zip, and the
    callback payload carries 'pdf_filename' instead of 'zip_filename'."""
    callback_url = resolve_callback_url(receive_form)
    # print("callback_url", callback_url)
    work_dir = Path(tempfile.mkdtemp(prefix="csv_to_pdf_"))
    # print("work_dir", work_dir)
    log_ctx = f"pharmacy_id={pharmacy_id} csv_filename={csv_filename} env={receive_form}"
    logger.info(f"[generate-csv-to-pdf-report] BACKGROUND JOB STARTED | {log_ctx}")

    base_payload = {
        "request_type": "csv-to-pdf",
        "timestamp": int(time.time()),
        "pharmacy_id": pharmacy_id,
        "result_filename": pdf_filename,
        "csv_filename": csv_filename,
    }

    # print('base_payload',base_payload)

    try:
        csv_path = work_dir / "input.csv"

        download_csv(csv_url, csv_path)
        logger.info(f"[generate-csv-to-pdf-report] CSV downloaded ({csv_path.stat().st_size} bytes) | {log_ctx}")

        pdf_key = build_pdf_key_from_csv_url(csv_url)
        pdf_path = work_dir / Path(pdf_key).name
        # print('pdf_path', pdf_path)
        # print('csv_path', csv_path)
        generate_single_pdf(csv_path, pdf_path)
        logger.info(f"[generate-csv-to-pdf-report] PDF generated -> {pdf_path.name} | {log_ctx}")

        if STORAGE_BACKEND == "local":
            pdf_url = save_to_local(pdf_path, pdf_key)
        else:
            if S3_BUCKET and STORAGE_BACKEND == "s3":
                key_flag = 'pdf_file'
                pdf_url = upload_to_s3(pdf_path, S3_BUCKET, pdf_key, key_flag)
                # print("pdf_url", pdf_url)
            else:
                pdf_url = save_to_local(pdf_path, pdf_key)

        logger.info(f"[generate-csv-to-pdf-report] PDF saved -> {pdf_url} | {log_ctx}")

        print('base_payload', base_payload)
        send_callback(callback_url, base_payload)
        logger.info(f"[generate-csv-to-pdf-report] BACKGROUND JOB COMPLETED (success) | {log_ctx}")

    except HTTPException as e:
        logger.error(f"[generate-csv-to-pdf-report] BACKGROUND JOB FAILED (HTTPException {e.status_code}: {e.detail}) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e.detail),
        })
    except Exception as e:
        logger.exception(f"[generate-csv-to-pdf-report] BACKGROUND JOB FAILED (unexpected error) | {log_ctx}")
        send_callback(callback_url, {
            **base_payload,
            "status": "failed",
            "error": str(e),
        })
    finally:
        shutil.rmtree(work_dir, ignore_errors=True)


# --------------------------------------------------------------------------
# ENDPOINTS
# --------------------------------------------------------------------------
@app.post("/generate-report", response_model=GenerateReportResponse, dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_report(req: GenerateReportRequest, background_tasks: BackgroundTasks, request: Request):
    """Validates csv_url, kicks off the CSV -> PDF -> zip -> upload pipeline
    in the background, and returns immediately. Once the background job
    finishes (success or failure), the result is POSTed to whichever Laravel
    endpoint matches the csv_url's environment segment (local/stage -> stage,
    live -> live)."""
    csv_url = str(req.csv_url)
    client_ip = request.client.host if request.client else "unknown"

    logger.info(f"[generate-report] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    try:
        # Cheap synchronous parse + validation only — no download/PDF/zip work
        # happens here, and this also confirms we know where to send the callback.
        receive_form, pharmacy_id, csv_filename, zip_filename = parse_csv_url_parts(csv_url)
    except HTTPException as e:
        logger.error(f"[generate-report] REQUEST REJECTED from {client_ip} | csv_url={csv_url} | "
                     f"{e.status_code}: {e.detail}")
        raise

    logger.info(
        f"[generate-report] REQUEST VALIDATED | env={receive_form} pharmacy_id={pharmacy_id} "
        f"csv_filename={csv_filename} zip_filename={zip_filename}"
    )

    background_tasks.add_task(
        process_report_background,
        csv_url,
        receive_form,
        pharmacy_id,
        csv_filename,
        zip_filename,
    )

    logger.info(f"[generate-report] Background job queued | pharmacy_id={pharmacy_id} csv_filename={csv_filename}")

    return GenerateReportResponse(status="success")


@app.post("/generate-csv-to-pdf-report", response_model=GenerateReportResponse, dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_csv_to_pdf_report(req: GenerateReportRequest, background_tasks: BackgroundTasks, request: Request):
    """Same request/response/callback-routing contract as /generate-report,
    but converts the whole CSV into ONE combined landscape PDF (via
    generate_single_pdf()) instead of one PDF per drug + a zip. The PDF is
    stored alongside the CSV in the same report_files/ S3 (or local) folder.

    Request body is identical to /generate-report:
      {"csv_url": "https://.../local/pharmacies/<id>/report_files/<name>.csv"}

    Returns IMMEDIATELY: {"status": "success"}

    Once the background job finishes, the result is POSTed to whichever
    Laravel endpoint matches the csv_url's environment segment (same
    local/stage -> stage, live -> live routing as /generate-report):

      {
        "request_type": "csv-to-pdf",
        "timestamp": 1784021199,
        "pharmacy_id": "N0tZbTJJRTFSYVhHanpGSm1EbVFKdz09",
        "pdf_filename": "6a38ce059932a-cd-entries.pdf",
        "csv_filename": "6a38ce059932a-cd-entries.csv"
      }
    """
    csv_url = str(req.csv_url)
    client_ip = request.client.host if request.client else "unknown"

    logger.info(f"[generate-csv-to-pdf-report] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    try:
        # Reuses the same parser/validation as /generate-report — csv_url
        # structure and environment routing are identical, only the output
        # file extension differs (.pdf instead of .zip).
        receive_form, pharmacy_id, csv_filename, _zip_filename = parse_csv_url_parts(csv_url)
    except HTTPException as e:
        logger.error(f"[generate-csv-to-pdf-report] REQUEST REJECTED from {client_ip} | csv_url={csv_url} | "
                     f"{e.status_code}: {e.detail}")
        raise

    pdf_filename = f"{os.path.splitext(csv_filename)[0]}.pdf"

    logger.info(
        f"[generate-csv-to-pdf-report] REQUEST VALIDATED | env={receive_form} pharmacy_id={pharmacy_id} "
        f"csv_filename={csv_filename} pdf_filename={pdf_filename}"
    )

    background_tasks.add_task(
        process_csv_to_pdf_background,
        csv_url,
        receive_form,
        pharmacy_id,
        csv_filename,
        pdf_filename,
    )

    logger.info(f"[generate-csv-to-pdf-report] Background job queued | pharmacy_id={pharmacy_id} csv_filename={csv_filename}")

    return GenerateReportResponse(status="success")


@app.post("/generate-report/download", dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_report_download(csv_url: HttpUrl, request: Request, zip_filename: Optional[str] = None):
    """Same pipeline, but streams the zip back directly instead of uploading to S3.
    Useful for local testing without AWS credentials. This endpoint remains
    synchronous (no background task, no callback) since the point is to get
    the file back in the same request.

    zip_filename is OPTIONAL — if omitted, it's derived automatically from
    csv_url, e.g. ".../2906266-cd-entries.csv" -> "2906266-cd-entries.zip".
    """
    client_ip = request.client.host if request.client else "unknown"
    logger.info(f"[generate-report/download] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    work_dir = Path(tempfile.mkdtemp(prefix="cd_report_dl_"))
    csv_path = work_dir / "input.csv"
    pdf_dir = work_dir / "pdfs"
    pdf_dir.mkdir()

    try:
        download_csv(str(csv_url), csv_path)
        logger.info(f"[generate-report/download] CSV downloaded ({csv_path.stat().st_size} bytes) | csv_url={csv_url}")

        pdf_paths = generate_pdfs(csv_path, pdf_dir)
        if not pdf_paths:
            logger.error(f"[generate-report/download] No PDFs generated (empty CSV?) | csv_url={csv_url}")
            raise HTTPException(status_code=422, detail="No PDFs were generated from the CSV (empty file?).")

        logger.info(f"[generate-report/download] Generated {len(pdf_paths)} PDF(s) | csv_url={csv_url}")

        if not zip_filename:
            csv_filename = os.path.basename(urlparse(str(csv_url)).path)
            zip_filename = f"{os.path.splitext(csv_filename)[0]}.zip"
        elif not zip_filename.lower().endswith(".zip"):
            zip_filename = f"{zip_filename}.zip"

        zip_path = work_dir / zip_filename
        zip_pdfs(pdf_paths, zip_path)
        logger.info(f"[generate-report/download] Streaming zip back to client -> {zip_filename} | csv_url={csv_url}")

    except HTTPException:
        shutil.rmtree(work_dir, ignore_errors=True)
        raise
    except Exception as e:
        logger.exception(f"[generate-report/download] REQUEST FAILED (unexpected error) | csv_url={csv_url}")
        shutil.rmtree(work_dir, ignore_errors=True)
        raise HTTPException(status_code=500, detail=f"Failed to generate report: {e}")

    # NOTE: work_dir is intentionally not deleted before the file is streamed;
    # FileResponse streams from disk after this function returns. In production
    # add a BackgroundTask to clean up after the response is sent.
    return FileResponse(
        path=str(zip_path),
        filename=zip_filename,
        media_type="application/zip",
    )

@app.post("/generate-csv-to-pdf-report/download", dependencies=[Depends(verify_api_key), Depends(verify_domain)])
def generate_csv_to_pdf_report_download(csv_url: HttpUrl, request: Request, pdf_filename: Optional[str] = None):
    """Same pipeline as /generate-csv-to-pdf-report, but streams the single
    combined PDF back directly instead of uploading to S3 — the PDF
    equivalent of /generate-report/download. Useful for local testing
    without AWS credentials. Synchronous (no background task, no callback)
    since the point is to get the file back in the same request.

    pdf_filename is OPTIONAL — if omitted, it's derived automatically from
    csv_url, e.g. ".../2906266-cd-entries.csv" -> "2906266-cd-entries.pdf".
    """
    client_ip = request.client.host if request.client else "unknown"
    logger.info(f"[generate-csv-to-pdf-report/download] REQUEST RECEIVED from {client_ip} | csv_url={csv_url}")

    work_dir = Path(tempfile.mkdtemp(prefix="csv_to_pdf_dl_"))
    csv_path = work_dir / "input.csv"

    try:
        download_csv(str(csv_url), csv_path)
        logger.info(f"[generate-csv-to-pdf-report/download] CSV downloaded ({csv_path.stat().st_size} bytes) | csv_url={csv_url}")

        if not pdf_filename:
            csv_filename = os.path.basename(urlparse(str(csv_url)).path)
            pdf_filename = f"{os.path.splitext(csv_filename)[0]}.pdf"
        elif not pdf_filename.lower().endswith(".pdf"):
            pdf_filename = f"{pdf_filename}.pdf"

        pdf_path = work_dir / pdf_filename
        generate_single_pdf(csv_path, pdf_path)
        logger.info(f"[generate-csv-to-pdf-report/download] Streaming PDF back to client -> {pdf_filename} | csv_url={csv_url}")

    except HTTPException:
        shutil.rmtree(work_dir, ignore_errors=True)
        raise
    except Exception as e:
        logger.exception(f"[generate-csv-to-pdf-report/download] REQUEST FAILED (unexpected error) | csv_url={csv_url}")
        shutil.rmtree(work_dir, ignore_errors=True)
        raise HTTPException(status_code=500, detail=f"Failed to generate report: {e}")

    # NOTE: work_dir is intentionally not deleted before the file is streamed;
    # FileResponse streams from disk after this function returns. In production
    # add a BackgroundTask to clean up after the response is sent.
    return FileResponse(
        path=str(pdf_path),
        filename=pdf_filename,
        media_type="application/pdf",
    )

@app.get("/health")
def health():
    return JSONResponse({"status": "ok"})