Initial project setup: NF Hotel Data Analysis API
- Add project structure with domain-driven design organization - Implement FastAPI endpoints for hotel booking reports - Add data cleaning, statistics, and LLM report generation services - Include configuration management and security utilities - Add repository layer for bookings and metadata access - Setup testing framework with conftest.py - Include example files and documentation
This commit is contained in:
@@ -0,0 +1,48 @@
|
||||
from functools import lru_cache
|
||||
|
||||
from fastapi import Depends
|
||||
|
||||
from nf_hotel_api.core.config import Settings, get_settings
|
||||
from nf_hotel_api.domain.metadata import HotelMetadata
|
||||
from nf_hotel_api.repositories.metadata_repository import JsonHotelMetadataRepository
|
||||
from nf_hotel_api.services.cleaning import DataCleaningService
|
||||
from nf_hotel_api.services.llm_report import LLMReportService
|
||||
from nf_hotel_api.services.report import ReportService
|
||||
from nf_hotel_api.services.statistics import DescriptiveStatsService
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_hotel_metadata() -> HotelMetadata:
|
||||
return JsonHotelMetadataRepository(get_settings().metadata_path).load()
|
||||
|
||||
|
||||
def get_cleaning_service(
|
||||
metadata: HotelMetadata = Depends(get_hotel_metadata),
|
||||
) -> DataCleaningService:
|
||||
return DataCleaningService(metadata)
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_stats_service() -> DescriptiveStatsService:
|
||||
return DescriptiveStatsService()
|
||||
|
||||
|
||||
def get_llm_service(
|
||||
settings: Settings = Depends(get_settings),
|
||||
metadata: HotelMetadata = Depends(get_hotel_metadata),
|
||||
) -> LLMReportService:
|
||||
return LLMReportService(
|
||||
metadata=metadata,
|
||||
base_url=settings.llm_base_url,
|
||||
api_key=settings.llm_api_key,
|
||||
model=settings.llm_model,
|
||||
timeout_seconds=settings.llm_timeout_seconds,
|
||||
)
|
||||
|
||||
|
||||
def get_report_service(
|
||||
cleaning_service: DataCleaningService = Depends(get_cleaning_service),
|
||||
stats_service: DescriptiveStatsService = Depends(get_stats_service),
|
||||
llm_service: LLMReportService = Depends(get_llm_service),
|
||||
) -> ReportService:
|
||||
return ReportService(cleaning_service, stats_service, llm_service)
|
||||
@@ -0,0 +1,47 @@
|
||||
from fastapi import APIRouter, Depends, HTTPException, status
|
||||
|
||||
from nf_hotel_api.api.deps import get_report_service
|
||||
from nf_hotel_api.core.config import Settings, get_settings
|
||||
from nf_hotel_api.core.security import require_api_key
|
||||
from nf_hotel_api.domain.schemas import BookingBatch, ReportResponse
|
||||
from nf_hotel_api.repositories.booking_repository import (
|
||||
CsvBookingRepository,
|
||||
JsonBookingRepository,
|
||||
)
|
||||
from nf_hotel_api.services.llm_report import LLMServiceError
|
||||
from nf_hotel_api.services.report import ReportService
|
||||
|
||||
router = APIRouter(
|
||||
prefix="/report",
|
||||
tags=["report"],
|
||||
dependencies=[Depends(require_api_key)],
|
||||
)
|
||||
|
||||
|
||||
@router.post("/from-file", response_model=ReportResponse)
|
||||
async def generate_report_from_file(
|
||||
report_service: ReportService = Depends(get_report_service),
|
||||
settings: Settings = Depends(get_settings),
|
||||
) -> ReportResponse:
|
||||
"""Generate the analytics report from the bundled nf_hotel_bookings.csv file."""
|
||||
repository = CsvBookingRepository(settings.default_data_path, settings.csv_separator)
|
||||
try:
|
||||
return await report_service.generate(repository)
|
||||
except FileNotFoundError as exc:
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=str(exc)) from exc
|
||||
except LLMServiceError as exc:
|
||||
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
|
||||
|
||||
|
||||
@router.post("/from-json", response_model=ReportResponse)
|
||||
async def generate_report_from_json(
|
||||
batch: BookingBatch,
|
||||
report_service: ReportService = Depends(get_report_service),
|
||||
) -> ReportResponse:
|
||||
"""Generate the analytics report from booking records supplied as JSON."""
|
||||
records = [record.model_dump() for record in batch.records]
|
||||
repository = JsonBookingRepository(records)
|
||||
try:
|
||||
return await report_service.generate(repository)
|
||||
except LLMServiceError as exc:
|
||||
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
|
||||
@@ -0,0 +1,6 @@
|
||||
from fastapi import APIRouter
|
||||
|
||||
from nf_hotel_api.api.v1.endpoints import report
|
||||
|
||||
api_router = APIRouter(prefix="/api/v1")
|
||||
api_router.include_router(report.router)
|
||||
@@ -0,0 +1,48 @@
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
||||
from pydantic import field_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
# src/nf_hotel_api/core/config.py -> project root is three levels up.
|
||||
_PROJECT_ROOT = Path(__file__).resolve().parents[3]
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""Central application configuration, loaded from environment / .env."""
|
||||
|
||||
model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore")
|
||||
|
||||
# --- API security ---
|
||||
api_key: str
|
||||
allowed_origins: list[str] = ["http://localhost:3000"]
|
||||
|
||||
# --- Local data source (fallback / default dataset) ---
|
||||
default_data_path: Path = Path("data/nf_hotel_bookings.csv")
|
||||
csv_separator: str = ";"
|
||||
# Reference data: room types, standard prices and how many rooms exist.
|
||||
metadata_path: Path = Path("data/hotel_metadata.json")
|
||||
|
||||
@field_validator("default_data_path", "metadata_path")
|
||||
@classmethod
|
||||
def _resolve_relative_to_project_root(cls, value: Path) -> Path:
|
||||
# Anchored to the project root, not the process's current working
|
||||
# directory - otherwise this 404s whenever uvicorn/pytest is
|
||||
# launched from anywhere other than the repo root.
|
||||
return value if value.is_absolute() else _PROJECT_ROOT / value
|
||||
|
||||
# --- LLM (OpenAI-compatible local inference server, e.g. vLLM / LM Studio) ---
|
||||
llm_base_url: str = "http://localhost:1234/v1"
|
||||
llm_api_key: str = "not-needed"
|
||||
llm_model: str = "qwen3.8-whittle-moe-27b-a17.8b"
|
||||
# Reasoning models can spend minutes generating <think>/reasoning_content
|
||||
# tokens before the final answer - keep this generous.
|
||||
llm_timeout_seconds: float = 900.0
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_settings() -> Settings:
|
||||
# api_key (and other required fields) come from the environment / .env
|
||||
# at runtime via BaseSettings, not from constructor arguments - static
|
||||
# type checkers can't see that, hence the ignore.
|
||||
return Settings() # type: ignore[call-arg]
|
||||
@@ -0,0 +1,30 @@
|
||||
import secrets
|
||||
|
||||
from fastapi import Depends, HTTPException, Security, status
|
||||
from fastapi.security import APIKeyHeader
|
||||
|
||||
from nf_hotel_api.core.config import Settings, get_settings
|
||||
|
||||
_api_key_header = APIKeyHeader(name="X-API-Key", auto_error=False)
|
||||
|
||||
|
||||
class ApiKeyAuthenticator:
|
||||
"""Validates inbound requests against the configured API key.
|
||||
|
||||
Uses a constant-time comparison to avoid leaking key length/content via
|
||||
response-time side channels.
|
||||
"""
|
||||
|
||||
def __call__(
|
||||
self,
|
||||
api_key: str | None = Security(_api_key_header),
|
||||
settings: Settings = Depends(get_settings),
|
||||
) -> None:
|
||||
if not api_key or not secrets.compare_digest(api_key, settings.api_key):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED,
|
||||
detail="Invalid or missing API key",
|
||||
)
|
||||
|
||||
|
||||
require_api_key = ApiKeyAuthenticator()
|
||||
@@ -0,0 +1,34 @@
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class RoomType(BaseModel):
|
||||
"""One kind of room the hotel offers."""
|
||||
|
||||
size: str
|
||||
standard_price_per_night: float = Field(..., ge=0)
|
||||
# None = not known yet; anything that needs the count (e.g. occupancy)
|
||||
# must be skipped rather than guess.
|
||||
room_count: int | None = Field(default=None, ge=0)
|
||||
|
||||
|
||||
class HotelMetadata(BaseModel):
|
||||
"""Reference data about the hotel that is not part of the booking records.
|
||||
|
||||
``room_types`` is keyed by the ``assigned_room_type`` code used in the CSV.
|
||||
"""
|
||||
|
||||
hotel: str
|
||||
currency: str = "USD"
|
||||
room_types: dict[str, RoomType]
|
||||
|
||||
def describe(self) -> str:
|
||||
"""Plain-text summary of the room catalogue, for the LLM prompt."""
|
||||
lines = []
|
||||
for code, room in self.room_types.items():
|
||||
count = "unknown" if room.room_count is None else str(room.room_count)
|
||||
lines.append(
|
||||
f"- Room type {code}: {room.size}, standard price "
|
||||
f"{room.standard_price_per_night:g} {self.currency} per night, "
|
||||
f"number of rooms in hotel: {count}"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
@@ -0,0 +1,54 @@
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class Booking(BaseModel):
|
||||
"""A single hotel booking record, mirroring nf_hotel_bookings.csv.
|
||||
|
||||
Fields are intentionally loosely typed (str for dates/free-form values)
|
||||
because incoming JSON is treated as *raw* data that still needs to pass
|
||||
through the cleaning pipeline before analysis.
|
||||
"""
|
||||
|
||||
booking_id: int
|
||||
hotel: str
|
||||
is_canceled: int
|
||||
lead_time: int
|
||||
arrival_date_week_number: int
|
||||
booking_date: str
|
||||
arrival_date: str
|
||||
arrival_date_day_of_month: int
|
||||
stays_in_weekend_nights: int
|
||||
stays_in_week_nights: int
|
||||
adults: int
|
||||
children: int
|
||||
babies: int
|
||||
meal: str
|
||||
country: str
|
||||
market_segment: str
|
||||
is_repeated_guest: int
|
||||
previous_cancellations: int
|
||||
assigned_room_type: str
|
||||
booking_changes: int
|
||||
deposit_type: str
|
||||
agent: int
|
||||
customer_type: str
|
||||
required_car_parking_spaces: int
|
||||
total_of_special_requests: int
|
||||
# Optional: when omitted (or wrong) it is derived from assigned_room_type.
|
||||
# Spelling mirrors the CSV column header.
|
||||
prize_per_nigth: float | None = None
|
||||
|
||||
|
||||
class BookingBatch(BaseModel):
|
||||
"""Payload for submitting raw booking records as JSON."""
|
||||
|
||||
records: list[Booking] = Field(..., min_length=1)
|
||||
|
||||
|
||||
class ReportResponse(BaseModel):
|
||||
"""Result of a full analytics report: stats + LLM narrative."""
|
||||
|
||||
descriptive_stats: dict[str, dict[str, Any]]
|
||||
llm_report: str
|
||||
@@ -0,0 +1,28 @@
|
||||
from fastapi import FastAPI
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
|
||||
from nf_hotel_api.api.v1.router import api_router
|
||||
from nf_hotel_api.core.config import get_settings
|
||||
|
||||
settings = get_settings()
|
||||
|
||||
app = FastAPI(
|
||||
title="NF Hotel Analytics API",
|
||||
description="Secure API for descriptive statistics and LLM-generated reports over NF Hotel bookings.",
|
||||
version="0.1.0",
|
||||
)
|
||||
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=settings.allowed_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST"],
|
||||
allow_headers=["X-API-Key", "Content-Type"],
|
||||
)
|
||||
|
||||
app.include_router(api_router)
|
||||
|
||||
|
||||
@app.get("/health", tags=["health"])
|
||||
def health_check() -> dict[str, str]:
|
||||
return {"status": "ok"}
|
||||
@@ -0,0 +1,36 @@
|
||||
from abc import ABC, abstractmethod
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
class BookingRepository(ABC):
|
||||
"""Abstraction over "where booking records come from"."""
|
||||
|
||||
@abstractmethod
|
||||
def load(self) -> pd.DataFrame:
|
||||
"""Return the raw (unclean) bookings as a DataFrame."""
|
||||
|
||||
|
||||
class CsvBookingRepository(BookingRepository):
|
||||
"""Loads bookings from the on-disk NF Hotel CSV export."""
|
||||
|
||||
def __init__(self, path: Path, separator: str = ";") -> None:
|
||||
self._path = path
|
||||
self._separator = separator
|
||||
|
||||
def load(self) -> pd.DataFrame:
|
||||
if not self._path.exists():
|
||||
raise FileNotFoundError(f"Booking data file not found: {self._path}")
|
||||
return pd.read_csv(self._path, sep=self._separator)
|
||||
|
||||
|
||||
class JsonBookingRepository(BookingRepository):
|
||||
"""Loads bookings from JSON records supplied directly by the caller."""
|
||||
|
||||
def __init__(self, records: list[dict[str, Any]]) -> None:
|
||||
self._records = records
|
||||
|
||||
def load(self) -> pd.DataFrame:
|
||||
return pd.DataFrame.from_records(self._records)
|
||||
@@ -0,0 +1,15 @@
|
||||
from pathlib import Path
|
||||
|
||||
from nf_hotel_api.domain.metadata import HotelMetadata
|
||||
|
||||
|
||||
class JsonHotelMetadataRepository:
|
||||
"""Loads hotel reference data (room types, standard prices, room counts)."""
|
||||
|
||||
def __init__(self, path: Path) -> None:
|
||||
self._path = path
|
||||
|
||||
def load(self) -> HotelMetadata:
|
||||
if not self._path.exists():
|
||||
raise FileNotFoundError(f"Hotel metadata file not found: {self._path}")
|
||||
return HotelMetadata.model_validate_json(self._path.read_text(encoding="utf-8"))
|
||||
@@ -0,0 +1,176 @@
|
||||
import pandas as pd
|
||||
|
||||
from nf_hotel_api.domain.metadata import HotelMetadata
|
||||
|
||||
_DATE_COLUMNS = ("booking_date", "arrival_date")
|
||||
|
||||
_NUMERIC_COLUMNS = (
|
||||
"booking_id",
|
||||
"is_canceled",
|
||||
"lead_time",
|
||||
"arrival_date_week_number",
|
||||
"arrival_date_day_of_month",
|
||||
"stays_in_weekend_nights",
|
||||
"stays_in_week_nights",
|
||||
"adults",
|
||||
"children",
|
||||
"babies",
|
||||
"is_repeated_guest",
|
||||
"previous_cancellations",
|
||||
"booking_changes",
|
||||
"agent",
|
||||
"required_car_parking_spaces",
|
||||
"total_of_special_requests",
|
||||
)
|
||||
|
||||
_CATEGORICAL_COLUMNS = (
|
||||
"hotel",
|
||||
"meal",
|
||||
"country",
|
||||
"market_segment",
|
||||
"assigned_room_type",
|
||||
"deposit_type",
|
||||
"customer_type",
|
||||
)
|
||||
|
||||
# Accepted date layouts: ISO (JSON payloads) and day-first (the CSV export).
|
||||
_DATE_FORMATS = ("%Y-%m-%d", "%d-%m-%Y")
|
||||
|
||||
_PRICE_COLUMN = "prize_per_nigth"
|
||||
|
||||
_EMPTY_MARKERS = ("", "nan", "null", "none", "n/a", "na")
|
||||
|
||||
# Bookings with more occupants than this in a single field are treated as
|
||||
# data-entry errors (e.g. "55 adults" in one room) rather than real guests.
|
||||
_MAX_PLAUSIBLE_ADULTS = 10
|
||||
_MAX_PLAUSIBLE_CHILDREN = 5
|
||||
|
||||
|
||||
class DataCleaningService:
|
||||
"""Cleans raw booking data before it is analyzed.
|
||||
|
||||
Each private method covers one classic pandas cleaning concern, run in
|
||||
a fixed order via :meth:`clean`. Once the data is clean, :meth:`_add_pricing`
|
||||
derives room size and revenue using the hotel metadata.
|
||||
"""
|
||||
|
||||
def __init__(self, metadata: HotelMetadata) -> None:
|
||||
self._metadata = metadata
|
||||
|
||||
def clean(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
df = df.copy()
|
||||
df = self._fix_wrong_format(df)
|
||||
df = self._clean_empty_cells(df)
|
||||
df = self._fix_wrong_data(df)
|
||||
df = self._remove_duplicates(df)
|
||||
df = self._add_pricing(df)
|
||||
return df.reset_index(drop=True)
|
||||
|
||||
def _fix_wrong_format(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Coerce columns into their expected dtype (dates, numbers)."""
|
||||
for column in _DATE_COLUMNS:
|
||||
if column in df.columns:
|
||||
df[column] = self._parse_dates(df[column])
|
||||
|
||||
for column in _NUMERIC_COLUMNS:
|
||||
if column in df.columns:
|
||||
df[column] = pd.to_numeric(df[column], errors="coerce")
|
||||
|
||||
if _PRICE_COLUMN in df.columns:
|
||||
df[_PRICE_COLUMN] = pd.to_numeric(df[_PRICE_COLUMN], errors="coerce")
|
||||
|
||||
for column in _CATEGORICAL_COLUMNS:
|
||||
if column in df.columns:
|
||||
df[column] = df[column].astype("string").str.strip()
|
||||
|
||||
return df
|
||||
|
||||
@staticmethod
|
||||
def _parse_dates(series: pd.Series) -> pd.Series:
|
||||
"""Parse each value with the first matching layout in ``_DATE_FORMATS``.
|
||||
|
||||
A single ``pd.to_datetime`` call would infer one layout from the first
|
||||
value and turn every date in the other layout into NaT, which would
|
||||
then silently drop those rows.
|
||||
"""
|
||||
parsed = pd.Series(pd.NaT, index=series.index, dtype="datetime64[ns]")
|
||||
for date_format in _DATE_FORMATS:
|
||||
candidate = pd.to_datetime(series, format=date_format, errors="coerce")
|
||||
parsed = parsed.fillna(candidate)
|
||||
return parsed
|
||||
|
||||
def _clean_empty_cells(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Normalize placeholder/blank values to NA, then fill sensibly."""
|
||||
for column in _CATEGORICAL_COLUMNS:
|
||||
if column not in df.columns:
|
||||
continue
|
||||
lowered = df[column].str.lower()
|
||||
df.loc[lowered.isin(_EMPTY_MARKERS), column] = pd.NA
|
||||
df[column] = df[column].fillna("Unknown")
|
||||
|
||||
for column in _NUMERIC_COLUMNS:
|
||||
if column in df.columns:
|
||||
df[column] = df[column].fillna(0)
|
||||
|
||||
df = df.dropna(subset=[c for c in _DATE_COLUMNS if c in df.columns])
|
||||
return df
|
||||
|
||||
def _fix_wrong_data(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Correct implausible values without discarding the whole row."""
|
||||
if "adults" in df.columns:
|
||||
df["adults"] = df["adults"].clip(lower=0, upper=_MAX_PLAUSIBLE_ADULTS)
|
||||
if "children" in df.columns:
|
||||
df["children"] = df["children"].clip(lower=0, upper=_MAX_PLAUSIBLE_CHILDREN)
|
||||
if "babies" in df.columns:
|
||||
df["babies"] = df["babies"].clip(lower=0, upper=_MAX_PLAUSIBLE_CHILDREN)
|
||||
|
||||
# A booking with zero occupants in every guest field is invalid;
|
||||
# treat it as a single adult rather than dropping the record.
|
||||
guest_columns = [c for c in ("adults", "children", "babies") if c in df.columns]
|
||||
if guest_columns:
|
||||
no_guests = (df[guest_columns].sum(axis=1) == 0)
|
||||
if "adults" in df.columns:
|
||||
df.loc[no_guests, "adults"] = 1
|
||||
|
||||
for column in ("lead_time", "booking_changes", "previous_cancellations", "agent"):
|
||||
if column in df.columns:
|
||||
df[column] = df[column].clip(lower=0)
|
||||
|
||||
return df
|
||||
|
||||
def _remove_duplicates(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Drop repeated bookings, ignoring the (non-business) id column."""
|
||||
subset = [c for c in df.columns if c != "booking_id"]
|
||||
return df.drop_duplicates(subset=subset, keep="first")
|
||||
|
||||
def _add_pricing(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Add room size and revenue, and fill in missing prices.
|
||||
|
||||
``prize_per_nigth`` in the data is the price the customer actually
|
||||
paid, so it is kept as-is. Only a missing or negative price is replaced
|
||||
with the room type's standard price from the hotel metadata. Room types
|
||||
outside the metadata get no size and no invented price.
|
||||
"""
|
||||
if "assigned_room_type" not in df.columns:
|
||||
return df
|
||||
|
||||
room_types = self._metadata.room_types
|
||||
room_type = df["assigned_room_type"].astype("string").str.upper()
|
||||
standard_price = room_type.map(
|
||||
{code: room.standard_price_per_night for code, room in room_types.items()}
|
||||
).astype("float64")
|
||||
|
||||
paid_price = df.get(_PRICE_COLUMN, pd.Series(float("nan"), index=df.index))
|
||||
df[_PRICE_COLUMN] = paid_price.where(paid_price >= 0).fillna(standard_price)
|
||||
df["room_size"] = room_type.map(
|
||||
{code: room.size for code, room in room_types.items()}
|
||||
).fillna("Unknown")
|
||||
|
||||
nights_columns = [
|
||||
c for c in ("stays_in_weekend_nights", "stays_in_week_nights") if c in df.columns
|
||||
]
|
||||
if nights_columns:
|
||||
# Cancelled bookings bring in no money.
|
||||
billable = df["is_canceled"] == 0 if "is_canceled" in df.columns else True
|
||||
df["revenue"] = df[nights_columns].sum(axis=1) * df[_PRICE_COLUMN] * billable
|
||||
return df
|
||||
@@ -0,0 +1,86 @@
|
||||
import json
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from nf_hotel_api.domain.metadata import HotelMetadata
|
||||
|
||||
_PROMPT_TEMPLATE = """You work as a data analyst and in marketing to optimize hotel operations.
|
||||
User cannot interact with you so do not ask questions.
|
||||
Respond in markdown format.
|
||||
{hotel_context}We have extract descriptive analysis:\n
|
||||
{descriptive_analysis_data}"""
|
||||
|
||||
_HOTEL_CONTEXT_TEMPLATE = """Hotel reference data (room types the bookings refer to):
|
||||
{room_catalogue}
|
||||
The column prize_per_nigth is the price the customer actually paid per night.
|
||||
The column revenue is nights x prize_per_nigth for non-cancelled bookings.
|
||||
"""
|
||||
|
||||
# Reasoning models (e.g. Qwen3) may wrap their internal reasoning in
|
||||
# <think>...</think>; that content must never reach the API response.
|
||||
_THINK_BLOCK_PATTERN = re.compile(r"<think>.*?</think>", re.DOTALL | re.IGNORECASE)
|
||||
|
||||
|
||||
class LLMServiceError(RuntimeError):
|
||||
"""Raised when the LLM backend cannot produce a report."""
|
||||
|
||||
|
||||
class LLMReportService:
|
||||
"""Turns descriptive statistics into a marketing/ops narrative via an LLM."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
model: str,
|
||||
timeout_seconds: float,
|
||||
metadata: HotelMetadata | None = None,
|
||||
) -> None:
|
||||
self._metadata = metadata
|
||||
self._base_url = base_url.rstrip("/")
|
||||
self._api_key = api_key
|
||||
self._model = model
|
||||
self._timeout_seconds = timeout_seconds
|
||||
|
||||
async def generate_report(self, descriptive_stats: dict[str, Any]) -> str:
|
||||
hotel_context = (
|
||||
_HOTEL_CONTEXT_TEMPLATE.format(room_catalogue=self._metadata.describe())
|
||||
if self._metadata
|
||||
else ""
|
||||
)
|
||||
prompt = _PROMPT_TEMPLATE.format(
|
||||
hotel_context=hotel_context,
|
||||
descriptive_analysis_data=json.dumps(descriptive_stats, indent=2),
|
||||
)
|
||||
|
||||
payload = {
|
||||
"model": self._model,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"temperature": 0.3,
|
||||
}
|
||||
headers = {"Authorization": f"Bearer {self._api_key}"}
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=self._timeout_seconds) as client:
|
||||
response = await client.post(
|
||||
f"{self._base_url}/chat/completions",
|
||||
json=payload,
|
||||
headers=headers,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except httpx.HTTPError as exc:
|
||||
raise LLMServiceError(f"LLM backend request failed: {exc}") from exc
|
||||
|
||||
data = response.json()
|
||||
try:
|
||||
raw_content = data["choices"][0]["message"]["content"]
|
||||
except (KeyError, IndexError) as exc:
|
||||
raise LLMServiceError(f"Unexpected LLM response shape: {data}") from exc
|
||||
|
||||
return self._strip_thinking(raw_content)
|
||||
|
||||
@staticmethod
|
||||
def _strip_thinking(content: str) -> str:
|
||||
return _THINK_BLOCK_PATTERN.sub("", content).strip()
|
||||
@@ -0,0 +1,26 @@
|
||||
from nf_hotel_api.domain.schemas import ReportResponse
|
||||
from nf_hotel_api.repositories.booking_repository import BookingRepository
|
||||
from nf_hotel_api.services.cleaning import DataCleaningService
|
||||
from nf_hotel_api.services.llm_report import LLMReportService
|
||||
from nf_hotel_api.services.statistics import DescriptiveStatsService
|
||||
|
||||
|
||||
class ReportService:
|
||||
"""Application-layer use case: raw bookings -> full analytics report."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
cleaning_service: DataCleaningService,
|
||||
stats_service: DescriptiveStatsService,
|
||||
llm_service: LLMReportService,
|
||||
) -> None:
|
||||
self._cleaning_service = cleaning_service
|
||||
self._stats_service = stats_service
|
||||
self._llm_service = llm_service
|
||||
|
||||
async def generate(self, repository: BookingRepository) -> ReportResponse:
|
||||
raw_df = repository.load()
|
||||
clean_df = self._cleaning_service.clean(raw_df)
|
||||
descriptive_stats = self._stats_service.compute(clean_df)
|
||||
llm_report = await self._llm_service.generate_report(descriptive_stats)
|
||||
return ReportResponse(descriptive_stats=descriptive_stats, llm_report=llm_report)
|
||||
@@ -0,0 +1,27 @@
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
class DescriptiveStatsService:
|
||||
"""Computes descriptive statistics over cleaned booking data."""
|
||||
|
||||
def compute(self, df: pd.DataFrame) -> dict[str, dict[str, Any]]:
|
||||
df = df.drop(columns=["booking_id"], errors="ignore")
|
||||
described = df.describe(include="all")
|
||||
return {
|
||||
str(column): {
|
||||
str(stat): self._to_jsonable(value)
|
||||
for stat, value in described[column].items()
|
||||
if pd.notna(value)
|
||||
}
|
||||
for column in described.columns
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _to_jsonable(value: Any) -> Any:
|
||||
if isinstance(value, pd.Timestamp):
|
||||
return value.isoformat()
|
||||
if hasattr(value, "item"):
|
||||
return value.item()
|
||||
return value
|
||||
Reference in New Issue
Block a user