Initial project setup: NF Hotel Data Analysis API

- Add project structure with domain-driven design organization
- Implement FastAPI endpoints for hotel booking reports
- Add data cleaning, statistics, and LLM report generation services
- Include configuration management and security utilities
- Add repository layer for bookings and metadata access
- Setup testing framework with conftest.py
- Include example files and documentation
This commit is contained in:
2026-09-21 13:17:49 +08:00
commit 225d28feb7
39 changed files with 41693 additions and 0 deletions
View File
View File
+48
View File
@@ -0,0 +1,48 @@
from functools import lru_cache
from fastapi import Depends
from nf_hotel_api.core.config import Settings, get_settings
from nf_hotel_api.domain.metadata import HotelMetadata
from nf_hotel_api.repositories.metadata_repository import JsonHotelMetadataRepository
from nf_hotel_api.services.cleaning import DataCleaningService
from nf_hotel_api.services.llm_report import LLMReportService
from nf_hotel_api.services.report import ReportService
from nf_hotel_api.services.statistics import DescriptiveStatsService
@lru_cache
def get_hotel_metadata() -> HotelMetadata:
return JsonHotelMetadataRepository(get_settings().metadata_path).load()
def get_cleaning_service(
metadata: HotelMetadata = Depends(get_hotel_metadata),
) -> DataCleaningService:
return DataCleaningService(metadata)
@lru_cache
def get_stats_service() -> DescriptiveStatsService:
return DescriptiveStatsService()
def get_llm_service(
settings: Settings = Depends(get_settings),
metadata: HotelMetadata = Depends(get_hotel_metadata),
) -> LLMReportService:
return LLMReportService(
metadata=metadata,
base_url=settings.llm_base_url,
api_key=settings.llm_api_key,
model=settings.llm_model,
timeout_seconds=settings.llm_timeout_seconds,
)
def get_report_service(
cleaning_service: DataCleaningService = Depends(get_cleaning_service),
stats_service: DescriptiveStatsService = Depends(get_stats_service),
llm_service: LLMReportService = Depends(get_llm_service),
) -> ReportService:
return ReportService(cleaning_service, stats_service, llm_service)
View File
@@ -0,0 +1,47 @@
from fastapi import APIRouter, Depends, HTTPException, status
from nf_hotel_api.api.deps import get_report_service
from nf_hotel_api.core.config import Settings, get_settings
from nf_hotel_api.core.security import require_api_key
from nf_hotel_api.domain.schemas import BookingBatch, ReportResponse
from nf_hotel_api.repositories.booking_repository import (
CsvBookingRepository,
JsonBookingRepository,
)
from nf_hotel_api.services.llm_report import LLMServiceError
from nf_hotel_api.services.report import ReportService
router = APIRouter(
prefix="/report",
tags=["report"],
dependencies=[Depends(require_api_key)],
)
@router.post("/from-file", response_model=ReportResponse)
async def generate_report_from_file(
report_service: ReportService = Depends(get_report_service),
settings: Settings = Depends(get_settings),
) -> ReportResponse:
"""Generate the analytics report from the bundled nf_hotel_bookings.csv file."""
repository = CsvBookingRepository(settings.default_data_path, settings.csv_separator)
try:
return await report_service.generate(repository)
except FileNotFoundError as exc:
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=str(exc)) from exc
except LLMServiceError as exc:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
@router.post("/from-json", response_model=ReportResponse)
async def generate_report_from_json(
batch: BookingBatch,
report_service: ReportService = Depends(get_report_service),
) -> ReportResponse:
"""Generate the analytics report from booking records supplied as JSON."""
records = [record.model_dump() for record in batch.records]
repository = JsonBookingRepository(records)
try:
return await report_service.generate(repository)
except LLMServiceError as exc:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
+6
View File
@@ -0,0 +1,6 @@
from fastapi import APIRouter
from nf_hotel_api.api.v1.endpoints import report
api_router = APIRouter(prefix="/api/v1")
api_router.include_router(report.router)
View File
+48
View File
@@ -0,0 +1,48 @@
from functools import lru_cache
from pathlib import Path
from pydantic import field_validator
from pydantic_settings import BaseSettings, SettingsConfigDict
# src/nf_hotel_api/core/config.py -> project root is three levels up.
_PROJECT_ROOT = Path(__file__).resolve().parents[3]
class Settings(BaseSettings):
"""Central application configuration, loaded from environment / .env."""
model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore")
# --- API security ---
api_key: str
allowed_origins: list[str] = ["http://localhost:3000"]
# --- Local data source (fallback / default dataset) ---
default_data_path: Path = Path("data/nf_hotel_bookings.csv")
csv_separator: str = ";"
# Reference data: room types, standard prices and how many rooms exist.
metadata_path: Path = Path("data/hotel_metadata.json")
@field_validator("default_data_path", "metadata_path")
@classmethod
def _resolve_relative_to_project_root(cls, value: Path) -> Path:
# Anchored to the project root, not the process's current working
# directory - otherwise this 404s whenever uvicorn/pytest is
# launched from anywhere other than the repo root.
return value if value.is_absolute() else _PROJECT_ROOT / value
# --- LLM (OpenAI-compatible local inference server, e.g. vLLM / LM Studio) ---
llm_base_url: str = "http://localhost:1234/v1"
llm_api_key: str = "not-needed"
llm_model: str = "qwen3.8-whittle-moe-27b-a17.8b"
# Reasoning models can spend minutes generating <think>/reasoning_content
# tokens before the final answer - keep this generous.
llm_timeout_seconds: float = 900.0
@lru_cache
def get_settings() -> Settings:
# api_key (and other required fields) come from the environment / .env
# at runtime via BaseSettings, not from constructor arguments - static
# type checkers can't see that, hence the ignore.
return Settings() # type: ignore[call-arg]
+30
View File
@@ -0,0 +1,30 @@
import secrets
from fastapi import Depends, HTTPException, Security, status
from fastapi.security import APIKeyHeader
from nf_hotel_api.core.config import Settings, get_settings
_api_key_header = APIKeyHeader(name="X-API-Key", auto_error=False)
class ApiKeyAuthenticator:
"""Validates inbound requests against the configured API key.
Uses a constant-time comparison to avoid leaking key length/content via
response-time side channels.
"""
def __call__(
self,
api_key: str | None = Security(_api_key_header),
settings: Settings = Depends(get_settings),
) -> None:
if not api_key or not secrets.compare_digest(api_key, settings.api_key):
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Invalid or missing API key",
)
require_api_key = ApiKeyAuthenticator()
View File
+34
View File
@@ -0,0 +1,34 @@
from pydantic import BaseModel, Field
class RoomType(BaseModel):
"""One kind of room the hotel offers."""
size: str
standard_price_per_night: float = Field(..., ge=0)
# None = not known yet; anything that needs the count (e.g. occupancy)
# must be skipped rather than guess.
room_count: int | None = Field(default=None, ge=0)
class HotelMetadata(BaseModel):
"""Reference data about the hotel that is not part of the booking records.
``room_types`` is keyed by the ``assigned_room_type`` code used in the CSV.
"""
hotel: str
currency: str = "USD"
room_types: dict[str, RoomType]
def describe(self) -> str:
"""Plain-text summary of the room catalogue, for the LLM prompt."""
lines = []
for code, room in self.room_types.items():
count = "unknown" if room.room_count is None else str(room.room_count)
lines.append(
f"- Room type {code}: {room.size}, standard price "
f"{room.standard_price_per_night:g} {self.currency} per night, "
f"number of rooms in hotel: {count}"
)
return "\n".join(lines)
+54
View File
@@ -0,0 +1,54 @@
from typing import Any
from pydantic import BaseModel, Field
class Booking(BaseModel):
"""A single hotel booking record, mirroring nf_hotel_bookings.csv.
Fields are intentionally loosely typed (str for dates/free-form values)
because incoming JSON is treated as *raw* data that still needs to pass
through the cleaning pipeline before analysis.
"""
booking_id: int
hotel: str
is_canceled: int
lead_time: int
arrival_date_week_number: int
booking_date: str
arrival_date: str
arrival_date_day_of_month: int
stays_in_weekend_nights: int
stays_in_week_nights: int
adults: int
children: int
babies: int
meal: str
country: str
market_segment: str
is_repeated_guest: int
previous_cancellations: int
assigned_room_type: str
booking_changes: int
deposit_type: str
agent: int
customer_type: str
required_car_parking_spaces: int
total_of_special_requests: int
# Optional: when omitted (or wrong) it is derived from assigned_room_type.
# Spelling mirrors the CSV column header.
prize_per_nigth: float | None = None
class BookingBatch(BaseModel):
"""Payload for submitting raw booking records as JSON."""
records: list[Booking] = Field(..., min_length=1)
class ReportResponse(BaseModel):
"""Result of a full analytics report: stats + LLM narrative."""
descriptive_stats: dict[str, dict[str, Any]]
llm_report: str
+28
View File
@@ -0,0 +1,28 @@
from fastapi import FastAPI
from fastapi.middleware.cors import CORSMiddleware
from nf_hotel_api.api.v1.router import api_router
from nf_hotel_api.core.config import get_settings
settings = get_settings()
app = FastAPI(
title="NF Hotel Analytics API",
description="Secure API for descriptive statistics and LLM-generated reports over NF Hotel bookings.",
version="0.1.0",
)
app.add_middleware(
CORSMiddleware,
allow_origins=settings.allowed_origins,
allow_credentials=True,
allow_methods=["GET", "POST"],
allow_headers=["X-API-Key", "Content-Type"],
)
app.include_router(api_router)
@app.get("/health", tags=["health"])
def health_check() -> dict[str, str]:
return {"status": "ok"}
@@ -0,0 +1,36 @@
from abc import ABC, abstractmethod
from pathlib import Path
from typing import Any
import pandas as pd
class BookingRepository(ABC):
"""Abstraction over "where booking records come from"."""
@abstractmethod
def load(self) -> pd.DataFrame:
"""Return the raw (unclean) bookings as a DataFrame."""
class CsvBookingRepository(BookingRepository):
"""Loads bookings from the on-disk NF Hotel CSV export."""
def __init__(self, path: Path, separator: str = ";") -> None:
self._path = path
self._separator = separator
def load(self) -> pd.DataFrame:
if not self._path.exists():
raise FileNotFoundError(f"Booking data file not found: {self._path}")
return pd.read_csv(self._path, sep=self._separator)
class JsonBookingRepository(BookingRepository):
"""Loads bookings from JSON records supplied directly by the caller."""
def __init__(self, records: list[dict[str, Any]]) -> None:
self._records = records
def load(self) -> pd.DataFrame:
return pd.DataFrame.from_records(self._records)
@@ -0,0 +1,15 @@
from pathlib import Path
from nf_hotel_api.domain.metadata import HotelMetadata
class JsonHotelMetadataRepository:
"""Loads hotel reference data (room types, standard prices, room counts)."""
def __init__(self, path: Path) -> None:
self._path = path
def load(self) -> HotelMetadata:
if not self._path.exists():
raise FileNotFoundError(f"Hotel metadata file not found: {self._path}")
return HotelMetadata.model_validate_json(self._path.read_text(encoding="utf-8"))
+176
View File
@@ -0,0 +1,176 @@
import pandas as pd
from nf_hotel_api.domain.metadata import HotelMetadata
_DATE_COLUMNS = ("booking_date", "arrival_date")
_NUMERIC_COLUMNS = (
"booking_id",
"is_canceled",
"lead_time",
"arrival_date_week_number",
"arrival_date_day_of_month",
"stays_in_weekend_nights",
"stays_in_week_nights",
"adults",
"children",
"babies",
"is_repeated_guest",
"previous_cancellations",
"booking_changes",
"agent",
"required_car_parking_spaces",
"total_of_special_requests",
)
_CATEGORICAL_COLUMNS = (
"hotel",
"meal",
"country",
"market_segment",
"assigned_room_type",
"deposit_type",
"customer_type",
)
# Accepted date layouts: ISO (JSON payloads) and day-first (the CSV export).
_DATE_FORMATS = ("%Y-%m-%d", "%d-%m-%Y")
_PRICE_COLUMN = "prize_per_nigth"
_EMPTY_MARKERS = ("", "nan", "null", "none", "n/a", "na")
# Bookings with more occupants than this in a single field are treated as
# data-entry errors (e.g. "55 adults" in one room) rather than real guests.
_MAX_PLAUSIBLE_ADULTS = 10
_MAX_PLAUSIBLE_CHILDREN = 5
class DataCleaningService:
"""Cleans raw booking data before it is analyzed.
Each private method covers one classic pandas cleaning concern, run in
a fixed order via :meth:`clean`. Once the data is clean, :meth:`_add_pricing`
derives room size and revenue using the hotel metadata.
"""
def __init__(self, metadata: HotelMetadata) -> None:
self._metadata = metadata
def clean(self, df: pd.DataFrame) -> pd.DataFrame:
df = df.copy()
df = self._fix_wrong_format(df)
df = self._clean_empty_cells(df)
df = self._fix_wrong_data(df)
df = self._remove_duplicates(df)
df = self._add_pricing(df)
return df.reset_index(drop=True)
def _fix_wrong_format(self, df: pd.DataFrame) -> pd.DataFrame:
"""Coerce columns into their expected dtype (dates, numbers)."""
for column in _DATE_COLUMNS:
if column in df.columns:
df[column] = self._parse_dates(df[column])
for column in _NUMERIC_COLUMNS:
if column in df.columns:
df[column] = pd.to_numeric(df[column], errors="coerce")
if _PRICE_COLUMN in df.columns:
df[_PRICE_COLUMN] = pd.to_numeric(df[_PRICE_COLUMN], errors="coerce")
for column in _CATEGORICAL_COLUMNS:
if column in df.columns:
df[column] = df[column].astype("string").str.strip()
return df
@staticmethod
def _parse_dates(series: pd.Series) -> pd.Series:
"""Parse each value with the first matching layout in ``_DATE_FORMATS``.
A single ``pd.to_datetime`` call would infer one layout from the first
value and turn every date in the other layout into NaT, which would
then silently drop those rows.
"""
parsed = pd.Series(pd.NaT, index=series.index, dtype="datetime64[ns]")
for date_format in _DATE_FORMATS:
candidate = pd.to_datetime(series, format=date_format, errors="coerce")
parsed = parsed.fillna(candidate)
return parsed
def _clean_empty_cells(self, df: pd.DataFrame) -> pd.DataFrame:
"""Normalize placeholder/blank values to NA, then fill sensibly."""
for column in _CATEGORICAL_COLUMNS:
if column not in df.columns:
continue
lowered = df[column].str.lower()
df.loc[lowered.isin(_EMPTY_MARKERS), column] = pd.NA
df[column] = df[column].fillna("Unknown")
for column in _NUMERIC_COLUMNS:
if column in df.columns:
df[column] = df[column].fillna(0)
df = df.dropna(subset=[c for c in _DATE_COLUMNS if c in df.columns])
return df
def _fix_wrong_data(self, df: pd.DataFrame) -> pd.DataFrame:
"""Correct implausible values without discarding the whole row."""
if "adults" in df.columns:
df["adults"] = df["adults"].clip(lower=0, upper=_MAX_PLAUSIBLE_ADULTS)
if "children" in df.columns:
df["children"] = df["children"].clip(lower=0, upper=_MAX_PLAUSIBLE_CHILDREN)
if "babies" in df.columns:
df["babies"] = df["babies"].clip(lower=0, upper=_MAX_PLAUSIBLE_CHILDREN)
# A booking with zero occupants in every guest field is invalid;
# treat it as a single adult rather than dropping the record.
guest_columns = [c for c in ("adults", "children", "babies") if c in df.columns]
if guest_columns:
no_guests = (df[guest_columns].sum(axis=1) == 0)
if "adults" in df.columns:
df.loc[no_guests, "adults"] = 1
for column in ("lead_time", "booking_changes", "previous_cancellations", "agent"):
if column in df.columns:
df[column] = df[column].clip(lower=0)
return df
def _remove_duplicates(self, df: pd.DataFrame) -> pd.DataFrame:
"""Drop repeated bookings, ignoring the (non-business) id column."""
subset = [c for c in df.columns if c != "booking_id"]
return df.drop_duplicates(subset=subset, keep="first")
def _add_pricing(self, df: pd.DataFrame) -> pd.DataFrame:
"""Add room size and revenue, and fill in missing prices.
``prize_per_nigth`` in the data is the price the customer actually
paid, so it is kept as-is. Only a missing or negative price is replaced
with the room type's standard price from the hotel metadata. Room types
outside the metadata get no size and no invented price.
"""
if "assigned_room_type" not in df.columns:
return df
room_types = self._metadata.room_types
room_type = df["assigned_room_type"].astype("string").str.upper()
standard_price = room_type.map(
{code: room.standard_price_per_night for code, room in room_types.items()}
).astype("float64")
paid_price = df.get(_PRICE_COLUMN, pd.Series(float("nan"), index=df.index))
df[_PRICE_COLUMN] = paid_price.where(paid_price >= 0).fillna(standard_price)
df["room_size"] = room_type.map(
{code: room.size for code, room in room_types.items()}
).fillna("Unknown")
nights_columns = [
c for c in ("stays_in_weekend_nights", "stays_in_week_nights") if c in df.columns
]
if nights_columns:
# Cancelled bookings bring in no money.
billable = df["is_canceled"] == 0 if "is_canceled" in df.columns else True
df["revenue"] = df[nights_columns].sum(axis=1) * df[_PRICE_COLUMN] * billable
return df
+86
View File
@@ -0,0 +1,86 @@
import json
import re
from typing import Any
import httpx
from nf_hotel_api.domain.metadata import HotelMetadata
_PROMPT_TEMPLATE = """You work as a data analyst and in marketing to optimize hotel operations.
User cannot interact with you so do not ask questions.
Respond in markdown format.
{hotel_context}We have extract descriptive analysis:\n
{descriptive_analysis_data}"""
_HOTEL_CONTEXT_TEMPLATE = """Hotel reference data (room types the bookings refer to):
{room_catalogue}
The column prize_per_nigth is the price the customer actually paid per night.
The column revenue is nights x prize_per_nigth for non-cancelled bookings.
"""
# Reasoning models (e.g. Qwen3) may wrap their internal reasoning in
# <think>...</think>; that content must never reach the API response.
_THINK_BLOCK_PATTERN = re.compile(r"<think>.*?</think>", re.DOTALL | re.IGNORECASE)
class LLMServiceError(RuntimeError):
"""Raised when the LLM backend cannot produce a report."""
class LLMReportService:
"""Turns descriptive statistics into a marketing/ops narrative via an LLM."""
def __init__(
self,
base_url: str,
api_key: str,
model: str,
timeout_seconds: float,
metadata: HotelMetadata | None = None,
) -> None:
self._metadata = metadata
self._base_url = base_url.rstrip("/")
self._api_key = api_key
self._model = model
self._timeout_seconds = timeout_seconds
async def generate_report(self, descriptive_stats: dict[str, Any]) -> str:
hotel_context = (
_HOTEL_CONTEXT_TEMPLATE.format(room_catalogue=self._metadata.describe())
if self._metadata
else ""
)
prompt = _PROMPT_TEMPLATE.format(
hotel_context=hotel_context,
descriptive_analysis_data=json.dumps(descriptive_stats, indent=2),
)
payload = {
"model": self._model,
"messages": [{"role": "user", "content": prompt}],
"temperature": 0.3,
}
headers = {"Authorization": f"Bearer {self._api_key}"}
try:
async with httpx.AsyncClient(timeout=self._timeout_seconds) as client:
response = await client.post(
f"{self._base_url}/chat/completions",
json=payload,
headers=headers,
)
response.raise_for_status()
except httpx.HTTPError as exc:
raise LLMServiceError(f"LLM backend request failed: {exc}") from exc
data = response.json()
try:
raw_content = data["choices"][0]["message"]["content"]
except (KeyError, IndexError) as exc:
raise LLMServiceError(f"Unexpected LLM response shape: {data}") from exc
return self._strip_thinking(raw_content)
@staticmethod
def _strip_thinking(content: str) -> str:
return _THINK_BLOCK_PATTERN.sub("", content).strip()
+26
View File
@@ -0,0 +1,26 @@
from nf_hotel_api.domain.schemas import ReportResponse
from nf_hotel_api.repositories.booking_repository import BookingRepository
from nf_hotel_api.services.cleaning import DataCleaningService
from nf_hotel_api.services.llm_report import LLMReportService
from nf_hotel_api.services.statistics import DescriptiveStatsService
class ReportService:
"""Application-layer use case: raw bookings -> full analytics report."""
def __init__(
self,
cleaning_service: DataCleaningService,
stats_service: DescriptiveStatsService,
llm_service: LLMReportService,
) -> None:
self._cleaning_service = cleaning_service
self._stats_service = stats_service
self._llm_service = llm_service
async def generate(self, repository: BookingRepository) -> ReportResponse:
raw_df = repository.load()
clean_df = self._cleaning_service.clean(raw_df)
descriptive_stats = self._stats_service.compute(clean_df)
llm_report = await self._llm_service.generate_report(descriptive_stats)
return ReportResponse(descriptive_stats=descriptive_stats, llm_report=llm_report)
+27
View File
@@ -0,0 +1,27 @@
from typing import Any
import pandas as pd
class DescriptiveStatsService:
"""Computes descriptive statistics over cleaned booking data."""
def compute(self, df: pd.DataFrame) -> dict[str, dict[str, Any]]:
df = df.drop(columns=["booking_id"], errors="ignore")
described = df.describe(include="all")
return {
str(column): {
str(stat): self._to_jsonable(value)
for stat, value in described[column].items()
if pd.notna(value)
}
for column in described.columns
}
@staticmethod
def _to_jsonable(value: Any) -> Any:
if isinstance(value, pd.Timestamp):
return value.isoformat()
if hasattr(value, "item"):
return value.item()
return value