Files
Tirsvad 769c777b48 Add data generation scripts and public holidays service
- Add scripts for generating booking data, building metadata, and
  updating holidays
- Implement public_holidays.py service with holiday lookup functionality
- Update CSV data with expanded hotel booking records
- Add holidays.json dataset for public holiday dates
- Enhance LLM report service with improved formatting
- Update metadata domain model and API dependencies
- Add test coverage for dataset and public holidays
2026-09-21 14:13:39 +08:00

220 lines
10 KiB
Python

"""Generates the synthetic NF Hotel booking dataset (data/nf_hotel_bookings.csv).
These are NOT real bookings. The hotel is small (see room_count per room type
in data/hotel_metadata.json), so the simulation never sells more rooms of a
type than the hotel has on any night. Demand follows the calendar in the
metadata file (events near the hotel) and on Cambodian public holidays from
the `holidays` package.
Every multiplier below is a modelling ASSUMPTION, not a measurement - change
the constants and re-run to test other scenarios:
.venv/Scripts/python scripts/generate_bookings.py
The output is reproducible (fixed seed). A small share of deliberately dirty
rows (duplicates, impossible guest counts, blanks) is added at the end so the
cleaning pipeline has something to do; set DIRTY_SHARE = 0 to disable.
"""
import sys
from datetime import date, timedelta
from pathlib import Path
import numpy as np
import pandas as pd
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
from nf_hotel_api.domain.metadata import HotelMetadata # noqa: E402
from nf_hotel_api.services.public_holidays import PublicHolidayCalendar # noqa: E402
SEED = 2022
FIRST_ARRIVAL = date(2022, 1, 1)
LAST_ARRIVAL = date(2025, 12, 31)
OUTPUT = Path(sys.argv[1]) if len(sys.argv) > 1 else ROOT / "data" / "nf_hotel_bookings.csv"
METADATA_FILE = ROOT / "data" / "hotel_metadata.json"
# --- Demand model (assumptions) -------------------------------------------
BASE_REQUESTS_PER_DAY = 5.5 # booking requests arriving per day in a normal week
MONTH_FACTOR = { # cool/dry season busiest, rainy season quietest
1: 1.25, 2: 1.2, 3: 1.05, 4: 0.9, 5: 0.85, 6: 0.8,
7: 0.85, 8: 0.85, 9: 0.8, 10: 0.95, 11: 1.2, 12: 1.3,
}
WEEKDAY_FACTOR = {0: 0.95, 1: 0.95, 2: 0.95, 3: 1.0, 4: 1.15, 5: 1.2, 6: 1.0}
# Holiday effect on demand for a Phnom Penh hotel. Matched on the holiday name
# as spelled by the `holidays` package.
HOLIDAY_FACTORS = [
("Water Festival", 1.8), # Phnom Penh hosts the biggest celebrations
("Khmer New Year", 0.75), # capital empties as people travel to the provinces
("Pchum Ben", 0.8), # same: family and pagoda visits in home provinces
]
OTHER_HOLIDAY_FACTOR = 1.15
# Event effect, matched on the event name (first match wins).
EVENT_FACTORS = [
("ASEAN Summits", 2.2),
("SEA Games", 2.5),
("ASEAN Para Games", 1.7),
("Half Marathon", 1.3),
("AFC Challenge League", 1.3),
]
OTHER_EVENT_FACTOR = 1.2
# --- Guests and stays -----------------------------------------------------
STAY_NIGHTS = [1, 2, 3, 4, 5, 6, 7, 10]
STAY_WEIGHTS = [0.33, 0.30, 0.17, 0.08, 0.05, 0.03, 0.03, 0.01]
COUNTRIES = {
"Cambodia": 0.24, "China": 0.14, "Vietnam": 0.08, "United States": 0.08,
"France": 0.06, "United Kingdom": 0.06, "Australia": 0.05, "Japan": 0.05,
"South Korea": 0.05, "Thailand": 0.04, "Germany": 0.04, "Singapore": 0.03,
"Malaysia": 0.03, "Other": 0.05,
}
SEGMENTS = {"Online TA": 0.45, "Direct": 0.20, "Offline TA/TO": 0.12,
"Corporate": 0.09, "Groups": 0.08, "Complementary": 0.01}
# Price paid vs the standard price of the room type (assumption).
SEGMENT_PRICE_FACTOR = {"Corporate": 0.9, "Groups": 0.9, "Offline TA/TO": 0.95, "Complementary": 0.0}
PEAK_PRICE_FACTOR = 1.15 # applied when the arrival date has a demand factor >= 1.5
PEAK_DEMAND_THRESHOLD = 1.5
LARGE_ROOM_PREFERENCE = 0.85 # parties of 3+ or with children who ask for a large room
DIRTY_SHARE = 0.02
def demand_factor(day: date, metadata: HotelMetadata, holiday_names: dict[date, str]) -> float:
factor = MONTH_FACTOR[day.month] * WEEKDAY_FACTOR[day.weekday()]
holiday = holiday_names.get(day)
if holiday:
factor *= next((f for key, f in HOLIDAY_FACTORS if key in holiday), OTHER_HOLIDAY_FACTOR)
for event in metadata.nearby_events:
if event.covers(day):
factor *= next((f for key, f in EVENT_FACTORS if key in event.name), OTHER_EVENT_FACTOR)
break
return factor
def pick(rng: np.random.Generator, weights: dict[str, float]) -> str:
keys = list(weights)
probabilities = np.array(list(weights.values()), dtype=float)
return str(rng.choice(keys, p=probabilities / probabilities.sum()))
def simulate(metadata: HotelMetadata, rng: np.random.Generator) -> pd.DataFrame:
days = (LAST_ARRIVAL - FIRST_ARRIVAL).days + 1
horizon = days + 30 # stays that begin near the end run past LAST_ARRIVAL
capacity = {code: room.room_count for code, room in metadata.room_types.items()}
if any(count is None for count in capacity.values()):
raise SystemExit("room_count must be set for every room type in hotel_metadata.json")
booked = {code: np.zeros(horizon, dtype=int) for code in capacity}
holiday_names = PublicHolidayCalendar().for_years(FIRST_ARRIVAL.year, LAST_ARRIVAL.year)
small, large = "A", "B"
rows = []
for offset in range(days):
arrival = FIRST_ARRIVAL + timedelta(days=offset)
factor = demand_factor(arrival, metadata, holiday_names)
for _ in range(rng.poisson(BASE_REQUESTS_PER_DAY * factor)):
adults = int(rng.choice([1, 2, 3, 4], p=[0.25, 0.55, 0.12, 0.08]))
children = int(rng.choice([0, 1, 2], p=[0.85, 0.10, 0.05])) if adults >= 2 else 0
babies = int(rng.random() < 0.02)
wants_large = adults >= 3 or children > 0
preferred = large if rng.random() < (LARGE_ROOM_PREFERENCE if wants_large else 0.2) else small
nights = int(rng.choice(STAY_NIGHTS, p=STAY_WEIGHTS))
segment = pick(rng, SEGMENTS)
if segment == "Corporate":
nights = min(nights, 4)
deposit = pick(rng, {"No Deposit": 0.90, "Non Refund": 0.07, "Refundable": 0.03})
lead_mean = 45 if factor >= 1.3 else 22
lead_time = int(min(rng.exponential(lead_mean), 365))
cancel_probability = 0.20 * (0.3 if deposit == "Non Refund" else 1.0)
is_canceled = int(rng.random() < cancel_probability)
stay = slice(offset, offset + nights)
assigned = None
for code in (preferred, small if preferred == large else large):
if (booked[code][stay] < capacity[code]).all():
assigned = code
break
if assigned is None and not is_canceled:
continue # sold out on at least one night: request turned away
if assigned is None: # a cancelled request keeps its preferred type
assigned = preferred
if not is_canceled:
booked[assigned][stay] += 1
stay_dates = [arrival + timedelta(days=n) for n in range(nights)]
weekend_nights = sum(d.weekday() >= 5 for d in stay_dates)
standard = metadata.room_types[assigned].standard_price_per_night
price = standard * SEGMENT_PRICE_FACTOR.get(segment, 1.0)
if factor >= PEAK_DEMAND_THRESHOLD and segment != "Complementary":
price *= PEAK_PRICE_FACTOR
rows.append(
{
"hotel": metadata.hotel,
"is_canceled": is_canceled,
"lead_time": lead_time,
"arrival_date_week_number": arrival.isocalendar()[1],
"booking_date": arrival - timedelta(days=lead_time),
"arrival_date": arrival,
"arrival_date_day_of_month": arrival.day,
"stays_in_weekend_nights": weekend_nights,
"stays_in_week_nights": nights - weekend_nights,
"adults": adults,
"children": children,
"babies": babies,
"meal": pick(rng, {"BB": 0.70, "HB": 0.10, "FB": 0.02, "SC": 0.18}),
"country": pick(rng, COUNTRIES),
"market_segment": segment,
"is_repeated_guest": int(rng.random() < 0.08),
"previous_cancellations": int(rng.random() < 0.04),
"assigned_room_type": assigned,
"booking_changes": int(rng.choice([0, 1, 2], p=[0.88, 0.09, 0.03])),
"deposit_type": deposit,
"agent": int(rng.choice([0, 9, 14, 28, 40, 240], p=[0.35, 0.15, 0.15, 0.1, 0.1, 0.15])),
"customer_type": "Group Contract" if segment == "Groups" else pick(
rng, {"No Contract (Single)": 0.75, "No Contract (Group)": 0.17, "Contract (Single)": 0.08}
),
"required_car_parking_spaces": int(rng.random() < 0.05),
"total_of_special_requests": int(rng.choice([0, 1, 2, 3], p=[0.55, 0.28, 0.12, 0.05])),
"prize_per_nigth": round(price, 2),
}
)
df = pd.DataFrame(rows)
df = df.sort_values(["booking_date", "arrival_date"], kind="stable").reset_index(drop=True)
df.insert(0, "booking_id", np.arange(1, len(df) + 1))
return df
def add_dirty_rows(df: pd.DataFrame, rng: np.random.Generator) -> pd.DataFrame:
"""Append duplicates and corrupt a few values, like a real messy export."""
if DIRTY_SHARE <= 0:
return df
df = df.copy()
n = max(1, int(len(df) * DIRTY_SHARE))
corrupt = df.sample(n, random_state=SEED + 1).index
half = len(corrupt) // 2
df.loc[corrupt[:half], "adults"] = rng.choice([20, 55], size=half)
df.loc[corrupt[half:], ["adults", "children", "babies"]] = 0 # booking without guests
df["meal"] = df["meal"].astype(object)
df.loc[df.sample(n // 2, random_state=SEED + 2).index, "meal"] = ""
# Duplicates are copied last so they are exact copies (same values, new id)
# of rows as they appear in the export, including any corruption above.
duplicates = df.sample(n, random_state=SEED)
df = pd.concat([df, duplicates], ignore_index=True)
df = df.sort_values(["booking_date", "arrival_date"], kind="stable").reset_index(drop=True)
df["booking_id"] = np.arange(1, len(df) + 1)
return df
def main() -> None:
metadata = HotelMetadata.model_validate_json(METADATA_FILE.read_text(encoding="utf-8"))
rng = np.random.default_rng(SEED)
df = add_dirty_rows(simulate(metadata, rng), rng)
for column in ("booking_date", "arrival_date"):
df[column] = pd.to_datetime(df[column]).dt.strftime("%d-%m-%Y")
df.to_csv(OUTPUT, sep=";", index=False)
print(f"wrote {len(df)} rows to {OUTPUT}")
if __name__ == "__main__":
main()