Files
ewoooc/services/sales_analysis_query_service.py

392 lines
16 KiB
Python

"""Canonical, parameterized query contract for sales-analysis APIs."""
from __future__ import annotations
import math
from datetime import date, datetime, timedelta, timezone
from typing import Any, Mapping
from sqlalchemy import inspect, text
from services.analysis_period_service import parse_iso_date, parse_month
from utils.df_helpers import find_col
from utils.security import validate_table_name
ALLOWED_DATA_RANGES = {0, 1, 3, 6, 12}
TAIPEI_TZ = timezone(timedelta(hours=8))
SALES_TIME_TEXT_PATTERN = (
r"^([01][0-9]|2[0-3]):[0-5][0-9](:[0-5][0-9](\.[0-9]+)?)?$"
)
def _get_value(args: Mapping[str, Any], key: str, default: Any = "") -> Any:
getter = getattr(args, "get", None)
return getter(key, default) if callable(getter) else default
def _bounded_text(value: Any, field: str, *, max_length: int = 500) -> str:
result = str(value or "").strip()
if len(result) > max_length:
raise ValueError(f"{field} 長度不可超過 {max_length} 字元")
return result
def _optional_number(args: Mapping[str, Any], key: str) -> float | None:
raw = str(_get_value(args, key, "") or "").strip()
if not raw:
return None
try:
result = float(raw)
except (TypeError, ValueError) as exc:
raise ValueError(f"{key} 必須是數字") from exc
if not math.isfinite(result):
raise ValueError(f"{key} 必須是有限數字")
return result
def _optional_dimension(args: Mapping[str, Any], key: str) -> str | None:
raw = _bounded_text(_get_value(args, key, "all"), key)
return None if not raw or raw.lower() == "all" else raw
def _optional_bounded_int(
args: Mapping[str, Any],
key: str,
*,
minimum: int,
maximum: int,
) -> int | None:
raw = str(_get_value(args, key, "all") or "all").strip().lower()
if raw == "all":
return None
if not raw.isdigit() or not minimum <= int(raw) <= maximum:
raise ValueError(f"{key} 必須介於 {minimum}{maximum}")
return int(raw)
def normalize_sales_query_args(
args: Mapping[str, Any],
*,
default_data_range: int = 1,
) -> dict[str, Any]:
"""Validate and canonicalize all filter values shared by page APIs."""
range_raw = str(_get_value(args, "data_range", "") or "").strip()
if not range_raw:
data_range = default_data_range
elif not range_raw.isdigit() or int(range_raw) not in ALLOWED_DATA_RANGES:
raise ValueError("data_range 必須是 0、1、3、6 或 12")
else:
data_range = int(range_raw)
raw_start = str(_get_value(args, "start_date", "") or "").strip()
raw_end = str(_get_value(args, "end_date", "") or "").strip()
start = parse_iso_date(raw_start)
end = parse_iso_date(raw_end)
if raw_start and not start:
raise ValueError("start_date 必須是 YYYY-MM-DD")
if raw_end and not end:
raise ValueError("end_date 必須是 YYYY-MM-DD")
if start and end and start > end:
start, end = end, start
month_raw = str(_get_value(args, "month", "all") or "all").strip().lower()
month = None
if month_raw != "all":
parsed_month = parse_month(month_raw)
if not parsed_month:
raise ValueError("month 必須是 YYYY-MM")
month = parsed_month.strftime("%Y-%m")
return {
"data_range": data_range,
"start_date": start,
"end_date": end,
"category": _optional_dimension(args, "category"),
"brand": _optional_dimension(args, "brand"),
"vendor": _optional_dimension(args, "vendor"),
"activity": _optional_dimension(args, "activity"),
"payment": _optional_dimension(args, "payment"),
"month": month,
"dow": _optional_bounded_int(args, "dow", minimum=0, maximum=6),
"hour": _optional_bounded_int(args, "hour", minimum=0, maximum=23),
"min_price": _optional_number(args, "min_price"),
"max_price": _optional_number(args, "max_price"),
"min_margin": _optional_number(args, "min_margin"),
"max_margin": _optional_number(args, "max_margin"),
"keyword": _bounded_text(_get_value(args, "keyword", ""), "keyword", max_length=200),
}
def resolve_sales_query_columns(engine, table_name: str) -> dict[str, str | None]:
"""Resolve identifiers from the live table schema rather than request input."""
validate_table_name(table_name)
names = [column["name"] for column in inspect(engine).get_columns(table_name)]
return {
"date": find_col(names, ["日期", "交易日期", "Date", "Day"]),
"time": find_col(names, ["訂單時間", "成立時間", "下單時間", "購買時間", "時間", "Time", "Created"]),
"pid": find_col(names, ["商品ID", "Product ID", "i_code", "Item Code", "ID"]),
"name": find_col(names, ["商品名稱", "品名", "Name", "Product"]),
"brand": find_col(names, ["品牌", "Brand"]),
"vendor": find_col(names, ["廠商名稱", "Vendor Name", "廠商", "供應商", "Vendor", "Supplier"]),
"category": find_col(names, ["商品館", "館別", "分類", "Category"]),
"activity": find_col(names, ["折扣活動名稱", "折價券活動名稱", "滿額再折扣活動名稱", "活動", "Activity", "Campaign", "Promotion", "專案"]),
"discount_activity": find_col(names, ["折扣活動名稱"]),
"coupon_activity": find_col(names, ["折價券活動名稱"]),
"bonus_activity": find_col(names, ["滿額再折扣活動名稱"]),
"click_activity": find_col(names, ["點我再折扣"]),
"payment": find_col(names, ["付款方式", "付款", "Payment", "Pay"]),
"amount": find_col(names, ["銷售金額", "總業績", "業績", "金額", "Amount", "Sales", "Total"]),
"qty": find_col(names, ["銷售數量", "銷量", "數量", "Qty", "Quantity"]),
"cost": find_col(names, ["總成本", "成本", "Cost", "進價", "Cost Price", "Wholesale"]),
"profit": find_col(names, ["毛利", "Profit", "利潤"]),
"return_qty": find_col(names, ["退貨數量", "Return Qty", "退貨"]),
}
def quote_identifier(engine, identifier: str | None) -> str:
if not identifier:
raise ValueError("查詢所需欄位不存在")
return engine.dialect.identifier_preparer.quote(str(identifier))
def build_sales_metric_aggregate_sql(
engine,
columns: Mapping[str, str | None],
metric: str,
) -> tuple[str, str]:
"""Build one allowlisted aggregate shared by APIs and exports."""
amount = quote_identifier(engine, columns.get("amount"))
if metric in {"amount", "revenue"}:
return f"COALESCE(SUM(CAST({amount} AS REAL)), 0)", "銷售金額"
if metric == "qty":
qty = quote_identifier(engine, columns.get("qty"))
return f"COALESCE(SUM(CAST({qty} AS REAL)), 0)", "銷售數量"
if metric == "profit":
if columns.get("profit"):
profit = quote_identifier(engine, columns.get("profit"))
return f"COALESCE(SUM(CAST({profit} AS REAL)), 0)", "毛利金額"
cost = quote_identifier(engine, columns.get("cost"))
return (
f"COALESCE(SUM(CAST({amount} AS REAL)), 0) "
f"- COALESCE(SUM(CAST({cost} AS REAL)), 0)",
"毛利金額",
)
raise ValueError("metric 僅允許 amount、revenue、qty、profit")
def _normalised_date_sql(engine, quoted_date: str) -> tuple[str, str]:
if engine.dialect.name == "postgresql":
date_sql = f"LEFT(REPLACE(CAST({quoted_date} AS TEXT), '/', '-'), 10)"
return date_sql, f"LEFT({date_sql}, 7)"
date_sql = f"substr(replace(CAST({quoted_date} AS TEXT), '/', '-'), 1, 10)"
return date_sql, f"substr({date_sql}, 1, 7)"
def resolve_sales_date_bounds(
engine,
table_name: str,
columns: Mapping[str, str | None] | None = None,
) -> tuple[date | None, date | None]:
"""Return canonical source bounds for closing one-sided page ranges."""
table_name = validate_table_name(table_name)
resolved_columns = columns or resolve_sales_query_columns(engine, table_name)
quoted_date = quote_identifier(engine, resolved_columns.get("date"))
date_sql, _month_sql = _normalised_date_sql(engine, quoted_date)
table = quote_identifier(engine, table_name)
query = text(f"SELECT MIN({date_sql}), MAX({date_sql}) FROM {table}")
with engine.connect() as connection:
row = connection.execute(query).fetchone()
if not row:
return None, None
return parse_iso_date(row[0]), parse_iso_date(row[1])
def build_sales_where_clause(
engine,
columns: Mapping[str, str | None],
filters: Mapping[str, Any],
*,
period_start: date | str | None = None,
period_end: date | str | None = None,
apply_month: bool = True,
include_numeric: bool = True,
now: date | None = None,
) -> tuple[str, dict[str, Any]]:
"""Return a SQL fragment and bind parameters for the canonical filters."""
conditions: list[str] = []
params: dict[str, Any] = {}
explicit_period = period_start is not None or period_end is not None
start = parse_iso_date(period_start) if period_start is not None else filters.get("start_date")
end = parse_iso_date(period_end) if period_end is not None else filters.get("end_date")
if start and end and start > end:
start, end = end, start
if not explicit_period and not start and not end and int(filters.get("data_range") or 0) > 0:
end = now or datetime.now(TAIPEI_TZ).date()
start = end - timedelta(days=int(filters["data_range"]) * 30)
needs_date = bool(start or end or (apply_month and filters.get("month")) or filters.get("dow") is not None)
date_sql = month_sql = None
if needs_date:
quoted_date = quote_identifier(engine, columns.get("date"))
date_sql, month_sql = _normalised_date_sql(engine, quoted_date)
if start:
conditions.append(f"{date_sql} >= :filter_start_date")
params["filter_start_date"] = start.isoformat()
if end:
conditions.append(f"{date_sql} <= :filter_end_date")
params["filter_end_date"] = end.isoformat()
if apply_month and filters.get("month"):
conditions.append(f"{month_sql} = :filter_month")
params["filter_month"] = filters["month"]
dimension_columns = {
"category": "category",
"brand": "brand",
"vendor": "vendor",
"activity": "activity",
"payment": "payment",
}
for filter_name, column_name in dimension_columns.items():
value = filters.get(filter_name)
if value is None:
continue
quoted = quote_identifier(engine, columns.get(column_name))
if filter_name == "category" and value == "其他":
exclusions = filters.get("category_other_exclusions")
if exclusions is None:
raise ValueError("其他分類缺少排行範圍契約")
if not exclusions:
conditions.append("1 = 0")
continue
placeholders = []
for index, excluded in enumerate(exclusions):
param_name = f"filter_category_other_{index}"
placeholders.append(f":{param_name}")
params[param_name] = excluded
conditions.append(f"{quoted} NOT IN ({', '.join(placeholders)})")
continue
conditions.append(f"{quoted} = :filter_{filter_name}")
params[f"filter_{filter_name}"] = value
if filters.get("dow") is not None:
database_dow = (int(filters["dow"]) + 1) % 7
if engine.dialect.name == "postgresql":
conditions.append(
f"EXTRACT(DOW FROM TO_DATE({date_sql}, 'YYYY-MM-DD')) = :filter_db_dow"
)
params["filter_db_dow"] = database_dow
else:
conditions.append(f"strftime('%w', {date_sql}) = :filter_db_dow")
params["filter_db_dow"] = str(database_dow)
if filters.get("hour") is not None:
quoted_time = quote_identifier(engine, columns.get("time"))
if engine.dialect.name == "postgresql":
time_text = f"TRIM(CAST({quoted_time} AS TEXT))"
conditions.append(
"CASE WHEN "
f"{time_text} ~ '{SALES_TIME_TEXT_PATTERN}' "
f"THEN CAST(SUBSTRING({time_text} FROM 1 FOR 2) AS INTEGER) END "
"= :filter_hour"
)
else:
time_text = f"TRIM(CAST({quoted_time} AS TEXT))"
conditions.append(
f"CASE WHEN {time_text} GLOB '[0-2][0-9]:[0-5][0-9]*' "
"THEN CAST(strftime('%H', '2000-01-01 ' || "
f"{time_text}) AS INTEGER) END = :filter_hour"
)
params["filter_hour"] = int(filters["hour"])
keyword = filters.get("keyword")
if keyword:
keyword_columns = [columns.get(key) for key in ("name", "pid", "brand", "vendor")]
keyword_conditions = [
f"LOWER(CAST({quote_identifier(engine, column)} AS TEXT)) "
"LIKE LOWER(:filter_keyword) ESCAPE '\\'"
for column in keyword_columns
if column
]
if not keyword_conditions:
raise ValueError("目前資料來源沒有可搜尋欄位")
conditions.append(f"({' OR '.join(keyword_conditions)})")
escaped_keyword = (
keyword.replace("\\", "\\\\")
.replace("%", "\\%")
.replace("_", "\\_")
)
params["filter_keyword"] = f"%{escaped_keyword}%"
if include_numeric:
amount = columns.get("amount")
qty = columns.get("qty")
if filters.get("min_price") is not None or filters.get("max_price") is not None:
price_sql = (
f"CAST({quote_identifier(engine, amount)} AS REAL) / "
f"NULLIF(CAST({quote_identifier(engine, qty)} AS REAL), 0)"
)
for bound in ("min_price", "max_price"):
value = filters.get(bound)
if value is not None:
operator = ">=" if bound == "min_price" else "<="
conditions.append(f"{price_sql} {operator} :filter_{bound}")
params[f"filter_{bound}"] = value
if filters.get("min_margin") is not None or filters.get("max_margin") is not None:
quoted_amount = quote_identifier(engine, amount)
if columns.get("profit"):
profit_sql = f"CAST({quote_identifier(engine, columns['profit'])} AS REAL)"
elif columns.get("cost"):
profit_sql = (
f"CAST({quoted_amount} AS REAL) - "
f"CAST({quote_identifier(engine, columns['cost'])} AS REAL)"
)
else:
raise ValueError("目前資料來源沒有毛利或成本欄位")
margin_sql = f"({profit_sql} * 100.0 / NULLIF(CAST({quoted_amount} AS REAL), 0))"
for bound in ("min_margin", "max_margin"):
value = filters.get(bound)
if value is not None:
operator = ">=" if bound == "min_margin" else "<="
conditions.append(f"{margin_sql} {operator} :filter_{bound}")
params[f"filter_{bound}"] = value
return " AND ".join(conditions), params
def prepare_sales_query_context(
engine,
table_name: str,
args: Mapping[str, Any],
*,
default_data_range: int = 1,
) -> tuple[dict[str, Any], dict[str, str | None]]:
"""Resolve canonical filters, live columns and the dynamic Other bucket."""
table_name = validate_table_name(table_name)
filters = normalize_sales_query_args(args, default_data_range=default_data_range)
columns = resolve_sales_query_columns(engine, table_name)
if filters.get("category") != "其他":
return filters, columns
ranking_filters = dict(filters)
ranking_filters["category"] = None
where_sql, params = build_sales_where_clause(engine, columns, ranking_filters)
category = quote_identifier(engine, columns.get("category"))
amount = quote_identifier(engine, columns.get("amount"))
table = quote_identifier(engine, table_name)
where_suffix = f" AND {where_sql}" if where_sql else ""
query = text(f"""
SELECT {category} AS category_name
FROM {table}
WHERE {category} IS NOT NULL {where_suffix}
GROUP BY {category}
ORDER BY SUM(CAST({amount} AS REAL)) DESC, CAST({category} AS TEXT) ASC
LIMIT 12
""")
with engine.connect() as connection:
rows = connection.execute(query, params).fetchall()
filters["category_other_exclusions"] = [str(row[0]) for row in rows]
return filters, columns