infra: hadoop historical sales loader (HDFS partitioned by year)
This commit is contained in:
@@ -0,0 +1,72 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Generate historical sales data and load it into HDFS (partitioned by year).
|
||||||
|
|
||||||
|
Runs on the Hadoop master node (has hdfs client + datanode access). Writes
|
||||||
|
CSV partitions under HDFS_BASE so the data is browsable in the Command Center
|
||||||
|
HDFS view and can later be exposed as a Hive/Iceberg external table in Trino.
|
||||||
|
"""
|
||||||
|
import csv
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timedelta
|
||||||
|
|
||||||
|
HDFS_BASE = os.getenv("HDFS_BASE", "/data/historical/sales_orders")
|
||||||
|
ROWS_PER_YEAR = int(os.getenv("ROWS_PER_YEAR", "5000"))
|
||||||
|
YEARS = [int(y) for y in os.getenv("YEARS", "2020,2021,2022,2023,2024").split(",")]
|
||||||
|
|
||||||
|
REGIONS = ["EU", "APAC", "LATAM", "NA", "EMEA"]
|
||||||
|
CHANNELS = ["STORE", "ONLINE", "MOBILE", "B2B"]
|
||||||
|
CURRENCIES = ["EUR", "USD", "GBP", "JPY", "CNY"]
|
||||||
|
STATUSES = ["SHIPPED", "PENDING", "CANCELLED", "RETURNED", "DELIVERED"]
|
||||||
|
HEADER = ["order_id", "customer_id", "product_id", "region", "sales_channel",
|
||||||
|
"order_ts", "amount", "currency", "order_status"]
|
||||||
|
|
||||||
|
|
||||||
|
def hdfs(*args):
|
||||||
|
subprocess.run(["hdfs", "dfs", *args], check=True)
|
||||||
|
|
||||||
|
|
||||||
|
def gen_year(year):
|
||||||
|
start = datetime(year, 1, 1)
|
||||||
|
rows = []
|
||||||
|
for _ in range(ROWS_PER_YEAR):
|
||||||
|
ts = start + timedelta(days=random.randint(0, 364), hours=random.randint(0, 23),
|
||||||
|
minutes=random.randint(0, 59))
|
||||||
|
rows.append([
|
||||||
|
str(uuid.uuid4()),
|
||||||
|
random.randint(1, 100000),
|
||||||
|
random.randint(1, 5000),
|
||||||
|
random.choice(REGIONS),
|
||||||
|
random.choice(CHANNELS),
|
||||||
|
ts.strftime("%Y-%m-%d %H:%M:%S"),
|
||||||
|
round(random.uniform(10.0, 10000.0), 2),
|
||||||
|
random.choice(CURRENCIES),
|
||||||
|
random.choice(STATUSES),
|
||||||
|
])
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
print(f"Loading historical sales into hdfs://{HDFS_BASE} for years {YEARS}")
|
||||||
|
total = 0
|
||||||
|
for year in YEARS:
|
||||||
|
rows = gen_year(year)
|
||||||
|
with tempfile.NamedTemporaryFile("w", suffix=".csv", delete=False, newline="") as f:
|
||||||
|
w = csv.writer(f)
|
||||||
|
w.writerow(HEADER)
|
||||||
|
w.writerows(rows)
|
||||||
|
local = f.name
|
||||||
|
hdfs_dir = f"{HDFS_BASE}/year={year}"
|
||||||
|
hdfs("-mkdir", "-p", hdfs_dir)
|
||||||
|
hdfs("-put", "-f", local, f"{hdfs_dir}/part-0.csv")
|
||||||
|
os.unlink(local)
|
||||||
|
total += len(rows)
|
||||||
|
print(f" year={year}: {len(rows)} rows -> {hdfs_dir}/part-0.csv")
|
||||||
|
print(f"Done. {total} historical rows across {len(YEARS)} years in HDFS.")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user