File size: 4,206 Bytes
b5fd89f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
"""Generate the synthetic sample dataset shipped with this Space.

Everything here is fabricated with a fixed seed — no real, customer, or
proprietary data is used anywhere in this project. Re-run to regenerate:

    python make_sample_data.py
"""

import csv
import random
from datetime import date, timedelta
from pathlib import Path

SEED = 20260811
ROWS = 2000
OUT = Path(__file__).parent / "data" / "sample_sales.csv"

REGIONS = {
    "North America": ["United States", "United States", "Canada"],
    "LATAM": ["Mexico", "Mexico", "Colombia", "Argentina", "Chile"],
    "EMEA": ["Spain", "Germany", "United Kingdom"],
}
REGION_WEIGHTS = [0.45, 0.35, 0.20]

CATALOG = {
    "Electronics": [
        ("Wireless Headphones", 89.99),
        ("4K Monitor", 329.00),
        ("Mechanical Keyboard", 119.50),
        ("Smart Speaker", 59.00),
    ],
    "Home & Kitchen": [
        ("Espresso Machine", 249.00),
        ("Air Fryer", 129.99),
        ("Cookware Set", 179.00),
        ("Blender", 74.50),
    ],
    "Apparel": [
        ("Running Jacket", 94.00),
        ("Denim Jeans", 68.00),
        ("Merino Socks", 22.00),
        ("Rain Shell", 139.00),
    ],
    "Sports": [
        ("Yoga Mat", 45.00),
        ("Dumbbell Set", 159.00),
        ("Trail Backpack", 112.00),
        ("Water Bottle", 28.00),
    ],
    "Beauty": [
        ("Skincare Set", 82.00),
        ("Hair Dryer", 96.00),
        ("Perfume", 115.00),
        ("Electric Razor", 68.50),
    ],
}
CATEGORY_WEIGHTS = [0.28, 0.22, 0.20, 0.16, 0.14]

CHANNELS = ["Online", "Retail Store", "Partner"]
CHANNEL_WEIGHTS = [0.58, 0.28, 0.14]

SEGMENTS = ["Consumer", "Small Business", "Enterprise"]
SEGMENT_WEIGHTS = [0.62, 0.26, 0.12]

# Retail seasonality: quiet start, holiday spike in Nov/Dec.
MONTH_WEIGHTS = [0.60, 0.58, 0.70, 0.75, 0.80, 0.80,
                 0.85, 0.80, 0.90, 1.00, 1.60, 1.80]

START = date(2025, 1, 1)
DAYS = 365


def main() -> None:
    rng = random.Random(SEED)
    day_weights = [MONTH_WEIGHTS[(START + timedelta(days=d)).month - 1]
                   for d in range(DAYS)]

    rows = []
    for i in range(ROWS):
        order_date = START + timedelta(
            days=rng.choices(range(DAYS), weights=day_weights)[0]
        )
        region = rng.choices(list(REGIONS), weights=REGION_WEIGHTS)[0]
        country = rng.choice(REGIONS[region])
        category = rng.choices(list(CATALOG), weights=CATEGORY_WEIGHTS)[0]
        product, base_price = rng.choice(CATALOG[category])
        channel = rng.choices(CHANNELS, weights=CHANNEL_WEIGHTS)[0]
        segment = rng.choices(SEGMENTS, weights=SEGMENT_WEIGHTS)[0]

        quantity = rng.choices([1, 2, 3, 4, 6, 10],
                               weights=[46, 24, 12, 8, 6, 4])[0]
        if segment == "Enterprise":
            quantity *= rng.choice([2, 3, 5])

        # Prices drift a little by market; discounts cluster around promos.
        unit_price = round(base_price * rng.uniform(0.94, 1.08), 2)
        discount_pct = rng.choices(
            [0.00, 0.05, 0.10, 0.15, 0.25],
            weights=[52, 18, 15, 9, 6],
        )[0]
        if order_date.month in (11, 12) and rng.random() < 0.45:
            discount_pct = max(discount_pct, 0.15)

        revenue = round(quantity * unit_price * (1 - discount_pct), 2)

        rows.append({
            "order_id": f"ORD-{100000 + i}",
            "order_date": order_date.isoformat(),
            "region": region,
            "country": country,
            "channel": channel,
            "customer_segment": segment,
            "category": category,
            "product": product,
            "quantity": quantity,
            "unit_price": unit_price,
            "discount_pct": discount_pct,
            "revenue": revenue,
        })

    rows.sort(key=lambda r: r["order_date"])

    OUT.parent.mkdir(parents=True, exist_ok=True)
    with OUT.open("w", newline="", encoding="utf-8") as fh:
        writer = csv.DictWriter(fh, fieldnames=list(rows[0]))
        writer.writeheader()
        writer.writerows(rows)

    total = sum(r["revenue"] for r in rows)
    print(f"wrote {len(rows)} rows to {OUT}  (total revenue {total:,.2f})")


if __name__ == "__main__":
    main()