""" generate_data.py ----------------- Generates a synthetic e-commerce product catalog with 2000+ unique entries and saves it as products.csv. This CSV is the "custom database" the RAG chatbot retrieves from. Run: python generate_data.py """ import csv import random random.seed(42) # --------------------------------------------------------------------------- # Catalog taxonomy: category -> subcategories -> base product name templates # --------------------------------------------------------------------------- CATALOG = { "Electronics": { "subcats": ["Smartphones", "Laptops", "Headphones", "Smartwatches", "Cameras", "Tablets", "Speakers", "Gaming Consoles", "Monitors", "Power Banks"], "price_range": (25, 2200), "brands": ["Voltek", "Nimbus", "Zentro", "Quantex", "Pulsar", "Orbitron", "Skyline", "Nexura", "Vantek", "Corevolt"], }, "Clothing": { "subcats": ["T-Shirts", "Jeans", "Jackets", "Dresses", "Sweaters", "Activewear", "Shorts", "Formal Shirts", "Hoodies", "Coats"], "price_range": (8, 250), "brands": ["Urban Thread", "Loomcraft", "Northgate", "Willow & Ash", "StrideWear", "Fablink", "Merino Co.", "CasualEdge", "DrapeLine", "Threadfox"], }, "Home & Kitchen": { "subcats": ["Cookware", "Blenders", "Coffee Makers", "Bedding", "Vacuum Cleaners", "Storage Bins", "Cutlery Sets", "Lamps", "Air Purifiers", "Dinnerware"], "price_range": (10, 600), "brands": ["HearthPro", "Domesta", "Kitchcraft", "CozyNest", "PureAir", "Lumina Home", "ChefMate", "TidyBox", "SilverEdge", "WarmHearth"], }, "Beauty & Personal Care": { "subcats": ["Moisturizers", "Shampoos", "Perfumes", "Makeup Kits", "Electric Razors", "Hair Dryers", "Skincare Serums", "Sunscreens", "Lip Balms", "Face Masks"], "price_range": (5, 150), "brands": ["Glowvia", "PureSkin", "Belle Aura", "Dermalux", "Velvetone", "NatureGlow", "SilkRoute", "LumiCare", "EssenceLab", "RadiantHue"], }, "Sports & Outdoors": { "subcats": ["Yoga Mats", "Dumbbells", "Tents", "Bicycles", "Running Shoes", "Backpacks", "Water Bottles", "Fitness Trackers", "Camping Stoves", "Skateboards"], "price_range": (10, 900), "brands": ["TrailBlaze", "PeakForm", "IronGrit", "SummitGear", "Velocore", "RidgeLine", "ActivFit", "TerraTrek", "FlexCore", "AltitudePro"], }, "Books": { "subcats": ["Fiction", "Non-Fiction", "Science Fiction", "Biography", "Self-Help", "Children's Books", "Mystery & Thriller", "Cookbooks", "History", "Poetry"], "price_range": (4, 45), "brands": ["Penfield Press", "Storyhouse", "Chapter & Verse", "Inkwell Editions", "Lanternlight Books", "Northpage", "Quillmark", "Bright Leaf Publishing", "Cobblestone Press", "Willow Bind"], }, "Toys & Games": { "subcats": ["Building Blocks", "Board Games", "Action Figures", "Puzzles", "Dolls", "Remote Control Cars", "Educational Toys", "Card Games", "Plush Toys", "Outdoor Play Sets"], "price_range": (5, 180), "brands": ["Funkidoo", "Brightblox", "PlayNest", "Wonderrific", "Tinker Toys Co.", "GigglePatch", "Kudo Kids", "Puzzlewise", "Playscape", "JoyForge"], }, "Grocery & Gourmet": { "subcats": ["Coffee & Tea", "Snacks", "Spices", "Cereal", "Olive Oils", "Pasta & Grains", "Chocolates", "Honey & Preserves", "Nuts & Seeds", "Sauces"], "price_range": (3, 60), "brands": ["Harvest Table", "Golden Pantry", "Rustic Roots", "PureField", "Meadowbrook", "SpiceHaven", "Orchard Gold", "GrainWorks", "Savory Bay", "Farmstead Co."], }, "Office & Stationery": { "subcats": ["Notebooks", "Pens", "Desk Organizers", "Backpacks", "Printers", "Office Chairs", "Whiteboards", "Planners", "Sticky Notes", "Desk Lamps"], "price_range": (2, 400), "brands": ["Deskly", "Penmark", "Officia", "Craftline", "NotaBene", "GridWorks", "ClearDesk", "InkPoint", "OrganizeIt", "Brightfold"], }, "Pet Supplies": { "subcats": ["Dog Food", "Cat Toys", "Pet Beds", "Leashes & Collars", "Aquarium Kits", "Grooming Kits", "Bird Cages", "Pet Carriers", "Litter Boxes", "Chew Toys"], "price_range": (4, 220), "brands": ["Pawsome", "Furry Friend Co.", "WhiskerWorks", "TailWag", "NestleNook", "Barkline", "PetHaven", "ScratchCraft", "Critter Corner", "Loyal Paws"], }, } ADJECTIVES = ["Premium", "Classic", "Compact", "Deluxe", "Ergonomic", "Portable", "Wireless", "Eco-Friendly", "Heavy-Duty", "Ultra-Light", "All-Season", "Professional", "Everyday", "Signature", "Modern", "Vintage-Style", "Advanced", "Essential", "Rugged", "Smart"] COLORS = ["Black", "White", "Navy Blue", "Charcoal Grey", "Forest Green", "Sunset Orange", "Crimson Red", "Sand Beige", "Slate Blue", "Blush Pink", "Graphite", "Ivory"] MATERIALS = ["stainless steel", "brushed aluminum", "organic cotton", "recycled polyester", "genuine leather", "BPA-free plastic", "tempered glass", "bamboo", "high-density foam", "merino wool", "silicone", "anodized alloy"] FEATURES = [ "long battery life", "a sleek minimalist design", "industry-leading durability", "fast and reliable performance", "an intuitive user experience", "excellent value for money", "top-rated customer reviews", "easy maintenance", "a comfortable fit for all-day use", "energy-efficient operation", "quick setup with no tools required", "a lightweight build", "water-resistant construction", "adjustable settings for personalized use", "a timeless design that fits any space", "reinforced stitching for extra durability", "noise isolation for an immersive experience", "a non-slip grip for added safety", ] TAG_POOL = ["bestseller", "new arrival", "eco-friendly", "limited edition", "on sale", "top rated", "staff pick", "gift idea", "trending", "budget friendly", "premium choice", "family favorite"] def make_description(name, category, subcat, brand, color, material, adj): feat1, feat2 = random.sample(FEATURES, 2) templates = [ f"The {name} from {brand} combines {material} construction with {feat1}. " f"Designed for fans of {subcat.lower()}, this {adj.lower()} pick offers {feat2}, " f"making it a standout choice in our {category} collection.", f"Meet the {name} — a {adj.lower()} {subcat.lower()[:-1] if subcat.endswith('s') else subcat.lower()} " f"built from {material}. It delivers {feat1} and {feat2}, " f"perfect for anyone shopping our {category} lineup.", f"{brand} presents the {name}, finished in {color.lower()} with premium {material}. " f"Customers love its {feat1}, and it also offers {feat2}.", ] return random.choice(templates) def generate_products(min_count=2000): rows = [] product_id = 1 for category, meta in CATALOG.items(): subcats = meta["subcats"] brands = meta["brands"] low, high = meta["price_range"] # ~ enough combos per category to comfortably exceed 2000 total across 10 categories per_category_target = 210 for _ in range(per_category_target): subcat = random.choice(subcats) brand = random.choice(brands) adj = random.choice(ADJECTIVES) color = random.choice(COLORS) material = random.choice(MATERIALS) # Singular-ish name from subcat base_noun = subcat[:-1] if subcat.endswith("s") and not subcat.endswith("ss") else subcat name = f"{brand} {adj} {base_noun}" price = round(random.uniform(low, high), 2) rating = round(random.uniform(3.3, 5.0), 1) num_reviews = random.randint(3, 4800) stock = random.choice(["In Stock"] * 8 + ["Low Stock"] * 2 + ["Out of Stock"]) tags = ", ".join(random.sample(TAG_POOL, k=random.randint(1, 3))) description = make_description(name, category, subcat, brand, color, material, adj) rows.append({ "product_id": f"P{product_id:05d}", "name": name, "category": category, "subcategory": subcat, "brand": brand, "price_usd": price, "rating": rating, "num_reviews": num_reviews, "stock_status": stock, "color": color, "material": material, "tags": tags, "description": description, }) product_id += 1 random.shuffle(rows) # De-duplicate exact name collisions by appending a variant suffix seen = {} for r in rows: key = r["name"] if key in seen: seen[key] += 1 r["name"] = f'{r["name"]} (Style {seen[key]})' else: seen[key] = 1 if len(rows) < min_count: raise RuntimeError(f"Only generated {len(rows)} rows, need {min_count}+") return rows def main(): rows = generate_products(2000) fieldnames = ["product_id", "name", "category", "subcategory", "brand", "price_usd", "rating", "num_reviews", "stock_status", "color", "material", "tags", "description"] with open("products.csv", "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=fieldnames) writer.writeheader() writer.writerows(rows) print(f"Generated {len(rows)} products -> products.csv") if __name__ == "__main__": main()