"""Synthetic vineyard / winery DataFrames. Twelve generators spanning the whole operation -- vineyard block records, field time-series, lab chemistry, cellar, soil, pest scouting, sensors, sensory panel, logistics, sales, canopy management, and barrel aging. WHY SO MANY: the student model must learn to READ the schema preview it is given and reference those columns. With only a few generators it can instead memorise one fixed column vocabulary ("vineyard plot" -> df['brix']) and still drive the training loss to zero -- then fail on any unseen DataFrame. Distinct column names per generator is what forces the model to actually read the preview. SEEDING: every generator takes `seed` and builds its OWN Generator, so a call is reproducible from (name, seed) alone. This matters because generate_raw.py builds a df to write the preview, and execute.py later rebuilds it to validate the code -- they must agree. A module-level shared RNG would advance between those two calls and hand back different data. """ import numpy as np import pandas as pd BLOCKS = ["North Slope", "River Bench", "Hilltop", "Old Vines", "Clay Flat", "East Terrace"] VARIETIES = ["Cabernet Sauvignon", "Chardonnay", "Pinot Noir", "Syrah", "Merlot", "Sauvignon Blanc"] ROOTSTOCK = ["101-14", "3309C", "SO4", "1103P", "Riparia Gloire"] DISEASES = ["Powdery Mildew", "Downy Mildew", "Botrytis", "None"] def block_vintage_df(n=220, seed=0): """One row per block x vintage: climate, canopy, yield, and quality at harvest.""" rng = np.random.default_rng(seed) gdd = rng.normal(1650, 220, n).round(0).clip(900, 2600) # warmer seasons ripen harder: brix tracks GDD, acid falls away as sugar climbs brix = (18 + (gdd - 900) / 1700 * 6 + rng.normal(0, 0.9, n)).round(1).clip(18, 30) return pd.DataFrame({ "block": rng.choice(BLOCKS, n), "variety": rng.choice(VARIETIES, n), "rootstock": rng.choice(ROOTSTOCK, n), "vintage": rng.integers(2012, 2025, n), "gdd": gdd, # growing degree days "rainfall_mm": rng.normal(520, 160, n).round(1).clip(120, 1100), "irrigation_mm": rng.normal(180, 70, n).round(0).clip(0, 400), "yield_tonnes_per_acre": rng.normal(4.2, 1.6, n).round(2).clip(0.5, 9), "brix": brix, # sugar "ph": rng.normal(3.55, 0.18, n).round(2).clip(3.0, 4.1), "titratable_acidity_gL": (13 - brix * 0.25 + rng.normal(0, 0.7, n)).round(2).clip(3, 11), "disease": rng.choice(DISEASES, n, p=[0.28, 0.17, 0.15, 0.40]), "vine_age_years": rng.integers(3, 60, n), }) def phenology_df(n=260, seed=0): """Time series across a growing season: berry weight, sugar, acid accumulation.""" rng = np.random.default_rng(seed) dates = pd.date_range("2024-04-01", periods=n // 4 + 1, freq="W") reps = np.tile(dates, 4)[:n] # week index drives ripening: sugar up, acid down, veraison sigmoid wk = np.array([(d - dates[0]).days / 7 for d in reps]) span = max(wk.max(), 1) return pd.DataFrame({ "date": reps, "block": rng.choice(BLOCKS, n), "variety": rng.choice(VARIETIES, n), "berry_weight_g": (0.4 + 1.4 * wk / span + rng.normal(0, 0.15, n)).round(3).clip(0.2, 2.6), "brix": (6 + 18 * wk / span + rng.normal(0, 1.2, n)).round(1).clip(5, 28), "titratable_acidity_gL": (19 - 12 * wk / span + rng.normal(0, 1.0, n)).round(2).clip(3, 20), "veraison_pct": (100 / (1 + np.exp(-(wk - span / 2) * 0.8)) + rng.normal(0, 6, n)).round(0).clip(0, 100), }) def berry_chem_df(n=200, seed=0): """Lab chemistry samples: phenolics, anthocyanins, tannin by variety/block.""" rng = np.random.default_rng(seed) phenolics = rng.normal(55, 15, n).round(1).clip(10, 110) return pd.DataFrame({ "sample_id": np.arange(1, n + 1), "block": rng.choice(BLOCKS, n), "variety": rng.choice(VARIETIES, n), # anthocyanin and tannin are both phenolic fractions -> they co-vary "anthocyanin_mg_g": (phenolics * 0.022 + rng.normal(0, 0.22, n)).round(3).clip(0.1, 3.0), "total_phenolics_au": phenolics, "tannin_mg_g": (phenolics * 0.032 + rng.normal(0, 0.35, n)).round(2).clip(0.3, 4.5), "ph": rng.normal(3.55, 0.18, n).round(2).clip(3.0, 4.1), "brix": rng.normal(24.5, 2.2, n).round(1).clip(18, 30), }) def cellar_ferment_df(n=240, seed=0): """Daily tank readings through alcoholic fermentation.""" rng = np.random.default_rng(seed) day = rng.integers(0, 16, n) sugar = (230 * np.exp(-0.19 * day) + rng.normal(0, 6, n)).round(1).clip(0, 260) return pd.DataFrame({ "tank_id": rng.choice([f"T{i:02d}" for i in range(1, 19)], n), "yeast_strain": rng.choice(["EC-1118", "RC-212", "D254", "BM4x4", "Native"], n), "day_of_ferment": day, "residual_sugar_gL": sugar, # every gram of sugar consumed becomes roughly 1/17 % alcohol "alcohol_pct": ((230 - sugar) / 17.0 + rng.normal(0, 0.25, n)).round(2).clip(0, 16), "must_temp_c": rng.normal(26, 3.5, n).round(1).clip(12, 35), "cap_temp_c": rng.normal(29, 4.0, n).round(1).clip(14, 40), "free_so2_ppm": rng.normal(28, 9, n).round(0).clip(0, 60), "volatile_acidity_gL": rng.normal(0.42, 0.14, n).round(3).clip(0.05, 1.2), "punchdowns_per_day": rng.integers(0, 4, n), }) def soil_survey_df(n=180, seed=0): """Soil pit descriptions and lab results by depth horizon.""" rng = np.random.default_rng(seed) clay = rng.normal(28, 11, n).round(1).clip(3, 65) return pd.DataFrame({ "pit_id": rng.choice([f"P{i:03d}" for i in range(1, 41)], n), "soil_series": rng.choice(["Bale Loam", "Pleasanton", "Sobrante", "Hambright", "Yolo Silt"], n), "drainage_class": rng.choice(["Well drained", "Moderately well", "Somewhat poor", "Excessive"], n), "horizon_depth_cm": rng.choice([15, 30, 45, 60, 90, 120], n), "clay_pct": clay, "sand_pct": (90 - clay + rng.normal(0, 7, n)).round(1).clip(5, 92), "organic_matter_pct": rng.normal(2.1, 0.8, n).round(2).clip(0.2, 6.0), # clay holds cations -> CEC rises with clay fraction "cec_meq_100g": (clay * 0.42 + rng.normal(0, 2.2, n)).round(1).clip(2, 40), "soil_ph": rng.normal(6.4, 0.6, n).round(2).clip(4.5, 8.4), "available_water_mm_m": rng.normal(135, 35, n).round(0).clip(40, 240), }) def pest_scouting_df(n=280, seed=0): """Weekly scouting walks: pest pressure per row.""" rng = np.random.default_rng(seed) incidence = rng.gamma(2.0, 5.5, n).round(1).clip(0, 100) return pd.DataFrame({ "scout_date": np.tile(pd.date_range("2024-05-01", periods=n // 7 + 1, freq="W"), 7)[:n], "block": rng.choice(BLOCKS, n), "row_number": rng.integers(1, 61, n), "pest_name": rng.choice(["Vine Mealybug", "Leafhopper", "Spider Mite", "European Grapevine Moth", "Thrips"], n), "incidence_pct": incidence, "severity_index": (incidence / 25 + rng.normal(0, 0.4, n)).round(2).clip(0, 5), "trap_count": rng.poisson(6, n), "beneficials_count": rng.poisson(3, n), "threshold_exceeded": incidence > 20, "spray_applied": rng.random(n) < 0.3, }) def irrigation_sensor_df(n=300, seed=0): """Hourly-to-daily sensor telemetry for irrigation scheduling.""" rng = np.random.default_rng(seed) moisture = rng.normal(24, 6, n).round(2).clip(6, 42) return pd.DataFrame({ "timestamp": pd.date_range("2024-06-01", periods=n, freq="6h"), "sensor_id": rng.choice([f"S-{i:02d}" for i in range(1, 13)], n), "block": rng.choice(BLOCKS, n), "soil_moisture_vwc": moisture, # drier soil -> more negative (more stressed) stem water potential "stem_water_potential_bar": (-18 + moisture * 0.33 + rng.normal(0, 1.1, n)).round(2).clip(-20, -2), "canopy_temp_c": rng.normal(29, 5, n).round(1).clip(12, 46), "air_temp_c": rng.normal(26, 6, n).round(1).clip(8, 44), "relative_humidity_pct": rng.normal(52, 16, n).round(0).clip(8, 99), "et0_mm": rng.gamma(3, 1.6, n).round(2).clip(0.2, 14), "valve_open": rng.random(n) < 0.22, }) def sensory_panel_df(n=260, seed=0): """Blind tasting panel scores, one row per taster x wine.""" rng = np.random.default_rng(seed) fruit = rng.normal(6.4, 1.5, n).round(1).clip(1, 10) structure = rng.normal(6.0, 1.6, n).round(1).clip(1, 10) return pd.DataFrame({ "wine_code": rng.choice([f"W{i:03d}" for i in range(1, 25)], n), "taster_id": rng.choice([f"J{i:02d}" for i in range(1, 13)], n), "flight": rng.choice(["Flight A", "Flight B", "Flight C"], n), "aroma_intensity": rng.normal(6.1, 1.7, n).round(1).clip(1, 10), "fruit_score": fruit, "tannin_score": structure, "acidity_score": rng.normal(6.2, 1.4, n).round(1).clip(1, 10), "finish_seconds": rng.gamma(4, 3.2, n).round(0).clip(2, 60), # overall is mostly a blend of fruit and structure, plus taster noise "overall_rating": (fruit * 0.5 + structure * 0.4 + rng.normal(0, 0.6, n)).round(1).clip(1, 10), "would_purchase": rng.random(n) < 0.45, }) def harvest_logistics_df(n=240, seed=0): """Pick-day operations: crews, bins, transport to the crush pad.""" rng = np.random.default_rng(seed) crew = rng.integers(6, 25, n) return pd.DataFrame({ "pick_date": np.tile(pd.date_range("2024-08-20", periods=n // 6 + 1, freq="D"), 6)[:n], "block": rng.choice(BLOCKS, n), "crew_id": rng.choice(["Crew Alpha", "Crew Bravo", "Crew Charlie", "Crew Delta"], n), "crew_size": crew, "pick_method": rng.choice(["Hand", "Machine"], n, p=[0.72, 0.28]), # more pickers -> more bins, with diminishing returns and day-to-day noise "bins_filled": (crew * 1.9 + rng.normal(0, 4, n)).round(0).clip(2, 70), "kg_per_hour": rng.normal(410, 120, n).round(0).clip(80, 900), "transport_km": rng.normal(14, 7, n).round(1).clip(0.5, 45), "fruit_temp_c": rng.normal(17, 5, n).round(1).clip(4, 34), "wait_time_min": rng.gamma(2.5, 14, n).round(0).clip(0, 180), "mog_pct": rng.normal(2.4, 1.3, n).round(2).clip(0, 12), # material other than grapes }) def wine_sales_df(n=320, seed=0): """Bottle sales by channel and region.""" rng = np.random.default_rng(seed) price = rng.normal(38, 14, n).round(2).clip(9, 130) return pd.DataFrame({ "order_date": pd.date_range("2023-01-01", periods=n, freq="D"), "sku": rng.choice([f"SKU-{i:03d}" for i in range(1, 19)], n), "region": rng.choice(["Napa", "Sonoma", "Oregon", "Export EU", "Export Asia"], n), "channel": rng.choice(["Tasting Room", "Wine Club", "Distributor", "Online"], n), "unit_price_usd": price, # cheaper bottles move in larger volumes "bottles_sold": (rng.gamma(3, 22, n) * (60 / price)).round(0).clip(1, 900), "discount_pct": rng.choice([0, 5, 10, 15, 20, 25], n, p=[.4, .16, .16, .12, .1, .06]), "shipping_cost_usd": rng.normal(22, 9, n).round(2).clip(0, 80), "club_member": rng.random(n) < 0.38, }) def canopy_pruning_df(n=200, seed=0): """Dormant pruning and canopy architecture measurements per vine.""" rng = np.random.default_rng(seed) shoots = rng.integers(12, 60, n) return pd.DataFrame({ "vine_id": np.arange(1000, 1000 + n), "block": rng.choice(BLOCKS, n), "trellis_type": rng.choice(["VSP", "Lyre", "Head-trained", "Quadrilateral Cordon"], n), "pruning_method": rng.choice(["Spur", "Cane", "Minimal"], n), "shoot_count": shoots, "bud_count": (shoots * 1.4 + rng.normal(0, 3, n)).round(0).clip(8, 100), # leaf area scales with how many shoots the vine carries "leaf_area_m2": (shoots * 0.14 + rng.normal(0, 0.6, n)).round(2).clip(0.5, 12), "cane_weight_kg": rng.normal(0.85, 0.3, n).round(3).clip(0.1, 2.5), "internode_length_cm": rng.normal(7.5, 2.0, n).round(1).clip(2, 16), "cluster_count": rng.integers(8, 55, n), }) def barrel_aging_df(n=220, seed=0): """Barrel inventory and extraction chemistry during elevage.""" rng = np.random.default_rng(seed) months = rng.integers(0, 25, n) return pd.DataFrame({ "barrel_id": rng.choice([f"B{i:04d}" for i in range(1, 121)], n), "cooper": rng.choice(["Taransaud", "Seguin Moreau", "Francois Freres", "Nadalie", "World Cooperage"], n), "oak_origin": rng.choice(["French", "American", "Hungarian"], n), "toast_level": rng.choice(["Light", "Medium", "Medium Plus", "Heavy"], n), "months_in_barrel": months, "barrel_age_fills": rng.integers(1, 6, n), # oak compounds extract over time in barrel "vanillin_ppb": (months * 21 + rng.normal(0, 45, n)).round(0).clip(0, 700), "oak_lactone_ppb": (months * 14 + rng.normal(0, 38, n)).round(0).clip(0, 500), "color_intensity_au": rng.normal(11.5, 3.0, n).round(2).clip(2, 22), "topping_volume_l": rng.gamma(2, 0.9, n).round(2).clip(0, 9), }) GENERATORS = [ block_vintage_df, phenology_df, berry_chem_df, cellar_ferment_df, soil_survey_df, pest_scouting_df, irrigation_sensor_df, sensory_panel_df, harvest_logistics_df, wine_sales_df, canopy_pruning_df, barrel_aging_df, ] def df_preview(df, n=5): """The EXACT schema string the model sees at train AND inference time. Keep this the single source of truth -- any drift between train/infer formatting pushes the model out of distribution. """ dtypes = ", ".join(f"{c} ({df[c].dtype})" for c in df.columns) return f"Columns and dtypes:\n {dtypes}\nSample rows:\n{df.head(n).to_string(index=False)}"