import pandas as pd from ml_engine import MLContext # MLContext class used to train models and generate pipeline code def test_classification_train_and_code(): # tests classification training and generated pipeline code df = pd.DataFrame( # create a sample dataset for the classification test { "age": [21, 42, 33, 51, 19, 45, 37, 62, 28, 55, 31, 48, 26, 39, 57, 23, 44, 35, 52, 29, 41, 60, 24, 46], "segment": ["a", "b"] * 12, "income": ["low", "high"] * 12, } ) ctx = MLContext(df, "unit.csv") # initialize the ML context with the test dataset and source filename result = ctx.train_candidate("income", "Logistic Regression", "classification", test_size=0.25) # train a logistic regression classifier assert result["status"] == "trained" # verify that the model was trained successfully assert result["problem_type"] == "classification" # verify that the detected problem type is classification assert "accuracy" in result["metrics"] # verify that accuracy is included in the training metrics code = ctx.generate_pipeline_code("income", "Logistic Regression", "classification") # Generate reusable classification pipeline code assert "ColumnTransformer" in code # verify that preprocessing uses a ColumnTransformer assert "LogisticRegression" in code # verify that the generated pipeline uses Logistic Regression