{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":2564,"databundleVersionId":29456,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndf=pd.read_csv(\"/kaggle/input/DontGetKicked/training.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:26.532138Z","iopub.execute_input":"2025-03-22T13:38:26.532506Z","iopub.status.idle":"2025-03-22T13:38:27.444514Z","shell.execute_reply.started":"2025-03-22T13:38:26.532479Z","shell.execute_reply":"2025-03-22T13:38:27.443263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set 'RefId' as the index in df first\ndf = df.set_index('RefId')\n\n# Then define target and inputs\ntarget = df['IsBadBuy']\ninputs = df.drop(columns=['IsBadBuy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:27.445777Z","iopub.execute_input":"2025-03-22T13:38:27.446157Z","iopub.status.idle":"2025-03-22T13:38:27.483150Z","shell.execute_reply.started":"2025-03-22T13:38:27.446127Z","shell.execute_reply":"2025-03-22T13:38:27.481928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\n# split into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(inputs, target, test_size=0.20, random_state=1)\n\nX_train.shape,X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:27.485375Z","iopub.execute_input":"2025-03-22T13:38:27.485797Z","iopub.status.idle":"2025-03-22T13:38:28.234563Z","shell.execute_reply.started":"2025-03-22T13:38:27.485754Z","shell.execute_reply":"2025-03-22T13:38:28.233507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# test","metadata":{}},{"cell_type":"code","source":"# import numpy as np\n# def frequency_table(variable):\n#     unique_elements,counts=np.unique(variable.dropna(),return_counts=True)\n#     percentage=(counts/len(variable)*100)\n#     for i , j , k in zip(unique_elements,counts,percentage):\n#         print(f\"{i} :count{j} , percentage: {k:.2f}\")\n#     return\n\n# for col in categorical_fields:\n#     print(f\"Frequency Table for {col}:\")\n#     frequency_table(inputs[col])\n#     print(\"-\" * 40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.236045Z","iopub.execute_input":"2025-03-22T13:38:28.236590Z","iopub.status.idle":"2025-03-22T13:38:28.240512Z","shell.execute_reply.started":"2025-03-22T13:38:28.236557Z","shell.execute_reply":"2025-03-22T13:38:28.239411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef initial_preproc(data):\n    processed_data = data.copy()\n    # List of columns to drop\n    columns_to_drop = [\"PurchDate\",\"VehYear\",\"Model\",\"Trim\",\"SubModel\",\"WheelTypeID\",\"BYRNO\",\"VNZIP1\",\"VNST\", \"PRIMEUNIT\", \"AUCGUART\"]\n    \n    # Drop the columns\n    processed_data = processed_data.drop(columns=columns_to_drop)\n    # Replace 'NOT AVAIL' in the 'Color' column with NaN\n    processed_data['Color'] = processed_data['Color'].replace('NOT AVAIL', np.nan)\n    processed_data['Transmission'] = processed_data['Transmission'].replace(['Manual'], 'MANUAL')\n    \n    # Define a threshold for frequency (1% of total data)\n    threshold=1\n    for x in ('Make','Color'):\n        \n        unique_elements, counts = np.unique(processed_data[x].dropna(), return_counts=True)\n        percentage = (counts / len(processed_data[x])) * 100\n        rare_classes = [elem for elem, pct in zip(unique_elements, percentage) if pct < threshold]\n        processed_data[x] = processed_data[x].replace(rare_classes, 'OTHER')\n        # print(f\"\\nUpdated Frequency Table for {x}:\")\n        # frequency_table(processed_data[x])\n        \n    return processed_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.241632Z","iopub.execute_input":"2025-03-22T13:38:28.242023Z","iopub.status.idle":"2025-03-22T13:38:28.266614Z","shell.execute_reply.started":"2025-03-22T13:38:28.241979Z","shell.execute_reply":"2025-03-22T13:38:28.265488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = initial_preproc(X_train)\nX_test = initial_preproc(X_test)\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.267842Z","iopub.execute_input":"2025-03-22T13:38:28.268261Z","iopub.status.idle":"2025-03-22T13:38:28.558240Z","shell.execute_reply.started":"2025-03-22T13:38:28.268221Z","shell.execute_reply":"2025-03-22T13:38:28.557211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns = X_train.columns\ncategorical_fields = [\n    \"Auction\",\n    \"Make\",\n    \"Color\",\n    \"Transmission\",\n    \"WheelType\",\n    \"Nationality\",\n    \"Size\",\n    \"TopThreeAmericanName\",\n    \"IsOnlineSale\"\n]\ncontinuous_fields = [col for col in X_train.columns if col not in categorical_fields]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.559157Z","iopub.execute_input":"2025-03-22T13:38:28.559416Z","iopub.status.idle":"2025-03-22T13:38:28.564669Z","shell.execute_reply.started":"2025-03-22T13:38:28.559394Z","shell.execute_reply":"2025-03-22T13:38:28.563633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Handle Out-of-Range","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndef range_consistency(data, target):\n    # Define ranges for each column\n    column_ranges = {\n    'VehicleAge': (0,30),\n    'VehOdo': (0,120000),\n    'MMRAcquisitionAuctionAveragePrice': (800,46000),\n    'MMRAcquisitionAuctionCleanPrice': (1000,46000),\n    'MMRAcquisitionRetailAveragePrice': (1000,46000),\n    'MMRAcquisitonRetailCleanPrice': (1000,46000),\n    'MMRCurrentAuctionAveragePrice': (300,46000),\n    'MMRCurrentAuctionCleanPrice': (400,46000),\n    'MMRCurrentRetailAveragePrice': (800,46000),\n    'MMRCurrentRetailCleanPrice': (1000,46000),\n    'VehBCost': (1000,46000),\n    'WarrantyCost': (400,8000)\n    }\n\n    # Iterate through each column and fill NaN values outside the defined range\n    for column, (min_val, max_val) in column_ranges.items():\n        data[column] = data[column].apply(lambda x: x if min_val <= x <= max_val else None)\n\n    target = target.replace([':0', \"'0'\"], '0')\n        \n    return data, target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.567985Z","iopub.execute_input":"2025-03-22T13:38:28.568306Z","iopub.status.idle":"2025-03-22T13:38:28.586090Z","shell.execute_reply.started":"2025-03-22T13:38:28.568274Z","shell.execute_reply":"2025-03-22T13:38:28.584995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = range_consistency(X_train, y_train)[0]\nX_test = range_consistency(X_test, y_test)[0]\n\ny_train = range_consistency(X_train, y_train)[1]\ny_test = range_consistency(X_test, y_test)[1]\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:28.588222Z","iopub.execute_input":"2025-03-22T13:38:28.588612Z","iopub.status.idle":"2025-03-22T13:38:29.226209Z","shell.execute_reply.started":"2025-03-22T13:38:28.588575Z","shell.execute_reply":"2025-03-22T13:38:29.225175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# feature screening\n","metadata":{}},{"cell_type":"code","source":"def feature_screening(data, min_cv=0.1, mode_threshold=99, distinct_threshold=90):\n    processed_data = data.copy()\n   # Define a minimum value for coefficient of variation\n    min_cv = min_cv\n\n    # Calculate the coefficient of variation for each column\n    cv_values = processed_data[continuous_fields].std() / processed_data[continuous_fields].mean()\n\n    # Filter out columns with CV less than 0.1\n    screen_cv =  cv_values[cv_values < min_cv].index.tolist()\n\n\n    # Define a threshold for the dominant category percentage\n    mode_threshold = mode_threshold\n\n    # Calculate the percentage of the mode category for each column\n    mode_category = (processed_data[categorical_fields].apply(lambda x: x.value_counts().max() / len(x)) * 100)\n\n    # Select columns where the mode category percentage is greater than the threshold\n    screen_mode = mode_category[mode_category > mode_threshold].index.tolist()\n\n\n    # Set a threshold for excluding columns \n    distinct_threshold = distinct_threshold\n\n    # Calculate the percentage of distinct categories in categorical variables\n    distinct_percentage = (processed_data[categorical_fields].apply(lambda x: x.dropna().nunique() / x.count()) * 100)\n\n    # Select categorical columns based on distinct percentage threshold\n    screen_distinct = distinct_percentage[distinct_percentage > distinct_threshold].index.tolist()\n\n    screened_features  = list(set(screen_cv + screen_mode + screen_distinct))\n     \n    return screened_features ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:29.227118Z","iopub.execute_input":"2025-03-22T13:38:29.227430Z","iopub.status.idle":"2025-03-22T13:38:29.234565Z","shell.execute_reply.started":"2025-03-22T13:38:29.227405Z","shell.execute_reply":"2025-03-22T13:38:29.233151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drop_list = feature_screening(X_train, min_cv=0.1, mode_threshold=99, distinct_threshold=90)\n\nX_train = X_train.drop(drop_list, axis=1)\nX_test = X_test.drop(drop_list, axis=1)\n\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:29.235627Z","iopub.execute_input":"2025-03-22T13:38:29.235976Z","iopub.status.idle":"2025-03-22T13:38:29.435536Z","shell.execute_reply.started":"2025-03-22T13:38:29.235941Z","shell.execute_reply":"2025-03-22T13:38:29.434480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# outliers","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\n\ndef outlier_handling_replace(data, contamination=0.01):\n    inputs_iso = data.copy()\n    # Discard rows with NaN values\n    inputs_iso = inputs_iso.dropna()\n    \n    # Apply Z-score scaling to numerical columns\n    scaler = StandardScaler()\n    inputs_iso[continuous_fields] = scaler.fit_transform(inputs_iso[continuous_fields])\n    \n    # Apply label encoding to categorical columns\n    label_encoder = LabelEncoder()\n    inputs_iso[categorical_fields] = inputs_iso[categorical_fields].apply(label_encoder.fit_transform)\n    \n    # Fit Isolation Forest model\n    clf = IsolationForest(contamination=contamination, random_state=42)\n    clf.fit(inputs_iso)\n    \n    # Predict outliers\n    outliers = clf.predict(inputs_iso)\n    \n    # Add the outlier predictions to your DataFrame\n    inputs_iso['outlier'] = outliers\n    \n    # Identify outliers\n    outlier_index = inputs_iso[inputs_iso['outlier'] == -1].index   \n    return outlier_index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:29.436681Z","iopub.execute_input":"2025-03-22T13:38:29.437101Z","iopub.status.idle":"2025-03-22T13:38:29.866786Z","shell.execute_reply.started":"2025-03-22T13:38:29.437010Z","shell.execute_reply":"2025-03-22T13:38:29.865698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outlier_index = outlier_handling_replace(X_train, contamination=0.01)\n\nX_train = X_train.drop(outlier_index.tolist())\n\ny_train = y_train.drop(outlier_index.tolist())\nX_train.shape, y_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:29.867870Z","iopub.execute_input":"2025-03-22T13:38:29.868231Z","iopub.status.idle":"2025-03-22T13:38:33.775429Z","shell.execute_reply.started":"2025-03-22T13:38:29.868195Z","shell.execute_reply":"2025-03-22T13:38:33.773974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# missing values","metadata":{}},{"cell_type":"code","source":"columns_to_check=[\"MMRAcquisitionAuctionAveragePrice\",\n                  \"MMRAcquisitionAuctionCleanPrice\",\n                  \"MMRAcquisitionRetailAveragePrice\",\n                  \"MMRAcquisitonRetailCleanPrice\",\n                  \"MMRCurrentAuctionAveragePrice\",\n                  \"MMRCurrentAuctionCleanPrice\",\n                  \"MMRCurrentRetailAveragePrice\",\n                  \"MMRCurrentRetailCleanPrice\"]\nX_train[\"num_missing_values\"]=X_train[columns_to_check].isnull().sum(axis=1)\nvalid_indices = X_train[X_train[\"num_missing_values\"] < 4].index\nX_train = X_train.loc[valid_indices].drop(columns=[\"num_missing_values\"])\ny_train = y_train.loc[valid_indices]\nX_train.shape, y_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:33.776540Z","iopub.execute_input":"2025-03-22T13:38:33.776952Z","iopub.status.idle":"2025-03-22T13:38:33.832858Z","shell.execute_reply.started":"2025-03-22T13:38:33.776913Z","shell.execute_reply":"2025-03-22T13:38:33.831688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Row cleaning","metadata":{}},{"cell_type":"code","source":"def missing_row_report(data, missrow=12):\n    processed_data = data.copy()\n    \n\n    # Create a new column with the number of missing values in each row\n    processed_data['Num_Missing_Values'] = processed_data.isnull().sum(axis=1)\n\n    discard_missing_row = processed_data[processed_data['Num_Missing_Values'] > missrow].index.tolist()\n\n    return discard_missing_row","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:33.833890Z","iopub.execute_input":"2025-03-22T13:38:33.834185Z","iopub.status.idle":"2025-03-22T13:38:33.839184Z","shell.execute_reply.started":"2025-03-22T13:38:33.834162Z","shell.execute_reply":"2025-03-22T13:38:33.838206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"discard_missing_row = missing_row_report(X_train, missrow=35)\n\nX_train = X_train.drop(discard_missing_row)\ny_train = y_train.drop(discard_missing_row)\n\nX_train.shape, y_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:33.840438Z","iopub.execute_input":"2025-03-22T13:38:33.840840Z","iopub.status.idle":"2025-03-22T13:38:33.932756Z","shell.execute_reply.started":"2025-03-22T13:38:33.840805Z","shell.execute_reply":"2025-03-22T13:38:33.931740Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## column cleaning","metadata":{}},{"cell_type":"code","source":"def missing_col_report(data, misscol=50):\n    processed_data = data.copy()\n    # Report on count and percentage of missing values in each column\n    missing_values_report = pd.DataFrame({\n        'Column': processed_data.columns,\n        'Missing Values': processed_data.isnull().sum(),\n        'Percentage Missing': processed_data.isnull().mean() * 100\n        })\n    discard_missing_col = missing_values_report[missing_values_report['Percentage Missing'] > misscol].index.tolist()\n    return discard_missing_col","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:33.933523Z","iopub.execute_input":"2025-03-22T13:38:33.933791Z","iopub.status.idle":"2025-03-22T13:38:33.939342Z","shell.execute_reply.started":"2025-03-22T13:38:33.933769Z","shell.execute_reply":"2025-03-22T13:38:33.938196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"discard_missing_col = missing_col_report(X_train, misscol=50)\n\nX_train = X_train.drop(discard_missing_col, axis=1)\nX_test = X_test.drop(discard_missing_col, axis=1)\n\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:33.940409Z","iopub.execute_input":"2025-03-22T13:38:33.940821Z","iopub.status.idle":"2025-03-22T13:38:34.046593Z","shell.execute_reply.started":"2025-03-22T13:38:33.940781Z","shell.execute_reply":"2025-03-22T13:38:34.045584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## impute missing values","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import  SimpleImputer\n\ndef missing_imputer(train, test):\n    \n    continuous = train.select_dtypes(exclude=['object','category']).columns.tolist()\n    categorical = train.select_dtypes(include=['object','category']).columns.tolist()\n\n    # Define imputation strategies for each subset of columns\n    cat_imputer = SimpleImputer(strategy='most_frequent')\n    cont_imputer = SimpleImputer(strategy='median')\n    \n    try:\n\n    # Impute missing values\n        train[continuous] = cont_imputer.fit_transform(train[continuous])\n        train[categorical] = cat_imputer.fit_transform(train[categorical])\n    \n        test[continuous] = cont_imputer.transform(test[continuous])\n        test[categorical] = cat_imputer.transform(test[categorical])\n\n    except:\n        test[continuous] = cont_imputer.transform(test[continuous])\n        test[categorical] = cat_imputer.transform(test[categorical])\n        \n    return train, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:34.047591Z","iopub.execute_input":"2025-03-22T13:38:34.047943Z","iopub.status.idle":"2025-03-22T13:38:34.070875Z","shell.execute_reply.started":"2025-03-22T13:38:34.047918Z","shell.execute_reply":"2025-03-22T13:38:34.069294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test = missing_imputer(X_train, X_test)\n\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:34.072279Z","iopub.execute_input":"2025-03-22T13:38:34.072639Z","iopub.status.idle":"2025-03-22T13:38:34.371551Z","shell.execute_reply.started":"2025-03-22T13:38:34.072609Z","shell.execute_reply":"2025-03-22T13:38:34.370436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Discretize Features","metadata":{}},{"cell_type":"code","source":"pip install scorecardbundle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:34.372494Z","iopub.execute_input":"2025-03-22T13:38:34.372800Z","iopub.status.idle":"2025-03-22T13:38:40.481990Z","shell.execute_reply.started":"2025-03-22T13:38:34.372774Z","shell.execute_reply":"2025-03-22T13:38:40.480722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom scorecardbundle.feature_discretization import ChiMerge as cm\n\nchi_merge_list = ['VehBCost', 'WarrantyCost']\n\ndef discretizer(train, test, y, chi_list):\n\n    trans_cm = cm.ChiMerge(max_intervals=5, min_intervals=1, decimal=3,output_dataframe=True)\n    trans_cm.fit(train[chi_list], y.astype('int').squeeze()) \n\n    # Add -inf to the beginning of each array\n    boundaries_dict = {key: np.insert(boundaries, 0, -np.inf) for key, boundaries in trans_cm.boundaries_.items()}\n\n    # Iterate through the dictionary and add new columns to data\n    for key, boundaries in boundaries_dict.items():\n        column_name = f\"{key}_cat_cm\"\n        train[column_name] = pd.cut(train[key], bins=boundaries, labels=False, right=False)\n        test[column_name] = pd.cut(test[key], bins=boundaries, labels=False, right=False)\n        \n    return train, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:40.483180Z","iopub.execute_input":"2025-03-22T13:38:40.483490Z","iopub.status.idle":"2025-03-22T13:38:40.498661Z","shell.execute_reply.started":"2025-03-22T13:38:40.483464Z","shell.execute_reply":"2025-03-22T13:38:40.497648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test = discretizer(X_train, X_test, y_train, chi_merge_list)\n\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:38:40.503439Z","iopub.execute_input":"2025-03-22T13:38:40.503821Z","iopub.status.idle":"2025-03-22T13:39:54.217230Z","shell.execute_reply.started":"2025-03-22T13:38:40.503790Z","shell.execute_reply":"2025-03-22T13:39:54.215819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# transformation using the Box-Cox","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\n# List of features to transform\nselected_features =  ['VehBCost', 'WarrantyCost']\n\ndef transform(train, test, trans_list):\n    # Iterate through selected features\n    for feature in trans_list:\n        # Check if the feature contains negative values\n        has_negative_values = (train[feature] <= 0).any()\n\n        # Choose the appropriate transformation method\n        if has_negative_values:\n            transformer = PowerTransformer(method='yeo-johnson', standardize=False)\n        else:\n            transformer = PowerTransformer(method='box-cox', standardize=False)\n\n        # Fit and transform the feature, and store the result in the new DataFrame\n        train[f\"{feature}_transformed\"] = transformer.fit_transform(train[[feature]])\n        test[f\"{feature}_transformed\"] = transformer.transform(test[[feature]])\n        \n    return train, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:39:54.218970Z","iopub.execute_input":"2025-03-22T13:39:54.219371Z","iopub.status.idle":"2025-03-22T13:39:54.226689Z","shell.execute_reply.started":"2025-03-22T13:39:54.219319Z","shell.execute_reply":"2025-03-22T13:39:54.225302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test = transform(X_train, X_test, selected_features)\n\nX_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:39:54.227997Z","iopub.execute_input":"2025-03-22T13:39:54.228445Z","iopub.status.idle":"2025-03-22T13:39:54.878670Z","shell.execute_reply.started":"2025-03-22T13:39:54.228396Z","shell.execute_reply":"2025-03-22T13:39:54.877802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"code","source":"skewed_list=['VehBCost','WarrantyCost']\ntransformed_list=['VehBCost_transformed','WarrantyCost_transformed']\ndiscretized_list=['VehBCost_cat_cm','WarrantyCost_cat_cm']\nX_train['IsOnlineSale']=X_train['IsOnlineSale'].astype('int')\ncategorical = X_train.select_dtypes(include=['object','category']).columns.tolist()\ncontinuous = [i for i in X_train.columns if i not in skewed_list + transformed_list + discretized_list + categorical]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:39:54.879572Z","iopub.execute_input":"2025-03-22T13:39:54.879892Z","iopub.status.idle":"2025-03-22T13:39:54.895177Z","shell.execute_reply.started":"2025-03-22T13:39:54.879868Z","shell.execute_reply":"2025-03-22T13:39:54.894125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\n\nfrom sklearn.preprocessing import  OneHotEncoder, OrdinalEncoder, StandardScaler, MinMaxScaler\n\nfrom sklearn.feature_selection import SelectKBest, f_regression, mutual_info_regression, f_classif, mutual_info_classif, RFECV\nfrom sklearn.decomposition import PCA, KernelPCA\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.tree import DecisionTreeClassifier\n\none_hot_encoder = OneHotEncoder(drop='first', handle_unknown='ignore', sparse_output=False)\n\nz_score = StandardScaler()\nmin_max = MinMaxScaler()\nwrapper = RFECV(estimator=DecisionTreeClassifier(random_state=29), step=1, min_features_to_select=10, cv=5, n_jobs=-1)\npca = PCA(n_components=2, random_state=717)\nlda = LinearDiscriminantAnalysis(n_components=1)\nkpca = KernelPCA(n_components=3, kernel='rbf', random_state=717)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:39:54.896115Z","iopub.execute_input":"2025-03-22T13:39:54.896495Z","iopub.status.idle":"2025-03-22T13:39:54.985606Z","shell.execute_reply.started":"2025-03-22T13:39:54.896466Z","shell.execute_reply":"2025-03-22T13:39:54.984592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the preprocessing steps for numerical and categorical features separately\n\nnumerical_preprocessing_1 = Pipeline(steps=[\n    ('scaler', min_max)])  \n    \nnominal_preprocessing_1 = Pipeline(steps=[\n    ('nominal', one_hot_encoder),  \n    ('scaler', min_max)]) \n\n\n# Define the ColumnTransformer for numerical and categorical features\npreprocessor_1 = ColumnTransformer(transformers=[\n    ('num', numerical_preprocessing_1, discretized_list+continuous),\n    ('nom', nominal_preprocessing_1, categorical),\n]) \n\n\npipeline_1 = Pipeline(steps=[\n    ('preprocessor', preprocessor_1),\n    ('wrapper', wrapper),\n    ('model',DecisionTreeClassifier(random_state=17))])\n\n\n\n# Train the pipeline\npipe_1 = pipeline_1.fit(X_train, y_train)\nprint(f\"Optimal number of features: {wrapper.n_features_}\")\npipe_1[:-1].get_feature_names_out().tolist()\n\n# Use the pipeline for prediction or other tasks\npredictions_1 = pipe_1.predict(X_test)\n\nfrom sklearn.metrics import accuracy_score, classification_report\naccuracy_discrete_wrapper = accuracy_score(y_test, predictions_1)\nprint(\"Accuracy:\", accuracy_discrete_wrapper)\n\n# نمایش گزارش طبقه‌بندی (Precision, Recall, F1-Score)\nprint(\"\\nClassification Report:\\n\", classification_report(y_test, predictions_1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:39:54.986648Z","iopub.execute_input":"2025-03-22T13:39:54.986983Z","iopub.status.idle":"2025-03-22T13:42:56.249055Z","shell.execute_reply.started":"2025-03-22T13:39:54.986956Z","shell.execute_reply":"2025-03-22T13:42:56.247893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the preprocessing steps for numerical and categorical features separately\n\nnumerical_preprocessing_2 = Pipeline(steps=[\n    ('scaler', z_score),\n    ('pca',pca)])  \n    \nnominal_preprocessing_2 = Pipeline(steps=[\n    ('nominal', one_hot_encoder),  \n    ('scaler', z_score)]) \n\n\n# Define the ColumnTransformer for numerical and categorical features\npreprocessor_2 = ColumnTransformer(transformers=[\n    ('num', numerical_preprocessing_2, transformed_list+continuous),\n    ('nom', nominal_preprocessing_2, categorical),\n]) \n\n\npipeline_2 = Pipeline(steps=[\n    ('preprocessor', preprocessor_2),\n    ('wrapper', wrapper),\n    ('model',DecisionTreeClassifier(random_state=17))])\n\n\n\n# Train the pipeline\npipe_2 = pipeline_2.fit(X_train, y_train)\nprint(f\"Optimal number of features: {wrapper.n_features_}\")\npipe_2[:-1].get_feature_names_out().tolist()\n\n# Use the pipeline for prediction or other tasks\npredictions_2 = pipe_2.predict(X_test)\n\nfrom sklearn.metrics import accuracy_score, classification_report\naccuracy_transformed_pca_wrapper = accuracy_score(y_test, predictions_2)\nprint(\"Accuracy:\", accuracy_transformed_pca_wrapper)\n\n# نمایش گزارش طبقه‌بندی (Precision, Recall, F1-Score)\nprint(\"\\nClassification Report:\\n\", classification_report(y_test, predictions_2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:42:56.250296Z","iopub.execute_input":"2025-03-22T13:42:56.250728Z","iopub.status.idle":"2025-03-22T13:43:53.446474Z","shell.execute_reply.started":"2025-03-22T13:42:56.250667Z","shell.execute_reply":"2025-03-22T13:43:53.445270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the preprocessing steps for numerical and categorical features separately\n\nnumerical_preprocessing_3 = Pipeline(steps=[\n    ('scaler', z_score),\n    ('lda',lda)])  \n    \nnominal_preprocessing_3 = Pipeline(steps=[\n    ('nominal', one_hot_encoder),  \n    ('scaler', z_score)]) \n\n\n# Define the ColumnTransformer for numerical and categorical features\npreprocessor_3 = ColumnTransformer(transformers=[\n    ('num', numerical_preprocessing_3, transformed_list+continuous),\n    ('nom', nominal_preprocessing_3, categorical),\n]) \n\n\npipeline_3 = Pipeline(steps=[\n    ('preprocessor', preprocessor_3),\n    ('wrapper', wrapper),\n    ('model',DecisionTreeClassifier(random_state=17))])\n\n\n\n# Train the pipeline\npipe_3 = pipeline_3.fit(X_train, y_train)\nprint(f\"Optimal number of features: {wrapper.n_features_}\")\npipe_3[:-1].get_feature_names_out().tolist()\n\n# Use the pipeline for prediction or other tasks\npredictions_3 = pipe_3.predict(X_test)\n\nfrom sklearn.metrics import accuracy_score, classification_report\naccuracy_transformed_lda_wrapper = accuracy_score(y_test, predictions_3)\nprint(\"Accuracy:\", accuracy_transformed_lda_wrapper)\n\n# نمایش گزارش طبقه‌بندی (Precision, Recall, F1-Score)\nprint(\"\\nClassification Report:\\n\", classification_report(y_test, predictions_3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-22T13:44:19.551574Z","iopub.execute_input":"2025-03-22T13:44:19.552015Z","iopub.status.idle":"2025-03-22T13:44:59.014259Z","shell.execute_reply.started":"2025-03-22T13:44:19.551982Z","shell.execute_reply":"2025-03-22T13:44:59.012939Z"}},"outputs":[],"execution_count":null}]}