{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier, HistGradientBoostingClassifier\nfrom sklearn.metrics import log_loss, multilabel_confusion_matrix, roc_curve\n\nimport sys\nimport copy\nimport time\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-10-21T01:02:00.512918Z","iopub.execute_input":"2022-10-21T01:02:00.513728Z","iopub.status.idle":"2022-10-21T01:02:02.280380Z","shell.execute_reply.started":"2022-10-21T01:02:00.513617Z","shell.execute_reply":"2022-10-21T01:02:02.279083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMPORT_COLS = [\n    'ball_pos_x', 'ball_pos_y', 'ball_pos_z', 'ball_vel_x',\n    'ball_vel_y', 'ball_vel_z', 'p0_pos_x', 'p0_pos_y', 'p0_pos_z',\n    'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost', 'p1_pos_x', 'p1_pos_y',\n    'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z', 'p1_boost', 'p2_pos_x',\n    'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z', 'p2_boost',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z',\n    'p3_boost', 'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y',\n    'p4_vel_z', 'p4_boost', 'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x',\n    'p5_vel_y', 'p5_vel_z', 'p5_boost', 'boost0_timer', 'boost1_timer',\n    'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer',\n    'team_A_scoring_within_10sec','team_B_scoring_within_10sec'\n]\n\nDATA_COLS = [\n    'ball_pos_x', 'ball_pos_y', 'ball_pos_z', 'ball_vel_x',\n    'ball_vel_y', 'ball_vel_z', 'p0_pos_x', 'p0_pos_y', 'p0_pos_z',\n    'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost', 'p1_pos_x', 'p1_pos_y',\n    'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z', 'p1_boost', 'p2_pos_x',\n    'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z', 'p2_boost',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z',\n    'p3_boost', 'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y',\n    'p4_vel_z', 'p4_boost', 'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x',\n    'p5_vel_y', 'p5_vel_z', 'p5_boost', 'boost0_timer', 'boost1_timer',\n    'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer'\n]\n\nLABEL_COLS = [\n    \"team_A_scoring_within_10sec\",\n    \"team_B_scoring_within_10sec\"\n]\n\nRANDOM_SEED = 49","metadata":{"execution":{"iopub.status.busy":"2022-10-21T01:02:02.282304Z","iopub.execute_input":"2022-10-21T01:02:02.282698Z","iopub.status.idle":"2022-10-21T01:02:02.295609Z","shell.execute_reply.started":"2022-10-21T01:02:02.282665Z","shell.execute_reply":"2022-10-21T01:02:02.292920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def import_csv(csv):\n    \n    # Import csv\n    train_set = pd.read_csv(f\"../input/tabular-playground-series-oct-2022/train_{csv}.csv\",usecols=IMPORT_COLS,dtype=\"float32\")\n    train_set = train_set.dropna()\n#     train_set = train_set.sample(frac=0.1)    # To be deleted once all is verified\n    \n    # Change labels to integer\n    train_set[[\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"]] = train_set[[\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"]].astype(\"int8\")\n    \n    return train_set","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.204014Z","iopub.execute_input":"2022-10-19T20:25:15.204406Z","iopub.status.idle":"2022-10-19T20:25:15.226313Z","shell.execute_reply.started":"2022-10-19T20:25:15.204370Z","shell.execute_reply":"2022-10-19T20:25:15.224216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_weights(df):\n    weights = (df[\"team_A_scoring_within_10sec\"] + df[\"team_B_scoring_within_10sec\"]).value_counts()/len(df)\n    return weights","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.229507Z","iopub.execute_input":"2022-10-19T20:25:15.229975Z","iopub.status.idle":"2022-10-19T20:25:15.243972Z","shell.execute_reply.started":"2022-10-19T20:25:15.229934Z","shell.execute_reply":"2022-10-19T20:25:15.242299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_data(df):\n    # Create data and labels\n    data = df.drop(LABEL_COLS,axis=1)\n    label = df[LABEL_COLS]\n    \n    return data, label","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.246343Z","iopub.execute_input":"2022-10-19T20:25:15.246741Z","iopub.status.idle":"2022-10-19T20:25:15.256315Z","shell.execute_reply.started":"2022-10-19T20:25:15.246707Z","shell.execute_reply":"2022-10-19T20:25:15.254963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_data(scaler):\n    # Import test set\n    test = pd.read_csv(\"../input/tabular-playground-series-oct-2022/test.csv\",dtype=\"float32\")\n     \n    \n    # Scale data\n    test_scaled = scaler.transform(test[DATA_COLS])\n    \n    # Convert infinite/nan number to small number\n    test_scaled = np.nan_to_num(test_scaled)\n    \n    return test, test_scaled","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.257859Z","iopub.execute_input":"2022-10-19T20:25:15.258225Z","iopub.status.idle":"2022-10-19T20:25:15.270925Z","shell.execute_reply.started":"2022-10-19T20:25:15.258193Z","shell.execute_reply":"2022-10-19T20:25:15.269202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prep_data(data,label):\n    \n    # Scale data\n    scaler = StandardScaler()\n    X = scaler.fit_transform(data)\n    y = label\n    \n    # Train test split\n    X_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.33, random_state=RANDOM_SEED, stratify=y)\n    \n    del data\n    \n    return scaler, X_train, X_dev, y_train, y_dev, label","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.272834Z","iopub.execute_input":"2022-10-19T20:25:15.273751Z","iopub.status.idle":"2022-10-19T20:25:15.285106Z","shell.execute_reply.started":"2022-10-19T20:25:15.273701Z","shell.execute_reply":"2022-10-19T20:25:15.283986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Loop over csv's\nfor csv in range(10):\n    # Import csv\n    train_set = import_csv(csv)\n    print(f\"Train_{csv} imported\")\n    \n    # Create weights to compensate imbalanced dataset\n    weights = create_weights(train_set)\n    print(f\"Class Weights: {dict(weights)}\")\n    data, label = train_data(train_set)\n    \n    # Scale date\n    scaler, X_train, X_dev, y_train, y_dev, label = prep_data(data, label)\n    print(f\"Data has been scaled\")\n    classes=[np.unique(label)]\n    \n    # Fit model\n    estimator = HistGradientBoostingClassifier(loss='binary_crossentropy',max_depth=15,max_iter=100,warm_start=True,random_state=RANDOM_SEED)\n#     ExtraTreesClassifier(warm_start=True, class_weight=dict(weights),random_state=RANDOM_SEED)\n#     LogisticRegression(penalty=\"elasticnet\",l1_ratio=0.5,solver=\"saga\",warm_start=True,class_weight=[dict(weights),dict(weights)],random_state=RANDOM_SEED)\n#     RandomForestClassifier(n_estimators=50,warm_start=True,class_weight=dict(weights),random_state=RANDOM_SEED))\n    model = MultiOutputClassifier(estimator=estimator)\n    model.classes_ = 2\n    print(\"Fit model.....\")\n    model.fit(X_train, y_train)\n    y_pred = model.predict_proba(X_dev)\n    logloss = log_loss(y_dev[\"team_A_scoring_within_10sec\"],y_pred[0]) + log_loss(y_dev[\"team_B_scoring_within_10sec\"],y_pred[1])\n    print(f\"Log_loss: {logloss}\")\n    msg = f\"Train_{csv} completed\\n\\n\"\n    sys.stdout.write(\"\\r\"+msg)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:25:15.286935Z","iopub.execute_input":"2022-10-19T20:25:15.287810Z","iopub.status.idle":"2022-10-19T20:38:08.785386Z","shell.execute_reply.started":"2022-10-19T20:25:15.287754Z","shell.execute_reply":"2022-10-19T20:38:08.783599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import and process the test set\ntest, test_scaled = test_data(scaler)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:08.787946Z","iopub.execute_input":"2022-10-19T20:38:08.788395Z","iopub.status.idle":"2022-10-19T20:38:19.087400Z","shell.execute_reply.started":"2022-10-19T20:38:08.788328Z","shell.execute_reply":"2022-10-19T20:38:19.086215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run the test data through the model\nresults = model.predict_proba(test_scaled)\n# results = np.nan_to_num(results)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:19.092377Z","iopub.execute_input":"2022-10-19T20:38:19.092742Z","iopub.status.idle":"2022-10-19T20:38:19.696647Z","shell.execute_reply.started":"2022-10-19T20:38:19.092709Z","shell.execute_reply":"2022-10-19T20:38:19.694596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create output submission file\noutput = pd.DataFrame({\n    'id': np.nan_to_num(test[\"id\"]).astype(\"int\"),\n    'team_A_scoring_within_10sec': results[0][:,1],\n    'team_B_scoring_within_10sec': results[1][:,1]\n})","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:19.698535Z","iopub.execute_input":"2022-10-19T20:38:19.699500Z","iopub.status.idle":"2022-10-19T20:38:19.737280Z","shell.execute_reply.started":"2022-10-19T20:38:19.699427Z","shell.execute_reply":"2022-10-19T20:38:19.727973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:19.745139Z","iopub.execute_input":"2022-10-19T20:38:19.746370Z","iopub.status.idle":"2022-10-19T20:38:19.804768Z","shell.execute_reply.started":"2022-10-19T20:38:19.746322Z","shell.execute_reply":"2022-10-19T20:38:19.803441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(x=output[\"team_A_scoring_within_10sec\"], data=output)\nsns.kdeplot(x=output[\"team_B_scoring_within_10sec\"], data=output)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:19.805920Z","iopub.execute_input":"2022-10-19T20:38:19.806587Z","iopub.status.idle":"2022-10-19T20:38:25.274750Z","shell.execute_reply.started":"2022-10-19T20:38:19.806556Z","shell.execute_reply":"2022-10-19T20:38:25.273567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-19T20:38:25.275937Z","iopub.execute_input":"2022-10-19T20:38:25.276277Z","iopub.status.idle":"2022-10-19T20:38:27.909403Z","shell.execute_reply.started":"2022-10-19T20:38:25.276230Z","shell.execute_reply":"2022-10-19T20:38:27.908244Z"},"trusted":true},"execution_count":null,"outputs":[]}]}