{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Input data using dask to create partitions","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n#Load the dictionary (For a look only)\ndata_dic = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")\n#Load train data\ndata = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:46.580884Z","iopub.execute_input":"2024-12-03T13:41:46.581344Z","iopub.status.idle":"2024-12-03T13:41:47.677273Z","shell.execute_reply.started":"2024-12-03T13:41:46.581292Z","shell.execute_reply":"2024-12-03T13:41:47.676284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import dask.dataframe as dd\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Explain variable if needed\ndata_dic.loc[data_dic[\"Field\"] == \"PCIAT-PCIAT_18\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:47.679080Z","iopub.execute_input":"2024-12-03T13:41:47.679434Z","iopub.status.idle":"2024-12-03T13:41:47.689791Z","shell.execute_reply.started":"2024-12-03T13:41:47.679396Z","shell.execute_reply":"2024-12-03T13:41:47.688971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load test data and remove unmatch colums with Dtrain\ndata_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n# Retain \"sii\" column and matched columns\ncolumns_to_keep = set(data_test.columns).intersection(data.columns)\ncolumns_to_keep.add(\"sii\")  # Ensure \"sii\" is kept\n\n# Drop unmatched columns in Dtrain\ndata = data[list(columns_to_keep)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:47.690756Z","iopub.execute_input":"2024-12-03T13:41:47.691003Z","iopub.status.idle":"2024-12-03T13:41:47.710566Z","shell.execute_reply.started":"2024-12-03T13:41:47.690979Z","shell.execute_reply":"2024-12-03T13:41:47.709808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Tranformation","metadata":{}},{"cell_type":"code","source":"data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:47.728673Z","iopub.execute_input":"2024-12-03T13:41:47.729050Z","iopub.status.idle":"2024-12-03T13:41:47.762454Z","shell.execute_reply.started":"2024-12-03T13:41:47.729013Z","shell.execute_reply":"2024-12-03T13:41:47.761676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef Treat(data, remove_vars, acctivated = 1):\n    try:\n        data = data.drop(remove_vars, axis=1)\n    except:\n        data = data\n    #Drop NaN\n    if acctivated == 1:\n        data = data.dropna(axis=1, how='any')\n        return data\n    elif acctivated == 2:\n        data = data.fillna(100)\n        return data\n    else:\n        return data\n        \ndef Decoder(data):\n    #Decode to ASCII (optional) and change NaN to 0\n    for col in data.columns:\n        for i in range(len(data)):\n            x = data.iloc[i, data.columns.get_loc(col)] \n            # Check for NaN\n            if pd.isna(x):  \n                continue\n            # Skip numeric values\n            elif isinstance(x, (np.integer, np.floating, int, float)):\n                continue \n            # Convert each char to ASCII\n            elif isinstance(x, str):\n                ascii_values = [str(ord(char)) for char in x]  \n                concatenated = int(\"\".join(ascii_values)) \n                data.iloc[i, data.columns.get_loc(col)] = concatenated  \n        data[col] = data[col].astype(float)\n    return data\n\n\nremove_vars = ['id']\ntarget = ['sii']\n#Drop not needed variables\ndata = data.dropna(subset=target)\ndata = Treat(data, remove_vars, acctivated = 2)\ndata = Decoder(data)\ndata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:47.763505Z","iopub.execute_input":"2024-12-03T13:41:47.763827Z","iopub.status.idle":"2024-12-03T13:41:55.320263Z","shell.execute_reply.started":"2024-12-03T13:41:47.763784Z","shell.execute_reply":"2024-12-03T13:41:55.319394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n#Split data to train and test data set\ndef datasplit(data, rate=0):\n    if rate == 0 :\n        return data, data\n    # Add a random column for splitting\n    data['_split_index'] = random.sample(range(1, len(data) + 1), len(data))\n    data['_split_index'] = data['_split_index'] / len(data)\n    \n    # Split the data\n    Dtrain = data[data[\"_split_index\"] > rate].copy()\n    Dtest = data[data[\"_split_index\"] <= rate].copy()\n    \n    # Print summary\n    print(\"Total rows of the main data:\", len(data))\n    print(\"Total rows of the train data:\", len(Dtrain))\n    print(\"Total rows of the test data:\", len(Dtest))\n    \n    # Clean up temporary column\n    data.drop('_split_index', axis=1, inplace=True)\n    Dtrain.drop('_split_index', axis=1, inplace=True)\n    Dtest.drop('_split_index', axis=1, inplace=True)\n    \n    return Dtrain, Dtest\n\n# 0 for no split, 0.8 for 80% for train data set\nDtrain, Dtest = datasplit(data,0)\nDtrain.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:55.321460Z","iopub.execute_input":"2024-12-03T13:41:55.321734Z","iopub.status.idle":"2024-12-03T13:41:55.356168Z","shell.execute_reply.started":"2024-12-03T13:41:55.321706Z","shell.execute_reply":"2024-12-03T13:41:55.355357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report, roc_auc_score\n\n#Target of the prediction\ntarget_col = 'sii'\n\n# Split into features (X) and target (Y)\n\nY = Dtrain[target_col].astype(\"int\")\nX = Dtrain.drop(columns=[target_col])\n\n# Split into features (X) and target (y)\nY_hat = Dtest[target_col].astype(\"int\")\nX_hat = Dtest.drop(columns=[target_col])\n\ndef perfomance(Y_hat, Y_pred):\n    auc = roc_auc_score(Y_hat, Y_pred)\n    return accuracy_score(Y_hat, Y_pred), auc, classification_report(Y_hat, Y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:55.357298Z","iopub.execute_input":"2024-12-03T13:41:55.357577Z","iopub.status.idle":"2024-12-03T13:41:55.895143Z","shell.execute_reply.started":"2024-12-03T13:41:55.357552Z","shell.execute_reply":"2024-12-03T13:41:55.894433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.utils import to_categorical  # For one-hot encoding (if needed)\nimport numpy as np\n\n# Initialize variables\nmax_attempts = 1000  # Maximum number of retraining attempts\ntarget_accuracy = 1  # Target accuracy to stop training\nno_improvement_limit = 20  # Stop after  n attempts without improvement\nepochs = 1000\nbest_accuracy = 0.0  # Track the best accuracy\nbest_model_path = 'best_model.keras'  # Save the best model\n\nverbose = 0  # verbose = 0 to turn off the log, 1 to turn on\npatience = 20  # Patience time before ending the training on each attempt\n\nattempt = 0\nno_improvement_count = 0\n\n# Ensure Y is a 1D array of class labels (integer labels: [0, 1, 2, 3])\n# If Y is not already in this format, preprocess it.\n# Uncomment the line below if Y is in one-hot encoded format and needs conversion:\n# Y = np.argmax(Y, axis=1)\n\nwhile attempt < max_attempts:\n    attempt += 1\n    print(f\"Training Attempt: {attempt}\")\n\n    # Build the model\n    model = Sequential([\n        Input(shape=(X.shape[1],)),  # Input Layer\n        \n        Dense(300, activation='swish'),  # Hidden Layers\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        Dense(300, activation='swish'),\n        \n        Dense(4, activation='softmax')  # Output Layer for 4 classes\n    ])\n\n    # Compile the model\n    model.compile(optimizer=Adam(learning_rate=0.001), \n                  loss='sparse_categorical_crossentropy',  # Loss for integer labels\n                  metrics=['accuracy'])\n\n    # Callbacks\n    checkpoint = ModelCheckpoint('temp_model.keras', save_best_only=True, monitor='val_loss', mode='min', verbose=verbose)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=patience, restore_best_weights=True, verbose=verbose)\n\n    # Train the model\n    history = model.fit(\n        X, Y, epochs=epochs, batch_size=32, validation_split=0.2,\n        callbacks=[checkpoint, early_stopping],\n        verbose=verbose  # Show epoch progress\n    )\n\n    # Evaluate the model\n    loss, accuracy = model.evaluate(X_hat, Y_hat, verbose=verbose)\n    print(f\"Attempt {attempt} - Accuracy: {accuracy:.4f}\")\n\n    # Check for improvement\n    if accuracy > best_accuracy:\n        best_accuracy = accuracy\n        no_improvement_count = 0  # Reset counter\n        print(f\"New best accuracy: {best_accuracy:.4f}. Saving model.\")\n        model.save(best_model_path)  # Save as the best model\n    else:\n        no_improvement_count += 1\n\n    # Stop if no improvement after specified attempts\n    if no_improvement_count >= no_improvement_limit:\n        print(f\"No improvement for {no_improvement_limit} consecutive attempts. Stopping.\")\n        break\n\n    # Stop if target accuracy is achieved\n    if best_accuracy >= target_accuracy:\n        print(\"Target accuracy achieved. Stopping.\")\n        break\n\n# Load the best model\nprint(f\"Loading the best model with accuracy: {best_accuracy:.4f}\")\nbest_model = load_model(best_model_path)\n\n# Predict on test data\nY_pred_prob = best_model.predict(X_hat)\nY_pred = np.argmax(Y_pred_prob, axis=1)  # Convert probabilities to class labels\n\n# Evaluate predictions\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\n# Calculate performance metrics\nacc = accuracy_score(Y_hat, Y_pred)\ncla = classification_report(Y_hat, Y_pred)\nconf_matrix = confusion_matrix(Y_hat, Y_pred)\n\nprint(\"Test Accuracy:\", acc)\nprint(\"Classification Report:\\n\", cla)\nprint(\"Confusion Matrix:\\n\", conf_matrix)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T13:41:55.896314Z","iopub.execute_input":"2024-12-03T13:41:55.896685Z","iopub.status.idle":"2024-12-03T14:52:13.901156Z","shell.execute_reply.started":"2024-12-03T13:41:55.896645Z","shell.execute_reply":"2024-12-03T14:52:13.900263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test data\nY_pred_prob = best_model.predict(X_hat)\nY_pred = np.argmax(Y_pred_prob, axis=1)  # Convert probabilities to class labels\n\n# Evaluate predictions\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\n# Calculate performance metrics\nacc = accuracy_score(Y_hat, Y_pred)\ncla = classification_report(Y_hat, Y_pred)\nconf_matrix = confusion_matrix(Y_hat, Y_pred)\n\nprint(\"Test Accuracy:\", acc)\nprint(\"Classification Report:\\n\", cla)\nprint(\"Confusion Matrix:\\n\", conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:52:13.903251Z","iopub.execute_input":"2024-12-03T14:52:13.903742Z","iopub.status.idle":"2024-12-03T14:52:14.156561Z","shell.execute_reply.started":"2024-12-03T14:52:13.903715Z","shell.execute_reply":"2024-12-03T14:52:14.155674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Conclusion","metadata":{}},{"cell_type":"code","source":"#Extract columns\nextract_columns = [\"id\",\"sii\"]\n#Data preparation for prediction\ntraining_Test_data = data_test\ntraining_Test_data = Treat(training_Test_data, remove_vars, acctivated = 2)\ntraining_Test_data = Decoder(training_Test_data)\ntraining_Test_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:52:14.157560Z","iopub.execute_input":"2024-12-03T14:52:14.157862Z","iopub.status.idle":"2024-12-03T14:52:14.276629Z","shell.execute_reply.started":"2024-12-03T14:52:14.157813Z","shell.execute_reply":"2024-12-03T14:52:14.275639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Prediction\nY_pred_prob = best_model.predict(training_Test_data)\nY_pred = np.argmax(Y_pred_prob, axis=1)\ndata_test[target_col] = Y_pred\n\nsave_data = data_test[extract_columns]\nsave_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:52:14.277798Z","iopub.execute_input":"2024-12-03T14:52:14.278129Z","iopub.status.idle":"2024-12-03T14:52:14.597218Z","shell.execute_reply.started":"2024-12-03T14:52:14.278103Z","shell.execute_reply":"2024-12-03T14:52:14.596404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"save_data.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:52:14.598243Z","iopub.execute_input":"2024-12-03T14:52:14.598492Z","iopub.status.idle":"2024-12-03T14:52:14.605063Z","shell.execute_reply.started":"2024-12-03T14:52:14.598467Z","shell.execute_reply":"2024-12-03T14:52:14.604243Z"}},"outputs":[],"execution_count":null}]}