{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:15:53.950656Z","iopub.execute_input":"2024-12-06T18:15:53.951053Z","iopub.status.idle":"2024-12-06T18:15:58.138323Z","shell.execute_reply.started":"2024-12-06T18:15:53.950987Z","shell.execute_reply":"2024-12-06T18:15:58.136958Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  *Features EDA*","metadata":{}},{"cell_type":"code","source":"#libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:23:28.220420Z","iopub.execute_input":"2024-12-06T19:23:28.221325Z","iopub.status.idle":"2024-12-06T19:23:31.407180Z","shell.execute_reply.started":"2024-12-06T19:23:28.221276Z","shell.execute_reply":"2024-12-06T19:23:31.406010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:52:50.153198Z","iopub.execute_input":"2024-12-06T19:52:50.153620Z","iopub.status.idle":"2024-12-06T19:52:50.247023Z","shell.execute_reply.started":"2024-12-06T19:52:50.153583Z","shell.execute_reply":"2024-12-06T19:52:50.245874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display (train.head(10))\ndisplay (test.head(10))\n\nprint(f\"Train Dataset -> Rows: {train.shape[0]}, Columns: {train.shape[1]}\")\nprint(f\"Test Dataset  -> Rows: {test.shape[0]}, Columns: {test.shape[1]}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:05:06.395867Z","iopub.execute_input":"2024-12-06T20:05:06.397137Z","iopub.status.idle":"2024-12-06T20:05:06.452523Z","shell.execute_reply.started":"2024-12-06T20:05:06.397094Z","shell.execute_reply":"2024-12-06T20:05:06.451487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:07:03.618836Z","iopub.execute_input":"2024-12-06T20:07:03.619221Z","iopub.status.idle":"2024-12-06T20:07:03.632165Z","shell.execute_reply.started":"2024-12-06T20:07:03.619186Z","shell.execute_reply":"2024-12-06T20:07:03.630826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper Functions ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndef calculate_stats(data, columns):\n    # Ensure columns is a list\n    columns = [columns] if isinstance(columns, str) else columns\n    \n    def process_categorical(col):\n        counts = data[col].value_counts(dropna=False, sort=False)\n        percents = counts / len(data[col]) * 100\n        formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n        return pd.DataFrame({'count (%)': formatted})\n    \n    def process_numerical(col):\n        stats = data[col].describe().to_frame().T\n        stats['missing'] = data[col].isnull().sum()\n        stats.index.name = col\n        return stats\n    \n    # Generate stats for each column\n    stats = [\n        process_categorical(col) if pd.api.types.is_categorical_dtype(data[col]) or pd.api.types.is_object_dtype(data[col])\n        else process_numerical(col)\n        for col in columns\n    ]\n    \n    return pd.concat(stats, axis=0)\n?","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:13:13.722419Z","iopub.execute_input":"2024-12-06T20:13:13.722869Z","iopub.status.idle":"2024-12-06T20:13:13.731587Z","shell.execute_reply.started":"2024-12-06T20:13:13.722831Z","shell.execute_reply":"2024-12-06T20:13:13.730339Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Searching for the Target: The Quest for Insights\n\n","metadata":{}},{"cell_type":"code","source":"#retrieve featres from train and test\ntrain_features = set(train.columns)\n\ntest_features = set(test.columns)\n\n#View column names in the train dataset\nprint(\"Train dataset columns:\")\nprint(train.columns.tolist())\n\n#View column names in the test dataset\nprint(\"\\nTest dataset columns:\")\nprint(test.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:48:48.827238Z","iopub.execute_input":"2024-12-06T20:48:48.827677Z","iopub.status.idle":"2024-12-06T20:48:48.835956Z","shell.execute_reply.started":"2024-12-06T20:48:48.827639Z","shell.execute_reply":"2024-12-06T20:48:48.834417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#indentify features missing from the test set\nmissing_in_test = train_features - test_features\n\nprint(missing_in_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:48:29.169809Z","iopub.execute_input":"2024-12-06T20:48:29.170355Z","iopub.status.idle":"2024-12-06T20:48:29.178355Z","shell.execute_reply.started":"2024-12-06T20:48:29.170307Z","shell.execute_reply":"2024-12-06T20:48:29.176040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#find which features are related to the target variable\nrelated_features = []\nfor feature in missing_in_test:\n    if train[feature].dtype != 'object':  # For numerical features\n        correlation = train[feature].corr(train['sii'])\n        if abs(correlation) > 0.1:  # Adjust the threshold as needed\n            related_features.append((feature, correlation))\n    else:  # For categorical features\n        print(f\"Feature '{feature}' is categorical; consider statistical tests.\")\n\n# Display the related features\nprint(\"Features related to the target variable but missing in the test set:\")\nfor feature, correlation in related_features:\n    print(f\"{feature}: correlation with target = {correlation:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T20:51:57.656342Z","iopub.execute_input":"2024-12-06T20:51:57.656865Z","iopub.status.idle":"2024-12-06T20:51:57.683348Z","shell.execute_reply.started":"2024-12-06T20:51:57.656823Z","shell.execute_reply":"2024-12-06T20:51:57.681993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}