{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Second attempt","metadata":{}},{"cell_type":"markdown","source":"This is the second attempt. Here I will try to change my approach a bit since my first attempt.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:00:57.329779Z","iopub.execute_input":"2024-09-26T09:00:57.330452Z","iopub.status.idle":"2024-09-26T09:00:57.697402Z","shell.execute_reply.started":"2024-09-26T09:00:57.330404Z","shell.execute_reply":"2024-09-26T09:00:57.696372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:01:22.348342Z","iopub.execute_input":"2024-09-26T09:01:22.348747Z","iopub.status.idle":"2024-09-26T09:01:22.424428Z","shell.execute_reply.started":"2024-09-26T09:01:22.348705Z","shell.execute_reply":"2024-09-26T09:01:22.423454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:01:23.477236Z","iopub.execute_input":"2024-09-26T09:01:23.477627Z","iopub.status.idle":"2024-09-26T09:01:23.521627Z","shell.execute_reply.started":"2024-09-26T09:01:23.477587Z","shell.execute_reply":"2024-09-26T09:01:23.520718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:01:28.658777Z","iopub.execute_input":"2024-09-26T09:01:28.659150Z","iopub.status.idle":"2024-09-26T09:01:28.686313Z","shell.execute_reply.started":"2024-09-26T09:01:28.659113Z","shell.execute_reply":"2024-09-26T09:01:28.685381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will maintain the same approach removing PCIAT columns because I still don't think we need them for training.","metadata":{}},{"cell_type":"code","source":"train_cols = set(train_df.columns)\npciat_cols = [col for col in train_cols if 'PCIAT' in col]\ntrain_df = train_df.drop(columns=pciat_cols)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:02:48.660050Z","iopub.execute_input":"2024-09-26T09:02:48.660436Z","iopub.status.idle":"2024-09-26T09:02:48.698801Z","shell.execute_reply.started":"2024-09-26T09:02:48.660398Z","shell.execute_reply":"2024-09-26T09:02:48.697757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now, we have our training data without those PCIAT columns that were not in the test data. \n\nIn my first attempt, I removed columns with missing values, then removed the rows with missing `sii` and then remove more columns. Now I will try to remove first the rows with missing `sii`.","metadata":{}},{"cell_type":"code","source":"train_df = train_df.dropna(subset=[\"sii\"])\ntrain_df.shape, train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:26:47.636447Z","iopub.execute_input":"2024-09-26T09:26:47.636890Z","iopub.status.idle":"2024-09-26T09:26:47.656427Z","shell.execute_reply.started":"2024-09-26T09:26:47.636850Z","shell.execute_reply":"2024-09-26T09:26:47.655243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check the percentage of missing values now.","metadata":{}},{"cell_type":"code","source":"for col in train_df.columns:\n    miss_count = train_df[col].isna().sum()\n    percent_miss = (miss_count / len(train_df)) * 100\n    print(f\"{col}: {percent_miss:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:29:01.541184Z","iopub.execute_input":"2024-09-26T09:29:01.541925Z","iopub.status.idle":"2024-09-26T09:29:01.567558Z","shell.execute_reply.started":"2024-09-26T09:29:01.541872Z","shell.execute_reply":"2024-09-26T09:29:01.566580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Any column over 50% should be dropped.","metadata":{}},{"cell_type":"code","source":"cols_many_missing_values = [col for col in train_df.columns if (train_df[col].isna().sum() / len(train_df)) > 0.5]\ncols_many_missing_values","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:29:46.962406Z","iopub.execute_input":"2024-09-26T09:29:46.963460Z","iopub.status.idle":"2024-09-26T09:29:46.987538Z","shell.execute_reply.started":"2024-09-26T09:29:46.963404Z","shell.execute_reply":"2024-09-26T09:29:46.986338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns=cols_many_missing_values)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:29:58.158039Z","iopub.execute_input":"2024-09-26T09:29:58.158408Z","iopub.status.idle":"2024-09-26T09:29:58.196893Z","shell.execute_reply.started":"2024-09-26T09:29:58.158371Z","shell.execute_reply":"2024-09-26T09:29:58.195895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now I'm going to train again a Tabular Learner with this small tweaks. Don't expect to see much difference. First categorical and continuous columns.","metadata":{}},{"cell_type":"code","source":"categorical_cols = train_df.select_dtypes(include=[\"object\"]).columns\ncategorical_cols = categorical_cols[1:].to_list()\ncategorical_cols","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:03.909778Z","iopub.execute_input":"2024-09-26T09:33:03.910848Z","iopub.status.idle":"2024-09-26T09:33:03.920123Z","shell.execute_reply.started":"2024-09-26T09:33:03.910786Z","shell.execute_reply":"2024-09-26T09:33:03.918993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"continuous_cols = [col for col in train_df.columns if col not in categorical_cols]\ncontinuous_cols = continuous_cols[1:-1]\ncontinuous_cols","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:11.545339Z","iopub.execute_input":"2024-09-26T09:33:11.546175Z","iopub.status.idle":"2024-09-26T09:33:11.554837Z","shell.execute_reply.started":"2024-09-26T09:33:11.546119Z","shell.execute_reply":"2024-09-26T09:33:11.553763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's use a TabularLearner from fastai now.","metadata":{}},{"cell_type":"code","source":"from fastai.tabular.all import *","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:14.546572Z","iopub.execute_input":"2024-09-26T09:33:14.546963Z","iopub.status.idle":"2024-09-26T09:33:19.026364Z","shell.execute_reply.started":"2024-09-26T09:33:14.546928Z","shell.execute_reply":"2024-09-26T09:33:19.025358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits = RandomSplitter(valid_pct=0.2)(range_of(train_df))","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:19.028170Z","iopub.execute_input":"2024-09-26T09:33:19.028869Z","iopub.status.idle":"2024-09-26T09:33:19.058406Z","shell.execute_reply.started":"2024-09-26T09:33:19.028818Z","shell.execute_reply":"2024-09-26T09:33:19.057418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to = TabularPandas(train_df, procs=[Categorify, FillMissing, Normalize],\n                   cat_names=categorical_cols,\n                   cont_names=continuous_cols,\n                   y_names='sii',\n                   y_block=CategoryBlock,\n                   splits=splits)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:32.345062Z","iopub.execute_input":"2024-09-26T09:33:32.345465Z","iopub.status.idle":"2024-09-26T09:33:32.497831Z","shell.execute_reply.started":"2024-09-26T09:33:32.345429Z","shell.execute_reply":"2024-09-26T09:33:32.496604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = to.dataloaders(bs=64)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:33.082480Z","iopub.execute_input":"2024-09-26T09:33:33.082866Z","iopub.status.idle":"2024-09-26T09:33:33.172562Z","shell.execute_reply.started":"2024-09-26T09:33:33.082830Z","shell.execute_reply":"2024-09-26T09:33:33.171530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:34.681528Z","iopub.execute_input":"2024-09-26T09:33:34.682479Z","iopub.status.idle":"2024-09-26T09:33:34.791973Z","shell.execute_reply.started":"2024-09-26T09:33:34.682440Z","shell.execute_reply":"2024-09-26T09:33:34.790884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = tabular_learner(dls, metrics=accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:42.400513Z","iopub.execute_input":"2024-09-26T09:33:42.400916Z","iopub.status.idle":"2024-09-26T09:33:42.442858Z","shell.execute_reply.started":"2024-09-26T09:33:42.400879Z","shell.execute_reply":"2024-09-26T09:33:42.441863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(1)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:33:44.829204Z","iopub.execute_input":"2024-09-26T09:33:44.829713Z","iopub.status.idle":"2024-09-26T09:33:46.575367Z","shell.execute_reply.started":"2024-09-26T09:33:44.829647Z","shell.execute_reply":"2024-09-26T09:33:46.574327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Accuracy is a bit better. 53% instead of 49%. Let's try to submit this.","metadata":{}},{"cell_type":"code","source":"learn.show_results()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:34:53.233613Z","iopub.execute_input":"2024-09-26T09:34:53.234039Z","iopub.status.idle":"2024-09-26T09:34:53.343469Z","shell.execute_reply.started":"2024-09-26T09:34:53.234001Z","shell.execute_reply":"2024-09-26T09:34:53.342538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl = learn.dls.test_dl(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:34:56.834518Z","iopub.execute_input":"2024-09-26T09:34:56.835441Z","iopub.status.idle":"2024-09-26T09:34:56.911803Z","shell.execute_reply.started":"2024-09-26T09:34:56.835391Z","shell.execute_reply":"2024-09-26T09:34:56.910591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = learn.get_preds(dl=dl)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:35:04.392532Z","iopub.execute_input":"2024-09-26T09:35:04.392949Z","iopub.status.idle":"2024-09-26T09:35:04.433877Z","shell.execute_reply.started":"2024-09-26T09:35:04.392911Z","shell.execute_reply":"2024-09-26T09:35:04.432863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = []\ncount = 0\nfor pred in preds[0]:\n    max_idx = np.argmax(pred.numpy())\n    results.append((test_df['id'].iloc[count], str(max_idx)))\n    count += 1\n\nresults_df = pd.DataFrame(results, columns=['id', 'sii'])\nresults_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:35:49.579336Z","iopub.execute_input":"2024-09-26T09:35:49.580219Z","iopub.status.idle":"2024-09-26T09:35:49.590429Z","shell.execute_reply.started":"2024-09-26T09:35:49.580165Z","shell.execute_reply":"2024-09-26T09:35:49.589376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-09-26T09:35:50.945812Z","iopub.execute_input":"2024-09-26T09:35:50.946667Z","iopub.status.idle":"2024-09-26T09:35:52.126385Z","shell.execute_reply.started":"2024-09-26T09:35:50.946606Z","shell.execute_reply":"2024-09-26T09:35:52.124986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}