{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#https://www.kaggle.com/competitions/child-mind-institute-problematic-internet-use/overview\nimport numpy as np \nimport pandas as pd \nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom pandas.plotting import andrews_curves\n\n\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n#I have chosen this dataset because I am interested in the challenge of predicting the level of problematic internet usage shown by children based on their physical activity. \n#In an age where children are increasingly being exposed to internet usage, using machine learning to gain an understanding to how it is effecting their physical activity can be useful to warn parents to monitor their child's internet usage.\n\nprint(\"num features:\",train_data.shape[1])\nprint(\"num instances:\",train_data.shape[0])\n\n#There are 82 features and 3960 instances in the training dataset. ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-10-24T23:25:03.905167Z","iopub.execute_input":"2024-10-24T23:25:03.905672Z","iopub.status.idle":"2024-10-24T23:25:03.978328Z","shell.execute_reply.started":"2024-10-24T23:25:03.905627Z","shell.execute_reply":"2024-10-24T23:25:03.976818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.sciencedirect.com/science/article/abs/pii/S0020025523012537#:~:text=Neural%20networks%20have%20achieved%20excellent,by%20using%20multiple%20different%20imputers.\n#https://www.google.com/url?sa=t&rct=j&q=&esrc=s&source=web&cd=&ved=2ahUKEwiXteae-KeJAxWaLkQIHaMgJ8kQ-NANegQIPBAC&url=https%3A%2F%2Fresearch.fredhutch.org%2Fcontent%2Fdam%2Fstripe%2Fwu%2Ffiles%2FPublications%2F2018wires-svm.pdf&usg=AOvVaw0x_MohVfE05-aj3Z1arztZ&opi=89978449\nprint(train_data.describe())\n\n#From these statistics, I can see that some features such as the ssi, BMI, weight, and hight have missing data points. This means that models such as SVMs and neural networks won't be able to handle the data directly.  \n\n#What's interesting about this dataset is that it consists of two elements of study. The first element is physical activity which is comprised of data from a wrist-worn accelerometer, physical fitness assessments, and questionnaires.\n#The second element is internet usage behavior data such as the average time a child spends on the internet in a day.\n\n#The feature I am trying to predict in this competition is to predict the Severity Impairment Index (sii) which is a standard measure of problematic internet usage.\n\n#From observing the \"count\" of each feature, I can see that this dataset is not balanced. I will improve the balance by dropping features with significantly lower data points and potentially filling in features with slightly less data points with synthetic data.   ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:25:03.981194Z","iopub.execute_input":"2024-10-24T23:25:03.981680Z","iopub.status.idle":"2024-10-24T23:25:04.188920Z","shell.execute_reply.started":"2024-10-24T23:25:03.981631Z","shell.execute_reply":"2024-10-24T23:25:04.187259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Attempting to find correlations between the target sii feature and other features. \n\n#correlations = train_data.corr()['sii']\n#print(correlations)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:25:04.191052Z","iopub.execute_input":"2024-10-24T23:25:04.191969Z","iopub.status.idle":"2024-10-24T23:25:04.198827Z","shell.execute_reply.started":"2024-10-24T23:25:04.191906Z","shell.execute_reply":"2024-10-24T23:25:04.197261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://medium.com/codex/beyond-matplotlib-and-seaborn-python-data-visualization-tools-that-work-3ef7f8d1500e\n\n#histogram showing the levels for the Severity Impairment Index variabel: 0 for None, 1 for Mild, 2 for Moderate, and 3 for Severe\nfig, ax = plt.subplots(figsize=(12, 6))\n\nwith sns.axes_style(\"whitegrid\"):\n    viz = sns.histplot(data=train_data, x=\"sii\",  binwidth=.02, ax=ax)\n    viz.set_title(\"Histogram of Severity Impairment Index\")\n    viz.set_xlabel('sii')\n    viz","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:25:04.201143Z","iopub.execute_input":"2024-10-24T23:25:04.201631Z","iopub.status.idle":"2024-10-24T23:25:04.900183Z","shell.execute_reply.started":"2024-10-24T23:25:04.201568Z","shell.execute_reply":"2024-10-24T23:25:04.898772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scatter plot showing the increase of severe internet usage as a child ages. \nsns.FacetGrid(train_data, hue=\"Basic_Demos-Sex\", palette=\"husl\") \\\n   .map(plt.scatter, \"sii\", \"Basic_Demos-Age\") \\\n   .add_legend()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:25:04.903443Z","iopub.execute_input":"2024-10-24T23:25:04.903958Z","iopub.status.idle":"2024-10-24T23:25:05.578937Z","shell.execute_reply.started":"2024-10-24T23:25:04.903878Z","shell.execute_reply":"2024-10-24T23:25:05.577335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This prediction can't be solved without the use of machine learning because there is no clear understanding or set of rules for how the different features affect the target variable.  \n\n#This is a supervised learning problem because it involves labeled training data","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:25:05.580965Z","iopub.execute_input":"2024-10-24T23:25:05.581508Z","iopub.status.idle":"2024-10-24T23:25:05.588910Z","shell.execute_reply.started":"2024-10-24T23:25:05.581449Z","shell.execute_reply":"2024-10-24T23:25:05.587341Z"},"trusted":true},"execution_count":null,"outputs":[]}]}