{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-24T11:02:02.672230Z","iopub.execute_input":"2024-10-24T11:02:02.672997Z","iopub.status.idle":"2024-10-24T11:02:03.428344Z","shell.execute_reply.started":"2024-10-24T11:02:02.672959Z","shell.execute_reply":"2024-10-24T11:02:03.427401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport sklearn\nfrom bs4 import BeautifulSoup","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.430290Z","iopub.execute_input":"2024-10-24T11:02:03.430641Z","iopub.status.idle":"2024-10-24T11:02:03.435649Z","shell.execute_reply.started":"2024-10-24T11:02:03.430604Z","shell.execute_reply":"2024-10-24T11:02:03.434639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_sub = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\npath_data_dict ='/kaggle/input//child-mind-institute-problematic-internet-use/data_dictionary.csv'\npath_train = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\npath_test = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.436773Z","iopub.execute_input":"2024-10-24T11:02:03.437078Z","iopub.status.idle":"2024-10-24T11:02:03.444801Z","shell.execute_reply.started":"2024-10-24T11:02:03.437047Z","shell.execute_reply":"2024-10-24T11:02:03.443881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv (path_sub)\ndata_dict_df =pd.read_csv (path_data_dict)\ntrain_df =pd.read_csv (path_train)\ntest_df =pd.read_csv(path_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.447070Z","iopub.execute_input":"2024-10-24T11:02:03.447358Z","iopub.status.idle":"2024-10-24T11:02:03.510963Z","shell.execute_reply.started":"2024-10-24T11:02:03.447328Z","shell.execute_reply":"2024-10-24T11:02:03.510079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(sub_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.512021Z","iopub.execute_input":"2024-10-24T11:02:03.512299Z","iopub.status.idle":"2024-10-24T11:02:03.522885Z","shell.execute_reply.started":"2024-10-24T11:02:03.512269Z","shell.execute_reply":"2024-10-24T11:02:03.521889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(data_dict_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.524373Z","iopub.execute_input":"2024-10-24T11:02:03.524817Z","iopub.status.idle":"2024-10-24T11:02:03.541710Z","shell.execute_reply.started":"2024-10-24T11:02:03.524769Z","shell.execute_reply":"2024-10-24T11:02:03.540899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.542895Z","iopub.execute_input":"2024-10-24T11:02:03.543282Z","iopub.status.idle":"2024-10-24T11:02:03.581644Z","shell.execute_reply.started":"2024-10-24T11:02:03.543249Z","shell.execute_reply":"2024-10-24T11:02:03.580750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.582756Z","iopub.execute_input":"2024-10-24T11:02:03.583131Z","iopub.status.idle":"2024-10-24T11:02:03.625526Z","shell.execute_reply.started":"2024-10-24T11:02:03.583087Z","shell.execute_reply":"2024-10-24T11:02:03.624490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[sub_df.isnull()].count()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.626768Z","iopub.execute_input":"2024-10-24T11:02:03.627106Z","iopub.status.idle":"2024-10-24T11:02:03.636444Z","shell.execute_reply.started":"2024-10-24T11:02:03.627072Z","shell.execute_reply":"2024-10-24T11:02:03.635585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict_df[data_dict_df.isnull()].count()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.640219Z","iopub.execute_input":"2024-10-24T11:02:03.640692Z","iopub.status.idle":"2024-10-24T11:02:03.650396Z","shell.execute_reply.started":"2024-10-24T11:02:03.640654Z","shell.execute_reply":"2024-10-24T11:02:03.649319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df.isnull()].count()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.651538Z","iopub.execute_input":"2024-10-24T11:02:03.651836Z","iopub.status.idle":"2024-10-24T11:02:03.673703Z","shell.execute_reply.started":"2024-10-24T11:02:03.651804Z","shell.execute_reply":"2024-10-24T11:02:03.672880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[test_df.isnull()].count()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.674813Z","iopub.execute_input":"2024-10-24T11:02:03.675183Z","iopub.status.idle":"2024-10-24T11:02:03.687072Z","shell.execute_reply.started":"2024-10-24T11:02:03.675141Z","shell.execute_reply":"2024-10-24T11:02:03.686151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.688347Z","iopub.execute_input":"2024-10-24T11:02:03.688711Z","iopub.status.idle":"2024-10-24T11:02:03.699381Z","shell.execute_reply.started":"2024-10-24T11:02:03.688666Z","shell.execute_reply":"2024-10-24T11:02:03.698435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.700657Z","iopub.execute_input":"2024-10-24T11:02:03.701013Z","iopub.status.idle":"2024-10-24T11:02:03.713570Z","shell.execute_reply.started":"2024-10-24T11:02:03.700980Z","shell.execute_reply":"2024-10-24T11:02:03.712715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.714864Z","iopub.execute_input":"2024-10-24T11:02:03.715671Z","iopub.status.idle":"2024-10-24T11:02:03.737677Z","shell.execute_reply.started":"2024-10-24T11:02:03.715638Z","shell.execute_reply":"2024-10-24T11:02:03.736771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.738716Z","iopub.execute_input":"2024-10-24T11:02:03.739037Z","iopub.status.idle":"2024-10-24T11:02:03.751732Z","shell.execute_reply.started":"2024-10-24T11:02:03.739004Z","shell.execute_reply":"2024-10-24T11:02:03.750741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.752895Z","iopub.execute_input":"2024-10-24T11:02:03.753223Z","iopub.status.idle":"2024-10-24T11:02:03.761558Z","shell.execute_reply.started":"2024-10-24T11:02:03.753190Z","shell.execute_reply":"2024-10-24T11:02:03.760616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['PreInt_EduHx-computerinternet_hoursday'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.762645Z","iopub.execute_input":"2024-10-24T11:02:03.762957Z","iopub.status.idle":"2024-10-24T11:02:03.772941Z","shell.execute_reply.started":"2024-10-24T11:02:03.762924Z","shell.execute_reply":"2024-10-24T11:02:03.771911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dmap for internet user 0-30=None, 31-49 mild, 50-79=moderate, 80-100 = Severe\n\ndef categorize_internet_user(internet_user_score):\n  if 0 <= internet_user_score <= 30:\n    return 'None'\n  elif 31 <= internet_user_score <= 49:\n    return 'Mild'\n  elif 50 <= internet_user_score <= 79:\n    return 'Moderate'\n  elif 80 <= internet_user_score <= 100:\n    return 'Severe'\n  else:\n    return 'Unknown'\n\n# Apply the function to create a new column 'Internet Usage Category'\ntrain_df['Internet Usage Category'] = train_df['PreInt_EduHx-computerinternet_hoursday'].apply(categorize_internet_user)\n\n# Now you can analyze the trend of Internet Usage Category\ninternet_usage_category_counts = train_df['Internet Usage Category'].value_counts()\nprint(internet_usage_category_counts)\n\n# You can visualize the trend using a bar chart or other suitable plot\nplt.figure(figsize=(8, 6))\nsns.countplot(x='Internet Usage Category', data=train_df)\nplt.title('Distribution of Internet Usage Categories')\nplt.xlabel('Internet Usage Category')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:03.774074Z","iopub.execute_input":"2024-10-24T11:02:03.774483Z","iopub.status.idle":"2024-10-24T11:02:04.003862Z","shell.execute_reply.started":"2024-10-24T11:02:03.774441Z","shell.execute_reply":"2024-10-24T11:02:04.002955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert a column to datetime objects\ntrain_df['survey_date'] = pd.to_datetime(train_df['PreInt_EduHx-computerinternet_hoursday'], errors='coerce')\n\n# Example: Extract the year from the datetime column\n# Assuming 'PreInt_EduHx-Season' represents a season and is a string\n# Extract the season (e.g., 'Spring', 'Summer', 'Fall', 'Winter')\n# If it is a datetime column and you want the year, use: \n# train_df['survey_season'] = pd.to_datetime(train_df['PreInt_EduHx-Season'], errors='coerce').dt.year\ntrain_df['survey_season'] = train_df['PreInt_EduHx-Season'].str.extract('(Spring|Summer|Fall|Winter)')\n\n\n# Example: Group by year and count the number of records\n# Group by season instead of year\nseason_counts = train_df.groupby('survey_season')['PreInt_EduHx-computerinternet_hoursday'].count()\n\n# Example: Plot the trend\nplt.plot(season_counts.index, season_counts.values)\nplt.xlabel('Season')\nplt.ylabel('Number of Records')\nplt.title('Trend of Internet User')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.005288Z","iopub.execute_input":"2024-10-24T11:02:04.005713Z","iopub.status.idle":"2024-10-24T11:02:04.205819Z","shell.execute_reply.started":"2024-10-24T11:02:04.005670Z","shell.execute_reply":"2024-10-24T11:02:04.204920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'PreInt_EduHx-computerinternet_hoursday' to datetime objects\ntrain_df['internet_hoursday_datetime'] = pd.to_datetime(train_df['PreInt_EduHx-computerinternet_hoursday'], errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.207038Z","iopub.execute_input":"2024-10-24T11:02:04.207421Z","iopub.status.idle":"2024-10-24T11:02:04.214332Z","shell.execute_reply.started":"2024-10-24T11:02:04.207379Z","shell.execute_reply":"2024-10-24T11:02:04.213358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract hour from the 'PreInt_EduHx-computerinternet_hoursday' column\n# Assuming the column contains numeric values representing hours.\n# train_df['survey_hour'] = pd.to_numeric(train_df['PreInt_EduHx-computerinternet_hoursday'], errors='coerce')  # This line is redundant if the column is already numeric\n\n# If the column contains string data like \"3 hours\", you can extract the numeric part:\n# train_df['survey_hour'] = train_df['PreInt_EduHx-computerinternet_hoursday'].str.extract('(\\d+)').astype(int) # Remove this line as it's causing the error\n\n# Instead, if the column is numeric, simply copy the values to the new column:\ntrain_df['survey_hour'] = train_df['PreInt_EduHx-computerinternet_hoursday']\n\n# If the column is of mixed type (some strings, some numbers)\n# and you want to extract numeric values from string entries:\n# First, identify rows where the column is a string:\nstring_rows = train_df['PreInt_EduHx-computerinternet_hoursday'].apply(lambda x: isinstance(x, str))\n\n# Then, apply str.extract to those rows:\n#train_df.loc[string_rows, 'survey_hour'] = train_df.loc[string_rows, 'PreInt_EduHx-computerinternet_hoursday'].str.extract('(\\d+)').astype(int)\n\n# Finally, fill the remaining rows (numeric entries) with their original values:\ntrain_df.loc[~string_rows, 'survey_hour'] = train_df.loc[~string_rows, 'PreInt_EduHx-computerinternet_hoursday']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.215482Z","iopub.execute_input":"2024-10-24T11:02:04.215762Z","iopub.status.idle":"2024-10-24T11:02:04.228684Z","shell.execute_reply.started":"2024-10-24T11:02:04.215732Z","shell.execute_reply":"2024-10-24T11:02:04.227788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Assuming 'PreInt_EduHx-Season' represents a season and is a string\n# Extract the season (e.g., 'Spring', 'Summer', 'Fall', 'Winter')\n\n# Example: Group by season and count the number of records\nseason_counts = train_df.groupby('survey_season')['PreInt_EduHx-computerinternet_hoursday'].count()\n\n# Example: Plot the trend\nplt.plot(season_counts.index, season_counts.values)\nplt.xlabel('Season')\nplt.ylabel('Number of Records')\nplt.title('Trend of Internet User by Season')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.229742Z","iopub.execute_input":"2024-10-24T11:02:04.230129Z","iopub.status.idle":"2024-10-24T11:02:04.485726Z","shell.execute_reply.started":"2024-10-24T11:02:04.230081Z","shell.execute_reply":"2024-10-24T11:02:04.484891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dmap = {0:'Spring',1:'Summer',2:'Fall',3:'Winter'}\ntrain_df['survey_season']=train_df['survey_season'].map(dmap)\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.486976Z","iopub.execute_input":"2024-10-24T11:02:04.487337Z","iopub.status.idle":"2024-10-24T11:02:04.521233Z","shell.execute_reply.started":"2024-10-24T11:02:04.487295Z","shell.execute_reply":"2024-10-24T11:02:04.520210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'PreInt_EduHx-Season' represents a season and is a string\n# Extract the season (e.g., 'Spring', 'Summer', 'Fall', 'Winter')\n\n# Example: Group by season and count the number of records\nseason_counts = train_df.groupby('SDS-SDS_Total_T')['PreInt_EduHx-computerinternet_hoursday'].count()\n\n# Example: Plot the trend\nplt.plot(season_counts.index, season_counts.values)\nplt.xlabel('Season')\nplt.ylabel('Number of Records')\nplt.title('Trend of Internet User by Season')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.522340Z","iopub.execute_input":"2024-10-24T11:02:04.522688Z","iopub.status.idle":"2024-10-24T11:02:04.783337Z","shell.execute_reply.started":"2024-10-24T11:02:04.522653Z","shell.execute_reply":"2024-10-24T11:02:04.782448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  'Internet User' \nplt.figure(figsize=(15,5))\nbymonth = train_df.groupby('PreInt_EduHx-computerinternet_hoursday').count().reset_index()\nlp = sns.lineplot(x='PreInt_EduHx-computerinternet_hoursday', y= 'PCIAT-PCIAT_Total', data = bymonth, sort=False,markers = \"o\")\nax = lp.axes\nax.set_xlim(0,13)\nax.annotate('Max severity', color='red',\n            xy=(6, 1060), xycoords='data',\n            xytext=(0.8, 0.85), textcoords='axes fraction',\n            arrowprops=dict(facecolor='black', shrink=0.1),\n            horizontalalignment='right', verticalalignment='top')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:04.784591Z","iopub.execute_input":"2024-10-24T11:02:04.785345Z","iopub.status.idle":"2024-10-24T11:02:05.129407Z","shell.execute_reply.started":"2024-10-24T11:02:04.785310Z","shell.execute_reply":"2024-10-24T11:02:05.128500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the 'PCIAT-PCIAT_Total' column to string type before using .str.title()\ntrain_df['PreInt_EduHx-computerinternet_hoursday'] = train_df['PCIAT-PCIAT_Total'].astype(str).str.title()\nscore_freq = train_df['PCIAT-PCIAT_Total'].value_counts() # assuming traint_df was typo and correcting it\nscore_freq","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.130896Z","iopub.execute_input":"2024-10-24T11:02:05.131294Z","iopub.status.idle":"2024-10-24T11:02:05.145579Z","shell.execute_reply.started":"2024-10-24T11:02:05.131251Z","shell.execute_reply":"2024-10-24T11:02:05.144627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['SDS-SDS_Total_T'] = train_df['PCIAT-PCIAT_Total'].astype(str).str.title()\n# score_freq = traint_df['PCIAT-PCIAT_Total'].value_counts() # typo here, should be train_df\nscore_freq = train_df['PCIAT-PCIAT_Total'].value_counts()\nscore_freq","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.151883Z","iopub.execute_input":"2024-10-24T11:02:05.152541Z","iopub.status.idle":"2024-10-24T11:02:05.166994Z","shell.execute_reply.started":"2024-10-24T11:02:05.152509Z","shell.execute_reply":"2024-10-24T11:02:05.166110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nfrom wordcloud import WordCloud, STOPWORDS\n\n# Get the responses from the desired column, removing missing values, and converting to lowercase\nPreInt_EduHx_Season = train_df['PreInt_EduHx-computerinternet_hoursday'].dropna().str.lower().tolist()\n\n# Join the responses into a single string for wordcloud generation\nPreInt_EduHx_Season_text = ' '.join(PreInt_EduHx_Season)  \n\n# Define words to exclude from the wordcloud\nlist_stops = ('internet','Parent-Child Internet Addiction','Sleep Disturbance scale','Bio-electric Impedance Analysis','FitnessGram Child','Physical Measures')\n\n# Add the stop words to the default STOPWORDS set\nfor word in list_stops:\n    STOPWORDS.add(word)\n\n# Generate the wordcloud \nwordcloud = WordCloud(stopwords=STOPWORDS, background_color=\"white\").generate(PreInt_EduHx_Season_text)\n\n# Display the wordcloud\n# (You'll need to import a library like matplotlib to display the wordcloud)\nimport matplotlib.pyplot as plt\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.168119Z","iopub.execute_input":"2024-10-24T11:02:05.168451Z","iopub.status.idle":"2024-10-24T11:02:05.255541Z","shell.execute_reply.started":"2024-10-24T11:02:05.168419Z","shell.execute_reply":"2024-10-24T11:02:05.254664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning, module=\"ipykernel\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.256773Z","iopub.execute_input":"2024-10-24T11:02:05.257836Z","iopub.status.idle":"2024-10-24T11:02:05.262372Z","shell.execute_reply.started":"2024-10-24T11:02:05.257762Z","shell.execute_reply":"2024-10-24T11:02:05.261449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def categorize_internet_user(internet_user_score):\n  try:\n    # Attempt to convert to a numeric type\n    internet_user_score = float(internet_user_score)  \n  except ValueError:\n    # Handle cases where conversion fails (e.g., non-numeric strings)\n    return 'Unknown' \n\n  if 0 <= internet_user_score <= 30:\n    return 'None'\n  elif 31 <= internet_user_score <= 49:\n    return 'Mild'\n  elif 50 <= internet_user_score <= 79:\n    return 'Moderate'\n  elif 80 <= internet_user_score <= 100:\n    return 'Severe'\n  else:\n    return 'Unknown'\n\n# Apply the function to create a new column 'Internet Usage Category'\ntrain_df['Internet Usage Category'] = train_df['PreInt_EduHx-computerinternet_hoursday'].apply(categorize_internet_user)\n\n# Now you can analyze the trend of Internet Usage Category\ninternet_usage_category_counts = train_df['Internet Usage Category'].value_counts()\nprint(internet_usage_category_counts)\n\n# You can visualize the trend using a bar chart or other suitable plot\nplt.figure(figsize=(8, 6))\nsns.countplot(x='Internet Usage Category', data=train_df)\nplt.title('Distribution of Internet Usage Categories')\nplt.xlabel('Internet Usage Category')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.263773Z","iopub.execute_input":"2024-10-24T11:02:05.264445Z","iopub.status.idle":"2024-10-24T11:02:05.522370Z","shell.execute_reply.started":"2024-10-24T11:02:05.264397Z","shell.execute_reply":"2024-10-24T11:02:05.521460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.523372Z","iopub.execute_input":"2024-10-24T11:02:05.523635Z","iopub.status.idle":"2024-10-24T11:02:05.529140Z","shell.execute_reply.started":"2024-10-24T11:02:05.523607Z","shell.execute_reply":"2024-10-24T11:02:05.528307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.530341Z","iopub.execute_input":"2024-10-24T11:02:05.530680Z","iopub.status.idle":"2024-10-24T11:02:05.538672Z","shell.execute_reply.started":"2024-10-24T11:02:05.530644Z","shell.execute_reply":"2024-10-24T11:02:05.537800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sudo /etc/init.d/networking restart","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:05.539832Z","iopub.execute_input":"2024-10-24T11:02:05.540403Z","iopub.status.idle":"2024-10-24T11:02:06.613502Z","shell.execute_reply.started":"2024-10-24T11:02:05.540339Z","shell.execute_reply":"2024-10-24T11:02:06.612515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nimport ssl\n\ntry:\n    _create_unverified_https_context = ssl._create_unverified_context\nexcept AttributeError:\n    pass\nelse:\n    ssl._create_default_https_context = _create_unverified_https_context\n\nnltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:06.615092Z","iopub.execute_input":"2024-10-24T11:02:06.615440Z","iopub.status.idle":"2024-10-24T11:02:06.624706Z","shell.execute_reply.started":"2024-10-24T11:02:06.615405Z","shell.execute_reply":"2024-10-24T11:02:06.623764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenization and preprocessing for LDA\nfrom nltk.tokenize import word_tokenize\nimport nltk # Import nltk module\nnltk.download('punkt') # Download the necessary data for tokenization\n\n\ndef preprocess_text(text):\n  \"\"\"\n  Preprocesses text for LDA: tokenization, lowercase, removing punctuation, etc.\n  \"\"\"\n  tokens = word_tokenize(str(text))\n  tokens = [token.lower() for token in tokens if token.isalnum()] \n  return tokens\n\n# Example usage (assuming you have a text column called 'text_column' in train_df):\ntrain_df['tokens'] = train_df[['id', 'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n       'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n       'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10',\n       'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n       'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n       'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'SDS-Season',\n       'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday', 'sii','Internet Usage Category', 'survey_date', 'survey_season',\n       'internet_hoursday_datetime', 'survey_hour']].astype(str).apply(lambda row: preprocess_text(' '.join(row)), axis=1)\n\n# Now 'tokens' column in train_df contains a list of tokens for each text entry.\n\n# You can further refine the tokenization process, remove stopwords, perform stemming/lemmatization\n# as needed for your specific task and dataset. ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:06.626385Z","iopub.execute_input":"2024-10-24T11:02:06.626769Z","iopub.status.idle":"2024-10-24T11:02:10.085763Z","shell.execute_reply.started":"2024-10-24T11:02:06.626708Z","shell.execute_reply":"2024-10-24T11:02:10.084972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenization and preprocessing for LDA\nimport nltk # Import nltk module\nnltk.download('punkt') # Download the necessary data for tokenization\nfrom nltk.tokenize import word_tokenize\n\ndef preprocess_text(text):\n  \"\"\"\n  Preprocesses text for LDA: tokenization, lowercase, removing punctuation, etc.\n  \"\"\"\n  tokens = word_tokenize(str(text))\n  tokens = [token.lower() for token in tokens if token.isalnum()] \n  return tokens\n\n# Apply tokenization to the test dataset\n# Changed the apply to only apply to the relevant columns and not to the tokenized column\ntest_df['tokens'] = test_df[['id', 'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday']].astype(str).apply(lambda row: preprocess_text(' '.join(row)), axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:10.086962Z","iopub.execute_input":"2024-10-24T11:02:10.087272Z","iopub.status.idle":"2024-10-24T11:02:10.112806Z","shell.execute_reply.started":"2024-10-24T11:02:10.087240Z","shell.execute_reply":"2024-10-24T11:02:10.111736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare the vocabulary sizes\ntrain_vocabulary = set()\nfor tokens in train_df['tokens']:\n  train_vocabulary.update(tokens)\n\ntest_vocabulary = set()\nfor tokens in test_df['tokens']:\n  # Check if tokens is iterable before updating the set\n  if isinstance(tokens, (list, tuple, set)):  # Check if tokens is a list, tuple or set\n    test_vocabulary.update(tokens)\n  elif tokens is not None and not isinstance(tokens, float) and not isinstance(tokens, int):  # Check if tokens is a non-null, non-float, non-int value\n    test_vocabulary.add(str(tokens)) # Attempt to convert to string before adding to the vocabulary\n\nprint(\"Train Vocabulary Size:\", len(train_vocabulary))\nprint(\"Test Vocabulary Size:\", len(test_vocabulary))\n\n# Find common and unique tokens\ncommon_tokens = train_vocabulary.intersection(test_vocabulary)\ntrain_unique_tokens = train_vocabulary.difference(test_vocabulary)\ntest_unique_tokens = test_vocabulary.difference(train_vocabulary)\n\nprint(\"Number of Common Tokens:\", len(common_tokens))\nprint(\"Number of Unique Tokens in Train:\", len(train_unique_tokens))\nprint(\"Number of Unique Tokens in Test:\", len(test_unique_tokens))\n\n# You can also analyze the frequency distribution of tokens in both datasets\n# to get a better understanding of the differences.\nfrom collections import Counter\n\ntrain_token_counts = Counter([token for tokens in train_df['tokens'] for token in tokens])\ntest_token_counts = Counter([token for tokens in test_df['tokens'] for token in tokens])\n\nprint(\"Most common tokens in train:\", train_token_counts.most_common(10))\nprint(\"Most common tokens in test:\", test_token_counts.most_common(10))\n\n# Further analysis can include comparing the distribution of token lengths,\n# the presence of specific keywords or phrases, and other relevant metrics.","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:10.114304Z","iopub.execute_input":"2024-10-24T11:02:10.114930Z","iopub.status.idle":"2024-10-24T11:02:10.172881Z","shell.execute_reply.started":"2024-10-24T11:02:10.114885Z","shell.execute_reply":"2024-10-24T11:02:10.171916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Assuming you have already tokenized your text data in 'train_df' and 'test_df' \n# as demonstrated in the provided code.\n\n# Example: You can use the tokenized data to build a feature representation for your classification task.\n# One approach is to use TF-IDF (Term Frequency-Inverse Document Frequency) to represent the text as a vector.\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\n# Join the tokens back into strings for TF-IDF\ntrain_df['text_for_tfidf'] = train_df['tokens'].apply(lambda x: ' '.join(x))\ntest_df['text_for_tfidf'] = test_df['tokens'].apply(lambda x: ' '.join(x))\n\n# Create a TF-IDF vectorizer\nvectorizer = TfidfVectorizer()\n\n# Fit and transform the training data\ntrain_tfidf = vectorizer.fit_transform(train_df['text_for_tfidf'])\n\n# Transform the test data using the same vectorizer fitted on training data\ntest_tfidf = vectorizer.transform(test_df['text_for_tfidf'])\n\n# Now, 'train_tfidf' and 'test_tfidf' are matrices representing your text data as numerical vectors.","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:10.174283Z","iopub.execute_input":"2024-10-24T11:02:10.175382Z","iopub.status.idle":"2024-10-24T11:02:10.396328Z","shell.execute_reply.started":"2024-10-24T11:02:10.175337Z","shell.execute_reply":"2024-10-24T11:02:10.395355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\n\n# Filter out DeprecationWarnings \nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import log_loss\n\n# Assuming 'train_tfidf' and 'test_tfidf' are your feature matrices,\n# and 'train_df['Internet Usage Category']' is your target variable.\n\n# Create a LabelEncoder object\nlabel_encoder = LabelEncoder()\n\n# Fit the label encoder to your 'Internet Usage Category' column\nlabel_encoder.fit(train_df['Internet Usage Category'])\n\n# Transform the column using the fitted label encoder\ntrain_df['sii'] = label_encoder.transform(train_df['Internet Usage Category'])\n\n# View the mapping of labels to encoded values\nmapping = dict(zip(label_encoder.classes_, label_encoder.transform(label_encoder.classes_)))\nprint(mapping) \n# Convert the target variable to encoded values\ny_train_encoded = label_encoder.transform(train_df['Internet Usage Category'])\n# Split your training data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(train_tfidf, y_train_encoded, test_size=0.2, random_state=42)\n\n# Train a Logistic Regression model with hyperparameter tuning\nfrom sklearn.model_selection import GridSearchCV\n\nparam_grid = {\n    'C': [0.1, 1, 10],  # Regularization strength\n    'penalty': ['l2'],  # Regularization type\n    'solver': ['newton-cg', 'sag', 'saga', 'lbfgs'],  # Optimization algorithm\n    'max_iter': [500, 1000, 2000] # Maximum number of iterations\n}\n\nmodel = LogisticRegression(solver='sag')  # Specify multiclass handling\ngrid_search = GridSearchCV(model, param_grid, cv=5, scoring='neg_log_loss')\ngrid_search.fit(X_train, y_train)\n\n# Get the best model from the grid search\nbest_model = grid_search.best_estimator_\n\n# Make predictions on the validation set\ny_pred = best_model.predict(X_val)\n\n# Decode the predictions back to original labels\ny_pred_decoded = label_encoder.inverse_transform(y_pred)\ny_val_decoded = label_encoder.inverse_transform(y_val)\n\n# Evaluate the model's accuracy\ntraining_accuracy = best_model.score(X_train, y_train)\nvalidation_accuracy = best_model.score(X_val, y_val)\nprint(\"Training Accuracy:\", training_accuracy)\naccuracy = accuracy_score(y_val_decoded, y_pred_decoded)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:02:10.397586Z","iopub.execute_input":"2024-10-24T11:02:10.397919Z","iopub.status.idle":"2024-10-24T11:05:54.589420Z","shell.execute_reply.started":"2024-10-24T11:02:10.397879Z","shell.execute_reply":"2024-10-24T11:05:54.588290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'best_model' and 'test_tfidf' are defined from the previous code.\n\n# Make predictions on the test set\nsii_predictions = best_model.predict(test_tfidf)\n\n# Decode the predictions back to original labels (if needed)\n# sii_predictions_decoded = label_encoder.inverse_transform(sii_predictions)\n\nsii_predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:05:54.591164Z","iopub.execute_input":"2024-10-24T11:05:54.591971Z","iopub.status.idle":"2024-10-24T11:05:54.599126Z","shell.execute_reply.started":"2024-10-24T11:05:54.591920Z","shell.execute_reply":"2024-10-24T11:05:54.598043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'test_df' is your DataFrame and it has 'id' and 'sii' columns.\n# But \"sii\" is probably missing, so let's add it as a placeholder\n\n# Create a new 'sii' column filled with a default value (e.g., 0)\ntest_df['sii'] = sii_predictions  # Replace 0 with an appropriate default value if needed\n\n# Create a new DataFrame with only 'id' and 'sii' columns\nsubmission_df = test_df[['id', 'sii']]\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission_test.csv', index=False)\nprint(submission_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:05:54.600588Z","iopub.execute_input":"2024-10-24T11:05:54.601006Z","iopub.status.idle":"2024-10-24T11:05:54.615109Z","shell.execute_reply.started":"2024-10-24T11:05:54.600955Z","shell.execute_reply":"2024-10-24T11:05:54.614047Z"},"trusted":true},"execution_count":null,"outputs":[]}]}