{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T11:56:34.718996Z","iopub.execute_input":"2022-08-02T11:56:34.719783Z","iopub.status.idle":"2022-08-02T11:56:34.753327Z","shell.execute_reply.started":"2022-08-02T11:56:34.719684Z","shell.execute_reply":"2022-08-02T11:56:34.752470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import feature_extraction, linear_model, model_selection, preprocessing\n\ntrain_df = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest_df = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:34.754978Z","iopub.execute_input":"2022-08-02T11:56:34.755507Z","iopub.status.idle":"2022-08-02T11:56:36.094274Z","shell.execute_reply.started":"2022-08-02T11:56:34.755476Z","shell.execute_reply":"2022-08-02T11:56:36.093140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.095870Z","iopub.execute_input":"2022-08-02T11:56:36.096407Z","iopub.status.idle":"2022-08-02T11:56:36.119640Z","shell.execute_reply.started":"2022-08-02T11:56:36.096364Z","shell.execute_reply":"2022-08-02T11:56:36.118813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.122677Z","iopub.execute_input":"2022-08-02T11:56:36.123009Z","iopub.status.idle":"2022-08-02T11:56:36.138812Z","shell.execute_reply.started":"2022-08-02T11:56:36.122980Z","shell.execute_reply":"2022-08-02T11:56:36.137484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_df))\ntrain_df = train_df.drop_duplicates('text', keep='last')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.140454Z","iopub.execute_input":"2022-08-02T11:56:36.140926Z","iopub.status.idle":"2022-08-02T11:56:36.161534Z","shell.execute_reply.started":"2022-08-02T11:56:36.140873Z","shell.execute_reply":"2022-08-02T11:56:36.160660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train_df['target'])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.162822Z","iopub.execute_input":"2022-08-02T11:56:36.163599Z","iopub.status.idle":"2022-08-02T11:56:36.331405Z","shell.execute_reply.started":"2022-08-02T11:56:36.163565Z","shell.execute_reply":"2022-08-02T11:56:36.330223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.332687Z","iopub.execute_input":"2022-08-02T11:56:36.333522Z","iopub.status.idle":"2022-08-02T11:56:36.340972Z","shell.execute_reply.started":"2022-08-02T11:56:36.333489Z","shell.execute_reply":"2022-08-02T11:56:36.339974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,40))\nprint(f'Unique keywords num = {len(train_df.keyword.unique())}')\nprint(f'Unique keywords num = {len(test_df.keyword.unique())}')\nsns.countplot(y=train_df['keyword'], color=(0,0,1), label='Train')\nsns.countplot(y=test_df['keyword'], color=(1,0,0), label='Test')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:36.342245Z","iopub.execute_input":"2022-08-02T11:56:36.342678Z","iopub.status.idle":"2022-08-02T11:56:40.061226Z","shell.execute_reply.started":"2022-08-02T11:56:36.342631Z","shell.execute_reply":"2022-08-02T11:56:40.059980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,100))\nsns.countplot(data=train_df, y='keyword', hue='target')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:40.062765Z","iopub.execute_input":"2022-08-02T11:56:40.063103Z","iopub.status.idle":"2022-08-02T11:56:44.056483Z","shell.execute_reply.started":"2022-08-02T11:56:40.063074Z","shell.execute_reply":"2022-08-02T11:56:44.055449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()\ntest_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T14:33:18.722238Z","iopub.execute_input":"2022-08-02T14:33:18.723210Z","iopub.status.idle":"2022-08-02T14:33:18.737817Z","shell.execute_reply.started":"2022-08-02T14:33:18.723169Z","shell.execute_reply":"2022-08-02T14:33:18.736630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df ['location'] = train_df ['location'].fillna(train_df['location'].mode()[0])\ntrain_df ['keyword'] = train_df ['keyword'].fillna(train_df['keyword'].mode()[0])\n\ntest_df ['location'] = test_df ['location'].fillna(test_df['location'].mode()[0])\ntest_df ['keyword'] = test_df ['keyword'].fillna(test_df['keyword'].mode()[0])\n\ntrain_df.isnull().sum()\ntest_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T14:33:26.713536Z","iopub.execute_input":"2022-08-02T14:33:26.714788Z","iopub.status.idle":"2022-08-02T14:33:26.739989Z","shell.execute_reply.started":"2022-08-02T14:33:26.714737Z","shell.execute_reply":"2022-08-02T14:33:26.739055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_vectorizer = feature_extraction.text.CountVectorizer()\nexample_train_vectors = count_vectorizer.fit_transform(train_df['text'][0:5])\nprint(example_train_vectors[0].todense().shape)\nprint(example_train_vectors[0].todense())\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:44.059128Z","iopub.execute_input":"2022-08-02T11:56:44.059489Z","iopub.status.idle":"2022-08-02T11:56:44.071381Z","shell.execute_reply.started":"2022-08-02T11:56:44.059460Z","shell.execute_reply":"2022-08-02T11:56:44.070166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_vectors = count_vectorizer.fit_transform(train_df[\"text\"])\ntest_vectors = count_vectorizer.transform(test_df[\"text\"])\n\nclf = linear_model.RidgeClassifier()\nscores = model_selection.cross_val_score(clf, train_vectors, train_df[\"target\"], cv=3, scoring=\"f1\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:56:44.073331Z","iopub.execute_input":"2022-08-02T11:56:44.074170Z","iopub.status.idle":"2022-08-02T11:56:44.663432Z","shell.execute_reply.started":"2022-08-02T11:56:44.074123Z","shell.execute_reply":"2022-08-02T11:56:44.662157Z"},"trusted":true},"execution_count":null,"outputs":[]}]}