{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":7235670,"sourceType":"datasetVersion","datasetId":4190064},{"sourceId":7235683,"sourceType":"datasetVersion","datasetId":4190075}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Proyek PM Kelompok 11 - Stanford Ribonanza RNA Folding**","metadata":{"id":"hjnokqwFaH6O"}},{"cell_type":"markdown","source":"\n\n1.   Ita Anjelly P Sirait - 11420051\n2.   Gilbert Marpaung - 11421003\n3.   Dini H.J. Sipahutar - 11421051\n\n\n\n\n","metadata":{"id":"zx9QmMbuaX7N"}},{"cell_type":"markdown","source":"**Data Understanding**","metadata":{"id":"FXNTwkRla2_T"}},{"cell_type":"markdown","source":"**Unduh Library**","metadata":{"id":"X40unOcpax8i"}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"id":"43efrfW_bW1a","execution":{"iopub.status.busy":"2023-12-19T06:41:33.100448Z","iopub.execute_input":"2023-12-19T06:41:33.100819Z","iopub.status.idle":"2023-12-19T06:41:33.563755Z","shell.execute_reply.started":"2023-12-19T06:41:33.100790Z","shell.execute_reply":"2023-12-19T06:41:33.562551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load** **Data**","metadata":{"id":"2-ShFvKLYM2e"}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/train-data-quickstart/train_data_quickStart.csv')\ndf_test = pd.read_csv('/kaggle/input/datatest/test.csv')\n","metadata":{"id":"ArIApDyvbYHL","execution":{"iopub.status.busy":"2023-12-19T06:47:18.783776Z","iopub.execute_input":"2023-12-19T06:47:18.784190Z","iopub.status.idle":"2023-12-19T06:47:18.858084Z","shell.execute_reply.started":"2023-12-19T06:47:18.784143Z","shell.execute_reply":"2023-12-19T06:47:18.856712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.head())","metadata":{"id":"1VOOylv3K15G","outputId":"0af9ebed-11c9-4289-bf86-d4cfb3c12864","execution":{"iopub.status.busy":"2023-12-19T06:47:24.584477Z","iopub.execute_input":"2023-12-19T06:47:24.584951Z","iopub.status.idle":"2023-12-19T06:47:24.615627Z","shell.execute_reply.started":"2023-12-19T06:47:24.584912Z","shell.execute_reply":"2023-12-19T06:47:24.614336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Menampilkan Tipe Data**","metadata":{"id":"wkz6tRNYYfz9"}},{"cell_type":"code","source":"print(df_train.dtypes)","metadata":{"id":"-rIyrtrLVrm0","outputId":"b7495e17-a5ae-427a-a734-080f0604e377","execution":{"iopub.status.busy":"2023-12-19T06:47:33.135753Z","iopub.execute_input":"2023-12-19T06:47:33.136299Z","iopub.status.idle":"2023-12-19T06:47:33.146461Z","shell.execute_reply.started":"2023-12-19T06:47:33.136250Z","shell.execute_reply":"2023-12-19T06:47:33.144894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test.dtypes)","metadata":{"id":"SpXYO4ZfWrJ0","outputId":"bb0dfbeb-68ce-4cbe-f4a7-f922a249c2f8","execution":{"iopub.status.busy":"2023-12-19T06:47:37.373337Z","iopub.execute_input":"2023-12-19T06:47:37.373809Z","iopub.status.idle":"2023-12-19T06:47:37.382285Z","shell.execute_reply.started":"2023-12-19T06:47:37.373774Z","shell.execute_reply":"2023-12-19T06:47:37.380756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Mengeksplorasi Data Kategorik**","metadata":{"id":"UHsxhiTRYnIZ"}},{"cell_type":"code","source":"df_train.describe(exclude = np.number)","metadata":{"id":"9LLN150TWviF","outputId":"2d6f569d-5830-4c07-989a-f51dbcc4f3b5","execution":{"iopub.status.busy":"2023-12-19T06:47:41.215203Z","iopub.execute_input":"2023-12-19T06:47:41.215654Z","iopub.status.idle":"2023-12-19T06:47:41.335377Z","shell.execute_reply.started":"2023-12-19T06:47:41.215620Z","shell.execute_reply":"2023-12-19T06:47:41.333957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.describe(exclude = np.number)","metadata":{"id":"ec77PffvW4OK","outputId":"7377eb66-a4f1-4ac9-ab9f-16dac9e7fd98","execution":{"iopub.status.busy":"2023-12-19T06:47:47.087014Z","iopub.execute_input":"2023-12-19T06:47:47.088168Z","iopub.status.idle":"2023-12-19T06:47:47.104476Z","shell.execute_reply.started":"2023-12-19T06:47:47.088104Z","shell.execute_reply":"2023-12-19T06:47:47.102901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Preparation**","metadata":{"id":"R0N7i6GPY4Hb"}},{"cell_type":"code","source":"pip install pandas scikit-learn","metadata":{"id":"e6H-iFNRZWct","outputId":"94d27f7f-c758-4cf3-cf9b-6cffce800c8c","execution":{"iopub.status.busy":"2023-12-19T06:53:21.794542Z","iopub.execute_input":"2023-12-19T06:53:21.794923Z","iopub.status.idle":"2023-12-19T06:53:37.717247Z","shell.execute_reply.started":"2023-12-19T06:53:21.794894Z","shell.execute_reply":"2023-12-19T06:53:37.715467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler","metadata":{"id":"dJInZcc7bN3V","execution":{"iopub.status.busy":"2023-12-19T06:53:46.244905Z","iopub.execute_input":"2023-12-19T06:53:46.245585Z","iopub.status.idle":"2023-12-19T06:53:47.101519Z","shell.execute_reply.started":"2023-12-19T06:53:46.245541Z","shell.execute_reply":"2023-12-19T06:53:47.100182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ndf_train_hist = df_train.select_dtypes(exclude=['object'])","metadata":{"id":"cRfJWa6a6W0f","execution":{"iopub.status.busy":"2023-12-19T06:54:00.116790Z","iopub.execute_input":"2023-12-19T06:54:00.117283Z","iopub.status.idle":"2023-12-19T06:54:00.341674Z","shell.execute_reply.started":"2023-12-19T06:54:00.117244Z","shell.execute_reply":"2023-12-19T06:54:00.340166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_hist.hist(figsize=(80,60), alpha = 0.5, edgecolor='black',grid=False);","metadata":{"id":"d45uyimS6tEF","outputId":"37487c47-9d67-49d8-a2fa-6617979e90ad","execution":{"iopub.status.busy":"2023-12-19T06:54:05.856685Z","iopub.execute_input":"2023-12-19T06:54:05.857295Z","iopub.status.idle":"2023-12-19T06:55:37.802866Z","shell.execute_reply.started":"2023-12-19T06:54:05.857246Z","shell.execute_reply":"2023-12-19T06:55:37.800337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['sequence'].describe()","metadata":{"id":"YiIxWNlf7FM_","outputId":"2cfdc01a-6f5b-42de-d51c-4a160a116143","execution":{"iopub.status.busy":"2023-12-19T06:55:47.588887Z","iopub.execute_input":"2023-12-19T06:55:47.589333Z","iopub.status.idle":"2023-12-19T06:55:47.607038Z","shell.execute_reply.started":"2023-12-19T06:55:47.589302Z","shell.execute_reply":"2023-12-19T06:55:47.605354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nmissing_values = df_train[['sequence_id', 'sequence']].isnull().any()\n\nimputer = SimpleImputer(strategy='most_frequent')\n\nif missing_values.any():\n    df_train[['sequence_id', 'sequence']] = imputer.fit_transform(df_train[['sequence_id', 'sequence']])\n\n","metadata":{"id":"gwYpbwJF8xgB","execution":{"iopub.status.busy":"2023-12-19T06:56:21.230492Z","iopub.execute_input":"2023-12-19T06:56:21.230968Z","iopub.status.idle":"2023-12-19T06:56:21.243566Z","shell.execute_reply.started":"2023-12-19T06:56:21.230932Z","shell.execute_reply":"2023-12-19T06:56:21.241960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Data Cleaning\n#    Contoh: Mengganti jenis kelamin menjadi biner (0: False, 1: True)\ndf_train['reactivity_0001'] = df_train['reactivity_0001'].map({'False': 0, 'True': 1})\n","metadata":{"id":"zMJ-M_hNfjqV","execution":{"iopub.status.busy":"2023-12-19T06:56:27.335652Z","iopub.execute_input":"2023-12-19T06:56:27.336123Z","iopub.status.idle":"2023-12-19T06:56:27.344109Z","shell.execute_reply.started":"2023-12-19T06:56:27.336087Z","shell.execute_reply":"2023-12-19T06:56:27.342880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modeling**","metadata":{"id":"6A8RhUmeGeM8"}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"id":"BiiBnPvbNGXE","execution":{"iopub.status.busy":"2023-12-19T06:56:30.656681Z","iopub.execute_input":"2023-12-19T06:56:30.657121Z","iopub.status.idle":"2023-12-19T06:56:30.663258Z","shell.execute_reply.started":"2023-12-19T06:56:30.657078Z","shell.execute_reply":"2023-12-19T06:56:30.661891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.iloc[:, :-1].values\ny = df_train.iloc[:, 4].values","metadata":{"id":"_wI8EYEeKeGp","execution":{"iopub.status.busy":"2023-12-19T06:56:33.741637Z","iopub.execute_input":"2023-12-19T06:56:33.742126Z","iopub.status.idle":"2023-12-19T06:56:33.764384Z","shell.execute_reply.started":"2023-12-19T06:56:33.742089Z","shell.execute_reply":"2023-12-19T06:56:33.763170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.DataFrame(df_train)\nX_test = pd.DataFrame(df_test)","metadata":{"id":"vQKpLRlWAze2","execution":{"iopub.status.busy":"2023-12-19T06:56:36.870405Z","iopub.execute_input":"2023-12-19T06:56:36.870788Z","iopub.status.idle":"2023-12-19T06:56:36.877832Z","shell.execute_reply.started":"2023-12-19T06:56:36.870758Z","shell.execute_reply":"2023-12-19T06:56:36.876348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Assuming df_train is your original DataFrame\ndf_train = pd.DataFrame(X_train)\n\n# Identify columns with 'object' data type\nobject_columns = df_train.select_dtypes(include=['object']).columns\n\n# Apply label encoding to the 'object' column\nle = LabelEncoder()\ndf_train[object_columns[0]] = le.fit_transform(df_train[object_columns[0]])\n\n# Now, df_train[object_columns[0]] is converted to integers\n# Continue with your classifier fitting process","metadata":{"id":"xTy7JNJzaYF3","execution":{"iopub.status.busy":"2023-12-19T06:56:41.138834Z","iopub.execute_input":"2023-12-19T06:56:41.139336Z","iopub.status.idle":"2023-12-19T06:56:41.148443Z","shell.execute_reply.started":"2023-12-19T06:56:41.139298Z","shell.execute_reply":"2023-12-19T06:56:41.147569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\n\n# Assuming X is your NumPy array containing features\n# Specify the indices of categorical columns that need one-hot encoding\ncategorical_indices = [1]\n\n# Use SimpleImputer to handle NaN values (replace NaN with the most frequent value for all columns)\nimputer = SimpleImputer(strategy='most_frequent')\nX_imputed = imputer.fit_transform(X)\n\n# Create a ColumnTransformer\ntransformer = ColumnTransformer(\n    [('encoder', OneHotEncoder(), categorical_indices)],\n    remainder='passthrough'\n)\n\n# Transform the features\nX_transformed = transformer.fit_transform(X_imputed)\n\n# Print the transformed X to check\nprint(X_transformed)\n","metadata":{"id":"hzhAGmRDgRNJ","outputId":"fa9b5238-ca12-4480-d874-a7baac70e34c","execution":{"iopub.status.busy":"2023-12-19T06:57:02.470512Z","iopub.execute_input":"2023-12-19T06:57:02.470972Z","iopub.status.idle":"2023-12-19T06:57:02.573592Z","shell.execute_reply.started":"2023-12-19T06:57:02.470936Z","shell.execute_reply":"2023-12-19T06:57:02.571978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('sequence_id', axis=1)\ny = df_train['sequence_id']","metadata":{"id":"LHFDwZkwNK_E","execution":{"iopub.status.busy":"2023-12-19T06:57:06.981748Z","iopub.execute_input":"2023-12-19T06:57:06.982310Z","iopub.status.idle":"2023-12-19T06:57:06.994940Z","shell.execute_reply.started":"2023-12-19T06:57:06.982264Z","shell.execute_reply":"2023-12-19T06:57:06.993831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\n","metadata":{"id":"OjCzoDH2NiDE","outputId":"aaf5dbfb-c325-4c79-ce26-eff14891cc35","execution":{"iopub.status.busy":"2023-12-19T06:57:11.357743Z","iopub.execute_input":"2023-12-19T06:57:11.358292Z","iopub.status.idle":"2023-12-19T06:57:11.364816Z","shell.execute_reply.started":"2023-12-19T06:57:11.358254Z","shell.execute_reply":"2023-12-19T06:57:11.363531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.isnull().sum())\n","metadata":{"id":"amG3tdtQNjdu","outputId":"97a5158d-d252-4f7d-b27e-9e6955f2b06f","execution":{"iopub.status.busy":"2023-12-19T06:57:14.519859Z","iopub.execute_input":"2023-12-19T06:57:14.520354Z","iopub.status.idle":"2023-12-19T06:57:14.531030Z","shell.execute_reply.started":"2023-12-19T06:57:14.520314Z","shell.execute_reply":"2023-12-19T06:57:14.529818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Splitting dataset into training set and test set\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2, random_state=0)","metadata":{"id":"WabKQYWaNRsw","execution":{"iopub.status.busy":"2023-12-19T06:57:36.024535Z","iopub.execute_input":"2023-12-19T06:57:36.025049Z","iopub.status.idle":"2023-12-19T06:57:36.035974Z","shell.execute_reply.started":"2023-12-19T06:57:36.025012Z","shell.execute_reply":"2023-12-19T06:57:36.034973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"id":"7RlDx3gUCOdu","outputId":"7fe2a68c-5b48-4401-af39-28081cc06c16","execution":{"iopub.status.busy":"2023-12-19T06:57:39.967307Z","iopub.execute_input":"2023-12-19T06:57:39.967769Z","iopub.status.idle":"2023-12-19T06:57:40.005909Z","shell.execute_reply.started":"2023-12-19T06:57:39.967735Z","shell.execute_reply":"2023-12-19T06:57:40.004780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"id":"ho6Qe3qFCUe3","outputId":"c0601fea-888e-44a8-a92f-07ddc13cc1c8","execution":{"iopub.status.busy":"2023-12-19T06:57:43.986781Z","iopub.execute_input":"2023-12-19T06:57:43.987232Z","iopub.status.idle":"2023-12-19T06:57:43.997871Z","shell.execute_reply.started":"2023-12-19T06:57:43.987198Z","shell.execute_reply":"2023-12-19T06:57:43.996202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Making prediction\n#Fittimg classifier to the Training set\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom sklearn.model_selection import cross_val_score","metadata":{"id":"Qhhr4gbqVP46","execution":{"iopub.status.busy":"2023-12-19T06:57:55.146894Z","iopub.execute_input":"2023-12-19T06:57:55.147343Z","iopub.status.idle":"2023-12-19T06:57:55.153808Z","shell.execute_reply.started":"2023-12-19T06:57:55.147312Z","shell.execute_reply":"2023-12-19T06:57:55.152598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instantiate learning model(k=3)\nclassifier=KNeighborsClassifier(n_neighbors=3)","metadata":{"id":"5wwCyStMGg1q","execution":{"iopub.status.busy":"2023-12-19T06:57:59.110096Z","iopub.execute_input":"2023-12-19T06:57:59.110545Z","iopub.status.idle":"2023-12-19T06:57:59.116534Z","shell.execute_reply.started":"2023-12-19T06:57:59.110511Z","shell.execute_reply":"2023-12-19T06:57:59.115213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier\n\nX_train['sequence'] = X_train['sequence'].astype(str)\n\n# Identify numeric and categorical columns\nnumeric_cols = X_train.select_dtypes(include=['number']).columns\ncategorical_cols = X_train.select_dtypes(exclude=['number']).columns\n\n# Create transformers for numeric and categorical columns\nnumeric_transformer = SimpleImputer(strategy='mean')\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Create a column transformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('numeric', numeric_transformer, numeric_cols),\n        ('categorical', categorical_transformer, categorical_cols)\n    ])\n\n# Create a pipeline with preprocessing and classifier\nclassifier = Pipeline(steps=[('preprocessor', preprocessor),\n                               ('classifier', RandomForestClassifier())])\n\n# Fit the model\nclassifier.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-19T07:10:36.383482Z","iopub.execute_input":"2023-12-19T07:10:36.385055Z","iopub.status.idle":"2023-12-19T07:10:42.859191Z","shell.execute_reply.started":"2023-12-19T07:10:36.384987Z","shell.execute_reply":"2023-12-19T07:10:42.857924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\n# Ganti langkah classifier dengan KNN\nclassifier = Pipeline(steps=[('preprocessor', preprocessor),\n                               ('classifier', KNeighborsClassifier())])\n\n# Fit the model\nclassifier.fit(X_train, y_train)\n","metadata":{"id":"1NC867GoLLdB","outputId":"e28578c4-aa64-4e41-99ae-769458955442","execution":{"iopub.status.busy":"2023-12-19T07:10:46.594834Z","iopub.execute_input":"2023-12-19T07:10:46.596112Z","iopub.status.idle":"2023-12-19T07:10:46.718844Z","shell.execute_reply.started":"2023-12-19T07:10:46.596057Z","shell.execute_reply":"2023-12-19T07:10:46.717449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model\nclassifier.fit(X_train,y_train)","metadata":{"id":"2eMEVAnFGrkb","outputId":"6ea0981e-9bd4-46db-df68-9aec2e7f7838","execution":{"iopub.status.busy":"2023-12-19T07:10:50.635190Z","iopub.execute_input":"2023-12-19T07:10:50.635603Z","iopub.status.idle":"2023-12-19T07:10:50.758822Z","shell.execute_reply.started":"2023-12-19T07:10:50.635570Z","shell.execute_reply":"2023-12-19T07:10:50.757463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predicting the Test Set result\ny_pred = classifier.predict(X_test)","metadata":{"id":"0kQWFRc_H50w","execution":{"iopub.status.busy":"2023-12-19T07:10:59.434096Z","iopub.execute_input":"2023-12-19T07:10:59.434584Z","iopub.status.idle":"2023-12-19T07:10:59.527349Z","shell.execute_reply.started":"2023-12-19T07:10:59.434551Z","shell.execute_reply":"2023-12-19T07:10:59.526382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluating prediction building Confusion Matrix\ncm = confusion_matrix(y_test,y_pred)\ncm","metadata":{"id":"nThWoZl-Ho5W","outputId":"f3cba412-0f31-4d6b-97ae-e8aa4e1ca588","execution":{"iopub.status.busy":"2023-12-19T07:11:09.196262Z","iopub.execute_input":"2023-12-19T07:11:09.196712Z","iopub.status.idle":"2023-12-19T07:11:09.209524Z","shell.execute_reply.started":"2023-12-19T07:11:09.196680Z","shell.execute_reply":"2023-12-19T07:11:09.207723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Evaluation**","metadata":{"id":"jn0EAOqCqzP8"}},{"cell_type":"code","source":"accuracy = accuracy_score(y_test, y_pred)*100\nprint('Accuracy of our model is equal' + str(round(accuracy,2))+'%')","metadata":{"id":"DMZnDmQeow5b","outputId":"c649f67e-088a-4bb1-b615-92535919b2af","execution":{"iopub.status.busy":"2023-12-19T07:11:19.843679Z","iopub.execute_input":"2023-12-19T07:11:19.844128Z","iopub.status.idle":"2023-12-19T07:11:19.852549Z","shell.execute_reply.started":"2023-12-19T07:11:19.844093Z","shell.execute_reply":"2023-12-19T07:11:19.851384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score\n\n# Example with 'micro' average and zero_division='warn'\nprecision_micro = precision_score(y_test, y_pred, average='micro', zero_division='warn')\nprint('Precision (Micro): %.3f' % precision_micro)\n\n# Example with 'macro' average and zero_division='warn'\nprecision_macro = precision_score(y_test, y_pred, average='macro', zero_division='warn')\nprint('Precision (Macro): %.3f' % precision_macro)\n\n# Example with 'weighted' average and zero_division='warn'\nprecision_weighted = precision_score(y_test, y_pred, average='weighted', zero_division='warn')\nprint('Precision (Weighted): %.3f' % precision_weighted)\n","metadata":{"id":"7u118fRRIaE8","outputId":"2676c3a0-7cc0-4db1-c6d8-b4be78632c3f","execution":{"iopub.status.busy":"2023-12-19T07:11:51.077931Z","iopub.execute_input":"2023-12-19T07:11:51.078475Z","iopub.status.idle":"2023-12-19T07:11:51.100336Z","shell.execute_reply.started":"2023-12-19T07:11:51.078436Z","shell.execute_reply":"2023-12-19T07:11:51.098974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import recall_score\n\n# Example with 'micro' average\nrecall_micro = recall_score(y_test, y_pred, average='micro')\nprint('Recall (Micro): %.3f' % recall_micro)\n\n# Example with 'macro' average\nrecall_macro = recall_score(y_test, y_pred, average='macro')\nprint('Recall (Macro): %.3f' % recall_macro)\n\n# Example with 'weighted' average\nrecall_weighted = recall_score(y_test, y_pred, average='weighted')\nprint('Recall (Weighted): %.3f' % recall_weighted)\n","metadata":{"id":"C9qArpctJ-_Z","outputId":"1a2ea4b0-10b6-491d-fec4-abddd1626004","execution":{"iopub.status.busy":"2023-12-19T07:12:02.143708Z","iopub.execute_input":"2023-12-19T07:12:02.144124Z","iopub.status.idle":"2023-12-19T07:12:02.163436Z","shell.execute_reply.started":"2023-12-19T07:12:02.144093Z","shell.execute_reply":"2023-12-19T07:12:02.161750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\n# Example with 'micro' average\nf1_micro = f1_score(y_test, y_pred, average='micro')\nprint('F1 Score (Micro): %.3f' % f1_micro)\n\n# Example with 'macro' average\nf1_macro = f1_score(y_test, y_pred, average='macro')\nprint('F1 Score (Macro): %.3f' % f1_macro)\n\n# Example with 'weighted' average\nf1_weighted = f1_score(y_test, y_pred, average='weighted')\nprint('F1 Score (Weighted): %.3f' % f1_weighted)\n","metadata":{"id":"8GA0ijv4S3By","outputId":"e76c280f-9bd2-45ef-b21d-ad7e353b0b32","execution":{"iopub.status.busy":"2023-12-19T07:12:08.938515Z","iopub.execute_input":"2023-12-19T07:12:08.939397Z","iopub.status.idle":"2023-12-19T07:12:08.954860Z","shell.execute_reply.started":"2023-12-19T07:12:08.939358Z","shell.execute_reply":"2023-12-19T07:12:08.953685Z"},"trusted":true},"execution_count":null,"outputs":[]}]}