{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T12:49:08.706347Z","iopub.execute_input":"2022-08-13T12:49:08.707919Z","iopub.status.idle":"2022-08-13T12:49:08.740523Z","shell.execute_reply.started":"2022-08-13T12:49:08.707760Z","shell.execute_reply":"2022-08-13T12:49:08.739530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:49:09.263985Z","iopub.execute_input":"2022-08-13T12:49:09.265001Z","iopub.status.idle":"2022-08-13T12:49:09.309351Z","shell.execute_reply.started":"2022-08-13T12:49:09.264962Z","shell.execute_reply":"2022-08-13T12:49:09.308152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:49:11.403972Z","iopub.execute_input":"2022-08-13T12:49:11.404401Z","iopub.status.idle":"2022-08-13T12:49:11.438533Z","shell.execute_reply.started":"2022-08-13T12:49:11.404366Z","shell.execute_reply":"2022-08-13T12:49:11.437415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df.isna().sum() / len(df)) * 100","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:49:12.213544Z","iopub.execute_input":"2022-08-13T12:49:12.214033Z","iopub.status.idle":"2022-08-13T12:49:12.227304Z","shell.execute_reply.started":"2022-08-13T12:49:12.213990Z","shell.execute_reply":"2022-08-13T12:49:12.225870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns='Cabin', inplace=True)\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:49:22.697738Z","iopub.execute_input":"2022-08-13T12:49:22.698316Z","iopub.status.idle":"2022-08-13T12:49:22.718479Z","shell.execute_reply.started":"2022-08-13T12:49:22.698262Z","shell.execute_reply":"2022-08-13T12:49:22.717600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'] = df['Age'].fillna(np.mean(df['Age']))\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:51:30.246137Z","iopub.execute_input":"2022-08-13T12:51:30.246540Z","iopub.status.idle":"2022-08-13T12:51:30.262520Z","shell.execute_reply.started":"2022-08-13T12:51:30.246507Z","shell.execute_reply":"2022-08-13T12:51:30.261498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna()\ndf.info()\n# 'None', 'Unknown'\n# value_counts() -> mana yang banyak dari kategorinya, itu yang dipilih\n# divide datanya, berdasarkan kategorinya\n# Clustering -> pseudo-labelling","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:52:07.763436Z","iopub.execute_input":"2022-08-13T12:52:07.763844Z","iopub.status.idle":"2022-08-13T12:52:07.781425Z","shell.execute_reply.started":"2022-08-13T12:52:07.763811Z","shell.execute_reply":"2022-08-13T12:52:07.780149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:54:52.975886Z","iopub.execute_input":"2022-08-13T12:54:52.976685Z","iopub.status.idle":"2022-08-13T12:54:52.996676Z","shell.execute_reply.started":"2022-08-13T12:54:52.976642Z","shell.execute_reply":"2022-08-13T12:54:52.995749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:55:01.952590Z","iopub.execute_input":"2022-08-13T12:55:01.953005Z","iopub.status.idle":"2022-08-13T12:55:01.991145Z","shell.execute_reply.started":"2022-08-13T12:55:01.952975Z","shell.execute_reply":"2022-08-13T12:55:01.990216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Survived'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:56:14.998240Z","iopub.execute_input":"2022-08-13T12:56:14.998662Z","iopub.status.idle":"2022-08-13T12:56:15.006685Z","shell.execute_reply.started":"2022-08-13T12:56:14.998628Z","shell.execute_reply":"2022-08-13T12:56:15.005728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns='Name', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:56:59.609956Z","iopub.execute_input":"2022-08-13T12:56:59.610381Z","iopub.status.idle":"2022-08-13T12:56:59.617046Z","shell.execute_reply.started":"2022-08-13T12:56:59.610345Z","shell.execute_reply":"2022-08-13T12:56:59.616157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.set_index('PassengerId')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T12:57:31.398297Z","iopub.execute_input":"2022-08-13T12:57:31.399068Z","iopub.status.idle":"2022-08-13T12:57:31.414800Z","shell.execute_reply.started":"2022-08-13T12:57:31.399030Z","shell.execute_reply":"2022-08-13T12:57:31.413947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns='Ticket', inplace=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:06:29.073530Z","iopub.execute_input":"2022-08-13T13:06:29.074116Z","iopub.status.idle":"2022-08-13T13:06:29.095591Z","shell.execute_reply.started":"2022-08-13T13:06:29.074048Z","shell.execute_reply":"2022-08-13T13:06:29.094480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ndf['Sex'] = le.fit_transform(df['Sex'])\ndf['Embarked'] = le.fit_transform(df['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:08:30.647187Z","iopub.execute_input":"2022-08-13T13:08:30.647955Z","iopub.status.idle":"2022-08-13T13:08:30.688359Z","shell.execute_reply.started":"2022-08-13T13:08:30.647914Z","shell.execute_reply":"2022-08-13T13:08:30.687404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:08:35.700587Z","iopub.execute_input":"2022-08-13T13:08:35.701007Z","iopub.status.idle":"2022-08-13T13:08:35.716380Z","shell.execute_reply.started":"2022-08-13T13:08:35.700971Z","shell.execute_reply":"2022-08-13T13:08:35.715168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"y = mx + c\n\ny (target prediksi)\n\nm (gradient)\n\nx (prediktor)\n\nc (konstanta)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:09:17.896551Z","iopub.execute_input":"2022-08-13T13:09:17.896960Z","iopub.status.idle":"2022-08-13T13:09:17.946757Z","shell.execute_reply.started":"2022-08-13T13:09:17.896926Z","shell.execute_reply":"2022-08-13T13:09:17.945480Z"}}},{"cell_type":"code","source":"y = df['Survived']\nx = df.drop(columns='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:11:52.813683Z","iopub.execute_input":"2022-08-13T13:11:52.814073Z","iopub.status.idle":"2022-08-13T13:11:52.821367Z","shell.execute_reply.started":"2022-08-13T13:11:52.814040Z","shell.execute_reply":"2022-08-13T13:11:52.820202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nx_scaled = scaler.fit_transform(x)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:27:27.283686Z","iopub.execute_input":"2022-08-13T13:27:27.284100Z","iopub.status.idle":"2022-08-13T13:27:27.294295Z","shell.execute_reply.started":"2022-08-13T13:27:27.284051Z","shell.execute_reply":"2022-08-13T13:27:27.293163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:11:56.998766Z","iopub.execute_input":"2022-08-13T13:11:56.999226Z","iopub.status.idle":"2022-08-13T13:11:57.008946Z","shell.execute_reply.started":"2022-08-13T13:11:56.999180Z","shell.execute_reply":"2022-08-13T13:11:57.007967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:12:05.490150Z","iopub.execute_input":"2022-08-13T13:12:05.490551Z","iopub.status.idle":"2022-08-13T13:12:05.505574Z","shell.execute_reply.started":"2022-08-13T13:12:05.490520Z","shell.execute_reply":"2022-08-13T13:12:05.504541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split(x_scaled, y, random_state=42, test_size=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:35:37.863407Z","iopub.execute_input":"2022-08-13T13:35:37.863804Z","iopub.status.idle":"2022-08-13T13:35:37.871427Z","shell.execute_reply.started":"2022-08-13T13:35:37.863773Z","shell.execute_reply":"2022-08-13T13:35:37.870206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nmodel = LogisticRegression(penalty='elasticnet', max_iter=10000, solver='saga', l1_ratio=0.2)\nmodel.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:35:38.286963Z","iopub.execute_input":"2022-08-13T13:35:38.287374Z","iopub.status.idle":"2022-08-13T13:35:38.299144Z","shell.execute_reply.started":"2022-08-13T13:35:38.287341Z","shell.execute_reply":"2022-08-13T13:35:38.298142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ny_pred = model.predict(x_test)\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:35:38.812560Z","iopub.execute_input":"2022-08-13T13:35:38.813400Z","iopub.status.idle":"2022-08-13T13:35:38.824018Z","shell.execute_reply.started":"2022-08-13T13:35:38.813362Z","shell.execute_reply":"2022-08-13T13:35:38.822637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}