{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:49.497079Z","iopub.execute_input":"2025-04-23T19:40:49.497770Z","iopub.status.idle":"2025-04-23T19:40:50.660355Z","shell.execute_reply.started":"2025-04-23T19:40:49.497741Z","shell.execute_reply":"2025-04-23T19:40:50.659265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/nexar-collision-prediction/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/nexar-collision-prediction/test.csv\")\nsample= pd.read_csv(\"/kaggle/input/nexar-collision-prediction/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.661925Z","iopub.execute_input":"2025-04-23T19:40:50.662445Z","iopub.status.idle":"2025-04-23T19:40:50.679314Z","shell.execute_reply.started":"2025-04-23T19:40:50.662422Z","shell.execute_reply":"2025-04-23T19:40:50.678587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# preprocessing","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.680095Z","iopub.execute_input":"2025-04-23T19:40:50.680350Z","iopub.status.idle":"2025-04-23T19:40:50.690546Z","shell.execute_reply.started":"2025-04-23T19:40:50.680331Z","shell.execute_reply":"2025-04-23T19:40:50.689633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.692757Z","iopub.execute_input":"2025-04-23T19:40:50.693700Z","iopub.status.idle":"2025-04-23T19:40:50.709828Z","shell.execute_reply.started":"2025-04-23T19:40:50.693672Z","shell.execute_reply":"2025-04-23T19:40:50.708847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.710682Z","iopub.execute_input":"2025-04-23T19:40:50.710954Z","iopub.status.idle":"2025-04-23T19:40:50.739310Z","shell.execute_reply.started":"2025-04-23T19:40:50.710934Z","shell.execute_reply":"2025-04-23T19:40:50.738442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.740334Z","iopub.execute_input":"2025-04-23T19:40:50.740709Z","iopub.status.idle":"2025-04-23T19:40:50.748421Z","shell.execute_reply.started":"2025-04-23T19:40:50.740682Z","shell.execute_reply":"2025-04-23T19:40:50.747544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nimport pandas as pd\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.749730Z","iopub.execute_input":"2025-04-23T19:40:50.749964Z","iopub.status.idle":"2025-04-23T19:40:50.763306Z","shell.execute_reply.started":"2025-04-23T19:40:50.749946Z","shell.execute_reply":"2025-04-23T19:40:50.762365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = train['target']\nX = train.drop(columns=['target'])\ntest_ids = test['id']\n\n# Combine train and test for consistent preprocessing\ncombined = pd.concat([X, test.drop(columns=['id'])], axis=0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.764458Z","iopub.execute_input":"2025-04-23T19:40:50.764899Z","iopub.status.idle":"2025-04-23T19:40:50.782614Z","shell.execute_reply.started":"2025-04-23T19:40:50.764870Z","shell.execute_reply":"2025-04-23T19:40:50.781717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Handle missing values FIRST","metadata":{}},{"cell_type":"code","source":"# Fill numeric NaNs with median\nnumeric_cols = combined.select_dtypes(include=['int64', 'float64']).columns\nif len(numeric_cols) > 0:\n    num_imputer = SimpleImputer(strategy='median')\n    combined[numeric_cols] = num_imputer.fit_transform(combined[numeric_cols])\n\n# Fill categorical NaNs with 'MISSING' and then encode\ncategorical_cols = combined.select_dtypes(include=['object', 'category']).columns\nif len(categorical_cols) > 0:\n    combined[categorical_cols] = combined[categorical_cols].fillna('MISSING')\n    for col in categorical_cols:\n        le = LabelEncoder()\n        combined[col] = le.fit_transform(combined[col].astype(str))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.783625Z","iopub.execute_input":"2025-04-23T19:40:50.783885Z","iopub.status.idle":"2025-04-23T19:40:50.809846Z","shell.execute_reply.started":"2025-04-23T19:40:50.783865Z","shell.execute_reply":"2025-04-23T19:40:50.809070Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Verify NO NaN values remain","metadata":{}},{"cell_type":"code","source":"print(\"Missing values after preprocessing:\")\nprint(combined.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.812899Z","iopub.execute_input":"2025-04-23T19:40:50.813169Z","iopub.status.idle":"2025-04-23T19:40:50.827914Z","shell.execute_reply.started":"2025-04-23T19:40:50.813149Z","shell.execute_reply":"2025-04-23T19:40:50.827175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Split back into train and test sets","metadata":{}},{"cell_type":"code","source":"X_processed = combined.iloc[:len(X), :]\ntest_processed = combined.iloc[len(X):, :]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.828911Z","iopub.execute_input":"2025-04-23T19:40:50.829245Z","iopub.status.idle":"2025-04-23T19:40:50.844341Z","shell.execute_reply.started":"2025-04-23T19:40:50.829217Z","shell.execute_reply":"2025-04-23T19:40:50.843412Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Train-test split","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.845473Z","iopub.execute_input":"2025-04-23T19:40:50.845820Z","iopub.status.idle":"2025-04-23T19:40:50.861410Z","shell.execute_reply.started":"2025-04-23T19:40:50.845798Z","shell.execute_reply":"2025-04-23T19:40:50.860551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Train model - with additional NaN check","metadata":{}},{"cell_type":"code","source":"print(\"\\nFinal check before training:\")\nprint(\"NaN in X_train:\", X_train.isna().sum().sum())\nprint(\"NaN in y_train:\", y_train.isna().sum())\n\n# If there are still NaN values, we'll drop those rows\nif X_train.isna().sum().sum() > 0 or y_train.isna().sum() > 0:\n    print(\"\\nWarning: Dropping rows with remaining NaN values\")\n    non_nan_mask = ~X_train.isna().any(axis=1) & ~y_train.isna()\n    X_train = X_train[non_nan_mask]\n    y_train = y_train[non_nan_mask]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.862412Z","iopub.execute_input":"2025-04-23T19:40:50.863264Z","iopub.status.idle":"2025-04-23T19:40:50.880882Z","shell.execute_reply.started":"2025-04-23T19:40:50.863230Z","shell.execute_reply":"2025-04-23T19:40:50.879911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestClassifier(random_state=42)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:50.881824Z","iopub.execute_input":"2025-04-23T19:40:50.882118Z","iopub.status.idle":"2025-04-23T19:40:51.058400Z","shell.execute_reply.started":"2025-04-23T19:40:50.882084Z","shell.execute_reply":"2025-04-23T19:40:51.057531Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Evaluation","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(X_val)\nprint(f\"\\nValidation Accuracy: {accuracy_score(y_val, y_pred):.4f}\")\nprint(classification_report(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:51.059394Z","iopub.execute_input":"2025-04-23T19:40:51.059703Z","iopub.status.idle":"2025-04-23T19:40:51.084366Z","shell.execute_reply.started":"2025-04-23T19:40:51.059672Z","shell.execute_reply":"2025-04-23T19:40:51.083458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Predict on test set","metadata":{}},{"cell_type":"code","source":"test_predictions = model.predict(test_processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:51.085416Z","iopub.execute_input":"2025-04-23T19:40:51.085676Z","iopub.status.idle":"2025-04-23T19:40:51.106411Z","shell.execute_reply.started":"2025-04-23T19:40:51.085655Z","shell.execute_reply":"2025-04-23T19:40:51.105285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Create submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test_ids,\n    'target': test_predictions\n})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"\\n✅ Submission file created: submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T19:40:51.107321Z","iopub.execute_input":"2025-04-23T19:40:51.107668Z","iopub.status.idle":"2025-04-23T19:40:51.116941Z","shell.execute_reply.started":"2025-04-23T19:40:51.107628Z","shell.execute_reply":"2025-04-23T19:40:51.115968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}