{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:36:05.924706Z","iopub.execute_input":"2025-07-03T19:36:05.924894Z","iopub.status.idle":"2025-07-03T19:36:09.190108Z","shell.execute_reply.started":"2025-07-03T19:36:05.924872Z","shell.execute_reply":"2025-07-03T19:36:09.183837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Setup and Imports\n!pip install lightgbm pyarrow fastparquet -q\n\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\nprint(\"✅ Ready\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:36:09.191168Z","iopub.execute_input":"2025-07-03T19:36:09.191499Z","iopub.status.idle":"2025-07-03T19:36:16.487203Z","shell.execute_reply.started":"2025-07-03T19:36:09.191475Z","shell.execute_reply":"2025-07-03T19:36:16.480264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Data\n\ntry:\n    train_df = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/train.parquet\")\n    test_df = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/test.parquet\")\n    sample_submission_df = pd.read_parquet(\"/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet\")\n    print(\"✅ Data modified successfully\")\nexcept FileNotFoundError:\n    print(\"Error\")\n    # Create dummy dataframes to prevent further errors\n    train_df = pd.DataFrame()\n    test_df = pd.DataFrame()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:36:16.490314Z","iopub.execute_input":"2025-07-03T19:36:16.490693Z","iopub.status.idle":"2025-07-03T19:36:50.105569Z","shell.execute_reply.started":"2025-07-03T19:36:16.490667Z","shell.execute_reply":"2025-07-03T19:36:50.100949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing for Dates and Categories\n\nif not train_df.empty:\n    # --- NEW: Handle Date/Time Columns ---\n    datetime_features = train_df.select_dtypes(include=['datetime64']).columns.tolist()\n    \n    if datetime_features:\n        print(f\"Found datetime features: {datetime_features}\")\n        for col in datetime_features:\n            # Extract useful numerical features from the date\n            train_df[f'{col}_year'] = train_df[col].dt.year\n            train_df[f'{col}_month'] = train_df[col].dt.month\n            train_df[f'{col}_day'] = train_df[col].dt.day\n            train_df[f'{col}_dayofweek'] = train_df[col].dt.dayofweek # Monday=0, Sunday=6\n            \n            test_df[f'{col}_year'] = test_df[col].dt.year\n            test_df[f'{col}_month'] = test_df[col].dt.month\n            test_df[f'{col}_day'] = test_df[col].dt.day\n            test_df[f'{col}_dayofweek'] = test_df[col].dt.dayofweek\n            \n            # Drop the original datetime column\n            train_df = train_df.drop(col, axis=1)\n            test_df = test_df.drop(col, axis=1)\n        print(\"✅ Converted datetime features to numerical format.\")\n    \n    # --- EXISTING: Handle Categorical Columns ---\n    categorical_features = train_df.select_dtypes(include=['object']).columns.tolist()\n\n    if categorical_features:\n        print(f\"Found categorical features: {categorical_features}\")\n        for col in categorical_features:\n            le = LabelEncoder()\n            # Combine train and test data to ensure all categories are learned\n            combined_data = pd.concat([train_df[col], test_df[col]]).astype(str)\n            le.fit(combined_data)\n            train_df[col] = le.transform(train_df[col].astype(str))\n            test_df[col] = le.transform(test_df[col].astype(str))\n        print(\"✅ Converted categorical features to numerical format.\")\n        \n    print(\"\\nPreprocessing complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:36:50.108180Z","iopub.execute_input":"2025-07-03T19:36:50.108459Z","iopub.status.idle":"2025-07-03T19:54:00.806401Z","shell.execute_reply.started":"2025-07-03T19:36:50.108433Z","shell.execute_reply":"2025-07-03T19:54:00.801530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Training\n\nif not train_df.empty:\n    # Define features (X) and target (y)\n    features = [col for col in train_df.columns if col not in ['ranker_id', 'flight_id', 'selected']]\n    X_train = train_df[features]\n    y_train = train_df['selected']\n\n    # Initialize and train the LightGBM Classifier\n    lgbm_ranker = lgb.LGBMClassifier(\n        objective='binary',\n        metric='logloss',\n        n_estimators=1000,\n        learning_rate=0.05,\n        num_leaves=31,\n        random_state=42,\n        n_jobs=-1\n    )\n\n    print(\"🚀 ...Start model training\")\n    lgbm_ranker.fit(X_train, y_train)\n    print(\"✅ Model training completed\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:54:00.808984Z","iopub.execute_input":"2025-07-03T19:54:00.809240Z","iopub.status.idle":"2025-07-03T19:58:50.634201Z","shell.execute_reply.started":"2025-07-03T19:54:00.809216Z","shell.execute_reply":"2025-07-03T19:58:50.628021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediction and Ranking\n\nif not test_df.empty:\n    X_test = test_df[features]\n\n    # Predict the probability of being 'selected' (class 1)\n    probabilities = lgbm_ranker.predict_proba(X_test)[:, 1]\n\n    # Add the prediction scores to the test dataframe\n    test_df['score'] = probabilities\n\n    # Calculate the rank within each group based on the score\n    test_df['rank'] = test_df.groupby('ranker_id')['score'].rank(method='first', ascending=False).astype(int)\n    \n    print(\"✅ The results were successfully predicted and arranged\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:58:50.636194Z","iopub.execute_input":"2025-07-03T19:58:50.636461Z","iopub.status.idle":"2025-07-03T19:59:05.547603Z","shell.execute_reply.started":"2025-07-03T19:58:50.636436Z","shell.execute_reply":"2025-07-03T19:59:05.542221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if the test_df is empty\nprint(f\"Is test_df empty? {test_df.empty}\")\nprint(f\"Number of rows in test_df: {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T19:59:05.550362Z","iopub.execute_input":"2025-07-03T19:59:05.550692Z","iopub.status.idle":"2025-07-03T19:59:05.560822Z","shell.execute_reply.started":"2025-07-03T19:59:05.550664Z","shell.execute_reply":"2025-07-03T19:59:05.555581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# --- Step 1: Load Data using the correct function and path ---\n# This is the corrected line for your Kaggle Notebook.\n# We use read_parquet because the file ends with .parquet.\ntry:\n    path_to_file = \"/kaggle/input/aeroclub-recsys-2025/test.parquet\"\n    test_df = pd.read_parquet(path_to_file)\n    print(\"Test data loaded successfully from Parquet file!\")\n    print(\"Test data shape:\", test_df.shape)\nexcept Exception as e:\n    print(f\"An error occurred: {e}\")\n    print(\"Please double-check the file path and that the file exists.\")\n\n\n# --- Step 2: Generate your predictions ---\n# (Your model's code goes here)\n# For demonstration, we will create dummy predictions.\n# Ensure the number of predictions matches the number of rows in test_df\nif 'test_df' in locals():\n    # Make sure to use your actual model's prediction logic here\n    model_predictions = np.random.randint(0, 2, size=len(test_df))\n    predictions = model_predictions\n\n\n# --- Step 3: Create the submission file ---\n# This part will now work with your actual test data\nif 'predictions' in locals():\n    print(\"Creating submission file...\")\n\n    submission_df = pd.DataFrame({\n        \"Id\": test_df[\"Id\"], # Make sure your parquet file has a column named 'Id'\n        \"Predicted\": predictions\n    })\n\n    # The submission file is usually expected in CSV format\n    submission_df.to_csv(\"submission.csv\", index=False)\n    \n    print(\"Submission file 'submission.csv' created successfully!\")\n    print(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T20:19:04.023605Z","iopub.execute_input":"2025-07-03T20:19:04.023978Z","iopub.status.idle":"2025-07-03T20:19:16.885310Z","shell.execute_reply.started":"2025-07-03T20:19:04.023951Z","shell.execute_reply":"2025-07-03T20:19:16.879587Z"}},"outputs":[],"execution_count":null}]}