{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":105399,"databundleVersionId":12733338}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv\nimport math\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:12:26.341927Z","iopub.execute_input":"2025-09-01T01:12:26.343534Z","iopub.status.idle":"2025-09-01T01:12:27.783323Z","shell.execute_reply.started":"2025-09-01T01:12:26.343487Z","shell.execute_reply":"2025-09-01T01:12:27.782072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:12:27.785104Z","iopub.execute_input":"2025-09-01T01:12:27.785544Z","iopub.status.idle":"2025-09-01T01:12:27.791339Z","shell.execute_reply.started":"2025-09-01T01:12:27.785520Z","shell.execute_reply":"2025-09-01T01:12:27.790064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import dask.dataframe as dd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:12:27.792499Z","iopub.execute_input":"2025-09-01T01:12:27.793412Z","iopub.status.idle":"2025-09-01T01:12:32.693328Z","shell.execute_reply.started":"2025-09-01T01:12:27.793371Z","shell.execute_reply":"2025-09-01T01:12:32.691891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = dd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:12:32.695256Z","iopub.execute_input":"2025-09-01T01:12:32.695849Z","iopub.status.idle":"2025-09-01T01:12:32.849128Z","shell.execute_reply.started":"2025-09-01T01:12:32.695815Z","shell.execute_reply":"2025-09-01T01:12:32.848007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:12:32.852361Z","iopub.execute_input":"2025-09-01T01:12:32.852680Z","iopub.status.idle":"2025-09-01T01:13:07.987890Z","shell.execute_reply.started":"2025-09-01T01:12:32.852659Z","shell.execute_reply":"2025-09-01T01:13:07.985573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape[0].compute() #check rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:13:07.989237Z","iopub.execute_input":"2025-09-01T01:13:07.989992Z","iopub.status.idle":"2025-09-01T01:13:08.052530Z","shell.execute_reply.started":"2025-09-01T01:13:07.989935Z","shell.execute_reply":"2025-09-01T01:13:08.050841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape[1] #check columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:13:08.053487Z","iopub.execute_input":"2025-09-01T01:13:08.053895Z","iopub.status.idle":"2025-09-01T01:13:08.110717Z","shell.execute_reply.started":"2025-09-01T01:13:08.053860Z","shell.execute_reply":"2025-09-01T01:13:08.108270Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Data Cleaning\n\n1. create 3 datafames - no_null, nulls_less_50, nulls_more_50","metadata":{}},{"cell_type":"code","source":"def divide_df(df,chunk_size,threshold_ratio):\n    #get total number of rows\n    total_rows = df.shape[0].compute()\n\n    #create threshold\n    threshold = total_rows * threshold_ratio\n\n    #create number of times/chunks \n    chunk_number = math.ceil(len(df.columns)/chunk_size)\n\n    #create 3 lists to store each columns\n    cols_1 = []  #to store df with no null cvalues\n    cols_2 = []  #to store cols with less than 50% null values\n    cols_3 = []  #store cols with  more than 50% null values\n \n    for i in range(chunk_number):\n        #select null columns in each column chunks\n        null_cols = df.iloc[:,i*chunk_size:(i+1)*chunk_size].isnull().sum().compute()\n        \n        list1 = null_cols[null_cols == 0].index.tolist()\n        list2 = null_cols[(null_cols > 0) & (null_cols <= threshold)].index.tolist()\n        list3 = null_cols[null_cols > threshold].index.tolist()\n        \n        cols_1.extend(list1)\n        cols_2.extend(list2)\n        cols_3.extend(list3)\n\n\n    return cols_1,cols_2,cols_3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:13:08.112222Z","iopub.execute_input":"2025-09-01T01:13:08.113253Z","iopub.status.idle":"2025-09-01T01:13:08.130223Z","shell.execute_reply.started":"2025-09-01T01:13:08.113201Z","shell.execute_reply":"2025-09-01T01:13:08.127138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"no_nulls,less_50,more_50 = divide_df(train,30,0.5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:13:08.132604Z","iopub.execute_input":"2025-09-01T01:13:08.133515Z","iopub.status.idle":"2025-09-01T01:14:17.916592Z","shell.execute_reply.started":"2025-09-01T01:13:08.133470Z","shell.execute_reply":"2025-09-01T01:14:17.915293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1 = train[no_nulls]  #df with no null values\ndf2 = train[less_50]  #df with null values less than 50%\ndf3 = train[more_50]  #df with null values more fhan 50%","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:17.918098Z","iopub.execute_input":"2025-09-01T01:14:17.918397Z","iopub.status.idle":"2025-09-01T01:14:17.930062Z","shell.execute_reply.started":"2025-09-01T01:14:17.918377Z","shell.execute_reply":"2025-09-01T01:14:17.928427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### initial observation\n\n- 23 columns without missing values\n- 27columns with less than 50% missing values\n- 76 columns with more than 50% missing values","metadata":{}},{"cell_type":"code","source":"train_copy = train.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:17.931270Z","iopub.execute_input":"2025-09-01T01:14:17.931644Z","iopub.status.idle":"2025-09-01T01:14:17.961610Z","shell.execute_reply.started":"2025-09-01T01:14:17.931614Z","shell.execute_reply":"2025-09-01T01:14:17.960351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2['has_leg1'] = df2['legs1_arrivalAt'].notnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:17.963285Z","iopub.execute_input":"2025-09-01T01:14:17.963717Z","iopub.status.idle":"2025-09-01T01:14:17.999605Z","shell.execute_reply.started":"2025-09-01T01:14:17.963678Z","shell.execute_reply":"2025-09-01T01:14:17.998257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df2 = df2[['legs0_segments0_aircraft_code','legs0_segments0_arrivalTo_airport_city_iata',\n     'legs0_segments0_arrivalTo_airport_iata',\n     'legs0_segments0_baggageAllowance_quantity',\n     'legs0_segments0_baggageAllowance_weightMeasurementType',\n     'legs0_segments0_departureFrom_airport_iata',\n     'miniRules0_monetaryAmount','miniRules0_statusInfos',\n     'miniRules1_monetaryAmount','miniRules1_statusInfos','pricingInfo_isAccessTP',\n     'has_leg1']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.001284Z","iopub.execute_input":"2025-09-01T01:14:18.001671Z","iopub.status.idle":"2025-09-01T01:14:18.015210Z","shell.execute_reply.started":"2025-09-01T01:14:18.001633Z","shell.execute_reply":"2025-09-01T01:14:18.014043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg0_segment1'] = df3[\n'legs0_segments1_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.021110Z","iopub.execute_input":"2025-09-01T01:14:18.022568Z","iopub.status.idle":"2025-09-01T01:14:18.047042Z","shell.execute_reply.started":"2025-09-01T01:14:18.022518Z","shell.execute_reply":"2025-09-01T01:14:18.045722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg0_segment2'] = df3[\n'legs0_segments2_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.048296Z","iopub.execute_input":"2025-09-01T01:14:18.048599Z","iopub.status.idle":"2025-09-01T01:14:18.084011Z","shell.execute_reply.started":"2025-09-01T01:14:18.048575Z","shell.execute_reply":"2025-09-01T01:14:18.082082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg0_segment3'] = df3[\n'legs0_segments3_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.085537Z","iopub.execute_input":"2025-09-01T01:14:18.086248Z","iopub.status.idle":"2025-09-01T01:14:18.122414Z","shell.execute_reply.started":"2025-09-01T01:14:18.086216Z","shell.execute_reply":"2025-09-01T01:14:18.120916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg1_segment1'] = df3[\n'legs1_segments1_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.123652Z","iopub.execute_input":"2025-09-01T01:14:18.124402Z","iopub.status.idle":"2025-09-01T01:14:18.160142Z","shell.execute_reply.started":"2025-09-01T01:14:18.124369Z","shell.execute_reply":"2025-09-01T01:14:18.158316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg1_segment2'] = df3[\n'legs1_segments2_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.162161Z","iopub.execute_input":"2025-09-01T01:14:18.162669Z","iopub.status.idle":"2025-09-01T01:14:18.189347Z","shell.execute_reply.started":"2025-09-01T01:14:18.162626Z","shell.execute_reply":"2025-09-01T01:14:18.188244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3['has_leg1_segment3'] = df3[\n'legs1_segments3_aircraft_code'].isnull().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.190629Z","iopub.execute_input":"2025-09-01T01:14:18.191170Z","iopub.status.idle":"2025-09-01T01:14:18.220211Z","shell.execute_reply.started":"2025-09-01T01:14:18.191138Z","shell.execute_reply":"2025-09-01T01:14:18.218992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df3 = df3[['has_leg0_segment1','has_leg0_segment2',\n               'has_leg0_segment3','has_leg1_segment1',\n               'has_leg1_segment2','has_leg1_segment3']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.221120Z","iopub.execute_input":"2025-09-01T01:14:18.221403Z","iopub.status.idle":"2025-09-01T01:14:18.250007Z","shell.execute_reply.started":"2025-09-01T01:14:18.221375Z","shell.execute_reply":"2025-09-01T01:14:18.248478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy = dd.concat([df1,new_df2,new_df3],axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.251252Z","iopub.execute_input":"2025-09-01T01:14:18.251568Z","iopub.status.idle":"2025-09-01T01:14:18.296324Z","shell.execute_reply.started":"2025-09-01T01:14:18.251545Z","shell.execute_reply":"2025-09-01T01:14:18.295096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['miniRules0_statusInfos'] = train_copy['miniRules0_statusInfos'].where(\n    train_copy['miniRules0_monetaryAmount'] != 0, \n    0.0\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.297728Z","iopub.execute_input":"2025-09-01T01:14:18.298247Z","iopub.status.idle":"2025-09-01T01:14:18.310356Z","shell.execute_reply.started":"2025-09-01T01:14:18.298211Z","shell.execute_reply":"2025-09-01T01:14:18.308899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['miniRules1_statusInfos'] = train_copy['miniRules1_statusInfos'].where(\n    train_copy['miniRules1_monetaryAmount'] != 0, \n    0.0\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.311721Z","iopub.execute_input":"2025-09-01T01:14:18.312434Z","iopub.status.idle":"2025-09-01T01:14:18.351618Z","shell.execute_reply.started":"2025-09-01T01:14:18.312403Z","shell.execute_reply":"2025-09-01T01:14:18.350212Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_copy = train_copy.drop('legs0_segments0_arrivalTo_airport_city_iata',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.352907Z","iopub.execute_input":"2025-09-01T01:14:18.353292Z","iopub.status.idle":"2025-09-01T01:14:18.388306Z","shell.execute_reply.started":"2025-09-01T01:14:18.353264Z","shell.execute_reply":"2025-09-01T01:14:18.386885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy = train_copy.dropna(subset = [\n    'legs0_segments0_arrivalTo_airport_iata',\n'legs0_segments0_departureFrom_airport_iata',\n'legs0_segments0_aircraft_code']) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.389385Z","iopub.execute_input":"2025-09-01T01:14:18.389686Z","iopub.status.idle":"2025-09-01T01:14:18.418328Z","shell.execute_reply.started":"2025-09-01T01:14:18.389656Z","shell.execute_reply":"2025-09-01T01:14:18.416924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['pricingInfo_passengerCount'].value_counts().compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:18.419452Z","iopub.execute_input":"2025-09-01T01:14:18.419730Z","iopub.status.idle":"2025-09-01T01:14:23.341480Z","shell.execute_reply.started":"2025-09-01T01:14:18.419709Z","shell.execute_reply":"2025-09-01T01:14:23.340395Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I am dropping this column because it has just one value, so it is irrelevant to training a model","metadata":{}},{"cell_type":"code","source":"train_copy = train_copy.drop('pricingInfo_passengerCount',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.343819Z","iopub.execute_input":"2025-09-01T01:14:23.344455Z","iopub.status.idle":"2025-09-01T01:14:23.350443Z","shell.execute_reply.started":"2025-09-01T01:14:23.344423Z","shell.execute_reply":"2025-09-01T01:14:23.349438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask = {True:1,False:0}\n\ntrain_copy['bySelf'] = train_copy['bySelf'].replace(mask).astype(int)\ntrain_copy['isAccess3D'] = train_copy['isAccess3D'].replace(mask).astype(int)\ntrain_copy['isVip'] = train_copy['isVip'].replace(mask).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.351216Z","iopub.execute_input":"2025-09-01T01:14:23.351709Z","iopub.status.idle":"2025-09-01T01:14:23.388613Z","shell.execute_reply.started":"2025-09-01T01:14:23.351656Z","shell.execute_reply":"2025-09-01T01:14:23.387447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['legs0_segments0_flightNumber'] = train_copy[\n'legs0_segments0_flightNumber'].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.389717Z","iopub.execute_input":"2025-09-01T01:14:23.390156Z","iopub.status.idle":"2025-09-01T01:14:23.415599Z","shell.execute_reply.started":"2025-09-01T01:14:23.390123Z","shell.execute_reply":"2025-09-01T01:14:23.414195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I am choosing to drop the `sex` column, because it is not really descriptive. there is no way to tell from the available information which one is male and which one is female","metadata":{}},{"cell_type":"code","source":"train_copy = train_copy.drop('sex',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.417053Z","iopub.execute_input":"2025-09-01T01:14:23.417381Z","iopub.status.idle":"2025-09-01T01:14:23.441638Z","shell.execute_reply.started":"2025-09-01T01:14:23.417346Z","shell.execute_reply":"2025-09-01T01:14:23.440407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How to check if a categorical column is strongly predictive of the target column?\n\n**1. Look at Target Distribution per Category**\n\nCompute the mean target rate (selected) for each category.\n\nExample in pandas:","metadata":{}},{"cell_type":"code","source":"#df.groupby('departure_iata')['selected'].mean().sort_values()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.443203Z","iopub.execute_input":"2025-09-01T01:14:23.443547Z","iopub.status.idle":"2025-09-01T01:14:23.466163Z","shell.execute_reply.started":"2025-09-01T01:14:23.443524Z","shell.execute_reply":"2025-09-01T01:14:23.464524Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"If some airports have much higher or lower booking rates than the global average, it means the category is predictive.\n\nSuppose overall selection rate = 0.3 (30%)\n\n* If JFK = 0.75 and LHR = 0.05 → very predictive.\n\n* If all airports hover around 0.28–0.32 → not predictive.\n\n**2. Mutual Information (MI)**\n\nMeasures how much knowing the category reduces uncertainty about the target.","metadata":{}},{"cell_type":"code","source":"#from sklearn.feature_selection import mutual_info_classif\n#from sklearn.preprocessing import LabelEncoder\n\n#X = df[['departure_iata']].astype(str)  # categorical\n#y = df['selected']\n\n#le = LabelEncoder()\n#X_encoded = le.fit_transform(X['departure_iata']).reshape(-1,1)\n\n#mi = mutual_info_classif(X_encoded, y, discrete_features=True)\n#print(\"Mutual Information:\", mi[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.467430Z","iopub.execute_input":"2025-09-01T01:14:23.467916Z","iopub.status.idle":"2025-09-01T01:14:23.493438Z","shell.execute_reply.started":"2025-09-01T01:14:23.467890Z","shell.execute_reply":"2025-09-01T01:14:23.491865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Higher MI → stronger predictive power.\n\n* Near zero → category isn’t informative.\n\n**3. Chi-Square Test (for classification targets)**\n\nChecks if the distribution of the target depends on the category.","metadata":{}},{"cell_type":"code","source":"#import pandas as pd\n#from scipy.stats import chi2_contingency\n\n#contingency = pd.crosstab(df['departure_iata'], df['selected'])\n#chi2, p, _, _ = chi2_contingency(contingency)\n\n#print(\"Chi-square p-value:\", p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.494622Z","iopub.execute_input":"2025-09-01T01:14:23.495067Z","iopub.status.idle":"2025-09-01T01:14:23.526193Z","shell.execute_reply.started":"2025-09-01T01:14:23.495025Z","shell.execute_reply":"2025-09-01T01:14:23.524851Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* p < 0.05 → significant relationship between airport and booking.\n\n* p ≥ 0.05 → not really predictive.\n\n\n**4. Feature Importance (after encoding)**\n\nRun a quick tree-based model (like RandomForest or LightGBM) and check feature importance.\n\nIf the encoded IATA feature gets high importance, it’s predictive.\n\n**✅ Rule of thumb for your IATA codes:**\n\n* If most airports behave similarly → skip or just frequency encode.\n\n* If some airports behave very differently (e.g., a hub always gets selected) → target encode or group them.","metadata":{}},{"cell_type":"markdown","source":"## How to handle columns like airport code columns?\n\n**1. Target (Mean) Encoding**\n\n👉 Idea: Replace each category with the average value of the target for that category.\n\nExample:\n\nSuppose your target column is `selected` (1 = booked, 0 = not booked).\nAnd you have a column `departure_iata` with airports:\n\ndeparture_iata\t|selected|\n---|---|\nLOS|\t1\nLOS\t|0\nJFK\t|1\nLHR|\t0\nLHR|\t0\nDXB|\t1\n\n**Step 1: Compute the mean target per airport.**\n\n* LOS → (1 + 0)/2 = 0.5\n\n* JFK → (1)/1 = 1.0\n\n* LHR → (0 + 0)/2 = 0.0\n\n* DXB → (1)/1 = 1.0\n\n**Step 2: Replace airport code with that mean.**\n\ndeparture_iata|\tselected\t|target_encoded|\n--|---|---|\nLOS\t|1\t|0.5\nLOS\t|0\t|0.5\nJFK\t|1|\t1.0\nLHR\t|0|\t0.0\nLHR|\t0|\t0.0\nDXB\t|1|\t1.0\n\n**✅ Pros:**\n\n* Captures the relationship between the airport and the target directly.\n\n* Reduces dimensionality (just 1 column, no explosion).\n\n**⚠️ Cons:**\n\n* Risk of overfitting if some airports have very few samples (e.g., 1 route always booked → model thinks it’s always 100%).\n\n* Usually solved with smoothing (blend global mean with local mean).","metadata":{}},{"cell_type":"code","source":"#import category_encoders as ce\n\n#encoder = ce.TargetEncoder(cols=['legs0_segments0_departureFrom_airport_iata'])\n#df_encoded = encoder.fit_transform(df, df['selected'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.527400Z","iopub.execute_input":"2025-09-01T01:14:23.527897Z","iopub.status.idle":"2025-09-01T01:14:23.564639Z","shell.execute_reply.started":"2025-09-01T01:14:23.527868Z","shell.execute_reply":"2025-09-01T01:14:23.563251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**2. Frequency (Count) Encoding**\n\n👉 Idea: Replace each category with how often it appears in the dataset.\n\nUsing same data:\n\ndeparture_iata\t|selected\n---|---\nLOS\t|1\nLOS|\t0\nJFK|\t1\nLHR\t|0\nLHR|\t0\nDXB|\t1\n\n**Step 1: Count occurrences.**\n\n* LOS → 2\n\n* JFK → 1\n\n* LHR → 2\n\n* DXB → 1\n\n**Step 2: Replace airport code with its frequency.**\n\ndeparture_iata|\tselected\t|freq_encoded\n----|---|---\nLOS\t|1|\t2\nLOS\t|0|\t2\nJFK|\t1|\t1\nLHR\t|0\t|2\nLHR|\t0|\t2\nDXB|\t1|\t1\n\n**✅ Pros:**\n\n* Simple, fast, and safe.\n\n* Captures how “common” an airport is in your dataset.\n\n* Works well with tree-based models (they can split on frequency thresholds).\n\n**⚠️ Cons:**\n\n* Doesn’t directly capture relationship with target.\n\n* Just popularity measure (e.g., common airports vs rare ones).","metadata":{}},{"cell_type":"code","source":"#freq_map = df['legs0_segments0_departureFrom_airport_iata'].value_counts().to_dict()\n#df['iata_freq'] = df['legs0_segments0_departureFrom_airport_iata'].map(freq_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.566109Z","iopub.execute_input":"2025-09-01T01:14:23.566483Z","iopub.status.idle":"2025-09-01T01:14:23.597923Z","shell.execute_reply.started":"2025-09-01T01:14:23.566451Z","shell.execute_reply":"2025-09-01T01:14:23.596612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**🔑 When to use which:**\n\n* Target Encoding → great when the category is strongly predictive of the target (e.g., certain airports might almost always get selected).\n\n* Frequency Encoding → safe baseline when you just want to reduce dimensionality without leaking target info.\n\n**👉 Think of it like this:**\n\n* Target Encoding = category’s “success rate.”\n\n* Frequency Encoding = category’s “popularity.”","metadata":{}},{"cell_type":"code","source":"number_df = train_copy.select_dtypes(include='number')\nobject_df = train_copy.select_dtypes(exclude='number') #not just categories but datetimes as well","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.599186Z","iopub.execute_input":"2025-09-01T01:14:23.599610Z","iopub.status.idle":"2025-09-01T01:14:23.630435Z","shell.execute_reply.started":"2025-09-01T01:14:23.599559Z","shell.execute_reply":"2025-09-01T01:14:23.629239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train_copy['selected'].value_counts()/len(train_copy['selected'])).compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:14:23.631448Z","iopub.execute_input":"2025-09-01T01:14:23.632046Z","iopub.status.idle":"2025-09-01T01:15:06.024995Z","shell.execute_reply.started":"2025-09-01T01:14:23.632014Z","shell.execute_reply":"2025-09-01T01:15:06.023467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"iata_rates = train_copy.groupby(\n    'legs0_segments0_arrivalTo_airport_iata')[\n    'selected'].agg(['mean','count']).compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:06.026086Z","iopub.execute_input":"2025-09-01T01:15:06.026341Z","iopub.status.idle":"2025-09-01T01:15:11.752704Z","shell.execute_reply.started":"2025-09-01T01:15:06.026322Z","shell.execute_reply":"2025-09-01T01:15:11.751700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"iata_rates.sort_values(by='count',ascending=False).head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:11.754162Z","iopub.execute_input":"2025-09-01T01:15:11.754435Z","iopub.status.idle":"2025-09-01T01:15:11.769146Z","shell.execute_reply.started":"2025-09-01T01:15:11.754415Z","shell.execute_reply":"2025-09-01T01:15:11.767954Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 1. Global baseline\n\n* Global selection rate = 0.0058 (≈ 0.6%).\n\n* If a column is predictive, some groups should deviate meaningfully from this baseline.\n\n#### 2. What you found\n\n* The big airports with millions of samples (LED, SVO, VKO, AER, IST, DXB, etc.) all have means between 0.002–0.006, basically hugging the global average.\n→ That means for most of your dataset, this feature does not separate signal from noise.\n\n* The higher means you see (like TJM=0.023, PEE=0.009, CEK=0.014, etc.) come from smaller airports with lower counts (tens of thousands instead of millions).\n→ These might look “predictive,” but because they represent a small fraction of your total data, they may not generalize well.\n\n#### 3. The key insight\n\n* For a feature to be worth keeping, it should:\n\n    * show strong deviation from the global mean and\n\n    * cover a substantial portion of the dataset (not just tiny groups).\n\nIn your case:\n\n* High-count categories (millions of rows) are almost identical to the baseline → no signal.\n\n* Deviating categories are small (tens of thousands or less) → risk of overfitting.\n\n**So overall, this column isn’t giving much useful predictive power.**","metadata":{}},{"cell_type":"code","source":"depart_iata_rates = train_copy.groupby(\n'legs0_segments0_departureFrom_airport_iata')[\n    'selected'].agg(['mean','count']).compute()\n\ndepart_iata_rates.sort_values(by='count',ascending=False).head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:11.770356Z","iopub.execute_input":"2025-09-01T01:15:11.770619Z","iopub.status.idle":"2025-09-01T01:15:17.289545Z","shell.execute_reply.started":"2025-09-01T01:15:11.770600Z","shell.execute_reply":"2025-09-01T01:15:17.287941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How do you compute mutual information (MI) between a feature and a target variable?\n\nMutual Information (MI) is a way to measure how much knowing one variable reduces the uncertainty about another. In feature selection, it tells you how predictive a feature is of the target, without assuming linear relationships (unlike correlation).\n\n#### ⚡ How to compute MI in Python\n\nSince you’re working with categorical + binary target (selected), you can use scikit-learn’s `mutual_info_classif`.","metadata":{}},{"cell_type":"code","source":"#from sklearn.feature_selection import mutual_info_classif\n\n# Example: categorical column (IATA code) + target\n#X = train_copy[['legs0_segments0_departureFrom_airport_iata']]\n#y = train_copy['selected']\n\n# Convert categorical to numbers (label encoding, not one-hot)\n#X_encoded = X.astype('category').apply(lambda col: col.cat.codes)\n\n# Compute MI\n#mi = mutual_info_classif(X_encoded, y, discrete_features=True)\n\n#print(\"Mutual Information:\", mi[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.295648Z","iopub.execute_input":"2025-09-01T01:15:17.296073Z","iopub.status.idle":"2025-09-01T01:15:17.301151Z","shell.execute_reply.started":"2025-09-01T01:15:17.296046Z","shell.execute_reply":"2025-09-01T01:15:17.300013Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📖 How to interpret MI\n\n* MI = 0 → feature and target are independent (feature has no predictive power).\n\n* Higher MI > 0 → feature carries some information about the target.\n\n* There’s no absolute “good” threshold, but usually you compare MI across features and keep the strongest ones.\n\n#### ⚠ Important notes\n\n* MI doesn’t tell you whether the relationship is positive or negative — just that there’s some dependency.\n\n* Small categories (like your IATA outliers with <20k rows) can inflate MI slightly — so combine MI with your count filtering.\n\n* Works best when you compute it for all features at once to rank them.\n\nExample with multiple features:","metadata":{}},{"cell_type":"code","source":"#features = ['legs0_segments0_departureFrom_airport_iata', \n#            'legs0_segments0_arrivalTo_airport_iata',\n#            'sex', 'isVip', 'isAccess3D']\n\n#X = train_copy[features].astype('category').apply(lambda col: col.cat.codes)\n#y = train_copy['selected']\n\n#mi_scores = mutual_info_classif(X, y, discrete_features=True)\n\n#pd.Series(mi_scores, index=features).sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.302255Z","iopub.execute_input":"2025-09-01T01:15:17.302563Z","iopub.status.idle":"2025-09-01T01:15:17.331360Z","shell.execute_reply.started":"2025-09-01T01:15:17.302540Z","shell.execute_reply":"2025-09-01T01:15:17.330176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from sklearn.feature_selection import mutual_info_classif\n\n#cat_features = ['legs0_segments0_marketingCarrier_code',\n#               'legs0_segments0_operatingCarrier_code',\n#               'legs0_segments0_aircraft_code',\n#                'legs0_segments0_arrivalTo_airport_iata',\n#                'legs0_segments0_departureFrom_airport_iata']\n\n#X = train_copy[cat_features].map_partitions(\n#    lambda df: df.apply(lambda col: col.astype('category').cat.codes,axis=1)\n#)\n                                            \n#y = train_copy['selected']\n\n#mi_scores = mutual_info_classif(X, y, discrete_features=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.332898Z","iopub.execute_input":"2025-09-01T01:15:17.333260Z","iopub.status.idle":"2025-09-01T01:15:17.358384Z","shell.execute_reply.started":"2025-09-01T01:15:17.333232Z","shell.execute_reply":"2025-09-01T01:15:17.357084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Explanation:**\n\n* `map_partitions` makes sure Dask handles it partition by partition.\n\n* `col.astype('category').cat.codes` guarantees the column is category before using `.cat`.","metadata":{"execution":{"iopub.status.busy":"2025-08-26T13:56:15.822151Z","iopub.execute_input":"2025-08-26T13:56:15.823242Z","iopub.status.idle":"2025-08-26T13:56:57.140492Z","shell.execute_reply.started":"2025-08-26T13:56:15.823208Z","shell.execute_reply":"2025-08-26T13:56:57.139544Z"}}},{"cell_type":"code","source":"import datetime as dt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.359822Z","iopub.execute_input":"2025-09-01T01:15:17.360191Z","iopub.status.idle":"2025-09-01T01:15:17.381386Z","shell.execute_reply.started":"2025-09-01T01:15:17.360164Z","shell.execute_reply":"2025-09-01T01:15:17.380186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datetime_cols = ['legs0_arrivalAt', 'legs0_departureAt', \n                 'legs0_duration', 'legs0_segments0_duration', \n                 'requestDate']\n\nfor col in datetime_cols:\n    train_copy[col] = dd.to_datetime(train_copy[col], errors=\"coerce\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.382648Z","iopub.execute_input":"2025-09-01T01:15:17.383086Z","iopub.status.idle":"2025-09-01T01:15:17.428872Z","shell.execute_reply.started":"2025-09-01T01:15:17.383057Z","shell.execute_reply":"2025-09-01T01:15:17.427951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Great question ⚡ — they look similar but are very different when working with timedelta objects.\n\n##### 🔹 `.dt.total_seconds()`\n\n* Returns the entire duration expressed in seconds (can be fractional).\n\n* Includes days, hours, minutes, seconds all rolled into one number.\n\nExample:","metadata":{}},{"cell_type":"code","source":"#pd.to_timedelta(\"2 days 3 hours 5 minutes 10 seconds\").total_seconds()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.430223Z","iopub.execute_input":"2025-09-01T01:15:17.430539Z","iopub.status.idle":"2025-09-01T01:15:17.435249Z","shell.execute_reply.started":"2025-09-01T01:15:17.430514Z","shell.execute_reply":"2025-09-01T01:15:17.434188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"👉 183910.0 (2 days = 172800s, +3h = 10800s, +5m = 300s, +10s = 10 → total = 183910).\n\nThis is usually what you want for converting to hours, minutes, etc.\n\n#### 🔹 `.dt.seconds`\n\n* Only returns the seconds part of the timedelta within a single day.\n\n* Ignores the days completely.\n\nExample:","metadata":{}},{"cell_type":"code","source":"#td = pd.to_timedelta(\"2 days 3 hours 5 minutes 10 seconds\")\n#td.seconds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.436408Z","iopub.execute_input":"2025-09-01T01:15:17.437152Z","iopub.status.idle":"2025-09-01T01:15:17.460434Z","shell.execute_reply.started":"2025-09-01T01:15:17.437124Z","shell.execute_reply":"2025-09-01T01:15:17.459320Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"👉 11110 (that’s just 3h 5m 10s = 10800 + 300 + 10). The 2 days are dropped.\n\n#### ✅ Which to use?\n\n* Use `.dt.total_seconds()` if you want the true total length of the timedelta.\n\n* `.dt.seconds` is only useful when you specifically want the “time of day” leftover after whole days.","metadata":{}},{"cell_type":"code","source":"train_copy['legs0_duration(hr)'] = ((train_copy['legs0_arrivalAt'] - train_copy[\n 'legs0_departureAt']).dt.total_seconds()/3600).compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:17.461611Z","iopub.execute_input":"2025-09-01T01:15:17.461946Z","iopub.status.idle":"2025-09-01T01:15:36.826170Z","shell.execute_reply.started":"2025-09-01T01:15:17.461921Z","shell.execute_reply":"2025-09-01T01:15:36.824811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['day_of_departure'] = train_copy['legs0_departureAt'].dt.day.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:36.827560Z","iopub.execute_input":"2025-09-01T01:15:36.828078Z","iopub.status.idle":"2025-09-01T01:15:49.029108Z","shell.execute_reply.started":"2025-09-01T01:15:36.828048Z","shell.execute_reply":"2025-09-01T01:15:49.027057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['month_of_departure'] = train_copy[\n 'legs0_departureAt'].dt.month.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:15:49.030473Z","iopub.execute_input":"2025-09-01T01:15:49.031106Z","iopub.status.idle":"2025-09-01T01:16:04.054390Z","shell.execute_reply.started":"2025-09-01T01:15:49.031074Z","shell.execute_reply":"2025-09-01T01:16:04.053027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_departure_weekend(element):\n    if element == 5 or element == 6:\n        return 1\n    else:\n        return 0\n        \n\ntrain_copy['is_departure_weekend'] = (train_copy[\n 'legs0_departureAt'].dt.dayofweek).apply(is_departure_weekend)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:04.055591Z","iopub.execute_input":"2025-09-01T01:16:04.055878Z","iopub.status.idle":"2025-09-01T01:16:04.072073Z","shell.execute_reply.started":"2025-09-01T01:16:04.055858Z","shell.execute_reply":"2025-09-01T01:16:04.070532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['day_of_arrival'] = train_copy[\n 'legs0_arrivalAt'].dt.day.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:04.073287Z","iopub.execute_input":"2025-09-01T01:16:04.073654Z","iopub.status.idle":"2025-09-01T01:16:16.481968Z","shell.execute_reply.started":"2025-09-01T01:16:04.073620Z","shell.execute_reply":"2025-09-01T01:16:16.479352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['month_of_arrival'] = train_copy[\n 'legs0_arrivalAt'].dt.month.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:16.484773Z","iopub.execute_input":"2025-09-01T01:16:16.486563Z","iopub.status.idle":"2025-09-01T01:16:41.739601Z","shell.execute_reply.started":"2025-09-01T01:16:16.486512Z","shell.execute_reply":"2025-09-01T01:16:41.735905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_arrival_weekend(element):\n    if element == 5 or element == 6:\n        return 1\n    else:\n        return 0\n        \n\ntrain_copy['is_arrival_weekend'] = (train_copy[\n 'legs0_arrivalAt'].dt.dayofweek).apply(is_arrival_weekend)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:41.744623Z","iopub.execute_input":"2025-09-01T01:16:41.745263Z","iopub.status.idle":"2025-09-01T01:16:41.788037Z","shell.execute_reply.started":"2025-09-01T01:16:41.745222Z","shell.execute_reply":"2025-09-01T01:16:41.779635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['requestDate'] = dd.to_datetime(train_copy['requestDate'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:41.789880Z","iopub.execute_input":"2025-09-01T01:16:41.790675Z","iopub.status.idle":"2025-09-01T01:16:41.856609Z","shell.execute_reply.started":"2025-09-01T01:16:41.790591Z","shell.execute_reply":"2025-09-01T01:16:41.853029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['month_of_request'] = train_copy['requestDate'].dt.month.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:41.859033Z","iopub.execute_input":"2025-09-01T01:16:41.861017Z","iopub.status.idle":"2025-09-01T01:16:54.127073Z","shell.execute_reply.started":"2025-09-01T01:16:41.860817Z","shell.execute_reply":"2025-09-01T01:16:54.126058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['day_of_request'] = train_copy['requestDate'].dt.day.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:16:54.128297Z","iopub.execute_input":"2025-09-01T01:16:54.128619Z","iopub.status.idle":"2025-09-01T01:17:01.392530Z","shell.execute_reply.started":"2025-09-01T01:16:54.128594Z","shell.execute_reply":"2025-09-01T01:17:01.391225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy['hour_request_made'] = train_copy['requestDate'].dt.hour.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:17:01.393442Z","iopub.execute_input":"2025-09-01T01:17:01.393732Z","iopub.status.idle":"2025-09-01T01:17:08.185345Z","shell.execute_reply.started":"2025-09-01T01:17:01.393709Z","shell.execute_reply":"2025-09-01T01:17:08.183907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_weekend(row):\n    if row == 5 or row == 6:\n        return 1\n    else:\n        return 0\n\ntrain_copy['is_requestdate_weekend'] = (train_copy[\n                            'requestDate'].dt.dayofweek).apply(is_weekend).compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:17:08.186857Z","iopub.execute_input":"2025-09-01T01:17:08.187323Z","iopub.status.idle":"2025-09-01T01:17:21.776130Z","shell.execute_reply.started":"2025-09-01T01:17:08.187295Z","shell.execute_reply":"2025-09-01T01:17:21.774881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = train_copy.drop(['legs0_arrivalAt', 'legs0_departureAt', 'legs0_duration',\n       'legs0_segments0_duration', 'legs0_segments0_marketingCarrier_code',\n       'legs0_segments0_operatingCarrier_code', 'ranker_id', 'requestDate',\n       'searchRoute', 'legs0_segments0_aircraft_code',\n       'legs0_segments0_arrivalTo_airport_iata',\n       'legs0_segments0_departureFrom_airport_iata'], axis=1)\n\ntarget = train_copy['selected']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:17:21.777161Z","iopub.execute_input":"2025-09-01T01:17:21.777440Z","iopub.status.idle":"2025-09-01T01:17:21.785792Z","shell.execute_reply.started":"2025-09-01T01:17:21.777419Z","shell.execute_reply":"2025-09-01T01:17:21.784731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features.isnull().sum().compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T01:25:17.897732Z","iopub.execute_input":"2025-09-01T01:25:17.898191Z","iopub.status.idle":"2025-09-01T01:25:18.008167Z","shell.execute_reply.started":"2025-09-01T01:25:17.898155Z","shell.execute_reply":"2025-09-01T01:25:18.006621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}