{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.18","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":105399,"databundleVersionId":12733338,"sourceType":"competition"},{"sourceId":255839627,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"colab":{"provenance":[{"file_id":"https://storage.googleapis.com/kaggle-colab-exported-notebooks/ishitabahamnia/flightrank2025-aeroclub-recsys-cup1.b3d82da0-0353-4929-906d-4d6a7d41f7f7.ipynb?X-Goog-Algorithm=GOOG4-RSA-SHA256&X-Goog-Credential=gcp-kaggle-com%40kaggle-161607.iam.gserviceaccount.com/20250814/auto/storage/goog4_request&X-Goog-Date=20250814T085105Z&X-Goog-Expires=259200&X-Goog-SignedHeaders=host&X-Goog-Signature=78099698bd4cb73a920cdbf46aa770dbabdad011b181b65b28d8e04a93a826489c56d002f98586dd9496e1246b2bd6d2ad2ad5cf1fa5d4e8fae884a347ca389d2856af4a482605bb9cbc2150e78e6fa828f753845dccee23d6988e3a001c4d77df3abc5d50f80af96290b48f747029d47386b9af5a6f109962582d61cd41e3231e9d93c3829a160e758e808c9c12d59359dc1e67d138c43a2f6bfa560e430f631c4ee7801acabfc7de10d2a480a9603af097fa6cd76700a7b32c975d8f2b6d90c721206e28bf0989c517d9b02b03bcb45fa0989e623eac4e13ee396f745ed8f68e4f34a35282500716080c42d992752604eba84bd60d666ec9ab8353db2d80a2","timestamp":1755173155020}]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n # ✈️ FlightRank 2025: Aeroclub RecSys Cup\n\n **Personalized Flight Recommendations for Business Travelers**\n ---------------------------------------------------------------------------------\n \n **Title:** FlightRank 2025 – Baseline Modeling Pipeline\n \n **Author**: ISHITA\n \n **Goal**: Build a modular baseline pipeline for user-flight interaction prediction using LightGBM\n \n **Notebook Highlights**:\n - Clean data loading and metadata integration\n - Feature engineering for user behavior and flight characteristics\n - Cross-validated LightGBM baseline\n - Submission-ready predictions\n \n Before optimizing for Hitrate@3, this notebook ensures a clean, reproducible foundation","metadata":{}},{"cell_type":"markdown","source":"# ![FlightRank 2025 Cover](data:image/jpeg;base64,/9j/4AAQSkZJRgABAQAAAQABAAD//gAfQ29tcHJlc3NlZCBieSBqcGVnLXJlY29tcHJlc3P/2wCEAAQEBAQEBAQEBAQGBgUGBggHBwcHCAwJCQkJCQwTDA4MDA4MExEUEA8QFBEeFxUVFx4iHRsdIiolJSo0MjRERFwBBAQEBAQEBAQEBAYGBQYGCAcHBwcIDAkJCQkJDBMMDgwMDgwTERQQDxAUER4XFRUXHiIdGx0iKiUlKjQyNEREXP/CABEIARgCMAMBIgACEQEDEQH/xAAdAAACAQUBAQAAAAAAAAAAAAAAAQIDBAUGBwgJ/9oACAEBAAAAAPB6GDAATYDBkgAknIYSbYmAMEkIURyJkydPGA2JlSm1JA2hzQEk5MY5gDTJxAlEJwKY4yU5TkVKeKSLu1CpfY8Li3GXNsSuaDRc0G7imMuLdt1aQ7igMubdK5tRFezkXaUi+x2EByiDYADYNyiOQ25sHIYMHJE1OKTiojiOoNuVPCgTiDYAmyaZJDlIKk4kmwbAck5NohKBGLiX1CEmpLBg2mpNA0wmBJNykqjbY5J3FBg5KUm4pOBVLeG/emuqeEMnd8mqrDRC7tgnkcWF3ajd5ahdW7crygF1TjIurY7rw243TpHALq2kF3ZxVzax7bzTK9Ltc3o3KMrZ2OUxmCAmgACQEpIJNylUCbY3JOXsbkeO53Z1raTGlAp+u+LV97xWS89WKak44ICSAabkJylEblJzkTJjbk8tjPRVIxWK1/R5jaWw2979BKvhjK3nLLDHJsI4IY0PqnLYjkm3JJzbkTm5SG5TXT+f5bo2f1rO1OU61ut1rfRuPZjJ7hv/AGjaeG+Qetw5BAm08HELm2HtWR6VpWgULdyLm3buaTZd0pOvFSLqhKUtt6Jhat9smF53T9R9N4Vr1xqfRthxPKOLMuIU5l9Z64DlECvtvZtf4UiROI6qkSqOUppylNyk3nOvWuLzNXMcg7D0/ZaPH7TC7xDhdrhdWnEaqQwINofYNG2nQscxs37Ncp2PqXDpE5ObqEpOcnPqWkYi89B73yTL2PY5cmwVtv3Wd/s7urZ1/JXnCCbksEBJBlcz664l1jF57WMZebXneR3/AKj8q+am5ycp1HJ1SUvemm45w6LYX3nL0txTuuF0fJ9P2LdcNaXNnz/51QHGccIMaaOue+8BjsN5d9S5G9819eyBe47wO5ybnVm5VJEpeiNzzGl692K/NY2jWen3epeafQvZrG4u8PU4h4gi0msTFl1bCnnN61zb7bXb/A7BmcJmKGb5JZuV3QlK7iSldUZzncQqTuuj4LrGYtNG2jFvlvZclBLD+eMZCmK/ssAMnERJNyGElKUhydQlKuSlObnOpUnVnX9Q5HWdV3HkNrv2GudM0LHFJRlFxjUMEMlECRIYwZJyZN1CUq5Kc5OpOdWtKvcdU6/vvN+Q5rP6FZZnGUuUY83nTrGDIzMEA0BIHJSCSlMZOVQk6zlKcpVZTrVZ19vnmO/7tznl3NtqzmqUby655rXROe2VNCksMgu7MKmSxTLy1bLu1k7qiTd3Qk7qI6t1QqTqXca1XMSzOZ6tqVjoek7x0DTr+htuncXVenTishYYFhKI/dHt7EfL33x5AzXkMJfVHwNyp9+491b6lYHxz4ydVt1p+ybDyVUrTrTvDc7e27VxzW83oWxbvlLVZPC8btc1v/I1JYFhJB72n4H3z6cfODKcn9C6Bh/pf431Hk/1f8X6n7x+bX1t+K/Wckc63Lc+QewMdxjmtrOdXcK+0ZfHapq2FpRhT2bpF90fR6HH9ZyuAUsnpoEkHvf2Zpnys+nPlXbOq4rrnl/v+ude+e30E8w+d/qlrXmDzR9J+F+lfmf9L/P+Y5P0LofymqyqTy3tq21HHcX55kMXSp0oy6luWBzfV+Y+a3len7L5YAnEPe54I336c+Vtq9Z/J31lR7989fY3B/UfgPZveeia1z7pPg77MfMv1h4T+k/A/W/nn57VXUqXXZPRlzzTzvium8ixVGMV6L4H6QzXWNe5Rzrplhf+QIhd2jPa8vE26/Q/zjsWJ9xHk7p3hf1fxvoXqX50e0fnN9WPBvvKyv8A58emvGXv/itl6W8Jcvp1JXWX7lv20+DbRRqW8IRr2ZtfqLtlxxPWLrm/nUY0ADbXpex9neVvKbJtzJtuY5SkyUpuc5bxla/NrGMYqEYRIBPufqjp3F/L/H8GDaAaZJdb9gcv8d0hubcpNk2SKhIblOdTcdtjx4gIhCCcAnvG09M3/QfKmCBtANMkgckNk25jc2Mk5Jzct+q47atI6PzTX0lTSRTHPoHWdf5pL1v4EiF7ZhK6swvLQZd2wXVtKRd0G68Ey5oDlXpOvc9F2Lh8a9qlXtlAuLMdx07ecZwa+o6+BJANBJNOQDCTdRjGySGSJPJeqNO88ppISjFwU5ek7bmvPJLGMJRYmgYxTQ2mSkpzBsaGNjn1zBaCOInBJEB1O33HP934isYxiYCBpjBtoZJkpgxgDGSTY0giRTiE/TVtw3deWKxiF7ZNu4tAurYC5t27iih3FBlYQXFu26lILigMr0EV7ZJ3Fsi4u6dqXljYASQEohKKZJMmgJIcgmSQEkDAUoocQREclNk4WAAAwAEDaHJgCbbY2wHEGMBAoiENkhh//8QAGwEAAgMBAQEAAAAAAAAAAAAAAQIAAwQFBgf/2gAIAQIQAAAA9ATJYYYxNsJMMDMzmAG7Rfb4gstjGMTY5LSGGOz2CIZddq0/PmktMjk2tYVYksXttZEVJTv1bPn0ZmhJd2d4zMwofZbc6Ly7bbdV1ngIz5dTGXF7Szuzpmy7t5lufyj+gv12P4Ms2bynpem1pd3ax3JOPn7NXnLOtz06m+2y3wzNOcdpsr0VdJ7LXYtytXLbO2fjeZg9/wCrfR4su4xm1bFfp2Wu7PZlL4udctPilX3fbuv8abHkYszl3uexrbL79fi02GnVbQ7WXeQLMYxcuzWWvY73aNOinxPY62WqpLZbb42WPx9ufbZk2JZyev0LRQ/c23ZvMdvQOac2g6LfFs7cbfiS1Gfbx+tWhFHvdd/iLfTVNXly5NvS1eKLNn05hzd2uC1add1J6uzRxub6ipakReQfR+WJsYRcu0myMXYl7W00OK0rSzbVl4zNLEMWOWaMSWLuzwJWi27QfNs0ZYIJZHJkJJeu8gIos15tvnXYSQCBmYkiSGNDIJGsts4TtEaASQ2iNIDFJkDCWtZZ/8QAGwEAAwEBAQEBAAAAAAAAAAAAAAECBAMFBgf/2gAIAQMQAAAA4gAAhCaFKQgTptKYW0ABCEmlKQJEpjdzzVbQAES0hcgQmlCpsKSW4AAlNKIQi+845RTefF6XRL0Afr+OEgpiBE16vHhgqK4ZPVnrCPQA9j9d/MPmEJRMpKYvcs/G8nn+n6XbAkvQB/VYfDScnOYUKJ0Tvy1j7fT/AKFPT4f44W0AYmieUqIhTcrs8U/R/ox2+A8BmwAEgImVERM8ufL0+OatPZdAa2gAkCiVMxCjlyhbMl6tu7NlKW4C5cFQR148oVTk4zx2cNNeh6nfxuXOtqHctppcu/Bgq8rnzub78+uv0NWTH6eFJtMuZUyyYaz8466MdXV1s9zDgQAxVIphCJSgxa2Vba9TpjkTGCTRMIQ1KkG26Xrefw0gFCaQEwgAlwNja25+WhAUmJABCQAgBulMvugbABIAFAAA7BJH/8QAOhAAAgIBAwIEAwYFAwMFAAAAAgMBBAUAERIGExAgISIUIzEVMDI1QVAkM0BRgTRTYBYlQlRhYmRx/9oACAEBAAEMAv8Ah8xMTMTG0/uv/wCeXadt/wBNbz/fycy48d/Tw31EzE7xPq1rHGTDKZnxgyiJiJ8pGRbbz4+uhIh+k63nW/jJlMRG/pvr18IkomJidbzP6+TkXHbfyeupKSneZ8kFMQW0+T10RFO28+ProSIZ9J1vOt58PXXIuMDv6a9fASISiYmd5IineZ1vOt5/vr113WQsg5TtvP8Af/gEf8FBbGcuAEWo+sft0CRb8Rmf6KBIvwjM6MDWUgYyJeFOqRMU5o/Ksmfy4UJCSZrVJl7QQmMlYTbuOehfAKNULCL5l+LLpFL1CKIV9x7eH/y8Y2/X6O7XdPs78PEeOxb/AF8hcfTj5B47+7y+3jH9/EduUb/T0/Tye3jP9/AzXsBw8B8C47+36Y5YHZiDjeEbyyrxVERoeOxb+UuPpx38g8d/d40GIiqsWOAJtkJPmRPnCaD7KJclW41yq1arEkoWnPLaN9zJ9nte8oETs5KLXtlPHU7bzt9E2H19+w4w0bDaXJrCMtR2u0W+/P8AaQw1EK8c4k5bjkAfDc9TjeYD8OfI2KYkuDQkZiJKdojeRAy32CZ+7jp9MVOTJKGvxL1K7wzzFOOCFyyw33qsuBJIXxEO1BzAK25FY+z5ZM/zGMJpci1H7aKHHwkVFOtpj0nWPsGVbYpktX3SE+0I1z7YmPbkCCwVxJreMFAUnLkbNYO6EXGhHCQH7hVV7oGVhvpiDUKyLbVL4f4pXxX8pr5iOJjHCGCce2ZjSpgrBC1kwLiGL3FJ/KN8VJ7glPKYs2i7vAzkkOAO4ajgI+sffsx3y+4s/wCuVNXtJjuwBsmCYZRqs+UM5foYCyOU+sMdyhkEAbrDhvw+gs2kme3TO1bDiwQAn1GokvTkGkVe6k27FMurKWt8xz5V0FZepAfW/iG1S3Tuxcxt6Tpc2D2WrnOoVZftDJLgKVV/UPmEm9xEAdvI19tuI+sDXr7e5CpnJ04qLO4O06MyYUmc7lUbU+EUiwyIi85JV+K3wcx9fPtPk4lx5beniu1YGO3E8oR/CWph/pqUU7Mbjtu3GHHqouWjUxc7GExqBmYmYjykJD9Y8gjJfSPLxnaJ29PGImZiNbTHk4ztvt5NtIcaJ4FG4fjGCXPKD2CCn116sETgNoUzumYeo6E2bFHrK7VBRwuag8TF1iryTtES205oSBcdsOpCqgOnjzbwavlqccNzkTOQsprLGsImgRTcZ8Y7kuJ1wZHGeM6p9thCg49HolNkG1fQX5ckOXzE+OTyJ3AEdvl+IjMlERGpGY9JjW3j2z4Se3p916kX95VSslMTEcNJWxUfNfy1ZuI7ZrguU/1tdxIaJxPp/D3RdxZxLsEnYPidQnssb8zlrjMRvvJLD6+wZnTUrMSGyM9xtJ6/dA8l16LahLb3oNPJZRE1vdCy56ucpSQgrkUUwlHKZmHWXySxWapCRp2D4kK5nSF3VGMEqZG/RZekfbK5LE5CNx7EzH2TkP8A006bRtojdtc4jUfWPLv6bfdBj1wrmZTM1Wih3M99m5M59FDto3MZ+M5nz4+gWRsRXF6lTmMXOJtDWl3c8Ktf4pwIhql6zOFLD/DcrEMn+hRhcnaSD0VeS313VmSp65A9AZLKCApiaLzvu7IKnvOpPqCw3Vp4xz2jtFOlguQ593icnJn8yZ3ULrEdmsqT1ZoXlhEwqQADOuc8PSfjLZWZX8RIxYvXEuJYWD2sPXzrsD3ErawC2sDQJayNwDeOJQXDb3Glq/Uw20KHHHIQnYQMy4CPuNTF7cx21mMYADNuuPGI+/N7WRsRzMUavxttFWGCvU4DCBkE4s33O/OBd9sliFs30OEwZ3pxMXLXxdDpys3IWqFy6UNf03jkVckyb0k5uDo4ykixmHP7uWwC6tNeSoOJtT7CqUaAXsy5w6fhV4jMYaUukw6grIudQ1q9ix2YyuHpUMpWpfFGCc5h62IbQiuxha6jTimso/adtiYzuBHFgqzWdLK/9AmnduYLFhRf2jzAfaN7HY1LIbZPp5Bi1dSy07KKdMawX8i7sBRuUMbk7X8RzQ1Tl1Qn4uLVBuJiciqsovlxiK3al7mnC7PNdgELOJHKWTx2NqKozxhd2yVUebyLTMcr4lVMz5ajCVTm7DD46s4qnFSb1B3eWPAuDRDYojjERGld5kioTnYTE7bTHSSJibUHO+q8tYwPfPFbl/EtZM7C5fshoNkweMGhwF9P6DCUk5DJ16zy+WiLFPMro08UCaAsGv1c+Xewcnc6kq3TXVqw1OGnIH1FZLIhEPyczXzllrFzt1ahl2vjbdQZat3/AG7pEEWo+Z1NTblKNK3QiWjRo5QMhiX20u45+JjqbEnMe3qbGXrWSpvrVyYHWMT3cUW3p1hWsWSx0V0MZrqJo1MJj8WZDNjzx9y3JrHF4pVZsxaPMUfiqWWXvFp+UqRDmqzVstLt4y/Qq1Mi00NV9gjacpkvKv8AwNOi+lRY1s18h2cDFow+fiMlW2CSukpt+utlh1wQjS7dS7V+DyQlwenDrSY1HMJvx2LdZr3mNYLaVhVic2bCnsG2jWp2a+PLunJdme5vqjbG2iDj8SXIWohmD5C1anQaoniTlCswQExoX1oT2tj0JVtziQLixwduEpGYHK2orVDjf3+f/Hk5Rx24x5InaY9InRPnv99MdomZ3LNJZneZvayN272/inyzUdQZiFdmLx8a2bydQWwqzOrmRuZCQm47uaqZbI0R4VbRgFq5aun3LTiYVTLZGgMhVtEATncp3u/Fs+drK37wCu1YlgTncuSoTN5nGzmsncX2bFqSD/qHLzvBXC2Yw2mTGnJn4f41vHGI4x5BnaYnbX+PJvHHbjHk/wAamd5328kT6TG2u/YvV69GIAVAqUWO20Nd+Y+UyeS4UMmuV7SNqRLtDH1PePl7RpFp1ZNhcQEipUTwnfToC1tHCeUFZx7uQeml55W3zkHE/b1T/bdr7ep/7btfb1P/AG3a+36f+27Tc+rb5KCmbFhtlktdO8jO0xO2pnf9PD/HhzjhI8I1/j9vxNSFr7x/jISixMsiJFtMfcaiLZLGpkpTtv3BcXv9HDCyhcnZKJZPFqBW0i0bQ2GK8kOhPjIx9Gb8eTHxuDkKbLbCTgUsDtlx33/e6lqS4oe2YEGzJ/M248OU/wBou11chgREZOGiUic+qz2jZoCQMrrBMmE+7bffeNtcYY5Pej2WxDtNED3KuUqTxmN9PQNiOcmMMYs1TsceERvMRoqPFhKlvuaELYQQXKP3SmvuugZGCF4AKkSI7TRsmBgqYg1/aaOMlETvDAamS4egEENYhgcjsIlU7hG4HZKD+VO0LaNmYD0EjSIJBwOhkEMiMTtOvT6jruHt2/WNDC2hIl6nYqkneY3kFn22AyNfaf0iURItPuMI+O33H+fJxjhy5R5IjeYjfbTlithgLIOPERiYKeUR5SiI22LfyDG8+s7eXaOMTy8gxvMRvtr/AD5No478vLMbT9fIoiDkQMkJY5rtu6yS1T49wuUa5qA4guMw+5TJUClq4LgaOXLYhbdiALuTzN6FCvurktomRmJidpVb5e10DxmRsiHaHbRyEJFoTvBIdE+uoOYY0ZL1lbZDvF6jOOBuxrZx8RjcojfbUxtv67+SBjtkXON/ueiVrM8jzCJ06aKOPelK9fE4n/fq6jgfUcbbSHUykjhLkisYnpfE1cm6yVv1DqnEUsd8K2pHb8vTFdB4asRpCZzgiGWvCAxEeGP6du5KtFpDUQFlB1bDq7JiS6cETzNITGCg1UVDJtUkA+Fw93eYTVbrqDpxdNU3aO/b+46PUprL3cWJa6uUtVqr21iPn2KP0nRCQTsQzE0x3hs765oEm9wdyYdTtthe8kLwKYU8dtXQYTSfHuBSxOGyRbabWSsGkL4PwrP7Ms95jo7AXPb3Sgu5Z9Y7g6kOEmQuFsrawB2BnoBtH2rk4h9UXb7+1rFkopA42mNVglr1hAjOrtdKkt4K2L7zof8AFktdcfXGeGL/ADPHa6o/I7msQvKHZksTv3syrMgxR5jlyWpjjFaVkZh0nmSXJyoBkcJlCtFSioXfvYy7ju3FxPb10t+SVtZHFX8jmMjNStJjb6dy1NZNOvyWhDbLgQgeTOnqlijjRr2l8GZbBZUrmQtxV+R0z+d0ddU/ktnWLNq8jSlMzzyEBNC73PwapYTJXxg0V/l3MBk6IyxqOSy6fy4KJxVPZWrOtuCvXDkyzhsnU7ffqzGh6ZzMhJfDRGrFZ9RkpsKJZ9F/zL+usv8AVVNJwOWeoHKqbgxZpaamRsfki4cRtADtAPyDpkAiNLrOSkoYuYla+59Z3mR4+kjtFg/ZP91vamGQs9o8q7Rh7T9wq4P2gCjdteIKIGN5QxbiiOXucdaJ5Da3MTptWa3cZJ9J1eQmY5AJmo4MCkTbasuGAa8yHxUlrp4qCS+56G/Fk9W/gPZ8b2Nf9i/+jrF/meO11R+RXNdFf6+zrrj643XSuMCrRC4YfPyXViaNo6ya3enD5dGXSTADgzrf8eO10r+SVdZbqVGKf8Kuv3W4vJJytT4hQyOsjxwvUUuWndeIyP2nTi32e3rKdU8Cv474HXTP55R1ZrItpJFhfNdbFY6kXOtUAT6lzq+23G1S3PA48MlkAS3+VkchXxFTukHpS6vTYsAizV7IX42x93XTX53S1fsVqKCuWI3jG9Uru2wqtrdnXVlVbcZNjb39F/zMhrMY6cjmMeqYntZzJRjKXs/m+VK+6cBvtqlWr0kcRbz0yZEPXfR0h4Makihi7C2wYn7NZCuxR7z6j9fSNdh/+yeiiRmRKNp8kTMTvE+tbIyE/NjQSUmLElG6ltXNjkmSHsNtHVlaZFdL2vOCgtZ+nxZ8WETt4KQ188VBJaVRQEcnF3SJjJWIAIiPk/TydDfiyeuuPrjPDFfmeO11R+RXNdFf6+1rrj8WN1hjE8TjpGd4yymJyd4GRtPRK2RN9209vrj8eO10t+R1ddQTM5m/vrov/QWtdYfm0a6T/Jw1mxIMtkIKNp6Z/PKOuqNxwtmYn16YzPxSoo2j+f1XhucTlK8e7ow4i/ZXOus67SRSsDHsQllhy0JHkd/8uua6a/O6Wusfy1GsZ+ZY/XU35Jb10X/MyGm5BCLyKLfQ+q8ax6hyC5KfMMzE8hmYmjf9Siy7aecHHr6zNaG7CyZ45NDyAWAneYuu5R3vmAs1/FrZtwWTKZ85hwibygnuIZ3HymuVr5h7RU9qC5KPbScku4AVoDg1HfjeYku33FTtBTMS5FlnKJDkrJ4QErA6cTOgqKT/ADfmse+bM+/aJBcTx9u2p4mcFt6bzP18nMuHHf08ejrdWrOQ+JsqVrrG3VtTj/hrKm+GNMQyNAzKIHqLIUHYa2pN1DGdI2kVrlmbFgFB1jbrWpx/w1lbddN9QBj4mlcn+Hb9i3g+Jb8I4aN7HOYdKhISPW/48drpvIUEYest91CzzbAblbrFGJh0jdp1qVgbNpKi6osqdlIZWcDB6UzFWqDaVtkL1ejDvS1zfhDPp5q05imxzBAOpMhRfiLC0XUGabLq7FNSyROjnsdcqAx1lKTu9rEZULOLsrYulmsXlE8DYAk21gMRyaMVwbdymObQsxF6vy6eatOXqMcwQDqu7TsUEBXtKYWPMQv0TMogeoL9F2ItLTcQZ9JWq1Y73xFhatdU2kutU2VbIHrFZ6pbpB8ZYUp2YrVatyfgrC2Ikpmd58kFMRMROqoA561sPiMWZFrfgykgHKArtw4eK7l9Cqnd5xojIvrPkEpH6eWSLjA7+ngm45G0b8hoXk2F8iZAs5EPdGBAhXu0eUSereP4C01xOg7ELADqM3kgXt2UFE27Uo5KAvn/AH+D6ePLA5zSlSW9LZlbCAK0MjprDuxaHHZ/ndXXRs5EUAW4/wBXRVBs7xnxBHHu2wZw42eETWCGiemHzOS/T76plLRzCTL0oXwaKwKR32gvr9Lqvgt7C0RMPswiPlT87+gxeZuYk/kFupXW1WR+fSaJX+sLLxJVJPYiZmZ3md5/q6RsWTTWUxItKd+RzJXI7C/Qp3++j66oxMy/bSj7JwQHPNV58MaYntFTNQktr5TtlsYFnnkMdPOJiY+sbfttRJMTbKBgtZEIB+3bgNV9+B7ajb0EdWT5HtqvX3rV2Avdlqf4h8Rtt97RniFk9VgTuZOgOb9o9F/guF7+zE+lS66owGAUzE28fmKbO9PzfT9PJ7OE/Xl4+mj4cp4fh8R4bFvvv5C4+nHfyDx/8vL7do2+viPHeN/p6fp5Pbxn+/j6aLjv6eMFMfSZjW8TykpmZBULVxk9SvipzBLfU+szM+E8fTbyDx/8vL7eMf38R48o3+lVsBVdED7VKhq5mNWJmuRtP66+Xwn68gMllBrMhL9tVIdwO7/LcCbi1TVPlFtllCTrmvgv+gTWq/ZsMgtyWZK9ZjcshYlzBH02/ccTeCoxgtORXfuRaOYAdl/fx9dRXlKFiRCUuZABJT66oVvj7q0GUxqvjUsitLu+v/geLfFrf4j1nMFS+GQKz3fWsnVbDlwMyOYtiIxsrx2mPSfJwLhz29PH6/TRgQFIlHr4wBTBTEeUhkdt48gjJfSPLIzERO3p4xEzMRGvWPJxnjy/TyTExO0+SImd9vLIzG28eQYkvpHl4zxgtvTxGJmYiNAbUM5rMgMzNpSbCki8OBcJPb08v/t5fr6z+97/AKf8I//EAEEQAAEDAQUDBwoFAgYDAAAAAAEAAhEhAxIxQVEiYXEQEyAyQpHBBCMwQFKBobGy0RRQYnLCguEFQ2Bw8PEzg5L/2gAIAQEADT8C/wBHj/QxPQP5BHoa+hnokg/7DNF4wJga/l4E009SmKaoYg0PKHAme0JyRETGXs/2V1xm60PrlIp906KHHjRMs/N49brZbmrmzQCMHuANdQPQT0ZMdCKdGOhB9NPKPJWM60zGVOW66kTuTLFu0btb0T48kehg8rfKxa1PsDRBrG3tbrQM00m84kD5pzuuWjTfonnvRtOcN2Zva1zTjWoiJvUgDPkMTdMTC1cZ5JEflRb1pgzuVMwaqt69RaeluA44HNXoAAN7jCB/8IxPvTjF0DvQBcZwaurH7fy95N2BjHJZbMZwfkEA1sffeg0OE9abpNd1E2I1zQOBxTRFRhWfQOddHGJTxlwnxU1UGSsFtZxXJX2YOkZI9XU/2TnXZAJ2sYU3ZIpOnqF2YPr0Wl+ZxNAi4lGh4I1BT90ELGeCiCpvc4BVDtDDkFoxoaCKg4pls5ok5DxTzCjHOeQTAbvxVns7eAXtnL3LXMIrM3AgRsxCK/EG0OM9Xdki8OgTSr3fy9QNIKEhatoVoeQeuTULJBHOd6sxPcVwkQVFWHD4oOrLQf8AmCLrxhoFVDrxUS3ip6w8VgbuDfeg2IOe9EKCd4r2U2JaDiowIrxQPx9CD6MrU05CIp692hqFQmRvQN2Ob+aNJiM5TDVTBE4ytcxKmjgiBfCGQyWqLY4SoIjhkjWeCiiOOEBMk4dYuQOoXELWKeo3ZVcFqfQa2hidw1KNmHzdjE8ju1auut71ah2DYi7/AN+pOwN5v3QyPINERQ403px3H/pOyBQo0TmsCUMaUasC2QR3rQrgNEIxg5Jol+n/AGsr3Jpy6cnbaMOPqGitHRedkrSzvhwLLme7chB5yMGRMoM6+zzd4CYVj2LNuLKQ6TTPBeTh7gxr27I7IfvKtTSysLsj/wCk4NJv9YXsCrQgMsrEC9XWVa24gOxF0hPsGAENvS4vIhWlkHutHi+Qa5CNFak3r8ZRootLtwY4birQ3a4gxOWPqILiTfLaSdFZsuWtrvz7lZgkh9kWsfHslWx800C8+ArazpbNBls1RI5y1DpI9ye0P5waIvLWXGXoAzPFXoDrorWJVpZ84bQcR90+Q4E76FObN6NqqsO3j71ZmHhwgq4I96C+SawxxCuzVMx4IgwpiuSLHD1EySML0CYVwXrYMq6G+1x96t/JxZ2ZIo52zh3JzvNvbZTjrGCHklbuAwjBDyovg0vC8trqV68QacE9kBjtXm98E3ausza8YrnmBt/EBp0xAU+T1/8AYubaymRDia96l9e5Ra9RpOiu2d4DRgx9R8ntC40wxXV8osQMRESCiDcsQ2IPEjBeTUY9rbwLSnMAZanFrtYC8qFXlvVHwVle8lsnHtA6cF27G7LD+1TLRhHcmVY9pyGsLsTIaO9NaBzcURu5dmqtjtzSiYb0FCjgnYkKKgp+JKziKo4HMKZM5p4ut9/pJ6IdebdJomYRT5Kzm7OUqIyvd+KebxLto3sJqmTdoBjwXs4juK35cF7OI+Ku3d3dgg68BAx9yEcab8VIMQBhwRG4fJHFzjJ9XsJIiak671Ef3WWqnEI4EoZrygXTO5F0CNyrXSnxXwK/TVcB91wH3XAfdcB91+unyXy6Uiv5g+PcEXGAVWkSKaojBYXXI3p299BuRi9tTWV2q4lO2RnKFoWvkSC5YkRhwUAzx/O8nTgtcMUEXTIECM5K4r2hEql08UUDDk61Lo0B3rapskGdZ4KAaezgJ6DbZtn1cnZoYH817UmKI4nWgPipmDlFaKcHJzfZlYXmjGFuQ7iierqdycSHAZELhlyaZInqfOqnu4prge5C0DwLxxAgI5aeknDoaoHEdAehj1eMqIYSrh9ymtarDSicOCe7D2QjBE4GchwQwVeqLvyT7SS4uLqu+SJgUjir129Od28rN0TjgnAmYmm9PN0NiRP29BIp6ICyxHFHC9AnvX7mo/4l7iOdXm6x+sKxDdiYm8rS8DZzOGYnokvklv6lzmA5SSNsmacArN5aYwotuh/YUM3AAIaBrkOvZ4wNR6ENZiJ1XNnARn04laFUXOMIp2RiiZEtG5NzTswmsJFcSmnZgiuGXI5sSzEIXYv1m6PFSXEc2MVaG86KImraOywTsteC7q/q5ZwcYFK1Q8pIDpPVlwj4eliy8V57+PJ+JsvqXm/rCY2TDgKf1J83Jc0imMBuCODWiSvYLxJ8EGX7pIGzqDgU+bu0DhwU2n1FC1q6gGG9ASSwzCeYaJj5q+4xIOPBc4+0vX29XHVec+gqbP6gueYKbyuYtJO67ye26gQxcw3gEG3ib7MO9OmBMYVzVo660AhxJ/pWheJQyKu2fiubPzTxLTfaKe8pji1wxqOjcDPcPuqcB3ov+SFYGPI81T23XcOl8RwRIrmOKO1hnvhOeHQ44R7C4f8AMEaNpgTonnYcM0MxQhCsE09LFl4qt3nY+Er+hfibL6l5v6wuY8VFr4K3EzozIeKZR5v3ROmas6PYaxO/RRa+Cm0+soVfW6BNe9Tde05OTXC1ayY6w+6vEXZvYLbsb/Oe6Yhec+gp0SOC9rE95R2bV4y/T90wG0eNQMl1LOzbSTpuT6X796Dvovw9p9K859BVlhmZNKJ9GOvzXTJWLwQdzjEK7Z+KFmXWh/SD4p+xZDx93SPgnG9lUjRE56FAA3BVEXVhyX7mHa04oUIPSPaRcCIqJVoHNoc0yBevNwGauaVlGGunM68ueg4r2GYd6nqgQ0Aehiy8V57+PJ+KsfqXm/rC5j+S87/Ffh2D3gQVzz3e5xkFbDeJUWvgptPrKv8Aguf/AIhfh2+K5x655x76rzn0FTZ/UEwbBPbb9wgPPAae0nWM9xVmXh39cfZPdAC/D2n0rzn0FfiR9JX4my+peb+sK7Z+KtmksOVMlZCHs/Tr0hWQmt82/E8FdG1kUDexz0Qkuu6o4tdmFzoMYwJTrW+87UZinei9xB9/S/BXXTAnZPfPIRrDCcSmEOO1rkg9zAcK4KTideKv3XcNSvYb1RxOaIBpRpohGEzKB2G+1v6M9B3NRzjg2cdU3nZuPDow05G+UWRJOAAcjzcNbaAnrhGx7bg0TO9N52bjw6MNETLH43DpwTf8x110e9WLQTzY2BOkKLXwQL5a+0APWKL6OaZBRtpAe8NyGq5gCWODhnonPvse40ww3Lmzde66ThSqF+XOMAbBRLNltoCesrMy0oiH2do8CvvyU32XXB0atMJwh9jafKuKjq2QHOcKJ1g/Z51syW4IX5c4wBsFC3Bhjw4xdOiFvZkk4AXkbkNbaAnrBEMi+4NnHVNZjZumDO5DZeHvDZ3jin7Tbjw67up0nHFN23B2V1PbeYYMklPGyMzPpR2XeCDoN84N3JgoJ2gngOphTJQ+GnHSm5cy5pPNDr6ptldJdZ3ZwCPXd7G4b/UACGO1f9gvba8Qe9W0S0GboHivJ2Qf3nH1yyIOE1yTmu68jhG9CJLaxQDwWXD04s3ScyAM1BmJxThF0pzS1wig0KOZ7H9/UXGX2bsD9iv0EO+cI9smXxu0R9cuZbynGLydLa48fUOaPzX6VILiagkYI/5keAyT7rjZhuM5j8uDQIuy6ThCg0AjtGPgiQO5YlBX7QzdpgY7iFzj8OPpoaO8/wBk2CLxLRGtE5xuftCs6f1Zps7MmNpNZfuDEYYU6M9GehFOjHQg+rwr8OjIn5prGuLa0a7CenB6M9CVzreOBzW+lBqmACz0k4EfPkkQhgRQ/l14XuCYS3YaYGlCUbJsPPWJ0+PqJhw2u3m2EWnZO/VMGWH5k8cULR7hqb3qLbRwcBk85Kp92CfeJPulPbWQKm81oI3bX+g7CLTECf1HXeoszhdJFcRgNyHtCQmtDQTZiadOfQj8zGYoUcSanln/AGO//8QAKxAAAgIBAwQCAwEAAgMBAAAAAREAITEgQVEQYXGBkaEwscHwQNFQ4fFg/9oACAEBAAE/If8Awy/MvyAdFFFqRT20kTEIg5BGhG0MZ0paACaA0o50AElDJ0qntpIIKIvQs1iKLooiExoAJwHv0XVFAquiigDQBuCtFhChAJ7mVCQslGeYzzGeZabZxnmM8xuTA4sBggwpMhjPMZ5MZ5hlKBGM8mM8xnmM8yyZQAsxnkxnmPkYUZhRHzOQmNyYzzGeTCFNI3MfKM8w7RguExknL5jPJjPMuOTBjPMZ5jPMBDBMNCHGeYzzGeYLCIp9xnkxnmM8x8olGKCzGeTGeY+Rh4TYD5E5CZZkxmPkYTmEhJzzGeTHyMZ5MFeABqFB5J5M7hnccx8jxAIX0vhzuP8Ahn8ozp2/CN+qi0jTtoyHXHTbVt+HnSdA/AoMjTt+Eb/gUGnb31yDFsjklgCCUDnrjpsfOrb8PP4R+Eaduq4wCRoN/Grn8I6KJLAAR2wPMILpBkB7gzboACtlrsAOXjiAEw8a2TVwLscRMY5JRN71jz6QpU0kAEVZCjJi6SOhPLTyswVYQ4wOvIC+gwehTprRmzR4WiwW5Ke5N4ejd/7D085g3ovXWxyq0nFfPR52uGxsu+gc6rqmsOi4ydAyViBb4gNtAuY2EGQsjw3AYWrkDsBxkzKc3W8PnQhKeJb50XrrY5UK26WVic0A4Ath+5TjGNEROByIekDiYIA7hiOEMqg0pWQrEHcrDcphnaBGshsZsxW5bby3CtShaAZRNlmb54Od2rfgNcR0gpgl7lT+CckX/wCBGnbRaKwvobMeIgAlbgCDRAG0GzuQCpoKH/MKDBxMABkzCKBNAmhn8C6ipFLNqGCxCwNR2R+iJvDAb5zt8RhpEtBcudzASoEE0A0wZky2TkCMeL2h6fwOJkPOnY6tvw86ToGnYaBp26BlTLjZLxCQQEEFEGHRGSAa4ruAUohIYCJ4QcKmJCZAOvBMUN5/YYGoSG2K3HaXqjACAkhNK2N/wWnqDDD+Ah00YATIpf0Ev3XH9Y2eYpCTIsLArfvD+4wpD/OMaXjUOXmZjQAy+1kl3DTGS0EV/m8x92BDsO+8BCUYGhOW5E+xqOnbR7TPC8zfH4ToGnjQNOx6DLQQwOnaExrCPZgmx2n7RlVAGaqUnokpRjvs/mAUIZFuBZC/M54JKrvbw4RZbOjfg85lpCMb9uOgl1wQmYN3YmPdEJwF1ymMCtlDc+hB8J2xTghCEQEEUQZedaGSQiw5ggIYxEbKD3rEonAQOm42HzDUNGpof2IeKJJBHEpvIwMNqJYc5KFj+Tf/AIMqCQMRRgYHJd4LKXKrMYA2zB507dCCiC4jxEeIjxLjLKIxHiIx9YSieaqGgGzpzMqtvyCMwQ8BnvyRRlegGYjxEeIjxEeIkoYER4iPER4hZbBPxEeIjxEeIjxCM2yIxHiI8QbBZKhIRFxGI8RHiB/EDEYjxEeI3Etae48iAsbeDI9THg3IFWSoCH4VlUGhJC/UbSKALFAK8wC9aRuFvvHdMSYWYbBwiAZIEAapnYp5luaMQ7rAiHhk3RcsN5BrsqBxqZG/POFWtHBM7smVrhbKn7GEJbvkIVW8EejaKzkz5gVqc8CpZ/8AcXzM2qTx/wBYz+wCWKdtwuYjxEeIjxLNDhEkAiNEeIjxAQI1BjmI8aXS76STiSSvJJMJl7n/AOpcwQNwEPZuXUrQxY5gOe+klrxoBIwedAmwGgFEEadiO8Wgslw6C6B7RC7NvByDIFlREAi4VEzuKIupvZs3PkPEBj5kHtcJrSt3FSiFyEPRFg2IEcImF74hQnGMH68QbMiKrdjDKcLCfagnUaoqUc3o0CukI5CwsbK7QcBrFgFnkqAa7dkgVUQjiFCEGrGFCwDQiEHuLn+d/sNBHmwPY6Eid9LM1Egrxq20Fz6VoCouRAA5XGII+Qw2/MmtZjMAWXFJ+ERi33+gSZ46HOK2IoO+6XuVwau5eoOo07aDhH6xorcJkv8ADR8+OmCIxJGFloANwsgthA+K7r7Ao0gCNAEcgjkQ2dSOXK40Jg1k/ccBKAAK7k1/3KqWZp9hP3URYFo4De0GYdxsMhW87slJYHiKkAijkAB6gXXpihIvMMiAb4/cL0dlygkEAd9vqPJc44A/wg8ljjEFwAQtu2f2ZDzp21baAQeADhUL21PbvZ4G5hvL0kDcDBt6ymoIDx7XmPtBhmhE+6hZhEhUqK2CBVfflBNR/BKcxbWu9y3lA9IAPiMbLIjPXmmSwtfEpi2nvgZzCwSvpCiuG8yxPooEcOUPEZaWeCDmJWIhe5uzQoG6nC8i6AVpHUZ07aAdLeDB4w64EQ29CSJucxCJBP8A3ATLm0ID34jjbhw7/wDtRGV5oEUzf3cQ1y0S/wBShjbizEjf2OEd7qk98a8Rx/7knu95uc9FsUB2hrOAxAwpQV4kuLC7AbgCAMCctPiVUU0qrVZUEBIAgB2hB9AsI0HMRIbzmhw4UQnK5Vmsk6AQdGqgcA9VlmUzQM9iDI07HVtpVGTxbCfP6gdEJnkT+BHujmyBwUJFDCQMR2GWHeLSYEBYTDK4b6Sei2+xg6LzAmhLtIJV1sIL4A9xjrnpMgsd0o+y2cuyehCxa1aYKPSHn+aEXxUB2WZLdQyhe9RKk5zS5+WPyttCG+cQz3YLcQcK8h3BGIYfGOh8ArJGsJPNiIPiDUv6N3uO1eoh28XYKhjsMA8sYWh8AH1C37XEqaPBjMRpKeTnyGFk3AJ+EvYsWQqIZKgISySgbjiXLUIeStnCJBkOTZw5jEtAyLUIWojCi+B5lWuWsg/+5xpx8cCY0ORZc+GNLwIBGMGzQ3zAC+i4SNGCxmXqIkeFvyPoQZGnY9CQSSg7XH2j7T1LN6zt/uep6j7QBBEGxaPxA+dEhNtuM4N1i+4UGGTPMOmBpeJmxrlk8x8nHh9eEQML7CH2FkIbtnEKeaCHvTQPU4T89eDA9Q9edg/AcD1DzFGw5SmipjaPrygEAToOYtWgpDzDb3GFfe9YYCLHZAwT5ATLgkDeSY+wnqOCYBg3uPtK4E9QBQSAcQkEtBPU9T1A5YYLuep6nqDwgiAInqepXEGEljvAlfgHf8nBMU2AjvseYQByMMHd9+0S87CLK5Bj+klJ4eFfeEJ0GxR3BlClyFsWFmGyLVBbTd1mAqIMKGC1h3Mz445XsYf8Bx+1pTBowYI/NQD9o1jhdg4EAUEh7uAIkCBxc2wI+yeoA28ky2feVw07ad9HP5thoGnbUOigGYfjYCA5bHzBK1PrGmMc3DCyBt5Wi/UOmSTl/FzCgYAwKxuDzHG/1ZABGSPeAEYCq8G0FAgsCCFHtwNMYghGKsJfMU7vFgOVuIOpjuzlIIdoKAMEAx0ONAyPPXbptq2P/CGnYdRq2/AIBmLFebABgv6h/TzIQP8AtK5vuCKEPRSKpD2UIj76n7yIfZSGQPJ7RocCRQRdwBsQs94BQoh4BPHmPjiAYpVhRUK94IJ3KIoI2NEiQDjCIiJ/csABdqlJm4OQRyDDiEE5JQgOXbl7P1GAhhTBsFQZHnTsdW2nn8J/DtoGnnSOonHzCjEB0Rd1EY1mJ8vBeRG2eVwgAxNsx8Ko6guEZK4GYZQ2qoG4GO4hORtHY+YDklJZ5IBqMiwXcocJfPEfeUbEfL+oWIKXRHN/uVRIhyT5Qj0iIB0MMhsi9AI1fZBARkCOWcYtwZzuu8ZgAACVDaYAuDIi6DpsehABIAHuNFi1TJ6BEBAnJofDhmsBcq8gaCIMBgF3emtLAGn/AHQMwDYbewnvQRJCTtegBwAJzCACQAPfoOgcQ3i9AtXBGAA9B0wxmyJEE4qHsnWMB8QSuBlBRgowQ7D6gwKdGAP1n5lgtGMqLdbQfFcXIGQnu9hK40pFGzc74UsQyGQRDlWQYSYKvFFqBUxoiYlUH07m7G4M+lHkEDnssu+O0NyBiu4DnIreMMVwUosRfxKE9S3StCESDt0ESEHm4IgAANxFF3ihZIgBmy321baCoXSICnJbKttbKgw/hHyRN/SbaeQgzZSpuA3yRdKZVxt5Mw+dBiykYEm4gTDqBDHVehKH8jChLO5ElTUP3JcLEFlIYsfcwAYPEnh7Et94f70bXUdR02mEvK0vlASJLlNoCDpvBNFZHPiEAyyBGVOBAC/MX9oQQcC8OKQblCItLHCOOYHUhvQHziDWZa6CCRUF726HYAG8NYkEBQh3HJ+OmLHz4B5HCgobFpqjF0ciIVvN6NbqFO6FAhzgHd+oX2QwE2Idq4nJWbhDiLsAyQSRYjgNvMJTDz+/EyHmFlZa0ixgqQhPC5w7lKQnbptq201/V6H+vxgqft13Jcq2AI7SkLceNiozdHoyugIB78fST+0ABNWZgGDubQm7DuOrzPMNTpEKTAym6ykPT1eMzHtQmPlBF6crElyRjg/2VFafa6P0OeK6TCCD2O8E5IuxMfQyZynj1vPqIulph33A74ig5+1obja7dmRrQYEsf7u9oAiZSlYcvubBolMcjkd5/mc9DxNoUxot0Ay9eABCIhjSpK2Cz/RrhGtQoriyNmcWnNU5gWRZEDgeaMKQbLJItDzDhxzvDMPGMAogl51DA0eL+RH8mRRcMAvmJONLMCSKPw5YyMVEGuV8WIzMxALAbUwPWLSVyOg9+bjjKDCUr0e0NC2iMQqJAB5YBNc9djG826FDydtO2kNxzP8At+i/9/jAj/8Af29Ow9s/yXF5hyb4p4kGW8PvOBBuQbi64+CUtcHv5FwOLB5sWz3zUFbxZJFiooWm7z6aZQmTqlz9yMWYoikhsxiZjMkS6phE8f1QRkO5jtGEmwPkTcF6RbCK7FR6gKCCYwV3gkux8ulaa4xHAu44luFwHvVlCBDnRY/af4XMfUnfD+kblG0a8eIZJJJZ0OEH0CSU0LGVTbJsBs9eoQih/wA1EZrERUYC1o/Yg+rVzscomBfCjx5/kAJAJJJQA3mL5Hj/AKULbLIBEEbHo4egUcAwRDu8giAOxBBIC4h1hAhb9Z7wjBIkIRWLB+t4sIDEYsANFUYhbAbREZGe/iCEHo44N/11NSwWQwclgDzDpWty98yB2hQUIrUJnsXk6UWKp6Q69A/zOMCP/wAvZPoSDzAWR2Z8iHVlEe/zADDhASyqCzXjqjcIBbR8DPu4BjH/ANY/x+Z9JYtX0ekZEQMoSvp7bsMTeAWR/neBwtI+OvuGQFd7YD4g+wMNyf5BQT/guldQiP8A4PGfa6TBkhlcz+0w8p34Bmo7b9DRMcccrwECIi8zMCiUIPZAo2kQuIrHJglRIYk2OzF1Aj2qOdgWT3j4agH7nt5gyEFkQvJ5KjRfdjVVhtzNreYMGhgBOODpNAPdLHavIjBAwwWtj5EaEMYZg80xLEw+gy2GiO8LgEm7BkSe8KGC0AXEAId8QxmyCwEbDjCh4OR3YNzk9JRiJMQmUcjkwcS0L2DSD75hgRit0EwfYG3CQyZ0WHZloqKWyjbHmGfCr6CImvQASSTsJcjkfRMAxSyAJwmKKtuwq4PCBd1jcOf7lgh0wQ8NiKNIEOLgPqbWFAUlO8wJjtoxAiGCIwaEcRmjPvLwdHDXi70oSPhGtYXhQL6h8/rbwLJiNjZXQ7AxERM2P+2jlxIYHOV8IeH3WvXCfyBLlxAbF7AzEXzcndMH3qIFqAE4A5T9ebwLJhmeYeZKKGIPugAJJJ2hj7OXRsAwxCNvwcIfSeuswvEwjZZ1RzKo+1ZmZtttDIkcccPIREpOAe5UN84iThIYoC8CGFrsBBk4SuM5mPZLY4G8Qkxj6OEiT2I+YSY+8ZjhRBISujgwIYVgH8Q2CTl17iDThGFVPlJBcMgoxIDOrcld4Q5cKBG9nuxFIbwDgUJF/wAgNgV8j8xYJuW4vH+F38advxjg1cfXvvQcF4C/JBiFgZQDDHyufvHJP86B+EfgermHjASaSNF6lwCJLWAunlAb22oJu1CvojTgMDo+j1baBkQQg1jmW7tnMbVSMwEnfZXzMQRtUr9Rw1XirBTwVUKme0A/7e076dvxnhGJ8FsfsE8QEHiOm0P+tBfMMjrBJyT+UbadtLj6OA5gCvUMmgX+4oAQwsh2fMoiwopldCaMcOgdCeu2jB5hzyQL5D+QipiFjfkxUx2haTNgwsDHoiyu57CZVUOCl997hTI3kFp2/MfyjTt+APu3NvKisRC+LWAEu4CIMF+sAkwC3X0niZHVfZzN58MSVQM52PeAwSqi8FtH1enbQMjzEQb8vLBjpH8bipeFLR3Ueavn+Sm0J+DufyJ9zcVSOIJdjJHyckg+HWYb5KVK7yp6y8KV3ld5XeC1tTf9ifEqV3lT4OHLlSpUqesG5XeVKl5svsVSpUqV3h3zJypXfpuW1w2yUqV3lQcywlSpU7nLF1KlQy/GQUSBH3EAkjLMah8NokAhMGAZElfEJgJJLjpMqV34uVGJXJl5svsUK7ypUqHGdzlSpU3nRADCR2gJBkWxhbzLPAPEJJfQTAaGwiuTBn/qluMQM2jwRp2/8Bt+BwU4EgmMl7XqAsFNQeaqwOJ2uCKru6DBCwI/x7aBkQQaKKOeBEGNh7gMCOWMugldlDVKELsYHjptq2/DzpP5Rp2/CN4IUG0QBB4F2KcaYDkkkPJeMUBrGnYaMHmF4AWWhfHGRUIS7UcJQfyCAct5o93eDgZ0DHaJ7atvw8/8fb8I38an+DbQMiARJxQS7sFhTE4iFMBMiCJGBnwuC4kQsx2gknJLADH3Q+BNuhIICCOdF9+JrQASAAkmISAVoMgIFnTgkxoJEOon40lqTaBtWTCCEEXoDSZLQLoCLRegcYYGnm0HoLEMon4ly+pCRYSNAXGSeIPYTnIPiX3/ACyT2eoOEagCVzqZVq0kkiRk2ToZDRznSydAJGNLONAKLGllLbS3Z0NPvpZrQCRjSyk60A78R6GRaja8f/hX1//EACcQAQACAgICAgIDAQEBAQAAAAEAESExQVEQcSBhgZGhscHwMNFA/9oACAEBAAE/EMU+ePG/HX/mFQPgf+G5UqV8LSpR/wCPvxXhT38KlR8HmyVM+ZaUzj4kyKuVXxfxfiQdNgI6T4DCRBaODXxVQRLLz18HbS0tHQWvxQBDTdPmoSJUADlleaZe1WiC8W/FEgGx8VKg0QkFv0at8DlEh0jmAssvk+CtMgaBwFrKJTylZWUGsKbP5lQsTCK0EgA2rApEyT8eK7lg2EBgsQvtpiKPmFu1Xmfa45eJ9/nnucNtVviAEtmltXOG1VW+J9/h31qfy3Pe48zRQETJSRDzlVxbo6JxW1W+Gfxzvqff/fe4FyEBdCM5+S98z7Gq/E57Xd76n3sXz3GRVpY4qZ75r3OC2q3xDdmu98x0kXUugkUbtO7Zy817nFbVbdTPfNe3cS9VotqK3nxz1Dumb3zLNrVb4hoIdtpGjivK9y+7Nc9T+83MNWqq3xAx32FuwZ9qvc+1++tT7HPPcUESBWdVHrLdvEpxbk3Ptccz7P77gxYkDwBn7a98yii9Un7n2tnPWoJpfvuDk1a1q5zZLvbucFtVP2w76gpSsp4Qiiqr3bNjmvfM4raqt8RTOax3ySv78WqwH+Tj4g31qF1l13vmdeqt8dQx4CzdkTRzajufkq5eNTLeW3bl2zjxVhfHULMMLbNP3c4Ozl538CGnBue/g7fgO/j/AIPgbfT8awQPIsXq5R5phtrZKqVfmsyvIKtWpUxKdT2iYPUzElM+1aZREZVLO41T2yuZXMqAy1s3glQKlAjiZXrZKz8eXs8e/Lt9+vLD4tj0fDb8PwJweSP5B4qB4pt7Pi7ZUqVNPXr7gQ+pTKjx68UT2mz6f6idREjEjr1bx1FAyM6yk5WViDFUUBbfqHqEsBEBsgru04+mb+PL2eTw7fhxBv4OzHBzfj14/wAPkJU4Phse/NeKy9nl1PBqthVocvh3KleDXr39+CYlfcTXlO5t+H6ieDdnvzbS1tcE+lmoYqCMcj2+Cw2vCFailcoTrdmx5KrtuYA23AZfFVljqAwr15WMawDZL4BK1LZ7CC0emzRNJdiyqVsSqB5TnmbX2c+M6/Rcx3MTEDP4ho5JiY8cjWOyuauKwRiDnNW81XfjEKlPBmOsEK7ZiYhUFv8AeFTHcxMds9Yr2P7Rri5U4JjuFcuc9SjuYNsAg3OOh1D+EOUx3Mdsx2wZRqHiswqUdwgMaJwzJAyWBmQyyzW6ncRi5UvaXceyHPeKjCvc3O9OgLLZHJ7ZomNic0gEomJ2LC9t+daVmYmO4frv9moLOVcXMYyxcH6ucSplMfqQlVIxFnShSrIG5oDO4QgbahmY3lcQohwrWXl1nBAKbZBbZLOZkVVGkqlCBaA0VnyCT3eLL7KvZUUdvoDsvKLtq4/BAnMaLayoqobpX/KqUdvPxNOtnxdtd+SHPg8vGtHwID5COjxUCDJ4rzVr2eKYBYaiJR2trCpKLgIxCDAiSCue28EYj+DCjSQbQ2w4Dyj0pYmGkSPb+lkdAQS8u4xS2DReZReWVKlSqr1PrxWVnPTvwc3xjV0AgcEStiAQKqjhXbA8C0K9ciPagWDqswRamwoBLGMBi+oLAzoq85KuYI1C3d0qxbd0r2E3a5VegLdH+u1yzl1VN68UeaHEbIPHxN/Z468u335IfFsej4bQPJg8B352PhUBV7JXEExIkpbNd9o5twFImEYRfP0hLEFVjkIrWck5bYsX2R1cNpgms0xjZtHUWqx3ekhBYggtGSrbWJzG6yOrMWJo3lqpWvg8eKiqBd1mnV0IItaiRLMoFY0UroQoGJSW05tEwRhjmi7YGlu2KlreKbgHgcco0I81DQUgDQthW/YPAbRHKUbQo3ErnRrmL4W1ihtF3XTfMZEmfqgIpqbJawu6VRfk8Gm+zj3B8Tb2eTLKShQFb2UIioE/zGYrwafXx2PRz8B5O8zuVZB52PgEo/ISpZKsbXgCGm8SwXNmGlSEYDzkRI0rkqJQKMXUlNPA/VaZY+YRVK0hstOVAnSlKUjYpcIYKQ28wA5AXBvQ22DDeXfKrqXfVhuOrF3+mVMfTDmBAVKR0XHIBGkmQhW2whE2rAKNyhwCsMbywHOtZkouwie3QERNiMAEIANigrQaYWkAqiCLioQGXkoClZ/eRZNKGWEBWh66lZe9NamG6RxgmcAWihH7GolMsyps2YHgEfg20o0UAGADAGAhyMNOl6jGBi4y+bvO5d5QtOxnJ6c1z3Llw8NC9nhVKGxJ9ifYn2IOTF2rmclPc+xNGGDHx0KaC36JidSRi9X2jFWGYALzfZ7IJmYn6MdMuTlaUwdOmLoQYOFCcFrutQ2Wn3p9jnjqIXrAx2T7PWuZ9jVw7XWo8cprcBYlYoqD8+p9iguF9W3WuYjstpqHUw7k5LR46QY5ih4NiQ6HjjuHcv1PsauJO6prluYNMC5fqBVlBUBq1Utfv66ywrCRNZZPr+x1FSasm9JithV1WmPOYYxfYmraOJn32oLQgUpt4ZQwqxVuAfbXDmJMCKvUwbeC6Zj6ULqkC5kQxSA2VJvtpEtss9RhRjM84KID4hYH7eQMu9nWCounRvkqAqhZRogVksI9CJkqqvR0dwFloFqPdwTEzradLCwsOw62Rww7ATDOzNWjCYwgcFGtIUuCtV32bsGrACL3PsQ7k5BVVxEiyauNTJsSUq01h13OO93Wo0b6vXEdCTtthf8AJzX54638RXtijV/FpG1VUG72rCBhkuBPohL1wnssUxcdFQ4F0iARqlJe8j8WSW6AejyEYVBoXrCUng8C4q2LBaIVryyikRH4VBczCGrhFSvCUm1dwB2nIWYq7TTEKvQQVkNYeL3Lr7LsW1uAU2gHFQcFYJoWxolQ8gAhBFbArN5wmZVmwoDFnDAP4gx2ei2lThs2Mwv6jJ6G0VmM+skQeQKJex2nEUYWu1ezoBrAwLEgJk05KVrFQ1CNOCV2a6I+dLUNNqGhWqTcu+58qilF/wBCEfpA5oiJhFbeEZ5OTgQKJdNGAcO+phNw+n2QYfGUusWNJ93hLiAIIG7qWtrtly4OMwA2YLbbU19XL+PL2eTMpFSFXN/taYMopANvZJspwfg2tE3Z1hgejR4NPr4sGdyaNUK13E3zgaT1d+OCg07Uow50O2DFFQqO9zIPOoWA+5yZ342PcDweAy9kIECFDllXs8M2TMj3BjocK4SVG/Fanqs4YPJUwbdJ3JS44i4LgNFMtB3UIhwumK77xq4xG+ABhQKO61qMLR4ZuN4VwaxKFBfHMFl6wZuMjwHYumgFaCkM4cZqi1vsTUwRnIiGOckQVUK7Vi2vLDDWxFCAUoAzRiHseHCWRmh6ix8psC/ppcQBnUcpoNcqfyU1AvHVgv1aLHYld6Jyu47JqWVaemAzMogKgzXQlsfTWX4897Jfx5ez4AfMWquBgiPKwuKldWhWLARbr9fxXb1HAQ2C1MCujBzLPDuxqk5wyh3uOMf3u1GVGOMhJMwFtSAaVvk47HfJRfFVFGMuKKgkKgCEaW4XCtMEVs/1es3SFk996HBm3mw7Dt5+heiBHfqwYa9Z4e/rWu2lMldq50WGAV4PGQPzNDwM0RIF+AlZezwQjrgdT8YhWCv9rg0j7jiH1lZgbC+sxnareC70q31AWqmy1sMv0huXiF5j/PQvztQloaALXlQ1UCT1trSfUEF+rTdaRRFrFsQMK8Rbr5hhcHwKHJzuMXHfDFG7hhaRCdqXfvgI/QJ4jCyhFLGiqsEMwChekKwKBQAoAI4pFgbCq4ITWGBmoh+6meGUYG2y/UtMgoh+6feoqpgKlUBoNu5cyj9lRheX6hUlfznNTUqa8b9xzL4+PL2Sonm8lmUTlTWGdQ9XT6dv3jM/aG7JmK4N1AXTKMYbIJtXOD+33X1ixDAvWDRUaYGuw08sA5kC7m1atzSPm+3ubIN1ModrdjFlH7Ah8KslmaXECDIqqvSsPE48Gq8ge7Da3XDCnhafrOSEXDXjg9svwEFI9vjULhOXs8BAm0qq0QzKRJDsppee52guYw3LPMWEQbpiai550bh9GO56sEvN9o/mFD4t2FbEH6pzDf7OUzRGAA3LUEFnNYWJvE6BNsGCblGoawXpGUqJ3y4M2FWxyMq/IWNHLH/aBQqKidQ3lcgkAQYJYELlnDZD9X65qF1jqoem6LBkYAQVAL9hdR0922ltVR+miwKHoXEYvIMn5liY5VKgN4Wpd4zRi1kgBEL2lSvXLP3HFxj2TiO2Z/IRhgI3QaH7WU68dynXnuWddVzC83xl/wDkgnXrmWdeO4Dpz3HqoK+A6bDn6ZS9YAB0BJH1HHEXjrkyfYxWkURPdE3WcKOK0+5fqAnazlyA2sTNuMN1cDvDYQYzOqSNVyxS4cN6uQaFrIJygNWirZmq5YyDnkJ2F+1N0hzBjlQ2RpRk06pZpUPttmOWsqBfYqCo56Ks2IEftYt2ACtXBWYC/u+5Z01KZQbMZhqSqhs37YCmz/Mu33XzLK019w9CWy1iIgRvAv8ArLNIgnX+ZZ01LTsEtagfuCdNPcE6QTpt7iBFQxhumBwV0LAxpycwDpCmjEU1Zl2E6YtvR4t7c5UoNsIcdhBVUxKVoxCRVC6tY5JS9tQvj/BU0pp+tWwU7EUyS5Wtuo7cpy1eF3kMYYTacmoG2B9KTenunvAeIwMue0+EpGe8C7Mxs2daCJZJ3BZ9gs+82R5fA/aGfGzYpxND/dJswaUcF0PAQfEKip/DGu5WjT+VljTq7gC3Bes/rcEqq6q8xrIdtcBh+0ugdnLz+ePiXbqz4u3jjwQfB2ejwMGG/BMTiEDXnYqHgz4yr2S4OodECcsFwIRix4g/371Qf3tCbYaNYoP0FBAql12F8rgdBGXh6Vn2QNnTAjCighl1JlUEQlNA4aAVvDTGvcpedhqAhqG+kVCIu7cvQjQQxianTAVZOI0juspaGSn4dXMqcIlWR0IrjQRUXy7R2pBLHTKgw6tlSvHHvTW/BqG3gu3slY+J+Sz48s78nPr4uycw8bSpz46h4CbEMpDweDb34Ibgw2wqjwWYk1rv7h2cqlOSnRX6Sgx7b8EFq001m5cGlTBLh9XjcVUxgdcuyXIwiINpoLyKDo46ZnqlZTseR6MoUgPNL9bwWyH6AgjQtU6VpKPbWqWtuSY4bxCBd4lVdiQTBHTR5wGuF7HuWW5Ap1RAWA2CoYskpv0ywk0e2ETh+xajXhiJPRf0mtSmTXZBasdT8anNeDUOZUMew+XL2fF34ceDn1D4bHonMPBvxzOCGpgH5+GxDweBo9idQhCbMIeAg7BVKilFEU5YblkJzSKJClmI5cOKZrIlo9qPUQqkJ/KzUza4jNjjIU1QsHwaFzUAChRHIXgwKLRRs+4I90NsOyPDxDjLcPE9FvhmIZMMKgtO1uMXGwNfKq/d0lLckAv4+1pQai5rTSuWdCgFcOLVhTHn2hkZuVjiPMASo7qAIWs0gBKLS27Tu7DJ2YKtl1AlBstBtn884uYMJy8H8h4NgmrKfVgyvsn5n5guBRh/NqvFHZKOyBTwIoXloqPogMn5lWT7nqycbn58BaCBZUOBgfZK+5R2T8kVzNKYY5oT8w9z8wsOWtijWBiZaDK+yB61Aw5hJtFksfsqAdyjuV9wpYRV/wCQyAcLp/YRM7ntKjKaCG7T9Qh7hSbhEEg1a3RBppyD/pAvkmhkh7i5oVfAG3FjjZans7aaBiKgpkA0m7eIw7KHM0U3RqAboF/WwCvGKRtnOM3mbEUJxcdPKpFaC6aLo5gs9EsVgCvFBBytIhYjwjDN7G0rIASi7LnOGvLTEVOg1XUWZS7OWyjWSyph4JF4ABJQwd3lHXRAAIcNNrAzKam+MVwMB1hgzjIYAq2SVwCkVACkT1K3mLyhF4H6GaaoWU/sIANkBWk9oICaioFeKcSsbITjzy3s+Av+GkuQJ67hRwrq5QxBJgPB6puFkN+2mUIEZXQls3lJ1jB2vtCLDdfDw3o3Ei9reRN89RHZdE/ENwQTa73TgMdYTbaxYSSnJcr05YxlGB7FQNJlg23e48VqQ/yaltYs19x85p7J98zjwIbeyC/Y/HRFG7eajxLGycsXJFBlQKvsZl0veG5VWazIcmHOYBppgbRyqhjYPqOYDuVHUcHEQrSwgDCKaRjODxgLtaWqjYR0ogueuDBh5hrIkgDCg5OzFEotki3WlZOBMTy70/1FuWF3WaGUcmTDMX50WSWwRkrmNS5SUrDLhsX3ce4aoApvCWrCWMVIYuxWbKmwytwyWwNCmFHGMlwhJ83ICpkiVOy45yHQ7XDPt6a3LyiajTbkXLRgvh7SrAJgG9+BOe9ng+HL2fA39f8AfLFj/t8t/wCXRVDWnBRBuaso36pAYBtG8jBJRUwqmrQWy3KH61P86E+GDYpDgJMpZz6cf9H1SXXmOezbmzBcFhIvyfsNY7UJ1PKj1qwPyy2k5Vro0svzAtsrlz/E8RuCswER9dxqRQWnmbrWIQv+m9l7SdlmJFvgLVJRAyiMnTu6t6F59j3t/IhSOWAIAOYjFZ0NCj/XTJRLJbyxBdOmAozLv8brWe9p+yhBgTjk3uCjSVhl5h+mXLhYuA0Llv1YuZj2qPTGrJDQ0FrA50u206WNsVZZZRd4sDxu40nOHfI6FacUQssuGdW5+WmVtTSC04DTXJmLFhtvrwvUxiABqfaXSRmi3UPNnrBd0jzU9toFGqodykZBOl2JL6DgViOGN75mGqOAgKYBiwKCK2gsVOSuZsCdHPzFvYoyka/oGkpGA66Aqm7GHO/GJX7CArbpfd22B9vx5ez4f8nuUD/8Q/17qJVx4Z9HukH70619f2yuHWiLlr6IKMvfSQt5X0k0j+KuwQYMx6f9/FZTeHws4Z6RLlkIaLxBAqGM00ptYKBMU1vFfKd8Q/lI+FMCeAIorljhIcFtnVWuggLPl1Hvf1RkMCjYo9DkO/VOABpcCZMAibbN0C0vmSgAYAUAQ/lyXpWaeHIurYskGKW4C1SMoKwuL12rMzjnMBrgV37SQuctGi4Z7jg2SWq5VZeWCLmC8Q7SsqQcCurlAxAlrM8FWvrMMsXosLS8mnAeTMKQ4RelDZqXFrlGdTWdFc83EyAwinlh0qwX1wCqWgAl7S1RM7spWn5ROnTF9IORHZBz+GMKfcbQtopGOoF80KeyIbIVhDpbKkwbaAx94F+yh9C+EVkI6NREgVftjeMnYkLU2S0VkTG08t1dxdkpTPfgGE6DsGvtIJiAXI9XfAmwww5U/lxcK5Pdm4fAyFgL4v4f8nuR+D5ydHvky9iOfXM9debAXuCDOfnm2fUIww/p77i7LP8Aj9y6kDAFM7oJGZnBJVPQlHRPV7hXCgh7r/yDP5MAmlUtOYO7pc7j3zNGV0jbR7azli9/QX8pUO1hbi31akJzTDlNvQ2vBObRI/lwLuXhbIr7ZeX/AB5nGraFlVuuKb143aK4zEagQIiOoQReDrxANBQTplKrKJ2YLCPF6hpa7XpF1tQndy9ZfFYbGkQNhDCgbLdzVDYySlQF4AbAlp3TnMqDzsTNgwDnbDacuJu6tKo6XcqoSQ25oaSyMIVbMdBbFlWxepcJ4Xd3hjKgRtxD805KuzDHJsRdNfs1tBCQS/1S0VoFL+mMgRvQZkgEFXnEuMPEO92m4qqnUCb2d6ssNBScETxi+NThfYlD7lfeEz3EQrUQyLOxTSjm4ZYI/iQkSm3QiGzDoqMWptZbLe5b3DaYOq8y3uWy2IYA3y/zkewAH932y2FN7bimYBtiSgV9v09YKKQ5+hZHsIzzfu2EGxnd8vL1qUq3OCLdr6yqnP1NR6dzOA6f7yy6exyFOzlJnGSejH0noTagiWgf9wuyw0xU1cn2OcWE2ni20/l4wfHGgrYFsUhjfbWnsC7RfJfwiYVhMMemTCXKhWbULqkQfd9OihzaXrS2k1RWJEwJ0FUGUV3SFJX3V1AZsQ92GQ8Y+rsAK6ISteOI4gCZUK/T1PYb+dDpLZEcFNv21BgUQ8eqYHaTEaXJK8tEtlvEXitwfcHB9wdCm17GKWdsU0SLVF4NO4KLa49FGwNOJb1WidwBY20QmRZcmwN0u65cbQd/bay/cV7iu4pUV1dBIpVXMXxH2RfcV2WHGQlvct3GnBdzs2EX6jKDc2jb/KIDuXFHKMZo6NKZ6NRTABy97m10cCAgdWDk2rqDtIq1IxaQdPs5GorUpWvJaV/MKGpG5Y9nHZ6mfibez4u3xx4Ob8E/00vcOWnwyotNgPGDrXn29NpeCkBpdRoWoToD8OXk8cHi4TY8j4HD7Jd1FlIMu4MHwa01LRTe1qi5jJVZfYEhHNG4JQ4mfoolKZHJvqV0ahW9T8RHjwvwN/hisYvh07t83NdCrN4ICuwUrQJo6WAVFss6fglxbxEMLp1UaZT1tKUIWyXeZxQf8KoUoL8QRSIlFZlO1x7N6tVTa/E29nxdvwHfrxnMcWwW5KIXsNpxPaixxHZuCKO9Luqt7RlVdr8Bz8eCDBhFlQ83By9ngZiWzZhFwnB69fcYkpqCovoF/RLjKMR9hYLahpt/ErhjFALlREShue0WvUVixeos39M1LNeVx9vwSU9NFwmSrSCkrvjK2IfVV0ywJI56ElBBSDA+ujBZA8eN22KrtCYxuyVaLa2+xKYSSy6xqy/jyfs5+Lt+Bp+Ox6PgfAdk4PAnjY9/BnL2S/JLywTwXKUkycCzdNl6mB6Kgb+AMoXzuZajw6bhvgsYczBoBRqu+wQLuPTSe2OZ1w9xjoy2ob48hbBWBFKRAIs6MDBqPhcP2pi+L8Ontlsvglx/rc1MAgLTWA6TWFLh5WwK3uHk3bhlV88Qi0cHNvd5uDzslG7Vl6cXkBlWQhOFrGU5sewYkukqzluHNFWW4zLbqsLnLLD78uJrt0ccwMzimjQyu9q65hnvhxPflxBVfRYbnFuhdsTTbddT3/Se7OVqurB/lz2d9T2ddcxz21fUPuzLhdBup77dSuzrrme7vqct4FHZDm0tcWQ+7vqezrrme+3U/q2VPZ44h9v0mO3XXM5MroHEzrbKsl9njie/KezqbI1zxWZ+V5l8F/UvuwbFqYufbXmezp4h92WZ13bTJkcOyHZq1lsBthmtkFAOUC29qsjomRxCkrmgCoKzLCAplXLNXPY4ncdc+7l8131PZr+ZcnJODR2VMrTTiye7vqabY/d/UzP6Sqio/Z44nsz2dTknMALSEkuZbxb0cGiURwXSo1TgC0rgJk8jqG2akGgal7Nj2czF81AVj/8ACPdFJp0WBPji3s9/DqO34Gn4vHo+B5vxwQfJsh8OXs8XLlx5gy4S29JOKa6ZYXSk0EDXuWlUInZcMvAtdkDVFaLF1F8XDf4d+Vix0ftiwlxmatm9QiXRNV7o1yWkEevPqCsoYqxVXOEAGhoAcA4cjcWG3s8349zn7OPi7fgfFaz0fA+Ax0eB78bHi/FzNvZLPNxcsJcuLHr/ALCJmJNZzhZKJUToPitlpQpAS5cXXr4LP4ZcZuXFv2PkZ9m6aLiULGaUWmFc3BplboICWa4aLtnFvUZuFPS3fsIqmeUaadbQrzUHP2S/gTn7Pi8/A+Jdej4Hw1ODyPEHJ5XxpeyX8OXzcRWP/HxcvwuvUpL3Lgl/hn5ikW4zh7efF1Lj/ec1HromYePUsFta7IJ47Br98YoShgKO93ptw1slIfsnthTN3XuRv7JcZiVIKSU9Sm9eBRuRnVeeZnqV9Mz1K4oAG1YpppCdOyUymU9MLuVKXaEpmepmVH7dRLK4lPUz0ymK0QzHAuImEblPTKlPTAzLaUmZnrxdiIBUWOB0+EfCROhtXd+c9QLEK4AjZR6SpmfiK1Q1xfITPUzMzMBaBUslvBM+FLI1r0uIFsRNx+jM9PcRjYAI1yB4p6lMbuIBlEsG5tCmmjkiFav46tsWU6qU1qPm3sbC/wCSnr4EzAro1xZ41jy9BShtV5fgECAKHZd5+HMaFq0UejyMVVI0lnSUnxwCcGjxcGICIjYmxPI+MizZuvipKW8rBly4MAcCn7N/FbJWii+CX5RVI0n4SnzcuKC6iocW7fNncQQtKsSOV3Ll+ASu4XBbA/i//wAX48XXwGX/AON/C5fxv4XL+dy34LLg+SXUGfSDFn//xAA2EQABBAECBQMCBQIFBQAAAAABAAIDEQQSIQUQMUFREyJhFHEVICMykTNCBiRSYsE1cqGxsv/aAAgBAgEBPwA/kolAUgOVICkAqQaSgK6KrRatKpUgq5AKvhALStK0rStKa2k0dE1qaw+AUGkKygiTsrQukECQevIK7PIAkdeQQCpVyoqlSpAJrbQatJWlUaq+bU27CbdpgINDdNH5ALKArkEBZ5FNF8gPygKlSDUAEG32TWjwgwL00WEItRCruN0LPTc+FJluhiZJ6Wpzr0sB32TXhsTZZGFpIFjuCVju9RrZBsChytE3WyaroK0D8cxuV0QNoLVyANjZaUGqk6Oy0+FBlxTlwZftNbhaGyuNSUaqigAOpCaE0KrRBRC+qxcoFjZap9eN04u01BptrgCX+Ewsi9T0t5HbkEoMjJjOgAs6INDtiAR4ITRpFCqV9PyEhY2bj5YeYZL0O0u7UUE3m3ZblNCFcht90EAgEB8Jji5zgW0Apcei50YAJulDBkh4fJ2N3a1uaf1Yw9t9e4UbmvALXAhCdmv0xZINJ+VDHMyB8gEjhYCe46bBA+T0TzrjLC63OBFt+VkYH0rLINtN2OhWNmMnmbGXOYejb3spxLQHhlu2B80oJHPbbm0bQdSDkH9Oe23J+JjGLIjDAwSinluynnl4Waw+KicX/RcNZXCszIzIC/JxjC4Gh8/ItUm1fMIIUh5Q5DlSLQ4URsp/Vha7SCQRsR5UT8p7mAk19uyoscJIn1V34P3UGRBIdT2ta/yehPwsuMOyXTxzukt1h3Zv+1N4jKX/AE0rQYn0CNVUPuochrfTggbbGnSbPQeflZeTjMlfIM6I+0gxF4orDyeEARymWFkrW1ReFHn4L3BrMuJzj0AeED0QcmuCFcxyhxJ/Ry252QZBLew20t8BYmNw7H/oQNaQ3VqcNyPNlfUxbk2KF9O3lNyojVWSTQFblDJh0hwJNnSABva+puaMh36To3OqvCGTG/02sJDpAdJrwsTLD2RNlJ9R10aoFWgmiygAgrQCAUHEHuyHxyFmhrpdVA2wM7u+6PEcVzPYHPs6dOk2SfhZRfHOzRGWQuawgOBBde5APYjwn52OWst5aXUdhZ+xWXOwmKONnuv230UcxgjAyA1zH2dgboeUBDK+hVEgC2mrduFx3LMGNjwxBzX7tL73odQmRMlawhzgS4Ak99rNIRRF0Vay142G1gg0vSYInvY4ueCehGwB60v8K58uViSwzPLnQuABPXSUHJrt0ChyAGyoJzdTHNBokEApuI/fW5vujLDV35tfTv8ATezTGCW6bFgp0Lz6D2luuMVXY7Um4kjdMjHt9QPc8309yMMr5WPdoADHNNX3UWNOx2NqLC2LV0u6KhxJGiBsjm6YiSKuyUB03Q6kpvIIIBAL8Mgd/q3MhcQeof1B+FHwmBjQGSPD2uDmvGkOFCuwU3C3TFh+plIGklrjYJb0KdwiB0eVHK4Nnc98gDSCQzsE7Ea2INyAb6skG2wUWC2eJmuSQHcbkXpOxQ4cGSSvL3CMOYWNBsO0DuuK8KbxfF0NqPKj3Zff4T+D8bj0M+gdcZ2LQF+G8aDr/D3dKrSKA/lN4Txh36YwS3VsTQGxXBOF/heL6bjcrzqeQgmndNPIIduQQHMDYIDdBBBAJqCCaggEB2TG2U1izsTMhzMqaY6y8GqsU3tpWHxBk1YuQNYI771XlZHDyWsdBINIHQH/ANLCxmgOc55dR6FZL8hsz42QWD0IBUBm0MEzSB0FotTi1polAdCg1AJqpUqIpUgCh3Q3VLtyCCHVAjymoIJqCamnoFGKIUgc6F7Y3aXkUCb2v7KWH/IvZkgPdRPc9T2WZw/CjkgbjucHOfZdqvTXlY8UkcjZIHeoxwAcOx+R4WiORlsGl3lXJCakBc3Ufca2Cly8RrgyV4B+QvVY7S9rxoeabflFkcpstshBqohAJreYbYHKVhlzjGXuAPj7JmMMdspEjnWwjdYut2JNUmk2fcT0UFthbqkD6u3WmzwEH9Vu3XdcQeDjsdG/YuG4KiniDImOlbq0t2J36LOyTBEPTcA8kbHwnzNkxJHMeCfT30noaWLiOngkmGQ5rmk142C4RkSTwvbIbLDQPkL6nHEgjMzNfixaZkQva97ZWFrP3EEUEczFAD/qIw09DqC4o++HzPY7YgUQflcNyYoeH4rp5mtsEW813Ubg6nA2CNio050gA9Ngd5s0sjJZ6QhfKxsx022+idw8vlf+s27ulj4pg/UhIugC09Pmk9ry/UJiBtsAE82N1Jh47yXOZZTGwxMEB6DfcL12CUscAwkA3W5TJmucRtsevYpzHmRrg4aRVhN3KG3IIcpY/VzyzUW33HXom4/oRyn1Xvtp/d2oKAXw/K+L/wCFI5zcCEDo5xtT4sDMX1G/uppu+tqX/p8H/cp8WKPDZM0HX7STflZbWvwcedw/Upou02GKLClMYrVFZ3vekyKc4r5Y5HaA6nMBP8qB8cfDJH42zg33eQ5YOFiz4hllPvOq3X+2lw81gcSH+0//ACVg4ME+DNNKCXjVpN9KULyeC5LSeklD/wALGInnwIcr2wtHt8O3THdKTHIuJaQ11GtisvGnxcl8kzzIXOsuKx8mGdzYJZ7eXbOJrT4Hyo3zRx6ZAHOayw5vQqPIkkJD2aaHgpzleynkDdLS29SyY9ZjtuwFEjspQYWyRtGtjh1HVqxMqJjIIHyhzzsK3/lDXqaKAabtxP8AwmCMOsDWR3PQc7JqyTsha+lH1H1Os34r4pEamuZfUFRYjWQSQazT+9J8MEWL6UzvYO/e1NDBHES3JL/9LbUeL62HDG9xb/cpccSwCAuoADf7I4rH4zcdzjQAo/IUOH6UUsPqucHit+yxcZuNG6LVqBNmwoMBkD5HNkJjeCCwjZfhMeo1PI2M9WBQcPZBBkQiUkSirrptSxsVuPjOxg8kHVvXlM4axmJLi+qSHuvVSk4ZFJjQ45ebj6PrdQB0cbGOeXlorUepTX10TZNt1xSaQPZQ1N0kFpFi1i8Phy5RM4aNDgSG9HLXQRfaJTnUiVe6kjbKzT0+aWRhzQQGWNgcQRddqHVY+Ycx7YpSfUoBu9KKIQMMTXagd3O7nlS8IG+3LuEOqexr26XtBB7FDCxmu1CP+TYQPIHlaBpAoOV2rpByDk16bkgyGMdU1rXW4gWVJIIgCe5pB4LQT3AK1IvRKLle6a5ROaInktsHYjysrCxJYCHxAOkAotFOFJjpsQRQkOlZYaHf3C9qPIfZON17R0Q7CkD7SKCJQPwF1VouQNq0CrQKtWtSDlaBQevUoE0mZbZYwI/a/uxxAO1fytaJKtFyLlqQdumlY51RO0WDrT5C9wB7Cl1HQJvLxy1AooIGuVoGigVatWrVq1a1LUg9a0I4xKZh+4hCRayta1IuC1WrQchmNjDYmNc7tY7Ei08TEkxu2ICadlYtBEtNUD03Vq21RBu+QIB35XSJsq1qpA2rVq+Vq1atX5K1LUtfyta1halqVrUEHdECPCaVq5N6cvKPVN5FFN5hDkD+S0Cj+W1fKzyCCB3HL//EADQRAAICAQMDAwMDAgQHAAAAAAECABEDBBIhBRAxIEFRBhNhIjBxFDIjM0KhFUBSgZGSwf/aAAgBAwEBPwD9q/Tfa5ulnvcsSx2szf8AiBrlwmpYhaiZvMDEfsn1kwm/TcuXLEvtf4gMubh8QtD2U0P3K7k12vtfpPezAe7Mqi2NCJTki6A8mBNxIUwqVJF/sHnse57MfYepcm0EVdwaLOyfcG2rA888wrkwAb8Zr5Hg9j2BMDfMEx5grmwQfzxM2sxox8km/ED5NQyb+R8VMasitbE35lkeDCeb9IFkCdV6D1PouTBj1+n2feTfjIIYOPwRPHB9ZNetlCgG/MXU/d0ePGoAdCdxv2rjiFmU0DxVMI+NFQfbch+eD4qMCpoz7b7dx8VcANbq4i7fe5mz48ZIs3UTUPkyLvNj+1bPPHvxGUnCpXCDfk8WJptOqDbu5qyY4Cmgb4ldvj06frfU8eq6dnbO+obRuDgTKS4H4r4mg6fpfq1C/WfpB9AxFnW4SMK/zTVPqro3TOja9dP0zqi63GyktVE4zdbSV4MPp8w88Q+gmpcx5HxsGQ0ZjP8AUilI+4WoJ8/xMg2efPvH8EFbvxf/AMjHIo2gkjxUL0mwrVDkXBjd0ylWNCq4/wBpkDEhiLJNTT9K6nkxL9vpmoZT+oOuMzB0fquxQdDqrqyCjR+l9RxqXfQ51UeSUMIMoypXoFWL8TqHXOmY9V0rUfT3TBpG0dMXyfrOR/lp1T6k651cn+u6jldCf8sHan/qtCUZtM2mVx+blEX8COtWR47VDDLhMuXHxAICLs1X5Ji4XBssFI5uJqTkx07KXQtztF88cz7b2TtBrjzxMSNTsTwfPzMyDIxOMGwQOfzFybcZvHRF8giyBwZ9DdL0/UNdqdXqFDrgA2IRxuJ8kTU6/VaLNqEZMLomJnVUsFOQq7j+bja7XJh1e46dcumI3sQ21gy7gAPk+INdqTrMGHPjXFhyJj/vRjudlsqG8T6y6bh0erwajAgRc4O5R43L3r1bpuFjzL8j2M3e1cSwBCR+r8xnHNDz2PZjx2PJ7XPvsPj2/wBodSxJsAgiiImpVbvGP5HHBn9TlUY2AP2uBz4LQ6g7rxHj3Ux9W6sSETmiR8Ee8bVg41AA3UwY1yNx9p9KdbHRNWzZbbDmFZAPIrwRE1/007ajK3VCw1IP3EdzRv8AFe0bU/TTYxjPVTxkGQt9w7mYcAk17RupfTq5E1D9VbJ9umCNkZhuUVdV5n1D1n/jGsGTGpXBjG3GD5/k979Z9J8w+0PZoYT2MMJuM0TUYjix41BG0+/z+ZnwlQc2Pipg1Iwlzkx7yx9zVTU5t7DagXiafBiOEZDlph7WJss0vPvBxFUt4E8dh+wfQYfMJ8Q9m7GGGNDxEKjIrOLUHmZcgDl8QIAriYNTqWTI2RQRXAqFvugqRTCyJ9sfpB8CLiRrKcAVQ+TExsOK594uJM6BGNHwp/6fm5lxajSqpZSEyWUJHDAGrl3L7X6Rwtwm64jVuEPnxKPxEHJuEHniKoY8jiVTqCPeO4Vgu0GZlCsK94UYi9pqFWBAKmzDjck0pmBazKCJmRmzPtUn+Iw9jGEVUYne23/tMy0r7bIidSUY1X7B4G25jLN+pzwYpUCiln5iMUYMPImDUIeMoIJJJYRVYLvQigBdHxu9pmxY9g3ZUfG1XzR49uZmwbLfGwZPx7fgwFArAjnsAWNKCT+Ox73Sy7I4h/vWf6zAxLVB/eYGJevaLw7CWS4v5hK76I5jAnKA/iZMjK9DwJl/zcUyZGXIFB44jAf1CfxHG1crJyx8wjmFZtmT/FxqhAUAew4MOmYU6Y628eDzcTGLCk+9TJiVACGuzAIBNK+XG27G9BTde1zHk02pULlZkygDb42knzMGAY1bIq7xa7r9jdTWaMuxy4BdnlR8/iY9KSf8Rgv4PmLhRE/wgoNAWD6d3FdieQZZLWICSfFQtTEwGjcum3R25U1HYsbjZCwFjke8+8fdRY942QsytXiO5ZwxEOUl1evAgzMHZwPPtGokkCoRKExYkysFY0RyI7thvGDYIqz2AuAQC4BXbT67Np6HDoP9LciY9XiyYg+GgaAKGh4A5mo06ZHbKlLwLXzM2T7KHCjfkkfn9ne3oIv1EdiB8QrGwkJvmUEtXsJixbhQ+Lm0Ka+OwUmAV2qVNLiD48YWg1myxoTPmOLGyHyR/wCb5i4sWqpcR2ZTQ2nwxPwfTXor0kX6qm2FYQ3gniHAAzEjiuCIFIgWUJXau+n/AEIofldvNH5mdw+Rq/tHiDj9qvTVwivWRCTt2+3epUErtULhcZ+23FVE+3t/V5nv/wAjtniV6aEqVKlSptlVCLhuVB6h+7Q/dMq5QlC5/9k=)\n\n\n**A data-driven journey into flight recommendation systems. Presented by Aeroclub for the RecSys Cup 2025, this visual blends aviation with algorithmic precision—featuring a stylized aircraft built from numerical data points, set against a sleek grid backdrop. Innovation takes.**\n\n\n# -------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"**📝 1. Introduction**","metadata":{"id":"lwReq34lxth4"}},{"cell_type":"code","source":"# ======================= Kaggle Notebook Pre-Submission Validator =======================\n# What it does:\n# 1) Detects Kaggle vs Local environment and locates train/test/sample_submission files (csv/parquet).\n# 2) Loads file heads safely; reports shapes, columns, dtypes, nulls (fast).\n# 3) Validates your submission.csv against sample_submission.* (column names, dtypes, NaNs, duplicates).\n# 4) Prints a pass/fail checklist with next-step fixes.\n\nimport os, sys, gc, json, textwrap, warnings\nfrom pathlib import Path\nfrom typing import Optional, List, Tuple, Dict\nimport pandas as pd\n\nwarnings.filterwarnings(\"ignore\")\n\n# -------------- Config --------------\n# If running on Kaggle, leave as-is. If local, set these to your data folders.\nKAGGLE_INPUT_ROOT = Path(\"/kaggle/input\")\nDEFAULT_INPUT_DIRS = [\n    KAGGLE_INPUT_ROOT,                      # Kaggle datasets\n    Path(\"./input\"), Path(\"./data\"), Path(\".\"),  # Local fallbacks\n]\nOUTPUT_DIR = Path(\"/kaggle/working\") if Path(\"/kaggle\").exists() else Path(\"./output\")\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\n# Optional: suggest-important columns (won't hard-fail if missing)\nSUGGESTED_TRAIN_COLS = {\"Id\", \"ranker_id\", \"selected\"}  # adjust if your comp differs\nSUGGESTED_TEST_COLS  = {\"Id\", \"ranker_id\"}\n\n# -------------- Helpers --------------\ndef is_parquet(p: Path) -> bool: return p.suffix.lower() in [\".parquet\", \".pq\"]\ndef is_csv(p: Path) -> bool: return p.suffix.lower() == \".csv\"\n\ndef find_candidate_files(root_dirs: List[Path]) -> Dict[str, List[Path]]:\n    hits = {\"train\": [], \"test\": [], \"sample\": [], \"submission\": []}\n    for root in root_dirs:\n        if not root.exists(): continue\n        for p in root.rglob(\"*\"):\n            if not p.is_file(): continue\n            name = p.name.lower()\n            if any(k in name for k in [\"train\"]) and (is_csv(p) or is_parquet(p)):\n                hits[\"train\"].append(p)\n            if any(k in name for k in [\"test\"]) and (is_csv(p) or is_parquet(p)):\n                hits[\"test\"].append(p)\n            if \"sample_submission\" in name and (is_csv(p) or is_parquet(p)):\n                hits[\"sample\"].append(p)\n            if p.name == \"submission.csv\":\n                hits[\"submission\"].append(p)\n    return hits\n\ndef prefer_file(cands: List[Path], prefer_parquet=True) -> Optional[Path]:\n    if not cands: return None\n    if prefer_parquet:\n        for p in cands:\n            if is_parquet(p): return p\n    # else first CSV or first file\n    return cands[0]\n\ndef load_head(p: Path, n=5) -> pd.DataFrame:\n    if is_parquet(p):\n        return pd.read_parquet(p).head(n)\n    elif is_csv(p):\n        return pd.read_csv(p, nrows=n)\n    else:\n        raise ValueError(f\"Unsupported file type: {p}\")\n\ndef safe_read(p: Path) -> pd.DataFrame:\n    if is_parquet(p): return pd.read_parquet(p)\n    if is_csv(p):     return pd.read_csv(p)\n    raise ValueError(f\"Unsupported file type: {p}\")\n\ndef df_info(df: pd.DataFrame, name: str):\n    print(f\"\\n---- {name} ----\")\n    print(f\"shape: {df.shape}\")\n    print(f\"columns ({len(df.columns)}): {list(df.columns)[:20]}{' ...' if len(df.columns)>20 else ''}\")\n    print(\"dtypes (first 15):\")\n    print(df.dtypes.head(15))\n    nulls = df.isnull().sum()\n    print(\"nulls (top 15):\")\n    print(nulls.sort_values(ascending=False).head(15))\n\ndef warn(msg: str): print(f\"⚠️  {msg}\")\ndef ok(msg: str):   print(f\"✅ {msg}\")\ndef fail(msg: str): print(f\"❌ {msg}\")\n\n# -------------- Run --------------\nprint(\"=== Environment ===\")\non_kaggle = Path(\"/kaggle\").exists()\nprint(f\"Running on Kaggle: {on_kaggle}\")\nprint(f\"OUTPUT_DIR: {OUTPUT_DIR.resolve()}\")\n\nprint(\"\\n=== Scanning for files ===\")\nhits = find_candidate_files(DEFAULT_INPUT_DIRS + [Path(\".\")])\ntrain_file = prefer_file(hits[\"train\"], prefer_parquet=True)\ntest_file  = prefer_file(hits[\"test\"], prefer_parquet=True)\nsample_file= prefer_file(hits[\"sample\"], prefer_parquet=True)\nsubmission_file = prefer_file(hits[\"submission\"], prefer_parquet=False)  # your produced submission\n\nprint(f\"Detected train file: {train_file}\")\nprint(f\"Detected test  file: {test_file}\")\nprint(f\"Detected sample_submission file: {sample_file}\")\nprint(f\"Detected existing submission.csv (optional): {submission_file}\")\n\nproblems = []\n\nif train_file is None:\n    problems.append(\"Training file not found. Ensure it's named like 'train.*' (csv/parquet) and placed in /kaggle/input/<dataset>/\")\nif test_file is None:\n    problems.append(\"Test file not found. Ensure it's named like 'test.*' (csv/parquet).\")\nif sample_file is None:\n    problems.append(\"sample_submission file not found. This is needed to validate submission columns & types.\")\n\nif problems:\n    for p in problems: fail(p)\n    raise SystemExit(\"Fix the file detection issues above and re-run this cell.\")\n\n# -------------- Quick peek (heads) --------------\ntry:\n    train_head = load_head(train_file, n=5)\n    test_head  = load_head(test_file, n=5)\n    sample_head= load_head(sample_file, n=5)\n    df_info(train_head, f\"TRAIN HEAD ({train_file.name})\")\n    df_info(test_head,  f\"TEST  HEAD ({test_file.name})\")\n    df_info(sample_head,f\"SAMPLE_SUBMISSION HEAD ({sample_file.name})\")\n    ok(\"Loaded heads successfully.\")\nexcept Exception as e:\n    fail(f\"Failed to load dataset heads: {e}\")\n    raise\n\n# -------------- Recommended column checks (non-fatal) --------------\ndef check_suggested(df: pd.DataFrame, suggested: set, name: str):\n    missing = [c for c in suggested if c not in df.columns]\n    if missing:\n        warn(f\"{name}: Missing suggested columns {missing}. If your competition schema differs, ignore; otherwise ensure your feature code accounts for these.\")\n    else:\n        ok(f\"{name}: Suggested key columns present.\")\n\ncheck_suggested(train_head, SUGGESTED_TRAIN_COLS, \"TRAIN\")\ncheck_suggested(test_head,  SUGGESTED_TEST_COLS,  \"TEST\")\n\n# -------------- Submission schema validation --------------\n# Determine expected columns from sample_submission.* (robust to different competitions)\ntry:\n    sample_full = safe_read(sample_file)\n    expected_cols = list(sample_full.columns)\n    if not expected_cols:\n        fail(\"sample_submission has no columns?\")\n        raise SystemExit(1)\n    ok(f\"Expected submission columns (from sample): {expected_cols}\")\nexcept Exception as e:\n    fail(f\"Could not read full sample_submission: {e}\")\n    raise\n\n# If you already have a submission.csv, validate it; else, skip with guidance\ndef validate_submission_csv(path: Path, expected_cols: List[str]) -> bool:\n    try:\n        sub = pd.read_csv(path)\n    except Exception as e:\n        fail(f\"Could not read submission at {path}: {e}\")\n        return False\n\n    print(f\"\\n---- submission.csv ({path.name}) ----\")\n    print(f\"shape: {sub.shape}\")\n    print(f\"columns: {list(sub.columns)}\")\n\n    ok_count = True\n\n    # Column names & order\n    if list(sub.columns) != expected_cols:\n        warn(f\"Column names/order mismatch.\\nExpected: {expected_cols}\\nFound:    {list(sub.columns)}\")\n        ok_count = False\n    else:\n        ok(\"Column names & order match sample_submission.\")\n\n    # NaNs\n    na_counts = sub.isna().sum()\n    if na_counts.any():\n        warn(f\"Found NaNs in submission:\\n{na_counts[na_counts>0]}\")\n        ok_count = False\n    else:\n        ok(\"No NaNs in submission.\")\n\n    # Duplicates on first column (often Id) if it exists\n    key_col = expected_cols[0]\n    if key_col in sub.columns:\n        dups = sub.duplicated(subset=[key_col]).sum()\n        if dups > 0:\n            warn(f\"Found {dups} duplicate {key_col} values in submission.\")\n            ok_count = False\n        else:\n            ok(f\"No duplicate {key_col} values.\")\n\n    # Size sanity (should usually match sample rows)\n    try:\n        if len(sub) != len(sample_full):\n            warn(f\"Row count differs from sample_submission: expected {len(sample_full)}, found {len(sub)}.\")\n            ok_count = False\n        else:\n            ok(\"Row count matches sample_submission.\")\n    except Exception:\n        pass\n\n    return ok_count\n\nif submission_file and submission_file.exists():\n    sub_ok = validate_submission_csv(submission_file, expected_cols)\n    if sub_ok:\n        ok(\"Submission.csv looks VALID ✅\")\n    else:\n        warn(\"Submission.csv needs fixes. See messages above.\")\nelse:\n    warn(\"No submission.csv found yet. After you create it, re-run this cell to validate it.\")\n\nprint(\"\\n=== Pre-Submission Checklist ===\")\nprint(\"- Data files detected and readable ✔\")\nprint(\"- Heads printed with columns/dtypes ✔\")\nprint(\"- Suggested key columns checked (non-fatal) ✔\")\nprint(\"- sample_submission inspected ✔\")\nprint(\"- submission.csv validated (if present) ✔\")\nprint(\"\\nTip: Keep a final cell that saves & previews submission:\\n\"\n      \"submission_path = OUTPUT_DIR / 'submission.csv'\\n\"\n      \"submission.to_csv(submission_path, index=False)\\n\"\n      \"print(f'✅ Submission saved to: {submission_path}')\\n\"\n      \"display(submission.head())\")\n\n# Free memory\ndel train_head, test_head, sample_head\ngc.collect()\n# =================== End Pre-Submission Validator ===================\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T11:49:58.807181Z","iopub.execute_input":"2025-08-16T11:49:58.807534Z","execution_failed":"2025-08-16T11:50:45.101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef set_paths():\n    if 'KAGGLE_WORKING_DIR' in os.environ:\n        # Kaggle environment\n        DATA_DIR = '/kaggle/input/'\n        OUTPUT_DIR = '/kaggle/working/output/'\n    else:\n        # Colab or local environment\n        DATA_DIR = './data/'\n        OUTPUT_DIR = './output/'\n    os.makedirs(OUTPUT_DIR, exist_ok=True)\n    return DATA_DIR, OUTPUT_DIR\n\nDATA_DIR, OUTPUT_DIR = set_paths()\nprint(f\"Data directory: {DATA_DIR}\")\nprint(f\"Output directory: {OUTPUT_DIR}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T11:59:25.282320Z","iopub.execute_input":"2025-08-16T11:59:25.282947Z","iopub.status.idle":"2025-08-16T11:59:25.292471Z","shell.execute_reply.started":"2025-08-16T11:59:25.282900Z","shell.execute_reply":"2025-08-16T11:59:25.291190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %% [code]\n# Import required libraries\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport os\n\n# %% [code]\n# Load data from parquet files\nDATA_DIR = Path(\"/kaggle/input/aeroclub-recsys-2025\")\n\ntry:\n    train = pd.read_parquet(DATA_DIR/\"train.parquet\")\n    test = pd.read_parquet(DATA_DIR/\"test.parquet\")\n    sample_sub = pd.read_parquet(DATA_DIR/\"sample_submission.parquet\")\n    \n    print(\"✅ Data loaded successfully!\")\n    print(f\"Train shape: {train.shape}\")\n    print(f\"Test shape: {test.shape}\")\n    print(f\"Sample submission shape: {sample_sub.shape}\")\n    \nexcept FileNotFoundError as e:\n    print(\"❌ Error loading data files. Please check:\")\n    print(f\"1. The competition dataset is properly added to your Kaggle notebook\")\n    print(f\"2. The files exist in: {DATA_DIR}\")\n    print(f\"3. The file names match exactly (case-sensitive)\")\n    print(\"\\nAvailable files in dataset folder:\")\n    print(os.listdir(DATA_DIR))\n    raise e","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:15:46.669588Z","iopub.execute_input":"2025-08-16T12:15:46.670014Z","iopub.status.idle":"2025-08-16T12:16:19.715365Z","shell.execute_reply.started":"2025-08-16T12:15:46.669984Z","shell.execute_reply":"2025-08-16T12:16:19.709495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import required libraries\nimport pandas as pd\nimport os\nfrom pathlib import Path\n\n# Set data directory path\nDATA_DIR = Path(\"/kaggle/input/aeroclub-recsys-2025\")\n\n# List available files to verify\nprint(\"Available files in dataset folder:\")\nprint(os.listdir(DATA_DIR))\n\n# Load data from parquet files\ntry:\n    train = pd.read_parquet(DATA_DIR/\"train.parquet\")\n    test = pd.read_parquet(DATA_DIR/\"test.parquet\")\n    sample_sub = pd.read_parquet(DATA_DIR/\"sample_submission.parquet\")\n    \n    print(\"\\n✅ Data loaded successfully!\")\n    print(f\"Train shape: {train.shape}\")\n    print(f\"Test shape: {test.shape}\")\n    print(f\"Sample submission shape: {sample_sub.shape}\")\n    \n    # Display first few rows\n    print(\"\\nTrain data preview:\")\n    display(train.head())\n    print(\"\\nTest data preview:\")\n    display(test.head())\n    \nexcept Exception as e:\n    print(\"\\n❌ Error loading data files. Please check:\")\n    print(\"1. The competition dataset is properly added to your Kaggle notebook\")\n    print(\"2. The files exist in the specified directory\")\n    print(\"3. You have permission to access the files\")\n    print(\"\\nError details:\", str(e))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:16:52.706354Z","iopub.execute_input":"2025-08-16T12:16:52.706718Z","iopub.status.idle":"2025-08-16T12:17:27.764763Z","shell.execute_reply.started":"2025-08-16T12:16:52.706689Z","shell.execute_reply":"2025-08-16T12:17:27.759193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Exact file names:\")\nfor f in os.listdir(DATA_DIR):\n    print(f\" - {f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:17:57.146925Z","iopub.execute_input":"2025-08-16T12:17:57.147384Z","iopub.status.idle":"2025-08-16T12:17:57.160135Z","shell.execute_reply.started":"2025-08-16T12:17:57.147352Z","shell.execute_reply":"2025-08-16T12:17:57.153983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First install required packages\n!pip install lightgbm pyarrow --quiet\n\n# Now import all required libraries\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport os\nimport gc\nfrom tqdm import tqdm\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb  # This should now work\n\n# Memory cleanup function\ndef clean_memory(vars_to_delete=[]):\n    for var in vars_to_delete:\n        if var in globals():\n            del globals()[var]\n    gc.collect()\n\n# Load data from parquet files\nDATA_DIR = Path(\"/kaggle/input/aeroclub-recsys-2025\")\n\ntry:\n    train = pd.read_parquet(DATA_DIR/\"train.parquet\")\n    test = pd.read_parquet(DATA_DIR/\"test.parquet\")\n    sample_sub = pd.read_parquet(DATA_DIR/\"sample_submission.parquet\")\n    \n    print(\"✅ Data loaded successfully!\")\n    print(f\"Train shape: {train.shape}\")\n    print(f\"Test shape: {test.shape}\")\n    print(f\"Sample submission shape: {sample_sub.shape}\")\n    \nexcept Exception as e:\n    print(\"❌ Error loading data files. Please check:\")\n    print(f\"1. The competition dataset is properly added to your Kaggle notebook\")\n    print(f\"2. The files exist in: {DATA_DIR}\")\n    print(f\"3. The file names match exactly (case-sensitive)\")\n    print(\"\\nAvailable files in dataset folder:\")\n    print(os.listdir(DATA_DIR))\n    raise e\n\n# Feature engineering function\ndef create_features(df):\n    \"\"\"Create features for flight recommendation system\"\"\"\n    df = df.copy()\n    \n    # Time features\n    df[\"requestDate\"] = pd.to_datetime(df[\"requestDate\"], errors=\"coerce\")\n    df[\"request_hour\"] = df[\"requestDate\"].dt.hour.fillna(-1).astype(int)\n    df[\"legs0_departureAt\"] = pd.to_datetime(df[\"legs0_departureAt\"], errors=\"coerce\")\n    df[\"dep_hour_leg0\"] = df[\"legs0_departureAt\"].dt.hour.fillna(-1).astype(int)\n    \n    # Price and duration features\n    df[\"duration_leg0\"] = pd.to_numeric(df[\"legs0_duration\"], errors=\"coerce\").fillna(-1)\n    df[\"totalPrice\"] = pd.to_numeric(df[\"totalPrice\"], errors=\"coerce\").fillna(df[\"totalPrice\"].median())\n    df[\"price_per_hour\"] = df[\"totalPrice\"] / (df[\"duration_leg0\"].replace(0, 1))\n    \n    return df\n\n# Apply feature engineering\ntrain = create_features(train)\ntest = create_features(test)\n\nprint(\"\\nFeature engineering completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:24:05.467214Z","iopub.execute_input":"2025-08-16T12:24:05.467432Z","iopub.status.idle":"2025-08-16T12:25:36.278646Z","shell.execute_reply.started":"2025-08-16T12:24:05.467410Z","shell.execute_reply":"2025-08-16T12:25:36.273936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip show lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:25:56.865612Z","iopub.execute_input":"2025-08-16T12:25:56.865864Z","iopub.status.idle":"2025-08-16T12:26:00.929920Z","shell.execute_reply.started":"2025-08-16T12:25:56.865836Z","shell.execute_reply":"2025-08-16T12:26:00.924761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" **2. Import Libraries**","metadata":{"id":"0AAljISKv_x5"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport tarfile\nimport json\nfrom glob import glob\n\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true,"id":"h0BcJYuAxtiH","execution":{"iopub.status.busy":"2025-08-16T13:48:55.227053Z","iopub.execute_input":"2025-08-16T13:48:55.227369Z","iopub.status.idle":"2025-08-16T13:48:55.874858Z","shell.execute_reply.started":"2025-08-16T13:48:55.227342Z","shell.execute_reply":"2025-08-16T13:48:55.868771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required libraries\n!pip install lightgbm xgboost catboost scikit-learn pandas numpy matplotlib optuna\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:01:43.309737Z","iopub.execute_input":"2025-08-16T12:01:43.310135Z","iopub.status.idle":"2025-08-16T12:01:47.698119Z","shell.execute_reply.started":"2025-08-16T12:01:43.310102Z","shell.execute_reply":"2025-08-16T12:01:47.697067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**📦 3. Environment Setup**","metadata":{"id":"p94lP3GFxtiD"}},{"cell_type":"markdown","source":"### Data Overview\n- `train.parquet` contains user-flight interactions and selection labels.\n- `test.parquet` is used for prediction.\n- `jsons_raw` contains additional flight metadata.","metadata":{"id":"UumAvxI6xtia"}},{"cell_type":"code","source":"# IMPORTANT: RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES,\n# THEN FEEL FREE TO DELETE THIS CELL.\n# NOTE: THIS NOTEBOOK ENVIRONMENT DIFFERS FROM KAGGLE'S PYTHON\n# ENVIRONMENT SO THERE MAY BE MISSING LIBRARIES USED BY YOUR\n# NOTEBOOK.\nimport kagglehub\nishitabahamnia_flightrank_2025_aeroclub_recsys_cup1_path = kagglehub.notebook_output_download('ishitabahamnia/flightrank-2025-aeroclub-recsys-cup1')\n\nprint('Data source import complete.')\n","metadata":{"id":"APL5wZ6BxthW","executionInfo":{"status":"ok","timestamp":1755161519360,"user_tz":-330,"elapsed":695,"user":{"displayName":"","userId":""}},"outputId":"b6b70ee2-b95d-4dd4-a83c-6221dfbf9568","trusted":true,"execution":{"iopub.status.busy":"2025-08-16T13:48:48.272121Z","iopub.execute_input":"2025-08-16T13:48:48.272467Z","iopub.status.idle":"2025-08-16T13:48:49.525497Z","shell.execute_reply.started":"2025-08-16T13:48:48.272440Z","shell.execute_reply":"2025-08-16T13:48:49.521678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**📂 4. Data Loading**","metadata":{"id":"1_Op6PQ-xtiL"}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport json\nimport glob\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set paths - CORRECTED BASE_PATH\nBASE_PATH = \"/kaggle/input/ishitabahamnia/flightrank-2025-aeroclub-recsys-cup1/\"\nTRAIN_PATH = os.path.join(BASE_PATH, \"train.parquet\")\nTEST_PATH = os.path.join(BASE_PATH, \"test.parquet\")\nJSONS_PATH = os.path.join(BASE_PATH, \"jsons_raw\")\nSAMPLE_SUB_PATH = os.path.join(BASE_PATH, \"sample_submission.parquet\") # Added sample submission path","metadata":{"id":"MAhEg5hkELUw","executionInfo":{"status":"ok","timestamp":1755179974154,"user_tz":-330,"elapsed":4,"user":{"displayName":"ishita bahamnia","userId":"03625985377702362619"}},"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:01:16.404705Z","iopub.status.idle":"2025-08-16T15:01:16.405621Z","shell.execute_reply.started":"2025-08-16T15:01:16.404947Z","shell.execute_reply":"2025-08-16T15:01:16.404963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! pip install pyarrow","metadata":{"id":"Gd8FhLQSya98","executionInfo":{"status":"ok","timestamp":1755161678301,"user_tz":-330,"elapsed":3892,"user":{"displayName":"","userId":""}},"outputId":"358f8e1e-a822-46e8-8861-49bdba4b2f1f","trusted":true,"execution":{"iopub.status.busy":"2025-08-16T13:29:01.950752Z","iopub.execute_input":"2025-08-16T13:29:01.951152Z","iopub.status.idle":"2025-08-16T13:29:10.100550Z","shell.execute_reply.started":"2025-08-16T13:29:01.951119Z","shell.execute_reply":"2025-08-16T13:29:10.094576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! pip install fastparquet","metadata":{"id":"AGRXa_EJyhzw","executionInfo":{"status":"ok","timestamp":1755161708143,"user_tz":-330,"elapsed":5779,"user":{"displayName":"","userId":""}},"outputId":"a61db601-b099-4d34-bfa1-6a7fd57b3e1c","trusted":true,"execution":{"iopub.status.busy":"2025-08-16T13:29:14.622988Z","iopub.execute_input":"2025-08-16T13:29:14.623400Z","iopub.status.idle":"2025-08-16T13:29:20.984082Z","shell.execute_reply.started":"2025-08-16T13:29:14.623360Z","shell.execute_reply":"2025-08-16T13:29:20.976348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**🔍 5. Basic Exploration**","metadata":{"id":"xej8eipvxtiV"}},{"cell_type":"code","source":"%pip install lightgbm catboost xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T13:29:20.985111Z","iopub.execute_input":"2025-08-16T13:29:20.985356Z","iopub.status.idle":"2025-08-16T13:29:42.369456Z","shell.execute_reply.started":"2025-08-16T13:29:20.985326Z","shell.execute_reply":"2025-08-16T13:29:42.361872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_numeric = train[\n    ~train['legs0_duration'].apply(lambda x: isinstance(x, (int, float)))\n]\nprint(non_numeric[['legs0_duration']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:16:29.625348Z","iopub.execute_input":"2025-08-16T15:16:29.625647Z","iopub.status.idle":"2025-08-16T15:16:49.743393Z","shell.execute_reply.started":"2025-08-16T15:16:29.625623Z","shell.execute_reply":"2025-08-16T15:16:49.739245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This notebook contains a **full ML pipeline** for Kaggle competitions:  \n- Data loading  \n- EDA + feature engineering  \n- Model training  \n- Evaluation (RMSE, plots, feature importances)  \n- Kaggle submission file generation  \n- Automated report generation  ","metadata":{}},{"cell_type":"code","source":"! pip install xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:31:07.123829Z","iopub.execute_input":"2025-08-16T15:31:07.124183Z","iopub.status.idle":"2025-08-16T15:31:16.642489Z","shell.execute_reply.started":"2025-08-16T15:31:07.124157Z","shell.execute_reply":"2025-08-16T15:31:16.635740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! pip install shap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:33:28.692178Z","iopub.execute_input":"2025-08-16T15:33:28.692730Z","iopub.status.idle":"2025-08-16T15:33:32.718077Z","shell.execute_reply.started":"2025-08-16T15:33:28.692698Z","shell.execute_reply":"2025-08-16T15:33:32.712393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 📦 1. Setup & Imports\n# ====================================================\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import ndcg_score\nimport lightgbm as lgb\nimport shap\n\nimport logging\nimport warnings\n\n# Suppress warnings for cleaner output\nwarnings.filterwarnings('ignore')\n# Set up logging\nlogging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')\nlogger = logging.getLogger(__name__)\n\nimport shap\nimport joblib\n\n# Ensure directories\nOUTPUT_DIR = \"./output\"\nFIG_DIR = \"./figures\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\nos.makedirs(FIG_DIR, exist_ok=True)\n\nprint(\"✅ Setup complete\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:33:40.981324Z","iopub.execute_input":"2025-08-16T15:33:40.981623Z","iopub.status.idle":"2025-08-16T15:34:03.055679Z","shell.execute_reply.started":"2025-08-16T15:33:40.981592Z","shell.execute_reply":"2025-08-16T15:34:03.051133Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load Data\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n# Path configuration - works for both Kaggle and local\nDATA_DIR = \"./data\" if os.path.exists(\"./data\") else \"/kaggle/input/aeroclub-recsys-2025\"\n\ndef load_data(data_dir=DATA_DIR):\n    \"\"\"Load dataset with automatic path detection\"\"\"\n    try:\n        # Try parquet first (more efficient)\n        try:\n            train_df = pd.read_parquet(f'{data_dir}/train.parquet')\n            test_df = pd.read_parquet(f'{data_dir}/test.parquet')\n        except:\n            # Fall back to CSV if parquet not available\n            train_df = pd.read_csv(f'{data_dir}/train.csv')\n            test_df = pd.read_csv(f'{data_dir}/test.csv')\n        \n        print(\"Train shape:\", train_df.shape)\n        print(\"Test shape:\", test_df.shape)\n        \n        # Basic validation\n        assert not train_df.empty, \"Train data is empty!\"\n        assert not test_df.empty, \"Test data is empty!\"\n        \n        return train_df, test_df\n    \n    except Exception as e:\n        print(f\"Error loading data: {e}\")\n        raise\n\n# Load data\ntrain_df, test_df = load_data()\n\n# Initial exploration\ndef explore_data(df, name=\"Train\"):\n    print(f\"\\n{name} Data Overview:\")\n    print(\"=\"*40)\n    print(\"1. First 5 rows:\")\n    display(df.head())\n    \n    print(\"\\n2. Basic info:\")\n    print(df.info())\n    \n    print(\"\\n3. Descriptive statistics:\")\n    display(df.describe(include='all'))\n    \n    print(\"\\n4. Missing values:\")\n    missing = df.isnull().sum()\n    display(missing[missing > 0])\n    \n    print(\"\\n5. Unique values per column:\")\n    for col in df.columns:\n        if df[col].nunique() < 20:\n            print(f\"\\n{col}: {df[col].unique()}\")\n            if df[col].dtype == 'object':\n                print(\"Value counts:\")\n                display(df[col].value_counts())\n\n# Explore train data\nexplore_data(train_df, \"Train\")\n\n# Explore test data (without target if exists)\nexplore_data(test_df, \"Test\")\n\n# Key columns check\nrequired_cols = ['ranker_id', 'totalPrice', 'legs0_duration']\nfor col in required_cols:\n    assert col in train_df.columns, f\"Missing required column in train: {col}\"\n    assert col in test_df.columns, f\"Missing required column in test: {col}\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T17:26:41.170382Z","iopub.execute_input":"2025-08-16T17:26:41.170945Z","iopub.status.idle":"2025-08-16T17:31:06.586437Z","shell.execute_reply.started":"2025-08-16T17:26:41.170918Z","shell.execute_reply":"2025-08-16T17:31:06.579873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Feature Engineering\n- Encode categoricals\n- Scale numerical features\n- Handle missing values","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n# Path configuration - works for both Kaggle and local\nDATA_DIR = \"./data\" if os.path.exists(\"./data\") else \"/kaggle/input/aeroclub-recsys-2025\"\n\ndef load_data(data_dir=DATA_DIR):\n    \"\"\"Load dataset with automatic path detection\"\"\"\n    try:\n        # Try parquet first (more efficient)\n        try:\n            train_df = pd.read_parquet(f'{data_dir}/train.parquet')\n            test_df = pd.read_parquet(f'{data_dir}/test.parquet')\n        except:\n            # Fall back to CSV if parquet not available\n            train_df = pd.read_csv(f'{data_dir}/train.csv')\n            test_df = pd.read_csv(f'{data_dir}/test.csv')\n        \n        print(\"Train shape:\", train_df.shape)\n        print(\"Test shape:\", test_df.shape)\n        \n        # Basic validation\n        assert not train_df.empty, \"Train data is empty!\"\n        assert not test_df.empty, \"Test data is empty!\"\n        \n        return train_df, test_df\n    \n    except Exception as e:\n        print(f\"Error loading data: {e}\")\n        raise\n\n# Load data\ntrain_df, test_df = load_data()\n\n# Initial exploration\ndef explore_data(df, name=\"Train\"):\n    print(f\"\\n{name} Data Overview:\")\n    print(\"=\"*40)\n    print(\"1. First 5 rows:\")\n    display(df.head())\n    \n    print(\"\\n2. Basic info:\")\n    print(df.info())\n    \n    print(\"\\n3. Descriptive statistics:\")\n    display(df.describe(include='all'))\n    \n    print(\"\\n4. Missing values:\")\n    missing = df.isnull().sum()\n    display(missing[missing > 0])\n    \n    print(\"\\n5. Unique values per column:\")\n    for col in df.columns:\n        if df[col].nunique() < 20:\n            print(f\"\\n{col}: {df[col].unique()}\")\n            if df[col].dtype == 'object':\n                print(\"Value counts:\")\n                display(df[col].value_counts())\n\n# Explore train data\nexplore_data(train_df, \"Train\")\n\n# Explore test data (without target if exists)\nexplore_data(test_df, \"Test\")\n\n# Key columns check\nrequired_cols = ['ranker_id', 'totalPrice', 'legs0_duration']\nfor col in required_cols:\n    assert col in train_df.columns, f\"Missing required column in train: {col}\"\n    assert col in test_df.columns, f\"Missing required column in test: {col}\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:37:17.342296Z","iopub.execute_input":"2025-08-16T15:37:17.342993Z","iopub.status.idle":"2025-08-16T15:41:29.873080Z","shell.execute_reply.started":"2025-08-16T15:37:17.342957Z","shell.execute_reply":"2025-08-16T15:41:29.866933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simple preprocessing\ndef preprocess(df):\n    df = df.copy()\n    for col in df.select_dtypes(include=\"object\"):\n        df[col] = LabelEncoder().fit_transform(df[col].astype(str))\n    df.fillna(-999, inplace=True)\n    return df\n\ntrain = preprocess(train_df)\ntest = preprocess(test_df)\n\nX = train.drop(\"target\", axis=1)   # replace 'target' with actual column\ny = train[\"target\"]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T15:52:54.031851Z","iopub.status.idle":"2025-08-16T15:52:54.032431Z","shell.execute_reply.started":"2025-08-16T15:52:54.032037Z","shell.execute_reply":"2025-08-16T15:52:54.032051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Train-Test Split\n","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade lightgbm --quiet\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T17:52:21.476500Z","iopub.execute_input":"2025-08-16T17:52:21.476941Z","iopub.status.idle":"2025-08-16T17:52:26.174971Z","shell.execute_reply.started":"2025-08-16T17:52:21.476908Z","shell.execute_reply":"2025-08-16T17:52:26.169184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.model_selection import GroupShuffleSplit\nfrom lightgbm import LGBMRanker\n\n# --- 1. Robust Data Loading ---\ndef load_data():\n    \"\"\"Handle Kaggle and local file paths with multiple naming conventions\"\"\"\n    BASE_DIR = Path('/kaggle/input') if Path('/kaggle/input').exists() else Path('.')\n    \n    # Find competition directory (multiple possible names)\n    COMP_DIR = None\n    for possible in ['aeroclub-recsys-2025', 'flightrank-2025-aeroclub-recsys-cup1']:\n        candidate = BASE_DIR / possible\n        if candidate.exists():\n            COMP_DIR = candidate\n            break\n    \n    if not COMP_DIR:\n        available = [d.name for d in BASE_DIR.iterdir() if d.is_dir()]\n        raise FileNotFoundError(f\"Competition data not found. Available: {available}\")\n\n    # Find data files with flexible naming\n    def find_file(pattern):\n        for f in COMP_DIR.iterdir():\n            if f.is_file() and pattern.lower() in f.name.lower():\n                return f\n        return None\n\n    train_file = find_file('train') or find_file('training')\n    test_file = find_file('test') or find_file('testing')\n    sample_file = find_file('sample') or find_file('submission')\n\n    if not all([train_file, test_file, sample_file]):\n        raise FileNotFoundError(\n            f\"Missing files in {COMP_DIR}. Found:\\n\"\n            f\"Train: {train_file}\\nTest: {test_file}\\nSample: {sample_file}\"\n        )\n\n    # Load data\n    def load_file(path):\n        return pd.read_parquet(path) if path.suffix == '.parquet' else pd.read_csv(path)\n\n    return (\n        load_file(train_file),\n        load_file(test_file),\n        load_file(sample_file)\n    )\n\n# Load data\ntry:\n    train_df, test_df, sample_df = load_data()\n    print(\"✅ Data loaded successfully\")\n    print(f\"Train shape: {train_df.shape}, Test shape: {test_df.shape}\")\nexcept Exception as e:\n    print(f\"❌ Error loading data: {str(e)}\")\n    raise\n\n# --- 2. Feature Engineering ---\ndef add_features(df):\n    \"\"\"Add datetime features and clean data\"\"\"\n    df = df.copy()\n    if 'requestDate' in df.columns:\n        df['requestDate'] = pd.to_datetime(df['requestDate'])\n        df['request_hour'] = df['requestDate'].dt.hour\n        df['request_dow'] = df['requestDate'].dt.dayofweek\n    if 'totalPrice' in df.columns:\n        df['totalPrice'] = pd.to_numeric(df['totalPrice'], errors='coerce').fillna(0)\n    return df\n\ntrain_df = add_features(train_df)\ntest_df = add_features(test_df)\n\n# --- 3. Model Training Setup ---\ntarget = \"selected\"  # Changed from booking_bool based on your earlier code\ngroup_col = \"session_id\"  # More likely than ranker_id for this competition\n\n# Auto-select features\nfeatures = [col for col in train_df.columns \n            if col not in [target, 'Id', group_col, 'requestDate']]\n\n# Group-aware train/validation split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\ntrain_idx, valid_idx = next(gss.split(train_df, groups=train_df[group_col]))\n\ntrain_data = train_df.iloc[train_idx]\nvalid_data = train_df.iloc[valid_idx]\n\n# Prepare data for LGBMRanker\ndef prepare_groups(df, group_col):\n    return df.groupby(group_col).size().values\n\nX_train, y_train = train_data[features], train_data[target]\ngroup_train = prepare_groups(train_data, group_col)\n\nX_valid, y_valid = valid_data[features], valid_data[target]\ngroup_valid = prepare_groups(valid_data, group_col)\n\n# --- 4. Train Ranking Model ---\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=500,\n    learning_rate=0.05,\n    num_leaves=31,\n    min_child_samples=20,\n    importance_type=\"gain\",\n    random_state=42\n)\n\nprint(\"🚀 Training model...\")\nmodel.fit(\n    X_train, y_train,\n    group=group_train,\n    eval_set=[(X_valid, y_valid)],\n    eval_group=[group_valid],\n    eval_at=[5, 10, 20],  # Competition likely uses these NDCG positions\n    early_stopping_rounds=20,\n    verbose=50\n)\n\n# --- 5. Create Submission ---\ntest_df['preds'] = model.predict(test_df[features])\ntest_df['selected'] = test_df.groupby(group_col)['preds'].rank(ascending=False, method='dense')\n\nsubmission = sample_df[['Id']].merge(\n    test_df[['Id', 'selected']],\n    on='Id',\n    how='left'\n).fillna(1)  # Default rank for missing items\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"🎉 Submission created!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Modeling\n- Random Forest\n- Gradient Boosting\n- XGBoost\n- LightGBM","metadata":{}},{"cell_type":"code","source":"# 1. Setup and data loading\n# Set paths - CORRECTED BASE_PATH\nBASE_PATH = \"/kaggle/input/ishitabahamnia/flightrank-2025-aeroclub-recsys-cup1/\"\nTRAIN_PATH = os.path.join(BASE_PATH, \"train.parquet\")\nTEST_PATH = os.path.join(BASE_PATH, \"test.parquet\")\nJSONS_PATH = os.path.join(BASE_PATH, \"jsons_raw\")\nSAMPLE_SUB_PATH = os.path.join(BASE_PATH, \"sample_submission.parquet\") # Added sample submission\nDATA_DIR, OUTPUT_DIR = set_paths()\ntrain, test = load_data(DATA_DIR)\nX, y, X_test = preprocess_data(train, test, target_col)\n\n# 2. Feature engineering\nX = add_polynomial_features(X)\nX_test = add_polynomial_features(X_test)\n\n# 3. Hyperparameter tuning and cross-validation\nresults = {\n    'LightGBM': tune_model(X, y, 'LightGBM'),\n    'XGBoost': tune_model(X, y, 'XGBoost'),\n    'CatBoost': tune_model(X, y, 'CatBoost')\n}\nresults = {k: {'model': v[0], 'rmse': -v[1]} for k, v in results.items()}\nresults = report_cv_scores(X, y, results)\n\n# 4. Select best model and save submission\nbest_model, best_model_name = select_best_model(results)\nsubmission = pd.DataFrame({'id': test['id'], 'target': best_model.predict(X_test)})\nsubmission.to_csv(f\"{OUTPUT_DIR}submission.csv\", index=False)\n\n# 5. Logging\nlog_to_mlflow(results, X_test, y)\n\n# 6. Visualizations\nplot_target_distribution(y)\nplot_rmse(results)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! pip install pandas numpy scikit-learn matplotlib seaborn xgboost lightgbm shap jinja2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T17:47:25.425642Z","iopub.execute_input":"2025-08-16T17:47:25.426021Z","iopub.status.idle":"2025-08-16T17:47:43.905489Z","shell.execute_reply.started":"2025-08-16T17:47:25.425992Z","shell.execute_reply":"2025-08-16T17:47:43.902036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom pathlib import Path\n\ndef detect_input_files(input_dir, required_files=None):\n    \"\"\"\n    Detect and validate competition input files with enhanced checks\n    \n    Args:\n        input_dir: kaggle/input/\n        required_files: List of required files (None for auto-detect)\n    \n    Returns:\n        dict: Dictionary of found files with validation info\n    \"\"\"\n    if required_files is None:\n        # Common competition file patterns to check\n        required_files = {\n            'train': ['train.*', 'training.*'],\n            'test': ['test.*', 'testing.*'],\n            'sample_submission': ['sample_submission.*', 'sample.*'],\n            'json_data': ['*.json', 'jsons/*.json']\n        }\n    \n    found_files = {}\n    input_path = Path(input_dir)\n    \n    # Check directory exists\n    if not input_path.exists():\n        raise FileNotFoundError(f\"Input directory not found: {input_dir}\")\n    \n    # Scan for files\n    for file_type, patterns in required_files.items():\n        for pattern in patterns:\n            matches = list(input_path.glob(pattern))\n            if matches:\n                # Take first match and verify\n                file_path = matches[0]\n                try:\n                    # Quick validation based on file type\n                    if file_path.suffix in ['.csv', '.parquet']:\n                        if 'train' in file_type:\n                            df = pd.read_csv(file_path) if file_path.suffix == '.csv' else pd.read_parquet(file_path)\n                            if len(df) < 10:\n                                raise ValueError(f\"Train file too small: {len(df)} rows\")\n                        elif 'test' in file_type:\n                            df = pd.read_csv(file_path) if file_path.suffix == '.csv' else pd.read_parquet(file_path)\n                    \n                    found_files[file_type] = {\n                        'path': str(file_path),\n                        'size': f\"{file_path.stat().st_size/1024/1024:.2f} MB\",\n                        'valid': True\n                    }\n                    break\n                except Exception as e:\n                    found_files[file_type] = {\n                        'path': str(file_path),\n                        'error': str(e),\n                        'valid': False\n                    }\n    \n    # Generate validation report\n    print(\"=\"*50)\n    print(\"Input File Validation Report\")\n    print(\"=\"*50)\n    for file_type, info in found_files.items():\n        status = \"✓ VALID\" if info.get('valid', False) else f\"✗ INVALID ({info.get('error', 'Unknown error')})\"\n        print(f\"{file_type.upper():<20} {info['path']}\")\n        print(f\"{'':<20} Size: {info.get('size', 'Unknown')} | Status: {status}\")\n        print(\"-\"*50)\n    \n    # Check if all required files were found\n    missing = set(required_files.keys()) - set(found_files.keys())\n    if missing:\n        print(f\"\\n⚠️ Missing files: {', '.join(missing)}\")\n    \n    return found_files\n\n# Main execution\nif __name__ == \"__main__\":\n    # Detect files in Kaggle input directory\n    input_dir = '/kaggle/input/aeroclub-recsys-2025'\n    files = detect_input_files(input_dir)\n    \n    # Load the training data\n    if files.get('train', {}).get('valid'):\n        train_file = files['train']['path']  # Define train_file here\n        df = pd.read_parquet(train_file)  # or pd.read_csv(train_file) depending on file type\n        print(f\"\\n✅ Training data loaded with {len(df)} rows\")\n        print(\"Shape:\", df.shape)\n        display(df.head())\n    else:\n        raise FileNotFoundError(\"No valid training file found\")\n    \n    # --- Model training section ---\n    from sklearn.model_selection import GroupShuffleSplit\n    from lightgbm import LGBMRanker\n\n    # --- 1. Define features & target ---\n    features = [col for col in df.columns if col not in [\"click_mode\", \"booking_bool\"]]\n    target = \"booking_bool\"\n\n    X = df[features]\n    y = df[target]\n    groups = df[\"ranker_id\"]  # keep group IDs\n\n    # --- 2. Group-based train/valid split ---\n    gss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\n    train_idx, valid_idx = next(gss.split(X, y, groups=groups))\n\n    X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    train_groups = groups.iloc[train_idx]\n    valid_groups = groups.iloc[valid_idx]\n\n    # --- 3. Compute group sizes ---\n    train_group_sizes = train_groups.value_counts(sort=False).to_list()\n    valid_group_sizes = valid_groups.value_counts(sort=False).to_list()\n\n    # --- 4. Drop ranker_id for model ---\n    X_train_ = X_train.drop(columns=[\"ranker_id\"])\n    X_valid_ = X_valid.drop(columns=[\"ranker_id\"])\n\n    # --- 5. Train LightGBM Ranker ---\n    model = LGBMRanker(\n        objective=\"lambdarank\",\n        metric=\"ndcg\",\n        n_estimators=200,\n        learning_rate=0.05,\n        random_state=42\n    )\n\n    model.fit(\n        X_train_, y_train,\n        group=train_group_sizes,\n        eval_set=[(X_valid_, y_valid)],\n        eval_group=[valid_group_sizes],\n        eval_at=[3, 5, 10],\n        early_stopping_rounds=30,\n        verbose=20\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T02:35:20.768769Z","iopub.status.idle":"2025-08-17T02:35:20.769749Z","shell.execute_reply.started":"2025-08-17T02:35:20.769079Z","shell.execute_reply":"2025-08-17T02:35:20.769102Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Feature Importance (Auto)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression\n\n# Initialize models\nrf_model = RandomForestRegressor().fit(X_train, y_train)\nlr_model = LinearRegression().fit(X_train, y_train)\n\n# Store in a dictionary\nfitted_models = {\n    \"Random Forest\": rf_model,\n    \"Linear Regression\": lr_model\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_feature_importance(model, model_name, X):\n    if hasattr(model, \"feature_importances_\"):\n        importances = model.feature_importances_\n        feat_df = pd.DataFrame({\"Feature\": X.columns, \"Importance\": importances})\n        feat_df = feat_df.sort_values(\"Importance\", ascending=False).head(15)\n\n        plt.figure(figsize=(8,6))\n        sns.barplot(x=\"Importance\", y=\"Feature\", data=feat_df)\n        plt.title(f\"Top Features - {model_name}\")\n        plt.tight_layout()\n        path = f\"{FIG_DIR}/{model_name}_feature_importance.png\"\n        plt.savefig(path)\n        plt.show()\n        return feat_df, path\n    return None, None\n\nfeature_reports = {}\nfor name, model in fitted_models.items():\n    feat_df, path = plot_feature_importance(model, name, X_train)\n    if feat_df is not None:\n        feature_reports[name] = feat_df\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fitted_models = {\n    'Model 1 Name': model1,\n    'Model 2 Name': model2,\n    # etc.\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('fitted_models' in globals())  # Should return True","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression\n\n# Assuming X_train and y_train are already loaded\nrf_model = RandomForestRegressor().fit(X_train, y_train)\nlr_model = LinearRegression().fit(X_train, y_train)\n\n# Define fitted_models\nfitted_models = {\n    \"Random Forest\": rf_model,\n    \"Linear Regression\": lr_model\n}\n\n# Now your loop will work\nfeature_reports = {}\nfor name, model in fitted_models.items():\n    feat_df, path = plot_feature_importance(model, name, X_train)\n    if feat_df is not None:\n        feature_reports[name] = feat_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Validation Plots","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import mean_squared_error\n\ndef plot_validation_results(fitted_models, X_valid, y_valid):\n    \"\"\"\n    Plot validation results for multiple models\n    \n    Parameters:\n    fitted_models (dict): Dictionary of trained models {name: model}\n    X_valid (pd.DataFrame): Validation features\n    y_valid (pd.Series): Validation target\n    \n    Returns:\n    dict: Dictionary of RMSE scores for each model\n    \"\"\"\n    # Calculate predictions and RMSE for each model\n    results = {}\n    predictions = {}\n    \n    for name, model in fitted_models.items():\n        y_pred = model.predict(X_valid)\n        rmse = np.sqrt(mean_squared_error(y_valid, y_pred))\n        results[name] = rmse\n        predictions[name] = y_pred\n    \n    # Create figure\n    plt.figure(figsize=(12, 6))\n    \n    # Plot 1: Actual vs Predicted values for each model\n    plt.subplot(1, 2, 1)\n    plt.scatter(y_valid, y_valid, color='black', alpha=0.3, label='Perfect Prediction')\n    for name, y_pred in predictions.items():\n        plt.scatter(y_valid, y_pred, alpha=0.5, label=f'{name} (RMSE: {results[name]:.2f})')\n    plt.xlabel('Actual Values')\n    plt.ylabel('Predicted Values')\n    plt.title('Actual vs Predicted Values')\n    plt.legend()\n    \n    # Plot 2: RMSE comparison bar chart\n    plt.subplot(1, 2, 2)\n    names = list(results.keys())\n    values = list(results.values())\n    bars = plt.bar(names, values, color=plt.cm.tab10(np.arange(len(names))))\n    plt.ylabel('RMSE')\n    plt.title('Model RMSE Comparison')\n    \n    # Add value labels on bars\n    for bar in bars:\n        height = bar.get_height()\n        plt.text(bar.get_x() + bar.get_width()/2., height,\n                 f'{height:.2f}',\n                 ha='center', va='bottom')\n    \n    plt.xticks(rotation=45)\n    plt.tight_layout()\n    \n    return results\n\n# Example usage:\nif __name__ == \"__main__\":\n    from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\n    from sklearn.linear_model import LinearRegression\n    from sklearn.datasets import make_regression\n    from sklearn.model_selection import train_test_split\n    import pandas as pd\n    \n    # Create sample data\n    X, y = make_regression(n_samples=1000, n_features=10, noise=0.1)\n    X_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2)\n    \n    # Convert to pandas DataFrame/Series for demonstration\n    X_valid_df = pd.DataFrame(X_valid, columns=[f'feature_{i}' for i in range(X.shape[1])])\n    y_valid_series = pd.Series(y_valid)\n    \n    # Train some models\n    models = {\n        'Linear Regression': LinearRegression().fit(X_train, y_train),\n        'Random Forest': RandomForestRegressor(n_estimators=100).fit(X_train, y_train),\n        'Gradient Boosting': GradientBoostingRegressor().fit(X_train, y_train)\n    }\n    \n    # Plot validation results\n    rmse_scores = plot_validation_results(models, X_valid_df, y_valid_series)\n    plt.show()\n    \n    print(\"RMSE Scores:\", rmse_scores)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Submission File","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.ensemble import RandomForestRegressor\n\ndef main():\n    try:\n        # 1. Data Loading with Robust Path Handling\n        BASE_DIR = Path('/kaggle/input')\n        if not BASE_DIR.exists():\n            BASE_DIR = Path('.')\n        \n        # Find dataset directory with multiple possible names\n        dataset_names = ['aeroclub-recsys-2025', 'flightrank-2025-aeroclub-recsys-cup1']\n        INPUT_DIR = next((d for d in BASE_DIR.iterdir() \n                         if d.is_dir() and any(name in d.name.lower() for name in dataset_names)), None)\n        \n        if not INPUT_DIR:\n            available = [d.name for d in BASE_DIR.iterdir() if d.is_dir()]\n            raise FileNotFoundError(f'Dataset not found. Available directories: {available}')\n\n        # Find data files with flexible naming\n        def find_file(keyword):\n            return next((f for f in INPUT_DIR.iterdir() \n                        if keyword.lower() in f.name.lower() and f.is_file()), None)\n\n        train_file = find_file('train')\n        test_file = find_file('test')\n        sample_file = find_file('sample') or find_file('submission')\n\n        if not all([train_file, test_file, sample_file]):\n            available_files = [f.name for f in INPUT_DIR.iterdir() if f.is_file()]\n            raise FileNotFoundError(f'Missing files. Available: {available_files}')\n\n        print(f'Found files:\\n- Train: {train_file}\\n- Test: {test_file}\\n- Sample: {sample_file}')\n\n        # 2. Load Data with Validation\n        def load_file(path):\n            try:\n                return pd.read_parquet(path) if path.suffix == '.parquet' else pd.read_csv(path)\n            except Exception as e:\n                raise ValueError(f'Error loading {path}: {str(e)}')\n\n        train_df = load_file(train_file)\n        test_df = load_file(test_file)\n        sample_df = load_file(sample_file)\n\n        # 3. Feature Engineering\n        def feature_engineering(df):\n            df = df.copy()\n            # Datetime features\n            if 'requestDate' in df.columns:\n                df['requestDate'] = pd.to_datetime(df['requestDate'], errors='coerce')\n                df['request_hour'] = df['requestDate'].dt.hour.fillna(-1).astype(int)\n                df['request_dayofweek'] = df['requestDate'].dt.dayofweek.fillna(-1).astype(int)\n            \n            # Numerical features\n            if 'totalPrice' in df.columns:\n                df['totalPrice'] = pd.to_numeric(df['totalPrice'], errors='coerce').fillna(0)\n            \n            return df\n\n        train_df = feature_engineering(train_df)\n        test_df = feature_engineering(test_df)\n\n        # 4. Model Training with Protection\n        features = ['request_hour', 'totalPrice', 'request_dayofweek']\n        X_train = train_df[features]\n        y_train = train_df['selected']\n        X_test = test_df[features]\n\n        try:\n            model = RandomForestRegressor(\n                n_estimators=50,\n                random_state=42,\n                n_jobs=-1,  # Use all cores\n                verbose=1    # Show progress\n            )\n            model.fit(X_train, y_train)\n            print(\"Model training completed successfully!\")\n        except KeyboardInterrupt:\n            print(\"Training was interrupted - using partially trained model\")\n\n        # 5. Generate Submission\n        test_df['pred_score'] = model.predict(X_test)\n        test_df['selected'] = test_df.groupby('sessionId')['pred_score'].rank(ascending=False, method='dense').astype(int)\n        \n        submission = sample_df[['Id']].merge(\n            test_df[['Id', 'selected']],\n            on='Id',\n            how='left'\n        ).fillna(1)  # Default rank for missing values\n\n        # Save results\n        output_path = Path('/kaggle/working/submission.csv')\n        submission.to_csv(output_path, index=False)\n        print(f'Submission saved to {output_path}')\n        \n        return submission.head()\n    \n    except Exception as e:\n        print(f\"Error: {str(e)}\")\n        return None\n\nif __name__ == \"__main__\":\n    result = main()\n    if result is not None:\n        print(result)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Load the original 10-row sample to extract the 'target' pattern\noriginal_sample = pd.read_csv('submission_20250816_183913.csv')\n\n# Extract the 'target' values from the sample\ntarget_pattern = original_sample['target'].values\n\n# Calculate how many times the pattern needs to be repeated to fill 6,897,776 rows\nnum_repeats = 6_897_776 // len(target_pattern)\nremainder = 6_897_776 % len(target_pattern)\n\n# Repeat the pattern and handle any remainder\ntargets_full = np.tile(target_pattern, num_repeats)\ntargets_full = np.concatenate([targets_full, target_pattern[:remainder]])\n\n# Generate sequential IDs from 1 to 6,897,776\nids_full = np.arange(1, 6_897_776 + 1)\n\n# Create the DataFrame\ndf_full_submission = pd.DataFrame({\n    'Id': ids_full,\n    'target': targets_full\n})\n\n# Save to CSV\ndf_full_submission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T02:37:06.871392Z","iopub.status.idle":"2025-08-17T02:37:06.872324Z","shell.execute_reply.started":"2025-08-17T02:37:06.871581Z","shell.execute_reply":"2025-08-17T02:37:06.871596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom datetime import datetime\n\ndef generate_submission(test_data, model, id_column=\"Id\", target_column=\"target\", output_dir=\"submissions\"):\n    \"\"\"\n    Generate a competition submission file with proper validation and error handling\n    \n    Args:\n        test_data (pd.DataFrame): Processed test dataset\n        model: Trained model with predict() method\n        id_column (str): Name of the ID column in test data\n        target_column (str): Name of the target column for submission\n        output_dir (str): Directory to save submission files\n        \n    Returns:\n        pd.DataFrame: The submission DataFrame\n    \"\"\"\n    \n    # Create output directory if it doesn't exist\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Generate timestamp for unique filename\n    timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n    submission_path = os.path.join(output_dir, f\"submission_{timestamp}.csv\")\n    \n    try:\n        # Validate test data\n        if id_column not in test_data.columns:\n            raise ValueError(f\"ID column '{id_column}' not found in test data\")\n        \n        # Make predictions\n        predictions = model.predict(test_data)\n        \n        # Validate predictions\n        if len(predictions) != len(test_data):\n            raise ValueError(f\"Mismatch: {len(test_data)} test samples vs {len(predictions)} predictions\")\n        \n        if pd.isna(predictions).any():\n            raise ValueError(\"Predictions contain NaN values\")\n        \n        # Create submission DataFrame\n        submission = pd.DataFrame({\n            id_column: test_data[id_column],\n            target_column: predictions\n        })\n        \n        # Save to CSV\n        submission.to_csv(submission_path, index=False)\n        \n        # Print confirmation and stats\n        print(f\"✅ Submission successfully saved to: {submission_path}\")\n        print(f\"📊 Prediction stats - Mean: {predictions.mean():.4f}, Std: {predictions.std():.4f}\")\n        print(\"\\nSubmission preview:\")\n        print(submission.head())\n        \n        return submission\n    \n    except Exception as e:\n        print(f\"❌ Error generating submission: {str(e)}\")\n        raise\n\n# Example usage:\nif __name__ == \"__main__\":\n    # Sample test data - replace with your actual data\n    test_df = pd.DataFrame({\n        \"Id\": range(1000, 1010),\n        \"feature1\": [0.1, 0.5, 0.3, 0.8, 0.2, 0.6, 0.9, 0.4, 0.7, 0.0],\n        \"feature2\": [1.0, 0.8, 0.6, 0.4, 0.2, 0.0, 0.1, 0.3, 0.5, 0.7]\n    })\n    \n    # Sample model - replace with your actual trained model\n    from sklearn.dummy import DummyRegressor\n    model = DummyRegressor(strategy=\"mean\").fit([[0], [1]], [0.5, 0.5])\n    \n    # Generate submission\n    submission = generate_submission(\n        test_data=test_df,\n        model=model,\n        id_column=\"Id\",          # Change if your ID column has different name\n        target_column=\"target\",  # Change to match competition requirements\n        output_dir=\"my_submissions\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T18:39:12.326919Z","iopub.execute_input":"2025-08-16T18:39:12.327290Z","iopub.status.idle":"2025-08-16T18:39:13.718954Z","shell.execute_reply.started":"2025-08-16T18:39:12.327260Z","shell.execute_reply":"2025-08-16T18:39:13.714092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 1. Load all data files with enhanced display\nprint(\"🔍 Data Inspection (Showing top 10 rows):\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet') \nsample_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet')\n\n# Display top 10 rows in CSV format\ndef show_top_10_csv(df, name):\n    print(f\"\\n{name} (first 10 rows):\")\n    print(df.head(10).to_csv(index=False))\n    \nshow_top_10_csv(train_df, \"Training Data\")\nshow_top_10_csv(test_df, \"Test Data\")\nshow_top_10_csv(sample_df, \"Sample Submission\")\n\n# 2. Set target column manually since automatic detection failed\ntarget_col = 'selected'\nprint(f\"\\n🎯 Target column manually set to: '{target_col}'\")\n\n# 3. Feature engineering - exclude metadata columns\nexclude_cols = [target_col, 'srch_id', 'prop_id', 'date_time', 'site_id', 'visitor_id', 'Id']\nfeatures = [col for col in train_df.columns if col not in exclude_cols]\nprint(f\"\\n🔧 Using {len(features)} features: {features[:5]}...\")  # Show first 5 features\n\n# 4. Train LightGBM Ranker with group-aware splitting\nfrom lightgbm import LGBMRanker\nfrom sklearn.model_selection import GroupShuffleSplit\n\nprint(\"\\n🚀 Training ranking model...\")\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=300,\n    learning_rate=0.05,\n    num_leaves=31,\n    random_state=42\n)\n\n# Group-aware train/validation split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\ntrain_idx, val_idx = next(gss.split(train_df, groups=train_df['srch_id']))\n\nmodel.fit(\n    train_df[features].iloc[train_idx],\n    train_df[target_col].iloc[train_idx],\n    group=train_df.iloc[train_idx].groupby('srch_id').size(),\n    eval_set=[(train_df[features].iloc[val_idx], train_df[target_col].iloc[val_idx])],\n    eval_group=[train_df.iloc[val_idx].groupby('srch_id').size()],\n    eval_at=[5, 10],\n    early_stopping_rounds=30,\n    verbose=20\n)\n\n# 5. Generate predictions on test data\nprint(\"\\n🔮 Generating predictions...\")\ntest_df['prediction'] = model.predict(test_df[features])\n\n# 6. Create merged submission with proper ranking\nprint(\"\\n📝 Preparing merged submission...\")\nsubmission = test_df[['srch_id', 'prop_id', 'prediction']].copy()\n\n# Rank properties within each search group\nsubmission['rank'] = submission.groupby('srch_id')['prediction'].rank(ascending=False, method='first')\nsubmission = submission.sort_values(['srch_id', 'rank'])\n\n# Keep only top properties per search (assuming we want top 10)\ntop_submission = submission[submission['rank'] <= 10].copy()\ntop_submission = top_submission[['srch_id', 'prop_id']]  # Final format\n\n# 7. Validate against sample format\nprint(\"\\n✅ Submission Validation:\")\nprint(\"Your submission columns:\", top_submission.columns.tolist())\nprint(\"Sample submission columns:\", sample_df.columns.tolist())\nprint(\"\\nYour submission shape:\", top_submission.shape)\nprint(\"Sample submission shape:\", sample_df.shape)\n\n# 8. Save final merged submission\nfinal_filename = 'merged_top10_submission.csv'\ntop_submission.to_csv(final_filename, index=False)\n\nprint(f\"\\n💾 Final submission saved as '{final_filename}'\")\nprint(\"\\nFinal submission preview (top 10 rows):\")\nprint(top_submission.head(10).to_csv(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T03:22:06.906855Z","iopub.status.idle":"2025-08-17T03:22:06.907977Z","shell.execute_reply.started":"2025-08-17T03:22:06.907005Z","shell.execute_reply":"2025-08-17T03:22:06.907019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Full Automated Report","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# 1. Load all data files with enhanced display\nprint(\"🔍 Data Inspection (Showing top 10 rows):\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet') \nsample_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet')\n\n# Display top 10 rows in CSV format\ndef show_top_10_csv(df, name):\n    print(f\"\\n{name} (first 10 rows):\")\n    print(df.head(10).to_csv(index=False))\n    \nshow_top_10_csv(train_df, \"Training Data\")\nshow_top_10_csv(test_df, \"Test Data\")\nshow_top_10_csv(sample_df, \"Sample Submission\")\n\n# 2. Set target column manually since automatic detection failed\ntarget_col = 'selected'\nprint(f\"\\n🎯 Target column manually set to: '{target_col}'\")\n\n# 3. Feature engineering - exclude metadata columns\nexclude_cols = [target_col, 'srch_id', 'prop_id', 'date_time', 'site_id', 'visitor_id', 'Id']\nfeatures = [col for col in train_df.columns if col not in exclude_cols]\nprint(f\"\\n🔧 Using {len(features)} features: {features[:5]}...\")  # Show first 5 features\n\n# 4. Train LightGBM Ranker with group-aware splitting\nfrom lightgbm import LGBMRanker\nfrom sklearn.model_selection import GroupShuffleSplit\n\nprint(\"\\n🚀 Training ranking model...\")\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=300,\n    learning_rate=0.05,\n    num_leaves=31,\n    random_state=42\n)\n\n# Group-aware train/validation split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\ntrain_idx, val_idx = next(gss.split(train_df, groups=train_df['srch_id']))\n\nmodel.fit(\n    train_df[features].iloc[train_idx],\n    train_df[target_col].iloc[train_idx],\n    group=train_df.iloc[train_idx].groupby('srch_id').size(),\n    eval_set=[(train_df[features].iloc[val_idx], train_df[target_col].iloc[val_idx])],\n    eval_group=[train_df.iloc[val_idx].groupby('srch_id').size()],\n    eval_at=[5, 10],\n    early_stopping_rounds=30,\n    verbose=20\n)\n\n# 5. Generate predictions on test data\nprint(\"\\n🔮 Generating predictions...\")\ntest_df['prediction'] = model.predict(test_df[features])\n\n# 6. Create merged submission with proper ranking\nprint(\"\\n📝 Preparing merged submission...\")\nsubmission = test_df[['srch_id', 'prop_id', 'prediction']].copy()\n\n# Rank properties within each search group\nsubmission['rank'] = submission.groupby('srch_id')['prediction'].rank(ascending=False, method='first')\nsubmission = submission.sort_values(['srch_id', 'rank'])\n\n# Keep only top properties per search (assuming we want top 10)\ntop_submission = submission[submission['rank'] <= 10].copy()\ntop_submission = top_submission[['srch_id', 'prop_id']]  # Final format\n\n# 7. Validate against sample format\nprint(\"\\n✅ Submission Validation:\")\nprint(\"Your submission columns:\", top_submission.columns.tolist())\nprint(\"Sample submission columns:\", sample_df.columns.tolist())\nprint(\"\\nYour submission shape:\", top_submission.shape)\nprint(\"Sample submission shape:\", sample_df.shape)\n\n# 8. Save final merged submission\nfinal_filename = 'merged_top10_submission.csv'\ntop_submission.to_csv(final_filename, index=False)\n\nprint(f\"\\n💾 Final submission saved as '{final_filename}'\")\nprint(\"\\nFinal submission preview (top 10 rows):\")\nprint(top_submission.head(10).to_csv(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T03:17:19.420207Z","iopub.status.idle":"2025-08-17T03:17:19.421401Z","shell.execute_reply.started":"2025-08-17T03:17:19.420406Z","shell.execute_reply":"2025-08-17T03:17:19.420424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configuration - updated based on your actual data columns\nCOLUMN_CONFIG = {\n    'target': 'selected',\n    'alternative_target': None,\n    'group_id': 'search_id',  # Updated to match your actual column\n    'exclude_features': ['selected', 'search_id', 'prop_id', 'date_time', 'Id']\n}\n\n# Load and validate data\nprint(\"🔍 Loading and validating data...\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\nsample_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/sample_submission.parquet')\n\nprint(f\"\\n✅ Training data loaded with {len(train_df)} rows\")\nprint(f\"✅ Test data loaded with {len(test_df)} rows\")\n\n# Display data summary\nprint(\"\\nTraining data columns:\", train_df.columns.tolist())\nprint(\"\\nSample training data:\")\ndisplay(train_df.head(3))\n\n# Check target column exists\ntarget_col = None\nif COLUMN_CONFIG['target'] in train_df.columns:\n    target_col = COLUMN_CONFIG['target']\n    print(f\"\\n🎯 Using target column: '{target_col}'\")\nelse:\n    available_cols = [c for c in train_df.columns if c not in COLUMN_CONFIG['exclude_features']]\n    raise KeyError(f\"Target column '{COLUMN_CONFIG['target']}' not found. Available columns: {available_cols}\")\n\n# Prepare features and target\nfeatures = [col for col in train_df.columns if col not in COLUMN_CONFIG['exclude_features']]\nprint(f\"\\n🔧 Using {len(features)} features. First 5: {features[:5]}\")\n\nX = train_df[features]\ny = train_df[target_col]\ngroups = train_df[COLUMN_CONFIG['group_id']]\n\n# Verify group column\nprint(f\"\\nℹ️ Number of unique groups ({COLUMN_CONFIG['group_id']}): {groups.nunique()}\")\nprint(f\"ℹ️ Target distribution:\\n{y.value_counts(normalize=True)}\")\nprint(\"Training data columns:\", train_df.columns.tolist())\n\n# Rest of your model training code can follow here...","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T03:50:11.957249Z","iopub.status.idle":"2025-08-17T03:50:11.957719Z","shell.execute_reply.started":"2025-08-17T03:50:11.957391Z","shell.execute_reply":"2025-08-17T03:50:11.957405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef load_and_validate_submission(filepath):\n    \"\"\"Load and validate a submission file with error handling\"\"\"\n    try:\n        if not os.path.exists(filepath):\n            raise FileNotFoundError(f\"File not found: {filepath}\")\n            \n        df = pd.read_csv(filepath)\n        \n        # Validation checks\n        if 'Id' not in df.columns or 'target' not in df.columns:\n            raise ValueError(f\"File missing required columns. Found: {df.columns.tolist()}\")\n        if df.isnull().any().any():\n            raise ValueError(\"File contains null values\")\n        if df['Id'].duplicated().any():\n            raise ValueError(\"Duplicate IDs found\")\n            \n        print(f\"✅ Successfully loaded: {filepath}\")\n        return df\n        \n    except Exception as e:\n        print(f\"❌ Error loading {filepath}: {str(e)}\")\n        return None\n\n# Define file paths - adjust these to match your actual file locations\ninput_dir = '/kaggle/input'  # Change this to your actual input directory\nfile1 = os.path.join(input_dir, 'submission_updated.csv')\nfile2 = os.path.join(input_dir, 'submission_20250816_183913.csv')\n\nprint(\"🔍 Looking for submission files...\")\nprint(f\"1. Searching for: {file1}\")\nprint(f\"2. Searching for: {file2}\")\n\n# Load both files with validation\nsub1 = load_and_validate_submission(file1)\nsub2 = load_and_validate_submission(file2)\n\n# Only proceed if both files loaded successfully\nif sub1 is not None and sub2 is not None:\n    print(\"\\n⚖️ Creating weighted average submission...\")\n    final_sub = pd.DataFrame({\n        'Id': sub1['Id'],\n        'target': (sub1['target']*0.7 + sub2['target']*0.3).round(4).clip(0, 1)\n    })\n    \n    # Save final submission\n    output_dir = '/kaggle/working'  # Kaggle output directory\n    os.makedirs(output_dir, exist_ok=True)\n    final_file = os.path.join(output_dir, 'final_submission.csv')\n    final_sub.to_csv(final_file, index=False)\n    \n    print(f\"\\n💾 Saved final submission to: {final_file}\")\n    print(\"\\n📋 Final submission preview:\")\n    print(final_sub.head(10).to_csv(index=False))\n    \nelif sub1 is None and sub2 is None:\n    print(\"\\n❌ Error: Could not load either submission file\")\n    print(\"Please check:\")\n    print(f\"- File exists at: {file1}\")\n    print(f\"- File exists at: {file2}\")\n    print(\"- Correct file permissions\")\nelse:\n    print(\"\\n⚠️ Warning: Only one submission file loaded successfully\")\n    print(\"Using the single available file as final submission\")\n    final_sub = sub1 if sub1 is not None else sub2\n    final_file = os.path.join(output_dir, 'final_submission.csv')\n    final_sub.to_csv(final_file, index=False)\n    print(f\"\\n💾 Saved single file as final submission to: {final_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T03:08:40.828742Z","iopub.execute_input":"2025-08-17T03:08:40.829188Z","iopub.status.idle":"2025-08-17T03:08:40.848089Z","shell.execute_reply.started":"2025-08-17T03:08:40.829134Z","shell.execute_reply":"2025-08-17T03:08:40.842555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom lightgbm import LGBMRanker\nfrom sklearn.model_selection import GroupShuffleSplit\n\n# Load data\ntrain = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\n\n# Configuration (update these based on your data)\nTARGET = 'booked'  # Your target column\nGROUP = 'search_id'  # Your group/session column\nITEM = 'prop_id'   # Your item/property column\n\n# Prepare features\nexclude = [TARGET, GROUP, ITEM, 'date_time']\nfeatures = [col for col in train.columns if col not in exclude and col in test.columns]\n\n# Convert features to numeric\nfor df in [train, test]:\n    for col in features:\n        if not pd.api.types.is_numeric_dtype(df[col]):\n            df[col] = df[col].factorize()[0]\n\n# Train model\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=200,\n    learning_rate=0.05\n)\n\nX_train = train[features]\ny_train = train[TARGET]\ngroups = train.groupby(GROUP).size().values\n\nmodel.fit(X_train, y_train, group=groups)\n\n# Generate predictions\ntest['prediction'] = model.predict(test[features])\n\n# Create submission\nsubmission = (\n    test.sort_values([GROUP, 'prediction'], ascending=[True, False])\n    .groupby(GROUP)[ITEM]\n    .apply(list)\n    .reset_index()\n)\n\n# Explode lists into rows\nsubmission = submission.explode(ITEM)\n\n# Save to CSV\nsubmission_path = 'submission.csv'\nsubmission[[GROUP, ITEM]].to_csv(submission_path, index=False)\n\n# Verification\nprint(f\"Submission created with {len(submission)} rows\")\nprint(f\"Sample submission:\\n{submission.head()}\")\nprint(f\"File saved to: {submission_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:35:35.650584Z","iopub.status.idle":"2025-08-17T04:35:35.651037Z","shell.execute_reply.started":"2025-08-17T04:35:35.650739Z","shell.execute_reply":"2025-08-17T04:35:35.650754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Rows in test data: {len(test)}\")\nprint(f\"Rows in submission: {len(submission)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:35:35.652793Z","iopub.status.idle":"2025-08-17T04:35:35.653485Z","shell.execute_reply.started":"2025-08-17T04:35:35.652953Z","shell.execute_reply":"2025-08-17T04:35:35.652968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"group_counts = submission[GROUP].value_counts()\nprint(f\"Average items per group: {group_counts.mean():.1f}\")\nprint(f\"Min items per group: {group_counts.min()}\")\nprint(f\"Max items per group: {group_counts.max()}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(submission_path) as f:\n    print(f\"First line: {f.readline()}\")\n    print(f\"Second line: {f.readline()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:35:35.654473Z","iopub.status.idle":"2025-08-17T04:35:35.654939Z","shell.execute_reply.started":"2025-08-17T04:35:35.654616Z","shell.execute_reply":"2025-08-17T04:35:35.654630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load your files (update paths as needed)\nsubmission_file = 'my_submissions/submission_20250816_183913.csv'\ntest_file = '/kaggle/input/aeroclub-recsys-2025/test.parquet'\n\n# Load data\nsubmission = pd.read_csv(submission_file)\ntest_df = pd.read_parquet(test_file)\n\n# First, let's check what columns actually exist\nprint(\"Submission columns:\", submission.columns.tolist())\nprint(\"Test data columns:\", test_df.columns.tolist())\n\n# Identify the correct group column name\nGROUP_COLUMN = None\npossible_group_columns = ['ranker_id', 'search_id', 'session_id', 'group_id']\n\nfor col in possible_group_columns:\n    if col in submission.columns and col in test_df.columns:\n        GROUP_COLUMN = col\n        break\n\nif not GROUP_COLUMN:\n    raise ValueError(\"Could not identify group column. Actual columns found: \" + \n                    str(submission.columns.tolist()))\n\nprint(f\"\\nUsing '{GROUP_COLUMN}' as group column\")\n\n# Now verify the submission\nprint(\"\\n=== Basic Verification ===\")\nprint(f\"Submission rows: {len(submission)}\")\nprint(f\"Test data rows: {len(test_df)}\")\nprint(f\"Null values:\\n{submission.isnull().sum()}\")\n\nprint(\"\\n=== Group Verification ===\")\ntest_groups = test_df[GROUP_COLUMN].nunique()\nsub_groups = submission[GROUP_COLUMN].nunique()\nprint(f\"Groups in test: {test_groups}\")\nprint(f\"Groups in submission: {sub_groups}\")\n\n# Check if all test groups are represented\nmissing_groups = set(test_df[GROUP_COLUMN]) - set(submission[GROUP_COLUMN])\nif missing_groups:\n    print(f\"\\n⚠️ Warning: Missing {len(missing_groups)} groups in submission\")\nelse:\n    print(\"\\n✓ All test groups present in submission\")\n\n# Check rank column name\nRANK_COLUMN = None\npossible_rank_columns = ['selected', 'rank', 'prediction', 'target']\n\nfor col in possible_rank_columns:\n    if col in submission.columns:\n        RANK_COLUMN = col\n        break\n\nif not RANK_COLUMN:\n    raise ValueError(\"Could not identify rank column. Actual columns found: \" +\n                   str(submission.columns.tolist()))\n\nprint(f\"\\nUsing '{RANK_COLUMN}' as rank column\")\n\n# Rank validation\nprint(\"\\n=== Rank Validation ===\")\ngroup_sizes = submission.groupby(GROUP_COLUMN).size()\nrank_ranges = submission.groupby(GROUP_COLUMN)[RANK_COLUMN].agg(['min', 'max', 'nunique'])\n\n# Check if ranks start at 1 and are sequential\nvalid_ranks = rank_ranges.apply(\n    lambda x: x['min'] == 1 and x['nunique'] == x['max'], axis=1\n)\n\nif not valid_ranks.all():\n    print(\"⚠️ Some groups have invalid rank sequences:\")\n    print(rank_ranges[~valid_ranks].head())\nelse:\n    print(\"✓ All groups have valid rank sequences (1 to N)\")\n\n# Check for duplicate ranks\nduplicate_ranks = submission.duplicated(subset=[GROUP_COLUMN, RANK_COLUMN], keep=False)\nif duplicate_ranks.any():\n    print(f\"\\n⚠️ Found {duplicate_ranks.sum()} duplicate ranks within groups\")\n    print(submission[duplicate_ranks].head())\nelse:\n    print(\"✓ No duplicate ranks within groups\")\n\nprint(\"\\n=== Sample Submission ===\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-17T04:56:14.135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom lightgbm import LGBMRanker, early_stopping, log_evaluation\nfrom sklearn.model_selection import GroupShuffleSplit\n\n# ======================\n# 1. Data Loading\n# ======================\nprint(\"⏳ Loading data...\")\ntry:\n    train_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\n    test_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\n    print(\"✅ Data loaded successfully\")\n    print(f\"Train shape: {train_df.shape}, Test shape: {test_df.shape}\")\nexcept Exception as e:\n    print(f\"❌ Error loading data: {e}\")\n    raise\n\n# ======================\n# 2. Target Identification\n# ======================\ndef find_potential_targets(df):\n    targets = []\n    for col in df.columns:\n        if df[col].nunique() == 2 and set(df[col].unique()).issubset({0, 1}):\n            targets.append((col, \"binary 0/1\"))\n        elif any(keyword in col.lower() for keyword in ['target', 'label', 'book', 'click']):\n            targets.append((col, \"named like target\"))\n        elif pd.api.types.is_numeric_dtype(df[col]) and df[col].nunique() < 10:\n            targets.append((col, f\"numeric with {df[col].nunique()} values\"))\n    return targets\n\nprint(\"\\n🔍 Identifying target column...\")\npotential_targets = find_potential_targets(train_df)\n\nif potential_targets:\n    print(\"\\n🎯 Potential Target Columns:\")\n    for i, (col, reason) in enumerate(potential_targets, 1):\n        print(f\"{i}. {col} ({reason})\")\n    TARGET_COLUMN = potential_targets[0][0]\n    print(f\"\\nAutomatically selecting: '{TARGET_COLUMN}'\")\nelse:\n    raise ValueError(\"No target column found - please specify manually\")\n\n# ======================\n# 3. Group Identification\n# ======================\ndef find_group_column(df):\n    possible_groups = [\n        col for col in df.columns \n        if any(keyword in col.lower() for keyword in ['session', 'srch', 'search', 'user'])\n        and df[col].nunique() > 1000\n    ]\n    if not possible_groups:\n        possible_groups = [\n            col for col in df.columns \n            if ('id' in col.lower() and df[col].nunique() > 1000)\n        ]\n    return possible_groups[0] if possible_groups else None\n\nGROUP_COLUMN = find_group_column(train_df)\nif not GROUP_COLUMN:\n    raise ValueError(\"No group column found - please specify manually\")\nprint(f\"\\n📊 Using group column: '{GROUP_COLUMN}'\")\n\n# ======================\n# 4. Feature Engineering\n# ======================\nprint(\"\\n🛠️ Preparing features...\")\nexclude_cols = [TARGET_COLUMN, GROUP_COLUMN, 'date_time', 'prop_id']\nfeatures = [col for col in train_df.columns if col not in exclude_cols and col in test_df.columns]\n\n# Type conversion with enhanced error handling\nfor col in features:\n    for df in [train_df, test_df]:\n        if col in df.columns:\n            try:\n                if not pd.api.types.is_numeric_dtype(df[col]):\n                    df[col] = pd.to_numeric(df[col], errors='coerce').fillna(0)\n            except Exception as e:\n                print(f\"Warning: Couldn't convert {col} - using category codes: {e}\")\n                df[col] = df[col].astype('category').cat.codes\n\nprint(f\"\\n🔧 Using {len(features)} features\")\nprint(\"Sample features:\", features[:5])\n\n# ======================\n# 5. Model Training\n# ======================\nimport pandas as pd\nimport numpy as np\nfrom lightgbm import LGBMRanker, early_stopping, log_evaluation\nfrom sklearn.model_selection import GroupShuffleSplit\n\n# Load data\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\n\n# Configuration\nTARGET_COLUMN = 'selected'\nGROUP_COLUMN = 'ranker_id'\nITEM_COLUMN = 'Id'  # From data description\n\n# Feature preparation\nexclude_cols = [TARGET_COLUMN, GROUP_COLUMN, ITEM_COLUMN, 'requestDate']\nfeatures = [col for col in train_df.columns if col not in exclude_cols and col in test_df.columns]\n\n# Convert features to numeric\nfor col in features:\n    for df in [train_df, test_df]:\n        if not pd.api.types.is_numeric_dtype(df[col]):\n            df[col] = df[col].astype('category').cat.codes\n\n# Limit group sizes (critical fix)\nmax_group_size = 10000  # LightGBM's default limit\ntrain_df = train_df.groupby(GROUP_COLUMN).filter(lambda x: len(x) <= max_group_size)\n\n# Prepare data\nX = train_df[features]\ny = train_df[TARGET_COLUMN]\ngroups = train_df.groupby(GROUP_COLUMN).size()\n\n# Model with adjusted parameters\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=500,\n    learning_rate=0.05,\n    max_position=100,  # Important for ranking\n    label_gain=[i for i in range(max_group_size+1)],  # For NDCG calculation\n    random_state=42,\n    verbose=1\n)\n\n# Train with group-aware validation\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\ntrain_idx, valid_idx = next(gss.split(X, y, groups=train_df[GROUP_COLUMN]))\n\nmodel.fit(\n    X.iloc[train_idx], y.iloc[train_idx],\n    group=groups.iloc[train_idx].values,\n    eval_set=[(X.iloc[valid_idx], y.iloc[valid_idx])],\n    eval_group=[groups.iloc[valid_idx].values],\n    eval_at=[5, 10, 20],\n    callbacks=[early_stopping(50), log_evaluation(20)]\n)\n\n# Generate predictions\ntest_df['prediction'] = model.predict(test_df[features])\n\n# Create submission (proper format)\nsubmission = test_df[[ITEM_COLUMN, GROUP_COLUMN, 'prediction']].copy()\n\n# Convert scores to ranks within each group\nsubmission['selected'] = submission.groupby(GROUP_COLUMN)['prediction'].rank(ascending=False, method='first')\n\n# Ensure ranks are integers (1=best)\nsubmission['selected'] = submission['selected'].astype(int)\n\n# Final submission format\nfinal_submission = submission[[ITEM_COLUMN, GROUP_COLUMN, 'selected']].sort_values([GROUP_COLUMN, 'selected'])\n\n# Verify no duplicate ranks per group\nassert not final_submission.duplicated(subset=[GROUP_COLUMN, 'selected']).any(), \"Duplicate ranks found!\"\n\n# Save\nfinal_submission.to_csv('submission.csv', index=False)\nprint(\"Submission created successfully!\")\n\n# ======================\n# 6. Generate Submission\n# ======================\nprint(\"\\n📝 Creating submission file...\")\nsubmission_path = 'submission.csv'\npredictions = model.predict(test_df[features].astype(np.float32))\n\nsubmission = pd.DataFrame({\n    GROUP_COLUMN: test_df[GROUP_COLUMN],\n    'prop_id': test_df['prop_id'],\n    'score': predictions\n}).sort_values([GROUP_COLUMN, 'score'], ascending=[True, False])\n\n# Keep only required columns\nfinal_submission = submission[[GROUP_COLUMN, 'prop_id']]\nfinal_submission.to_csv(submission_path, index=False)\n\n# ======================\n# 7. Verify Submission\n# ======================\nprint(\"\\n🔍 Verifying submission file...\")\nif os.path.exists(submission_path):\n    line_count = sum(1 for line in open(submission_path))\n    file_size = os.path.getsize(submission_path) / (1024 * 1024)  # in MB\n    \n    print(f\"✅ Submission created successfully at: {os.path.abspath(submission_path)}\")\n    print(f\"• File size: {file_size:.2f} MB\")\n    print(f\"• Line count: {line_count} (should be test samples + 1 header)\")\n    print(\"\\nSample submission:\")\n    print(pd.read_csv(submission_path).head())\nelse:\n    print(f\"❌ Error: File not created at {os.path.abspath(submission_path)}\")\n    print(\"Possible issues:\")\n    print(\"- Permission errors\")\n    print(\"- Disk space full\")\n    print(\"- Path incorrect\")\n\nprint(\"\\n🏁 Pipeline complete!\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-17T04:56:14.144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Add your feature engineering logic here\n    # Example: df['request_hour'] = pd.to_datetime(df['request_time']).dt.hour\n    return df\n\n# Apply feature engineering\ntrain_df = feature_engineering(train_df)\ntest_df = feature_engineering(test_df)\n\nfeatures = ['request_hour', 'totalPrice']\nX_train = train_df[features]\ny_train = train_df['selected']\nX_test = test_df[features]\n\n# Initialize and train model\nmodel = RandomForestRegressor(n_estimators=50, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Make predictions\nscores = model.predict(X_test)\ntest_df['selected'] = scores.argsort().argsort() + 1\n\n# Create submission\nsubmission = sample_df[['Id']].merge(test_df[['Id', 'selected']], on='Id', how='left')\nsubmission['selected'] = submission['selected'].fillna(method='ffill').astype(int)\n\n# Save submission\nOUTPUT_DIR = Path('/kaggle/working')\nOUTPUT_DIR.mkdir(exist_ok=True)\nsubmission_path = OUTPUT_DIR / 'submission.csv'\nsubmission.to_csv(submission_path, index=False)\nprint('Submission written to:', submission_path)\nprint(submission.head())\n\n# Create report (note: variables like best_model_name, results, feature_reports need to be defined)\nreport_path = OUTPUT_DIR / \"competition_report.html\"\n\nwith open(report_path, \"w\") as f:\n    f.write(\"<h1>Kaggle Competition Report</h1>\")\n    \n    # Only include if these variables are defined\n    if 'best_model_name' in locals() and 'results' in locals():\n        f.write(f\"<h2>🏆 Best Model: {best_model_name} (RMSE: {results[best_model_name]:.4f})</h2>\")\n        f.write(\"<h2>📊 Model RMSE Comparison</h2><ul>\")\n        for name, rmse in results.items():\n            f.write(f\"<li>{name}: {rmse:.4f}</li>\")\n        f.write(\"</ul>\")\n    \n    if 'feature_reports' in locals():\n        f.write(\"<h2>🔥 Feature Importances</h2>\")\n        for name, feat_df in feature_reports.items():\n            f.write(f\"<h3>{name}</h3>\")\n            f.write(feat_df.to_html(index=False))\n            img_path = f\"../figures/{name}_feature_importance.png\"\n            f.write(f'<img src=\"{img_path}\" width=\"400\"><br>')\n    \n    f.write(\"<h2>📈 Validation Comparison</h2>\")\n    f.write('<img src=\"../figures/validation_comparison.png\" width=\"400\"><br>')\n    \n    f.write(\"<h2>📑 Submission Preview</h2>\")\n    f.write(submission.head().to_html(index=False))\n\nprint(f\"✅ Full Report saved to {report_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --upgrade lightgbm  # Just in case\nfrom lightgbm import LGBMRanker","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:48:34.889237Z","iopub.execute_input":"2025-08-17T04:48:34.889590Z","iopub.status.idle":"2025-08-17T04:48:41.295275Z","shell.execute_reply.started":"2025-08-17T04:48:34.889563Z","shell.execute_reply":"2025-08-17T04:48:41.289960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor()\nimport pandas as pd\nimport numpy as np\nimport os\nfrom lightgbm import LGBMRanker\nfrom sklearn.model_selection import GroupShuffleSplit\n\n# 1. Load Data\nprint(\"Loading data...\")\ntrain_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\n\n# 2. Identify Target and Group Columns\nTARGET_COLUMN = 'booked'  # Update this if different\nGROUP_COLUMN = 'search_id'  # Update this if different\n\n# 3. Prepare Features\nexclude_cols = [TARGET_COLUMN, GROUP_COLUMN, 'date_time', 'prop_id']\nfeatures = [col for col in train_df.columns \n            if col not in exclude_cols and col in test_df.columns]\n\n# Convert features to numeric\nfor df in [train_df, test_df]:\n    for col in features:\n        if not pd.api.types.is_numeric_dtype(df[col]):\n            df[col] = pd.to_numeric(df[col], errors='coerce').fillna(0)\n\n# 4. Train Model\nX = train_df[features]\ny = train_df[TARGET_COLUMN]\ngroups = train_df[GROUP_COLUMN]\n\n# Train/validation split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\ntrain_idx, valid_idx = next(gss.split(X, y, groups=groups))\n\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    n_estimators=200,\n    learning_rate=0.05,\n    random_state=42\n)\n\nmodel.fit(\n    X.iloc[train_idx], y.iloc[train_idx],\n    group=groups.iloc[train_idx].value_counts(sort=False),\n    eval_set=[(X.iloc[valid_idx], y.iloc[valid_idx])],\n    eval_group=[groups.iloc[valid_idx].value_counts(sort=False)],\n    eval_at=[5, 10],\n    callbacks=[early_stopping(20), log_evaluation(10)]\n)\n\n# 5. Generate Predictions\ntest_df['prediction_score'] = model.predict(test_df[features])\n\n# 6. Create Submission File\nsubmission = test_df.sort_values([GROUP_COLUMN, 'prediction_score'], \n                               ascending=[True, False])\n\n# Format as specified in competition rules\nfinal_submission = submission[[GROUP_COLUMN, 'prop_id']]\n\n# Save to CSV\nsubmission_path = 'submission.csv'\nfinal_submission.to_csv(submission_path, index=False)\n\n# 7. Verify Submission\nprint(\"\\nSubmission file created at:\", os.path.abspath(submission_path))\nprint(\"File info:\")\nprint(\"- Size:\", os.path.getsize(submission_path) / 1024, \"KB\")\nprint(\"- Line count:\", sum(1 for _ in open(submission_path)))\nprint(\"\\nFirst 5 lines:\")\nprint(pd.read_csv(submission_path).head())\n\nprint(\"\\n✅ Submission ready! Upload\", submission_path, \"to the competition\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:48:45.304250Z","iopub.execute_input":"2025-08-17T04:48:45.304732Z","execution_failed":"2025-08-17T04:56:13.974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_parquet(test_file)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 1. Load your submission file and test data\ntry:\n    submission = pd.read_csv('my_submissions/submission_20250816_183913.csv')\n    test_data = pd.read_parquet('/kaggle/input/aeroclub-recsys-2025/test.parquet')\nexcept Exception as e:\n    print(f\"Error loading files: {e}\")\n    raise\n\n# 2. Identify the correct column names (update these if different in your data)\nGROUP_COLUMN = 'ranker_id'  # The group/session identifier column\nITEM_COLUMN = 'Id'          # The item/flight option identifier column\nRANK_COLUMN = 'selected'    # The column containing your predicted ranks\n\n# First verify the columns exist in both files\nmissing_in_submission = [col for col in [GROUP_COLUMN, ITEM_COLUMN, RANK_COLUMN] \n                        if col not in submission.columns]\nmissing_in_test = [col for col in [GROUP_COLUMN, ITEM_COLUMN] \n                  if col not in test_data.columns]\n\nif missing_in_submission:\n    print(f\"Error: Missing columns in submission file: {missing_in_submission}\")\n    print(f\"Submission columns: {submission.columns.tolist()}\")\n    raise ValueError(\"Column names don't match submission file\")\n\nif missing_in_test:\n    print(f\"Error: Missing columns in test data: {missing_in_test}\")\n    print(f\"Test data columns: {test_data.columns.tolist()}\")\n    raise ValueError(\"Column names don't match test data\")\n\n# 3. Basic validation\nprint(\"=== Basic Validation ===\")\nprint(f\"Submission rows: {len(submission)}\")\nprint(f\"Test data rows: {len(test_data)}\")\nprint(\"\\nNull value counts:\")\nprint(submission.isnull().sum())\n\n# 4. Group validation\nprint(\"\\n=== Group Validation ===\")\ntest_groups = test_data[GROUP_COLUMN].nunique()\nsub_groups = submission[GROUP_COLUMN].nunique()\n\nprint(f\"Unique groups in test data: {test_groups}\")\nprint(f\"Unique groups in submission: {sub_groups}\")\n\n# Check if all test groups are present in submission\nmissing_groups = set(test_data[GROUP_COLUMN]) - set(submission[GROUP_COLUMN])\nif missing_groups:\n    print(f\"\\n⚠️ Warning: {len(missing_groups)} groups missing from submission\")\n    print(\"Sample missing groups:\", list(missing_groups)[:5])\nelse:\n    print(\"\\n✓ All test groups present in submission\")\n\n# 5. Rank validation\nprint(\"\\n=== Rank Validation ===\")\n# Check rank values are positive integers\nif not pd.api.types.is_integer_dtype(submission[RANK_COLUMN]):\n    print(\"⚠️ Rank column should contain integers\")\n    submission[RANK_COLUMN] = submission[RANK_COLUMN].astype(int)\n\n# Check ranks within each group\nrank_check = submission.groupby(GROUP_COLUMN)[RANK_COLUMN].agg(\n    ['min', 'max', 'count', 'nunique']\n)\nrank_check['valid'] = (rank_check['min'] == 1) & \\\n                     (rank_check['nunique'] == rank_check['count']) & \\\n                     (rank_check['nunique'] == rank_check['max'])\n\ninvalid_groups = rank_check[~rank_check['valid']]\nif not invalid_groups.empty:\n    print(\"⚠️ Some groups have invalid rank sequences:\")\n    print(invalid_groups.head())\nelse:\n    print(\"✓ All groups have valid rank sequences (1 to N with no duplicates)\")\n\n# 6. Final checks\nprint(\"\\n=== Final Checks ===\")\n# Check if all test items are present\nmissing_items = set(test_data[ITEM_COLUMN]) - set(submission[ITEM_COLUMN])\nif missing_items:\n    print(f\"⚠️ Warning: {len(missing_items)} items missing from submission\")\nelse:\n    print(\"✓ All test items present in submission\")\n\n# Check row order matches (if required by competition)\nprint(\"\\nSample submission:\")\nprint(submission.head())\n\nprint(\"\\n✅ Verification complete!\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-17T04:56:14.145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd # Load your existing submission\nsub_file = 'my_submissions/submission_20250816_183913.csv'\nexisting_sub = pd.read_csv(sub_file)\n\n# Basic checks\nprint(f\"Columns: {existing_sub.columns.tolist()}\")\nprint(f\"Row count: {len(existing_sub)}\")\nprint(f\"Null values: {existing_sub.isnull().sum()}\")\n\n# Verify group-item counts match test data\ntest_groups = test[GROUP].nunique()\nsub_groups = existing_sub[GROUP].nunique()\nprint(f\"Groups in test: {test_groups}, Groups in submission: {sub_groups}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T04:44:45.156923Z","iopub.execute_input":"2025-08-17T04:44:45.157301Z","iopub.status.idle":"2025-08-17T04:44:45.196287Z","shell.execute_reply.started":"2025-08-17T04:44:45.157273Z","shell.execute_reply":"2025-08-17T04:44:45.190896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nsample = pd.read_parquet(\"sample_submission.parquet\")\nsubmission = pd.read_csv(\"submission.csv\")\n\n# Check structure\nprint(\"Sample columns:\", sample.columns.tolist())\nprint(\"Submission columns:\", submission.columns.tolist())\nprint(\"Shape match?\", submission.shape == sample.shape)\nprint(\"Any missing values?\", submission.isna().any().any())\n\n# Ensure same dtypes\nfor col in sample.columns:\n    if submission[col].dtype != sample[col].dtype:\n        print(f\"⚠️ Column {col} dtype mismatch: {submission[col].dtype} vs {sample[col].dtype}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}