{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":5056,"databundleVersionId":868325}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:27.279920Z","iopub.execute_input":"2026-04-13T12:36:27.280251Z","iopub.status.idle":"2026-04-13T12:36:27.298650Z","shell.execute_reply.started":"2026-04-13T12:36:27.280223Z","shell.execute_reply":"2026-04-13T12:36:27.297817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:27.300132Z","iopub.execute_input":"2026-04-13T12:36:27.300439Z","iopub.status.idle":"2026-04-13T12:36:27.306007Z","shell.execute_reply.started":"2026-04-13T12:36:27.300410Z","shell.execute_reply":"2026-04-13T12:36:27.305181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read a million lines, but set the random_state using sample()\n\ndf_train = pd.read_csv('/kaggle/input/competitions/expedia-hotel-recommendations/train.csv', nrows=5_000_000)\n\ndf_train = df_train.sample(n=1_000_000, random_state=9)  # take a random 1 million\ndf_train = df_train.reset_index(drop=True)\n\ndf_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:27.307031Z","iopub.execute_input":"2026-04-13T12:36:27.307299Z","iopub.status.idle":"2026-04-13T12:36:43.721882Z","shell.execute_reply.started":"2026-04-13T12:36:27.307272Z","shell.execute_reply":"2026-04-13T12:36:43.721061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EDA\n\ndf_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:43.722935Z","iopub.execute_input":"2026-04-13T12:36:43.723190Z","iopub.status.idle":"2026-04-13T12:36:44.086824Z","shell.execute_reply.started":"2026-04-13T12:36:43.723164Z","shell.execute_reply":"2026-04-13T12:36:44.085849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:44.089744Z","iopub.execute_input":"2026-04-13T12:36:44.090789Z","iopub.status.idle":"2026-04-13T12:36:44.105329Z","shell.execute_reply.started":"2026-04-13T12:36:44.090742Z","shell.execute_reply":"2026-04-13T12:36:44.104472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Statistical description of numerical columns\n\ndf_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:44.106538Z","iopub.execute_input":"2026-04-13T12:36:44.107159Z","iopub.status.idle":"2026-04-13T12:36:44.849277Z","shell.execute_reply.started":"2026-04-13T12:36:44.107121Z","shell.execute_reply":"2026-04-13T12:36:44.848491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:44.850375Z","iopub.execute_input":"2026-04-13T12:36:44.850990Z","iopub.status.idle":"2026-04-13T12:36:45.203201Z","shell.execute_reply.started":"2026-04-13T12:36:44.850959Z","shell.execute_reply":"2026-04-13T12:36:45.202325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(df_train.isnull().sum() / len(df_train) * 100).round(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:45.204370Z","iopub.execute_input":"2026-04-13T12:36:45.204733Z","iopub.status.idle":"2026-04-13T12:36:45.563272Z","shell.execute_reply.started":"2026-04-13T12:36:45.204695Z","shell.execute_reply":"2026-04-13T12:36:45.562444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Missing values :\\n{df_train.isnull().sum()[df_train.isnull().sum()>0]}\")\n\n#  binary\nprint(f\"\\nBooking rate: {df_train['is_booking'].mean():.2%}\")\nprint(f\"Mobile rate: {df_train['is_mobile'].mean():.2%}\")\nprint(f\"Package rate: {df_train['is_package'].mean():.2%}\")\n\nprint(f\"\\nUnique hotel_markets: {df_train['hotel_market'].nunique()}\")\nprint(f\"Unique srch_destination_id: {df_train['srch_destination_id'].nunique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:45.564208Z","iopub.execute_input":"2026-04-13T12:36:45.564448Z","iopub.status.idle":"2026-04-13T12:36:46.297263Z","shell.execute_reply.started":"2026-04-13T12:36:45.564422Z","shell.execute_reply":"2026-04-13T12:36:46.296491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# how many entries are there for each hotel_market\n\nmarket_counts = df_train.groupby('hotel_market').size()\nmarket_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:46.298247Z","iopub.execute_input":"2026-04-13T12:36:46.298529Z","iopub.status.idle":"2026-04-13T12:36:46.323628Z","shell.execute_reply.started":"2026-04-13T12:36:46.298494Z","shell.execute_reply":"2026-04-13T12:36:46.322852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"srch_dest_counts = df_train.groupby('srch_destination_id').size()\nsrch_dest_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:46.324631Z","iopub.execute_input":"2026-04-13T12:36:46.325366Z","iopub.status.idle":"2026-04-13T12:36:46.353575Z","shell.execute_reply.started":"2026-04-13T12:36:46.325336Z","shell.execute_reply":"2026-04-13T12:36:46.352836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\n\nplt.subplot(1, 2, 1)\nmarket_counts.hist(bins=50)\nplt.title('Розподіл записів по hotel_market')\nplt.xlabel('Кількість записів')\nplt.ylabel('Кількість ринків')\n\nplt.subplot(1, 2, 2)\nsrch_dest_counts.hist(bins=50)\nplt.title('Розподіл записів по srch_destination_id')\nplt.xlabel('Кількість записів')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:46.354584Z","iopub.execute_input":"2026-04-13T12:36:46.354897Z","iopub.status.idle":"2026-04-13T12:36:46.793012Z","shell.execute_reply.started":"2026-04-13T12:36:46.354861Z","shell.execute_reply":"2026-04-13T12:36:46.792149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  select numeric columns for analysis for correlation matrix\n\nnumeric_cols = [\n    'is_booking', 'is_mobile', 'is_package',\n    'srch_adults_cnt', 'srch_children_cnt', 'srch_rm_cnt',\n    'orig_destination_distance', 'cnt',\n    'srch_destination_type_id'\n]\n\n# correlation matrix\ncorr_matrix = df_train[numeric_cols].corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True,fmt='.2f', cmap='coolwarm', square=True)\nplt.title('Correlation matrix')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:46.796269Z","iopub.execute_input":"2026-04-13T12:36:46.796900Z","iopub.status.idle":"2026-04-13T12:36:47.562808Z","shell.execute_reply.started":"2026-04-13T12:36:46.796870Z","shell.execute_reply":"2026-04-13T12:36:47.562023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.style.use('default')\nC1, C2, C3 = '#4361EE', '#F72585', '#4CC9F0'\n\nfig, axes = plt.subplots(2, 3, figsize=(16, 9))\nfig.suptitle('EDA', fontsize=16, fontweight='bold', y=1.01)\n \npanels = [\n    ('srch_adults_cnt','Adults per Search', C1),\n    ('srch_children_cnt', 'Children per Search', C2),\n    ('srch_rm_cnt', 'Rooms per Search', C3),\n    ('is_booking', 'Click(0) vs Booking(1)', C1),\n    ('hotel_continent', 'Hotel Continent', C2),\n    ('channel', 'Channel Distribution', C3),\n]\nfor ax, (col, title, color) in zip(axes.flat, panels):\n    vc = df_train[col].value_counts().sort_index()\n    ax.bar(vc.index.astype(str), vc.values, color=color, alpha=0.85, edgecolor='none')\n    ax.set_title(title, fontsize=11, fontweight='bold')\n    ax.set_xlabel(col, fontsize=9)\n    ax.grid(axis='y')\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:47.563772Z","iopub.execute_input":"2026-04-13T12:36:47.564084Z","iopub.status.idle":"2026-04-13T12:36:48.581524Z","shell.execute_reply.started":"2026-04-13T12:36:47.564046Z","shell.execute_reply":"2026-04-13T12:36:48.580790Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Adults per Search — переважає значення 2, типовий пошук — пара. Поїздки 4+ дорослих — це рідкість.\n\nChildren per Search — найбільше близько 800 тисяч - пошуків без дітей, 1-2 дитини є, але значно менше. Розподіл сильно правосторонній.\n\nRooms per Search — майже всі шукають 1 кімнату (~900k). \n\nClick(0) vs Booking(1) — приблизно 92% кліків, 8% бронювань, сильний дисбаланс.\n\nHotel Continent — континент 2 домінує (~520k), континенти 0 і 5 майже відсутні. Географічна концентрація попиту висока.\n\nChannel Distribution — канал 9 домінує (~540k). Канали 7, 8 майже не використовуються. \n\nЗагалом, дані сильно скошені по більшості ознак — схоже на типову поведінку для реальних даних такого типу, де більшість подій це кліки пар без дітей і замовляють одну кімнату.","metadata":{}},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:48.582548Z","iopub.execute_input":"2026-04-13T12:36:48.582867Z","iopub.status.idle":"2026-04-13T12:36:48.958105Z","shell.execute_reply.started":"2026-04-13T12:36:48.582830Z","shell.execute_reply":"2026-04-13T12:36:48.957235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **PIPELINE:**\n\n**groupby().agg()  →  fillna(median)  →  StandardScaler  →  KMeans / GMM**","metadata":{}},{"cell_type":"code","source":"# Aggregation: hotel_market and srch_destination_id\n\nagg_features = {\n    'is_booking'              : ['mean', 'sum'],\n    'is_mobile'               : 'mean',\n    'is_package'              : 'mean',\n    'srch_adults_cnt'         : 'mean',\n    'srch_children_cnt'       : 'mean',\n    'srch_rm_cnt'             : 'mean',\n    'cnt'                     : 'sum',  #  показує загальну активність ринку — скільки взаємодій генерується, популярність ринку.\n    'orig_destination_distance': 'mean',\n}\n \nagg_market = (df_train.groupby('hotel_market').agg(agg_features)\n              # Example: join ('is_booking', 'mean'), to 'is_booking_mean'\n               .pipe(lambda d: d.set_axis(['_'.join(c) for c in d.columns], axis=1))\n               .reset_index())  #  If the DataFrame has a MultiIndex, this method can remove one or more levels\n \nagg_destination   = (df_train.groupby('srch_destination_id').agg(agg_features)\n               .pipe(lambda d: d.set_axis(['_'.join(c) for c in d.columns], axis=1))\n               .reset_index())\n \nprint(f\"\\nagg_market shape : {agg_market.shape}\")\nprint(f\"agg_dest shape : {agg_destination.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:48.959156Z","iopub.execute_input":"2026-04-13T12:36:48.959474Z","iopub.status.idle":"2026-04-13T12:36:49.156654Z","shell.execute_reply.started":"2026-04-13T12:36:48.959428Z","shell.execute_reply":"2026-04-13T12:36:49.155831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_market.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.157851Z","iopub.execute_input":"2026-04-13T12:36:49.158616Z","iopub.status.idle":"2026-04-13T12:36:49.170509Z","shell.execute_reply.started":"2026-04-13T12:36:49.158586Z","shell.execute_reply":"2026-04-13T12:36:49.169692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_market.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.171473Z","iopub.execute_input":"2026-04-13T12:36:49.171727Z","iopub.status.idle":"2026-04-13T12:36:49.186333Z","shell.execute_reply.started":"2026-04-13T12:36:49.171701Z","shell.execute_reply":"2026-04-13T12:36:49.185571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_destination.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.187298Z","iopub.execute_input":"2026-04-13T12:36:49.187594Z","iopub.status.idle":"2026-04-13T12:36:49.207789Z","shell.execute_reply.started":"2026-04-13T12:36:49.187543Z","shell.execute_reply":"2026-04-13T12:36:49.207003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_destination.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.208865Z","iopub.execute_input":"2026-04-13T12:36:49.209152Z","iopub.status.idle":"2026-04-13T12:36:49.223958Z","shell.execute_reply.started":"2026-04-13T12:36:49.209126Z","shell.execute_reply":"2026-04-13T12:36:49.223162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature selection, same features for both groupings\n\nfeatures = [\n    'is_booking_mean',   # commercial importance of the segment\n    'is_mobile_mean',    # digital behaviour\n    'is_package_mean',   # bundled vs standalone booking tendency\n    'srch_adults_cnt_mean',  # travel party size (family vs business)\n    'srch_children_cnt_mean',  # family travel signal\n    'srch_rm_cnt_mean',        # booking scale\n    'orig_destination_distance_mean', # how far users travel (local vs international)\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.225095Z","iopub.execute_input":"2026-04-13T12:36:49.225352Z","iopub.status.idle":"2026-04-13T12:36:49.236244Z","shell.execute_reply.started":"2026-04-13T12:36:49.225327Z","shell.execute_reply":"2026-04-13T12:36:49.235536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Створюємо два датасети із середніми значеннями відібраних ознак із агрегованих вище датасетів\n# та заповнюємо пропуски в orig_destination_distance_mean медіанним значенням\n\nX_market = agg_market[features].fillna(agg_market[features].median())\nX_market.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.237144Z","iopub.execute_input":"2026-04-13T12:36:49.237392Z","iopub.status.idle":"2026-04-13T12:36:49.262481Z","shell.execute_reply.started":"2026-04-13T12:36:49.237368Z","shell.execute_reply":"2026-04-13T12:36:49.261636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_destination = agg_destination[features].fillna(agg_destination[features].median())\nX_destination.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.263449Z","iopub.execute_input":"2026-04-13T12:36:49.263746Z","iopub.status.idle":"2026-04-13T12:36:49.282866Z","shell.execute_reply.started":"2026-04-13T12:36:49.263711Z","shell.execute_reply":"2026-04-13T12:36:49.282056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_destination.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.283959Z","iopub.execute_input":"2026-04-13T12:36:49.284318Z","iopub.status.idle":"2026-04-13T12:36:49.289687Z","shell.execute_reply.started":"2026-04-13T12:36:49.284281Z","shell.execute_reply":"2026-04-13T12:36:49.289049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_market.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.290672Z","iopub.execute_input":"2026-04-13T12:36:49.291346Z","iopub.status.idle":"2026-04-13T12:36:49.308575Z","shell.execute_reply.started":"2026-04-13T12:36:49.291317Z","shell.execute_reply":"2026-04-13T12:36:49.307714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_destination.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.309719Z","iopub.execute_input":"2026-04-13T12:36:49.310183Z","iopub.status.idle":"2026-04-13T12:36:49.325559Z","shell.execute_reply.started":"2026-04-13T12:36:49.310146Z","shell.execute_reply":"2026-04-13T12:36:49.324690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# normalize the data—each feature will have a mean of 0 and a standard deviation of 1\n\n# це важливо особливо для K-means, так як приналежність до певного кластеру \n# визначається саме за допомогою евклідової відстані\n\nscaler = StandardScaler()\n\nX_market = scaler.fit_transform(X_market)  # тут отримуємо вже масив\nX_destination = scaler.fit_transform(X_destination)\n\nprint(X_market[:,1])\nprint(X_destination[:,1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.326684Z","iopub.execute_input":"2026-04-13T12:36:49.327030Z","iopub.status.idle":"2026-04-13T12:36:49.347108Z","shell.execute_reply.started":"2026-04-13T12:36:49.327002Z","shell.execute_reply":"2026-04-13T12:36:49.346198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import silhouette_score\n\n# OPTIMAL K  –  Elbow + Silhouette\n\n# Перш ніж приступити до самої кластеризації треба визначити оптимальне число кластерів\n# Є два методи: метод ліктя та метод силуету\n# Спочатку обчислю за допомогою функції\n\nk_range = range(2, 11)   # кількість кластерів, які будемо перебирати в циклі для знаходення оптимального числа\n\n\ndef find_optimal_k(X, sample=5000):\n    np.random.seed(9)\n    # із датасету srch_destination_id(21470) візьмемо тільки 5000 зразків, а із market - всі 2082\n    sample_idx = np.random.choice(len(X), min(sample, len(X)), replace=False)  # масив індексів\n    X_samples  = X[sample_idx]     # в новий датасет записуємо тільки вибрані рандомні рядки \n     \n    kmeans_inertias= []       # Сума квадратів відстаней до центроїдів, K-Means Elbow\n    kmeans_silhouettes = []   # Silhouette score, якість кластерів K-Means\n    gmm_bic_scores = []       # Bayesian Information Criterion, оптимальне k для GMM\n    gmm_silhouettes= []       # Silhouette score, якість кластерів GMM\n\n    for n_clusters in k_range:\n\n        # K-Means\n        kmeans_model = KMeans(n_clusters=n_clusters, random_state=9, n_init=10)\n        kmeans_cluster_labels = kmeans_model.fit_predict(X_samples)\n        kmeans_inertias.append(kmeans_model.inertia_)\n        kmeans_silhouettes.append(silhouette_score(X_samples, kmeans_cluster_labels))\n\n        # GMM\n        gmm_model = GaussianMixture(n_components=n_clusters, random_state=9)\n        gmm_model.fit(X_samples)\n        gmm_cluster_labels  = gmm_model.predict(X_samples)\n        gmm_bic_scores.append(gmm_model.bic(X_samples))\n        gmm_silhouettes.append(silhouette_score(X_samples, gmm_cluster_labels))\n\n    best_kmeans_k = list(k_range)[np.argmax(kmeans_silhouettes)]\n    best_gmm_k = list(k_range)[np.argmin(gmm_bic_scores)]\n\n    # повертаємо словник \n    return {\n        'best_kmeans_k'     : best_kmeans_k,\n        'best_gmm_k'        : best_gmm_k,\n        'kmeans_inertias'   : kmeans_inertias,\n        'kmeans_silhouettes': kmeans_silhouettes,\n        'gmm_bic_scores'    : gmm_bic_scores,\n        'gmm_silhouettes'   : gmm_silhouettes,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.348509Z","iopub.execute_input":"2026-04-13T12:36:49.348907Z","iopub.status.idle":"2026-04-13T12:36:49.357918Z","shell.execute_reply.started":"2026-04-13T12:36:49.348879Z","shell.execute_reply":"2026-04-13T12:36:49.357093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef show_plot(results, label):\n    ks = list(k_range)\n    \n    fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n    fig.suptitle(f'Optimal K selection – {label}', fontsize=15, fontweight='bold')\n    \n    # Elbow\n    axes[0].plot(ks, results['kmeans_inertias'], 'o-', color=C1, lw=2)\n    axes[0].set_title('K-Means Elbow (Inertia)', fontweight='bold')\n    axes[0].set_xlabel('k'); axes[0].set_ylabel('Inertia'); axes[0].grid(True)\n    \n    # Silhouette K-Means\n    axes[1].plot(ks, results['kmeans_silhouettes'], 's-', color=C2, lw=2)\n    axes[1].axvline(results['best_kmeans_k'], color=C2, ls=':', lw=1.5, label=f\"best k={results['best_kmeans_k']}\")\n    axes[1].set_title('K-Means Silhouette', fontweight='bold')\n    axes[1].set_xlabel('k'); axes[1].set_ylabel('Silhouette score')\n    axes[1].legend(); axes[1].grid(True)\n    \n    # BIC GMM\n    axes[2].plot(ks, results['gmm_bic_scores'], '^-', color=C3, lw=2)\n    axes[2].axvline(results['best_gmm_k'], color=C3, ls=':', lw=1.5, label=f\"best k={results['best_gmm_k']}\")\n    axes[2].set_title('GMM BIC (lower = better)', fontweight='bold')\n    axes[2].set_xlabel('k'); axes[2].set_ylabel('BIC')\n    axes[2].legend(); axes[2].grid(True)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.358995Z","iopub.execute_input":"2026-04-13T12:36:49.359287Z","iopub.status.idle":"2026-04-13T12:36:49.381802Z","shell.execute_reply.started":"2026-04-13T12:36:49.359253Z","shell.execute_reply":"2026-04-13T12:36:49.380845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_market = find_optimal_k(X_market)\nshow_plot(results_market, label='hotel_market')\n\nresults_dest = find_optimal_k(X_destination)\nshow_plot(results_dest, label='srch_destination_id')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:36:49.382866Z","iopub.execute_input":"2026-04-13T12:36:49.383220Z","iopub.status.idle":"2026-04-13T12:37:03.681294Z","shell.execute_reply.started":"2026-04-13T12:36:49.383193Z","shell.execute_reply":"2026-04-13T12:37:03.680494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# доступ до значень:\n\nprint(\"Optimal number of cluster for hotel_market:\")\nprint(f\"K-means: {results_market['best_kmeans_k']}\")\nprint(f\"GMM: {results_market['best_gmm_k']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:37:03.682299Z","iopub.execute_input":"2026-04-13T12:37:03.682601Z","iopub.status.idle":"2026-04-13T12:37:03.687646Z","shell.execute_reply.started":"2026-04-13T12:37:03.682567Z","shell.execute_reply":"2026-04-13T12:37:03.686892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Optimal number of cluster for srch_destination_id:\")\nprint(f\"K-means: {results_dest['best_kmeans_k']}\")\nprint(f\"GMM: {results_dest['best_gmm_k']}\")    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:37:03.688672Z","iopub.execute_input":"2026-04-13T12:37:03.688980Z","iopub.status.idle":"2026-04-13T12:37:03.707793Z","shell.execute_reply.started":"2026-04-13T12:37:03.688953Z","shell.execute_reply":"2026-04-13T12:37:03.706918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n#  Функція для кластеризації на повних стандартизовани згрупованих даних, а також для відображення графіків\n# по ринкам готелів та напрямкам пошуку\n\ndef cluster_and_visualize(feature_matrix, agg_df, group_col, best_kmeans_k, best_gmm_k, label):\n\n    # тут навчаємо на всіх даних \n\n    kmeans_model = KMeans(n_clusters=best_kmeans_k, random_state=9, n_init=10)\n    kmeans_labels = kmeans_model.fit_predict(feature_matrix)\n\n    gmm_model = GaussianMixture(n_components=best_gmm_k, random_state=9, max_iter=300)\n    gmm_model.fit(feature_matrix)\n    gmm_labels = gmm_model.predict(feature_matrix)\n    gmm_proba_per_point = gmm_model.predict_proba(feature_matrix).max(axis=1)\n    # predict_proba повертає матрицю (n_ринків × n_кластерів):\n    # [[0.8, 0.1, 0.1],   ← ринок №27:  80% належить до кластера 0\n\n    # Рахуємо silhouette на підвибірці\n    rng = np.random.default_rng(9)\n    sample_indices = rng.choice(len(feature_matrix), min(8000, len(feature_matrix)), replace=False)\n    silhouette_kmeans = silhouette_score(feature_matrix[sample_indices], kmeans_labels[sample_indices])\n    silhouette_gmm = silhouette_score(feature_matrix[sample_indices], gmm_labels[sample_indices])\n\n    print(f\"K-Means  k={best_kmeans_k}  Silhouette = {silhouette_kmeans:.4f}\")\n    print(f\"GMM k={best_gmm_k}  Silhouette = {silhouette_gmm:.4f}\")\n\n    # PCA: 7 ознак перетворюємо на 2\n    pca = PCA(n_components=2, random_state=9)\n    feature_matrix_2d = pca.fit_transform(feature_matrix)\n    explained_variance_ratio = pca.explained_variance_ratio_\n    # explained_variance_ratio → [0.45, 0.20] означає: PC1 пояснює 45% варіації даних, PC2 — 20%, разом 65%\n\n    # plot  \n    # беремо тільки вибірку \n    plot_sample_indices = rng.choice(len(feature_matrix_2d), min(6000, len(feature_matrix_2d)), replace=False)\n\n    fig, axes = plt.subplots(1, 2, figsize=(16, 7))\n    fig.suptitle(f'Clustering – {label} (PCA projection)', fontsize=14, fontweight='bold')\n\n    # K-Means: кожен кластер — окремий колір\n    for cluster_id in range(best_kmeans_k):\n        cluster_mask = kmeans_labels[plot_sample_indices] == cluster_id\n        axes[0].scatter(\n            feature_matrix_2d[plot_sample_indices][cluster_mask, 0],  # PC1\n            feature_matrix_2d[plot_sample_indices][cluster_mask, 1],  # PC2\n                    s=8, alpha=0.6, label=f'Cluster {cluster_id}')\n        \n    axes[0].set_title(f'K-Means k={best_kmeans_k} (silhouette={silhouette_kmeans:.3f})', fontweight='bold')\n    axes[0].set_xlabel(f'PC1 ({explained_variance_ratio[0]:.1%} variance)')\n    axes[0].set_ylabel(f'PC2 ({explained_variance_ratio[1]:.1%} variance)')\n    axes[0].legend(markerscale=2, fontsize=8)\n    axes[0].grid(True)\n\n    # GMM: колір = кластер, прозорість = впевненість моделі\n    # Точки з низькою ймовірністю (невпевнені) будуть прозорішими\n    for cluster_id in range(best_gmm_k):\n        cluster_mask = gmm_labels[plot_sample_indices] == cluster_id\n        cluster_confidence = gmm_proba_per_point[plot_sample_indices][cluster_mask]\n        axes[1].scatter(\n            feature_matrix_2d[plot_sample_indices][cluster_mask, 0],\n            feature_matrix_2d[plot_sample_indices][cluster_mask, 1],\n            s=8, alpha=np.clip(cluster_confidence, 0.2, 0.9), label=f'Cluster {cluster_id}')\n        \n    axes[1].set_title(f'GMM k={best_gmm_k} (silhouette={silhouette_gmm:.3f})', fontweight='bold')\n    axes[1].set_xlabel(f'PC1 ({explained_variance_ratio[0]:.1%} variance)')\n    axes[1].set_ylabel(f'PC2 ({explained_variance_ratio[1]:.1%} variance)')\n    axes[1].legend(markerscale=2, fontsize=8)\n    axes[1].grid(True)\n\n    plt.tight_layout()\n    plt.show()\n\n    # Додаємо мітки кластерів назад до агрегованого датафрейму\n    # щоб побачити які середні значення ознак у кожному кластері\n\n    agg_df_with_clusters = agg_df.copy()\n    agg_df_with_clusters['km_cluster'] = kmeans_labels\n    agg_df_with_clusters['gm_cluster'] = gmm_labels\n\n    print(f\"\\nK-Means cluster profiles (mean of features):\")\n    cluster_profiles = agg_df_with_clusters.groupby('km_cluster')[features].mean().round(4)\n    print(cluster_profiles.to_string())\n\n    return agg_df_with_clusters\n\n\nresult_market = cluster_and_visualize(X_market, agg_market, 'hotel_market',results_market['best_kmeans_k'],\n                                        results_market['best_gmm_k'], label='hotel_market')\n\nresult_dest = cluster_and_visualize(X_destination, agg_destination, 'srch_destination_id',results_dest['best_kmeans_k'],\n                                        results_dest['best_gmm_k'],label='srch_destination_id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T12:37:03.708809Z","iopub.execute_input":"2026-04-13T12:37:03.709129Z","iopub.status.idle":"2026-04-13T12:37:10.487124Z","shell.execute_reply.started":"2026-04-13T12:37:03.709100Z","shell.execute_reply":"2026-04-13T12:37:10.486300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Таблички під графіками необхідні для того, щоб зрозуміти, що за кластери ми отримали**","metadata":{}},{"cell_type":"markdown","source":"# Аналіз результатів групування за ринками готелів:\n\nCluster 0 — Найнижча конверсія (4%), але половина всіх пошуків — пакети (50%). Ринки де користувачі масово шукають тури \"все включено\", але довго вибирають. Типово для курортних напрямків. Пакетний туристичний ринок\n\nCluster 1 — Найвища конверсія (12%), найбільше дорослих (2.56) і кімнат (1.58). Ринки де бронюють групами — сім'ї або корпоративні поїздки. Груповий/сімейний ринок\n\nCluster 2 — Найбільша відстань (5003), мало дорослих (1.90), мало кімнат. Ринки з міжнародним трафіком одинаків або пар, які летять далеко. Середня конверсія (10%).Міжнародний індивідуальний ринок\n\nCluster 3 - Ринки з найвищою конверсією (14.7%) і найкоротшою відстанню (837). Сім'ї з дітьми, які бронюють самостійно без пакету, поблизу від дому.Найбільш комерційно цінний сегмент для готельного ринку.","metadata":{}},{"cell_type":"markdown","source":"# Аналіз результатів групування за ID напрямку пошуку:\n\nCluster 0 — Далекі пакетні поїздки 4940.2143, найнижча конверсія 0,07 тільки замовили, пакетний тур 0,29\n\nCluster 1 — локальні самостійні бронювання, відстань -  767.6145, досить висока конверсія 14%, майже без пакетів 0,03, тобто ділові поїздки або короткі вікенди.\n\nCluster 2 - шукають великими групами — розширені сім'ї або кілька сімей разом, кількість кімнат перевищує 2. Висока кількість і дорослих і дітей одночасно підтверджує саме сімейний груповий характер, бронюють самостійно без пакету.","metadata":{}},{"cell_type":"markdown","source":"# **K-Means vs GMM:**\n\n\n**hotel_market:**\n\nK-Means k=4, silhouette=0.220\nGMM k=7, silhouette=0.133\n\n**srch_destination_id:**\n\nK-Means k=3, silhouette=0.243\nGMM k=10, silhouette=-0.013\n\nУ GMM кластери хаотичні, перекриваються, можливо забагато кластерів. У K-Means хоча і немає дуже чіткого відділення, але краще можна зрозуміти розподіл.\n\nУ даній задачі K-Means виграв по обох групуваннях.","metadata":{}},{"cell_type":"markdown","source":"# Загальні враження:\n\nТак як пройшов вже деякий час від цієї теми, мабуть, найважче було зрозуміти для чого це треба. Дуже незручно працювати із даними у вигляді узагальнених чисел, незрозуміло, що саме за дані, що вони значать, постійно треба відволікатися на пояснення значень ознак.\n\nТільки із останніх табличок стало трохи зрозуміліше, що отримали на виході і для чого все це робилося. Візуалізація особисто для мене нічого не пояснила, ну або я її неправильно зробила. Ще можна було б спробувати виконати це за допомогою t-SNE, але вже немає часу.\n\nБез підтримки AI та різних статтей, де хоч щось пояснювалося, я б ніколи цього не писала, особливо останню функцію.\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}