{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.cluster import KMeans\nimport cudf\nfrom cuml.cluster import HDBSCAN\nimport umap\nfrom cuml.cluster import KMeans\nimport cupy as cp\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import MinMaxScaler\nfrom cuml.cluster import AgglomerativeClustering\nfrom sklearn.mixture import GaussianMixture","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:43:23.733190Z","iopub.execute_input":"2025-06-25T07:43:23.733893Z","iopub.status.idle":"2025-06-25T07:43:38.003456Z","shell.execute_reply.started":"2025-06-25T07:43:23.733863Z","shell.execute_reply":"2025-06-25T07:43:38.002754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet',\n                    engine = 'pyarrow')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:43:38.004124Z","iopub.execute_input":"2025-06-25T07:43:38.004652Z","iopub.status.idle":"2025-06-25T07:43:44.293279Z","shell.execute_reply.started":"2025-06-25T07:43:38.004631Z","shell.execute_reply":"2025-06-25T07:43:44.292568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:43:44.294035Z","iopub.execute_input":"2025-06-25T07:43:44.294321Z","iopub.status.idle":"2025-06-25T07:43:44.299256Z","shell.execute_reply.started":"2025-06-25T07:43:44.294288Z","shell.execute_reply":"2025-06-25T07:43:44.298563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.loc[:, ~df.isin([np.inf, -np.inf]).any()] # any() checks atleast one element\ndf.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:43:44.300010Z","iopub.execute_input":"2025-06-25T07:43:44.300209Z","iopub.status.idle":"2025-06-25T07:44:32.949112Z","shell.execute_reply.started":"2025-06-25T07:43:44.300192Z","shell.execute_reply":"2025-06-25T07:44:32.948475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop(columns = ['label'])\ny = df['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:44:32.949951Z","iopub.execute_input":"2025-06-25T07:44:32.950177Z","iopub.status.idle":"2025-06-25T07:44:33.982891Z","shell.execute_reply.started":"2025-06-25T07:44:32.950158Z","shell.execute_reply":"2025-06-25T07:44:33.981992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some cols has all inf values","metadata":{}},{"cell_type":"markdown","source":"# Working with a sample","metadata":{}},{"cell_type":"markdown","source":"## KMeans to cluster the data","metadata":{}},{"cell_type":"code","source":"df_sample = X.sample(n = 30000, random_state = 42)\nfeatures = df_sample.values\nscaler = MinMaxScaler()\nfeatures = scaler.fit_transform(features)\nkmeans = KMeans(n_clusters = 5, random_state = 42)\ncluster_labels = kmeans.fit_predict(features)\ndf_sample['cluster'] = cluster_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:44:33.985131Z","iopub.execute_input":"2025-06-25T07:44:33.985824Z","iopub.status.idle":"2025-06-25T07:44:59.923760Z","shell.execute_reply.started":"2025-06-25T07:44:33.985795Z","shell.execute_reply":"2025-06-25T07:44:59.923115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## UMAP to reduce the dimensions of data","metadata":{}},{"cell_type":"code","source":"umap_model = umap.UMAP(n_components = 2)\nX_umap = umap_model.fit_transform(features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:44:59.924566Z","iopub.execute_input":"2025-06-25T07:44:59.925050Z","iopub.status.idle":"2025-06-25T07:45:33.789478Z","shell.execute_reply.started":"2025-06-25T07:44:59.925021Z","shell.execute_reply":"2025-06-25T07:45:33.788840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nscatter = plt.scatter(\n    X_umap[:, 0],\n    X_umap[:, 1],\n    c=df_sample['cluster'],\n    cmap='Dark2',        \n    alpha=0.7,\n    s=10,\n    vmin=0,\n    vmax=4              \n)\nplt.xlabel('UMAP Dimension 1')\nplt.ylabel('UMAP Dimension 2')\nplt.title('K-Means Clusters Visualized with UMAP (sampled 30k points)')\nplt.colorbar(scatter, label='Cluster', ticks=range(5))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:33.790265Z","iopub.execute_input":"2025-06-25T07:45:33.790591Z","iopub.status.idle":"2025-06-25T07:45:34.390595Z","shell.execute_reply.started":"2025-06-25T07:45:33.790551Z","shell.execute_reply":"2025-06-25T07:45:34.389876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## HDBScan to cluster data","metadata":{}},{"cell_type":"code","source":"gdf = cudf.DataFrame.from_pandas(pd.DataFrame(features))\nclusterer = HDBSCAN(min_cluster_size = 7)\nhdb_labels = clusterer.fit_predict(gdf)\ndf_sample['hdbscan_cluster'] = hdb_labels.to_numpy()\n#get() method transfers the data from GPU memory to host memory as a NumPy array.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:34.391443Z","iopub.execute_input":"2025-06-25T07:45:34.391680Z","iopub.status.idle":"2025-06-25T07:45:37.987206Z","shell.execute_reply.started":"2025-06-25T07:45:34.391662Z","shell.execute_reply":"2025-06-25T07:45:37.986384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nscatter = plt.scatter(\n    X_umap[:, 0], X_umap[:, 1],\n    c=df_sample['hdbscan_cluster'],  \n    cmap='Dark2', s=10, alpha=0.7\n)\nplt.xlabel('UMAP Dimension 1')\nplt.ylabel('UMAP Dimension 2')\nplt.title('K-Means Clusters on UMAP-Reduced Data')\nplt.colorbar(scatter, label='Cluster')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:37.988103Z","iopub.execute_input":"2025-06-25T07:45:37.988341Z","iopub.status.idle":"2025-06-25T07:45:38.538139Z","shell.execute_reply.started":"2025-06-25T07:45:37.988318Z","shell.execute_reply":"2025-06-25T07:45:38.537411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agglo = AgglomerativeClustering(n_clusters = 40)\nagglo_labels = agglo.fit_predict(gdf)\ndf_sample['agglo_cluster'] = agglo_labels.to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:38.539056Z","iopub.execute_input":"2025-06-25T07:45:38.539296Z","iopub.status.idle":"2025-06-25T07:45:40.315147Z","shell.execute_reply.started":"2025-06-25T07:45:38.539278Z","shell.execute_reply":"2025-06-25T07:45:40.314537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nscatter = plt.scatter(\n    X_umap[:, 0], X_umap[:, 1],\n    c=df_sample['agglo_cluster'],  \n    cmap='Dark2', s=10, alpha=0.7\n)\nplt.xlabel('UMAP Dimension 1')\nplt.ylabel('UMAP Dimension 2')\nplt.title('K-Means Clusters on UMAP-Reduced Data')\nplt.colorbar(scatter, label='Cluster')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:40.315884Z","iopub.execute_input":"2025-06-25T07:45:40.316179Z","iopub.status.idle":"2025-06-25T07:45:40.867434Z","shell.execute_reply.started":"2025-06-25T07:45:40.316150Z","shell.execute_reply":"2025-06-25T07:45:40.866633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analyzing a tad-bit of main data","metadata":{}},{"cell_type":"code","source":"first_10 = df.iloc[:, :10]\n\n# Compute min, max, median for each column\nsummary = pd.DataFrame({\n    'min': first_10.min(),\n    'max': first_10.max(),\n    'median': first_10.median()\n})\n\nprint(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:40.868285Z","iopub.execute_input":"2025-06-25T07:45:40.868548Z","iopub.status.idle":"2025-06-25T07:45:40.941126Z","shell.execute_reply.started":"2025-06-25T07:45:40.868529Z","shell.execute_reply":"2025-06-25T07:45:40.940537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"negative_cols = [col for col in X.columns if (X[col] < 0).any()]\nprint(negative_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:40.941860Z","iopub.execute_input":"2025-06-25T07:45:40.942063Z","iopub.status.idle":"2025-06-25T07:45:41.595213Z","shell.execute_reply.started":"2025-06-25T07:45:40.942048Z","shell.execute_reply":"2025-06-25T07:45:41.594398Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Making test and train set, processing them separately to avoid data leakage and shuffling them as well","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:41.596049Z","iopub.execute_input":"2025-06-25T07:45:41.596620Z","iopub.status.idle":"2025-06-25T07:45:44.840091Z","shell.execute_reply.started":"2025-06-25T07:45:41.596584Z","shell.execute_reply":"2025-06-25T07:45:44.839194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:45:44.840959Z","iopub.execute_input":"2025-06-25T07:45:44.841206Z","iopub.status.idle":"2025-06-25T07:45:44.845033Z","shell.execute_reply.started":"2025-06-25T07:45:44.841188Z","shell.execute_reply":"2025-06-25T07:45:44.844423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib \n\nscalers = {}  # Dictionary to store the scaler for each column\nX_train_scaled = X_train.copy()\n\nfor col in X_train.columns:\n    if (X_train[col] < 0).any():\n        scaler = MinMaxScaler(feature_range=(-1, 1))\n    else:\n        scaler = MinMaxScaler(feature_range=(0, 1))\n    X_train_scaled[[col]] = scaler.fit_transform(X_train[[col]])\n    scalers[col] = scaler\n\njoblib.dump(scalers, 'column_scalers.pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:50:50.049977Z","iopub.execute_input":"2025-06-25T07:50:50.050648Z","iopub.status.idle":"2025-06-25T07:50:57.694078Z","shell.execute_reply.started":"2025-06-25T07:50:50.050620Z","shell.execute_reply":"2025-06-25T07:50:57.693409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_10 =X_train_scaled.iloc[:, :10]\n\n# Compute min, max, median for each column\nsummary = pd.DataFrame({\n    'min': first_10.min(),\n    'max': first_10.max(),\n    'median': first_10.median()\n})\n\nprint(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:51:02.729034Z","iopub.execute_input":"2025-06-25T07:51:02.729818Z","iopub.status.idle":"2025-06-25T07:51:02.808604Z","shell.execute_reply.started":"2025-06-25T07:51:02.729792Z","shell.execute_reply":"2025-06-25T07:51:02.807925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_scaled = X_test.copy()\nfor col in X_test.columns:\n    X_test_scaled[[col]] = scalers[col].transform(X_test[[col]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:51:06.010346Z","iopub.execute_input":"2025-06-25T07:51:06.011142Z","iopub.status.idle":"2025-06-25T07:51:08.303221Z","shell.execute_reply.started":"2025-06-25T07:51:06.011116Z","shell.execute_reply":"2025-06-25T07:51:08.302525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_10 =X_test_scaled.iloc[:, :10]\n\n# Compute min, max, median for each column\nsummary = pd.DataFrame({\n    'min': first_10.min(),\n    'max': first_10.max(),\n    'median': first_10.median()\n})\n\nprint(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:51:10.147206Z","iopub.execute_input":"2025-06-25T07:51:10.147548Z","iopub.status.idle":"2025-06-25T07:51:10.180752Z","shell.execute_reply.started":"2025-06-25T07:51:10.147524Z","shell.execute_reply":"2025-06-25T07:51:10.180130Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This might happen a lot probably in the actual test set","metadata":{}},{"cell_type":"code","source":"X_train_scaled_cudf = cudf.DataFrame.from_pandas(X_train_scaled)\n\nkmeans_gpu = KMeans(n_clusters=5, random_state=42)\nk_labels = kmeans_gpu.fit_predict(X_train_scaled_cudf)\n\nX_train_scaled[\"k_labels\"] = k_labels.to_numpy()\n\n# Save the trained GPU KMeans model\njoblib.dump(kmeans_gpu, \"kmeans_model_gpu.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:52:38.638924Z","iopub.execute_input":"2025-06-25T07:52:38.639655Z","execution_failed":"2025-06-25T07:54:59.225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_scaled_cudf = cudf.DataFrame.from_pandas(X_test_scaled)\n\n\nk_labels_test = kmeans_gpu.predict(X_test_scaled_cudf)\n\nX_test_scaled[\"k_labels\"] = k_labels_test.to_numpy()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_scaled.to_parquet(\"X_train_scaled.parquet\", index=False)\ny_train.to_frame().to_parquet(\"y_train.parquet\", index=False)\n\nX_test_scaled.to_parquet(\"X_test_scaled.parquet\", index=False)\ny_test.to_frame().to_parquet(\"y_test.parquet\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}