{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":11974159,"sourceType":"datasetVersion","datasetId":7530087}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_error, r2_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:46.851498Z","iopub.execute_input":"2025-05-27T19:35:46.852466Z","iopub.status.idle":"2025-05-27T19:35:47.279681Z","shell.execute_reply.started":"2025-05-27T19:35:46.852403Z","shell.execute_reply":"2025-05-27T19:35:47.278723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/train-crypto/train_crypto.csv')\nprint(df.info())\nprint(df['label'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:47.281055Z","iopub.execute_input":"2025-05-27T19:35:47.282051Z","iopub.status.idle":"2025-05-27T19:35:54.803260Z","shell.execute_reply.started":"2025-05-27T19:35:47.282027Z","shell.execute_reply":"2025-05-27T19:35:54.802509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['label'].hist(bins=50, figsize=(10, 5))\nplt.title(\"Label Histogram\")\nplt.xlabel(\"Label Value\")\nplt.ylabel(\"Frequency\")\nplt.grid(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:54.804035Z","iopub.execute_input":"2025-05-27T19:35:54.804341Z","iopub.status.idle":"2025-05-27T19:35:55.227616Z","shell.execute_reply.started":"2025-05-27T19:35:54.804317Z","shell.execute_reply":"2025-05-27T19:35:55.226835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_columns = df.columns[1:6]\n\ndf[sample_columns].hist(bins=50, figsize=(15, 8))\nplt.suptitle(\"Sample Feature Distributions\")\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:55.229398Z","iopub.execute_input":"2025-05-27T19:35:55.229677Z","iopub.status.idle":"2025-05-27T19:35:56.486695Z","shell.execute_reply.started":"2025-05-27T19:35:55.229658Z","shell.execute_reply":"2025-05-27T19:35:56.485726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = df['X100']\ny = df['label']\n\nplt.figure(figsize=(10, 5))\nplt.scatter(x, y, alpha=0.3, label='Data Points')\n\nslope, intercept = np.polyfit(x, y, 1)\nplt.plot(x, slope * x + intercept, color='red', label='Regression Line')\n\nplt.xlabel(\"X100\")\nplt.ylabel(\"Label\")\nplt.title(\"Label and X100 using regression line\")\nplt.legend()\nplt.grid(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:56.487657Z","iopub.execute_input":"2025-05-27T19:35:56.487972Z","iopub.status.idle":"2025-05-27T19:35:57.139153Z","shell.execute_reply.started":"2025-05-27T19:35:56.487952Z","shell.execute_reply":"2025-05-27T19:35:57.138234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.regplot(x='X400', y='label', data=df, scatter_kws={'alpha':0.3}, line_kws={\"color\":\"red\"})\nplt.title(\"X400 vs Label\")\nplt.grid(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:57.140382Z","iopub.execute_input":"2025-05-27T19:35:57.140674Z","iopub.status.idle":"2025-05-27T19:35:58.680385Z","shell.execute_reply.started":"2025-05-27T19:35:57.140654Z","shell.execute_reply":"2025-05-27T19:35:58.679509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 'timestamp' in df.columns:\n    df = df.drop(columns=['timestamp'])\n\ndf = df.replace([np.inf, -np.inf], np.nan)\ndf = df.fillna(df.mean(numeric_only=True))\n\nX = df.drop(columns=['label'])\ny = df['label']\nX = X.dropna(axis=1, how='all')\n\nX = X.fillna(X.mean(numeric_only=True))\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\npipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('model', LinearRegression())\n])\n\npipeline.fit(X_train, y_train)\n\ny_pred = pipeline.predict(X_test)\n\nprint(\"MSE:\", mean_squared_error(y_test, y_pred))\nprint(\"R^2 Score:\", r2_score(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:35:58.681372Z","iopub.execute_input":"2025-05-27T19:35:58.681726Z","iopub.status.idle":"2025-05-27T19:36:05.162906Z","shell.execute_reply.started":"2025-05-27T19:35:58.681695Z","shell.execute_reply":"2025-05-27T19:36:05.161552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"slope, intercept = np.polyfit(y_test, y_pred, 1)\nregression_line = slope * y_test + intercept\n\nplt.figure(figsize=(10, 5))\nplt.scatter(y_test, y_pred, alpha=0.3)\nplt.plot(y_test, regression_line, color='red', label='Regression Line')\nplt.xlabel(\"Actual Label\")\nplt.ylabel(\"Predicted Label\")\nplt.title(\"Actual vs Predicted Labels\")\nplt.grid(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T19:39:11.436203Z","iopub.execute_input":"2025-05-27T19:39:11.436992Z","iopub.status.idle":"2025-05-27T19:39:11.720677Z","shell.execute_reply.started":"2025-05-27T19:39:11.436965Z","shell.execute_reply":"2025-05-27T19:39:11.719803Z"}},"outputs":[],"execution_count":null}]}