{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":10228621,"sourceType":"datasetVersion","datasetId":6324106}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import required libraries\nimport pandas as pd\nimport numpy as np\n\n# Load only 100MB of the data\n# Use `nrows` to control the data size, adjust based on your needs.\nchunk_size = 1_000_000  # Roughly 1 million rows (~100MB, depends on columns)\ncompetition_path = \"/kaggle/input/amex-default-prediction/\"\n\n# Step 1: Load a sample chunk of the train_data\nprint(\"Loading a 100MB subset of the data...\")\ntrain_data_chunk = pd.read_csv(f\"{competition_path}train_data.csv\", nrows=chunk_size)\n\n# Check the data shape and first rows\nprint(f\"Data Shape: {train_data_chunk.shape}\")\ntrain_data_chunk.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:38:15.154804Z","iopub.execute_input":"2024-12-17T16:38:15.155167Z","iopub.status.idle":"2024-12-17T16:39:41.306057Z","shell.execute_reply.started":"2024-12-17T16:38:15.155136Z","shell.execute_reply":"2024-12-17T16:39:41.304999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Memory-efficient data types\ndef reduce_memory_usage(df):\n    \"\"\" Reduces memory usage of the dataframe by downcasting data types. \"\"\"\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type == 'float64':\n            df[col] = df[col].astype('float32')\n        elif col_type == 'int64':\n            df[col] = df[col].astype('int32')\n        elif col_type == 'object' and df[col].nunique() < df.shape[0] / 2:\n            df[col] = df[col].astype('category')\n    return df\n\nprint(\"Reducing memory usage...\")\ntrain_data_chunk = reduce_memory_usage(train_data_chunk)\n\n# Check memory usage after optimization\nprint(\"Memory usage after optimization:\")\ntrain_data_chunk.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:39:41.308645Z","iopub.execute_input":"2024-12-17T16:39:41.309764Z","iopub.status.idle":"2024-12-17T16:39:42.654922Z","shell.execute_reply.started":"2024-12-17T16:39:41.309711Z","shell.execute_reply":"2024-12-17T16:39:42.653500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical features list\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', \n                        'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n# Step 3: Calculate total unique elements in categorical features\nprint(\"Calculating unique values for categorical features...\")\nunique_counts = {}\nfor feature in categorical_features:\n    if feature in train_data_chunk.columns:\n        unique_counts[feature] = train_data_chunk[feature].nunique()\n    else:\n        unique_counts[feature] = \"Feature not found\"\n\n# Display unique value counts\nprint(\"Unique value counts for categorical features:\")\nfor feature, count in unique_counts.items():\n    print(f\"{feature}: {count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:39:42.656577Z","iopub.execute_input":"2024-12-17T16:39:42.656972Z","iopub.status.idle":"2024-12-17T16:39:42.751032Z","shell.execute_reply.started":"2024-12-17T16:39:42.656937Z","shell.execute_reply":"2024-12-17T16:39:42.749746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Update specified columns to categorical\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', \n                        'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nfor col in categorical_features:\n    if col in train_data_chunk.columns:\n        train_data_chunk[col] = train_data_chunk[col].astype('category')\n\nprint(\"Specified columns updated to categorical type.\\n\")\n\n# Step 2: Enhanced null percentage display with data types\ndef display_null_percentage_with_dtype(df):\n    \"\"\" Calculate and display percentage of null/empty values for each column along with data type. \"\"\"\n    null_percentage = df.isnull().sum() / len(df) * 100  # Calculate null percentages\n    null_percentage = null_percentage.reset_index()\n    null_percentage.columns = ['Column', 'Null_Percentage']\n    \n    # Add column data types\n    null_percentage['Data_Type'] = null_percentage['Column'].apply(lambda x: df[x].dtype)\n    \n    # Sort by null percentages\n    null_percentage = null_percentage.sort_values(by='Null_Percentage', ascending=False)\n    \n    # Print the results in chunks with limited width\n    print(\"Percentage of Null Values per Column (with Data Type):\")\n    print(\"=\"*70)\n    for i in range(0, len(null_percentage), 10):  # Print in chunks of 10 columns\n        chunk = null_percentage.iloc[i:i+10]\n        print(chunk.to_string(index=False))\n        print(\"=\"*70)\n\n# Execute the enhanced null percentage function\nprint(\"Exploring null/empty values with data types...\\n\")\ndisplay_null_percentage_with_dtype(train_data_chunk)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:39:42.752735Z","iopub.execute_input":"2024-12-17T16:39:42.753098Z","iopub.status.idle":"2024-12-17T16:39:43.197716Z","shell.execute_reply.started":"2024-12-17T16:39:42.753063Z","shell.execute_reply":"2024-12-17T16:39:43.196489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\ndef clean_data(df, cat_features):\n    \"\"\" Cleans the dataset by imputing missing values for numerical and categorical features. \"\"\"\n    # Separate numerical and categorical columns\n    numerical_cols = df.select_dtypes(include=['float32', 'int32']).columns\n    categorical_cols = cat_features\n    \n    # Impute numerical columns with median\n    print(\"Imputing numerical features with median...\")\n    num_imputer = SimpleImputer(strategy='median')\n    df[numerical_cols] = num_imputer.fit_transform(df[numerical_cols])\n    \n    # Impute categorical columns with mode\n    print(\"Imputing categorical features with mode...\")\n    cat_imputer = SimpleImputer(strategy='most_frequent')\n    for col in categorical_cols:\n        if col in df.columns:\n            # Use ravel() to flatten the result\n            df[col] = cat_imputer.fit_transform(df[[col]]).ravel()\n    \n    print(\"Data cleaning completed successfully!\")\n    return df\n\n# Call the cleaning function\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', \n                        'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n# Clean the train_data_chunk\ntrain_data_cleaned = clean_data(train_data_chunk, categorical_features)\n\n# Check the cleaned data\nprint(\"First 5 rows after cleaning:\")\nprint(train_data_cleaned.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:39:43.200468Z","iopub.execute_input":"2024-12-17T16:39:43.201020Z","iopub.status.idle":"2024-12-17T16:40:46.950163Z","shell.execute_reply.started":"2024-12-17T16:39:43.200980Z","shell.execute_reply":"2024-12-17T16:40:46.948364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_null_percentage_with_dtype(df):\n    \"\"\"\n    Calculate and display percentage of null/empty values for each column \n    along with their data types.\n    \"\"\"\n    # Calculate null percentage\n    null_percentage = df.isnull().sum() / len(df) * 100\n    null_percentage = null_percentage.reset_index()\n    null_percentage.columns = ['Column', 'Null_Percentage']\n\n    # Add column data types\n    null_percentage['Data_Type'] = null_percentage['Column'].apply(lambda x: df[x].dtype)\n\n    # Sort by null percentage\n    null_percentage = null_percentage.sort_values(by='Null_Percentage', ascending=False)\n\n    # Print the results in chunks for better readability\n    print(\"Percentage of Null Values per Column (with Data Type):\")\n    print(\"=\"*70)\n    for i in range(0, len(null_percentage), 10):  # Display 10 rows at a time\n        chunk = null_percentage.iloc[i:i+10]\n        print(chunk.to_string(index=False))\n        print(\"=\"*70)\n\n# Execute the function on the cleaned dataset\nprint(\"Displaying null values, column names, and data types for cleaned data...\\n\")\ndisplay_null_percentage_with_dtype(train_data_cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:46.952173Z","iopub.execute_input":"2024-12-17T16:40:46.952779Z","iopub.status.idle":"2024-12-17T16:40:47.462370Z","shell.execute_reply.started":"2024-12-17T16:40:46.952739Z","shell.execute_reply":"2024-12-17T16:40:47.461016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_date_features(df, date_column):\n    \"\"\"\n    Process date features to extract meaningful components and calculate differences.\n    \"\"\"\n    print(f\"Processing date features for column: {date_column}...\")\n\n    # Step 1: Force conversion of 'S_2' from category to string, then to datetime\n    print(\"Ensuring the date column is in datetime format...\")\n    df[date_column] = pd.to_datetime(df[date_column].astype(str), errors='coerce')\n\n    # Step 2: Extract date components\n    print(\"Extracting date components...\")\n    date_features = pd.DataFrame({\n        'year': df[date_column].dt.year,\n        'month': df[date_column].dt.month,\n        'day_of_week': df[date_column].dt.dayofweek,\n        'day_of_year': df[date_column].dt.dayofyear\n    }, index=df.index)\n\n    # Step 3: Calculate the difference in days relative to the earliest date (Day 1)\n    print(\"Calculating days since the earliest date for each customer_ID...\")\n    min_dates = df.groupby('customer_ID')[date_column].transform('min')\n    date_features['days_since_start'] = (df[date_column] - min_dates).dt.days + 1  # +1 makes the earliest date Day 1\n\n    # Step 4: Combine extracted features back into the original DataFrame\n    df = pd.concat([df, date_features], axis=1)\n\n    print(\"Date feature processing completed successfully!\")\n    return df\n\n# Call the function to process date features\ntrain_data_cleaned = process_date_features(train_data_cleaned, date_column='S_2')\n\n# Check the updated data\nprint(\"First 5 rows with date features:\")\nprint(train_data_cleaned.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:47.464434Z","iopub.execute_input":"2024-12-17T16:40:47.464993Z","iopub.status.idle":"2024-12-17T16:40:50.531165Z","shell.execute_reply.started":"2024-12-17T16:40:47.464940Z","shell.execute_reply":"2024-12-17T16:40:50.529922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_to_categorical(df, columns):\n    \"\"\"\n    Convert specified columns to categorical data type.\n    \"\"\"\n    for col in columns:\n        if col in df.columns:\n            df[col] = df[col].astype('category')\n            print(f\"Column '{col}' converted to categorical type.\")\n        else:\n            print(f\"Warning: Column '{col}' not found in the DataFrame.\")\n    return df\n\n# Convert year and month to categorical\ncategorical_columns = ['year', 'month']\ntrain_data_cleaned = convert_to_categorical(train_data_cleaned, categorical_columns)\n\n# Check the updated data types\nprint(\"\\nUpdated data types:\")\nprint(train_data_cleaned.dtypes[categorical_columns])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:50.533122Z","iopub.execute_input":"2024-12-17T16:40:50.533590Z","iopub.status.idle":"2024-12-17T16:40:50.571328Z","shell.execute_reply.started":"2024-12-17T16:40:50.533543Z","shell.execute_reply":"2024-12-17T16:40:50.569907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_column_info(df):\n    \"\"\"\n    Print all column names and their data types.\n    \"\"\"\n    print(\"Column Names and Data Types:\")\n    print(\"=\" * 60)\n    print(df.dtypes.to_string())  # Directly print all column names with dtypes\n    print(\"=\" * 60)\n\n# Call the function to display column info\nprint_column_info(train_data_cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:50.573122Z","iopub.execute_input":"2024-12-17T16:40:50.573677Z","iopub.status.idle":"2024-12-17T16:40:50.586179Z","shell.execute_reply.started":"2024-12-17T16:40:50.573539Z","shell.execute_reply":"2024-12-17T16:40:50.584518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of categorical features to be converted\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', \n                        'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ndef convert_to_categorical(df, columns):\n    \"\"\"\n    Convert specified columns to categorical data type.\n    \"\"\"\n    for col in columns:\n        if col in df.columns:\n            df[col] = df[col].astype('category')\n            print(f\"Column '{col}' converted to categorical.\")\n        else:\n            print(f\"Warning: Column '{col}' not found in the DataFrame.\")\n    return df\n\n# Convert the specified columns to categorical\ntrain_data_cleaned = convert_to_categorical(train_data_cleaned, categorical_features)\n\n# Verify the data types\nprint(\"\\nUpdated data types for categorical columns:\")\nprint(train_data_cleaned[categorical_features].dtypes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:50.588081Z","iopub.execute_input":"2024-12-17T16:40:50.588514Z","iopub.status.idle":"2024-12-17T16:40:50.888868Z","shell.execute_reply.started":"2024-12-17T16:40:50.588468Z","shell.execute_reply":"2024-12-17T16:40:50.887673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_duplicate_columns(df):\n    \"\"\"\n    Identify and remove duplicate columns in a DataFrame.\n    \"\"\"\n    print(\"Checking for duplicate columns...\")\n    \n    # Identify duplicate columns\n    duplicate_columns = []\n    for i in range(len(df.columns)):\n        for j in range(i + 1, len(df.columns)):\n            if df.iloc[:, i].equals(df.iloc[:, j]):\n                duplicate_columns.append(df.columns[j])\n\n    # Drop duplicate columns\n    if duplicate_columns:\n        print(f\"Duplicate columns found: {duplicate_columns}\")\n        df = df.drop(columns=duplicate_columns)\n        print(\"Duplicate columns removed successfully.\")\n    else:\n        print(\"No duplicate columns found.\")\n    \n    return df\n\n# Call the function to remove duplicate columns\ntrain_data_cleaned = remove_duplicate_columns(train_data_cleaned)\n\n# Verify the result\nprint(\"\\nUpdated DataFrame shape:\", train_data_cleaned.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:40:50.890492Z","iopub.execute_input":"2024-12-17T16:40:50.891248Z","iopub.status.idle":"2024-12-17T16:42:00.836487Z","shell.execute_reply.started":"2024-12-17T16:40:50.891212Z","shell.execute_reply":"2024-12-17T16:42:00.835215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_column_info(df):\n    \"\"\"\n    Print all column names and their data types.\n    \"\"\"\n    print(\"Column Names and Data Types:\")\n    print(\"=\" * 60)\n    print(df.dtypes.to_string())  # Directly print all column names with dtypes\n    print(\"=\" * 60)\n\n# Call the function to display column info\nprint_column_info(train_data_cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:42:00.838241Z","iopub.execute_input":"2024-12-17T16:42:00.838666Z","iopub.status.idle":"2024-12-17T16:42:00.850003Z","shell.execute_reply.started":"2024-12-17T16:42:00.838588Z","shell.execute_reply":"2024-12-17T16:42:00.848524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_cleaned.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:42:00.852004Z","iopub.execute_input":"2024-12-17T16:42:00.852477Z","iopub.status.idle":"2024-12-17T16:42:00.889432Z","shell.execute_reply.started":"2024-12-17T16:42:00.852429Z","shell.execute_reply":"2024-12-17T16:42:00.888085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_negative_columns_and_ranges(df):\n    \"\"\"\n    Identify columns with negative values and compute range (min, max) for all numeric columns.\n    \"\"\"\n    negative_columns = {}\n    column_ranges = {}\n\n    # Loop through all numeric columns\n    for col in df.select_dtypes(include=['number']).columns:\n        col_min = df[col].min()\n        col_max = df[col].max()\n        column_ranges[col] = (col_min, col_max)  # Store column range\n\n        # Check if the column has negative values\n        if col_min < 0:\n            negative_columns[col] = col_min\n\n    # Print results\n    print(\"\\nColumns with Negative Values:\")\n    if negative_columns:\n        for col, min_value in negative_columns.items():\n            print(f\"Column: {col}, Minimum Value: {min_value}\")\n    else:\n        print(\"No columns with negative values found.\")\n\n    print(\"\\nRange of All Numeric Columns:\")\n    for col, (min_val, max_val) in column_ranges.items():\n        print(f\"{col:30} -> Min: {min_val:10}, Max: {max_val:10}\")\n    \n    return negative_columns, column_ranges\n\n# Call the function\nnegative_columns, column_ranges = find_negative_columns_and_ranges(train_data_cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:42:00.894194Z","iopub.execute_input":"2024-12-17T16:42:00.894542Z","iopub.status.idle":"2024-12-17T16:42:02.341130Z","shell.execute_reply.started":"2024-12-17T16:42:00.894512Z","shell.execute_reply":"2024-12-17T16:42:02.339507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save to Parquet format\nprocessed_file_path = 'train_data_cleaned.parquet'\n\nprint(\"Saving the cleaned data in Parquet format...\")\ntrain_data_cleaned.to_parquet(processed_file_path, index=False)\nprint(f\"Data saved to {processed_file_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:50:34.009245Z","iopub.execute_input":"2024-12-17T16:50:34.009724Z","iopub.status.idle":"2024-12-17T16:50:45.894378Z","shell.execute_reply.started":"2024-12-17T16:50:34.009686Z","shell.execute_reply":"2024-12-17T16:50:45.892890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n\n# Create a download link\nFileLink(processed_file_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:50:58.174096Z","iopub.execute_input":"2024-12-17T16:50:58.174741Z","iopub.status.idle":"2024-12-17T16:50:58.184181Z","shell.execute_reply.started":"2024-12-17T16:50:58.174679Z","shell.execute_reply":"2024-12-17T16:50:58.182737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Save to Parquet format\nprocessed_file_path = 'train_data_cleaned.parquet'\n\nprint(\"Saving the cleaned data in Parquet format...\")\ntrain_data_cleaned.to_parquet(processed_file_path, index=False)\nprint(f\"Data saved to {processed_file_path}\")\n\n# Get the file size\nfile_size = os.path.getsize(processed_file_path) / (1024 * 1024)  # Convert bytes to MB\nprint(f\"File size: {file_size:.2f} MB\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:52:11.958929Z","iopub.execute_input":"2024-12-17T16:52:11.959410Z","iopub.status.idle":"2024-12-17T16:52:22.626294Z","shell.execute_reply.started":"2024-12-17T16:52:11.959368Z","shell.execute_reply":"2024-12-17T16:52:22.624851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install Kaggle CLI (if not already installed)\n!pip install kaggle\n\n# Create a metadata file for the dataset\n!mkdir -p kaggle_dataset\n!echo '{\"title\": \"train_data_cleaned\", \"id\": \"charansv/train-data-cleaned\", \"licenses\": [{\"name\": \"CC0-1.0\"}]}' > kaggle_dataset/dataset-metadata.json\n\n# Move the parquet file to the dataset folder\n!mv train_data_cleaned.parquet kaggle_dataset/\n\n# Create a new private dataset on Kaggle\n!kaggle datasets create -p kaggle_dataset --private\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:58:06.271548Z","iopub.execute_input":"2024-12-17T16:58:06.272801Z","iopub.status.idle":"2024-12-17T16:58:21.941036Z","shell.execute_reply.started":"2024-12-17T16:58:06.272751Z","shell.execute_reply":"2024-12-17T16:58:21.939207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:11:33.759038Z","iopub.execute_input":"2024-12-17T17:11:33.759529Z","iopub.status.idle":"2024-12-17T17:11:35.006333Z","shell.execute_reply.started":"2024-12-17T17:11:33.759493Z","shell.execute_reply":"2024-12-17T17:11:35.004664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /root/.kaggle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:01:55.347865Z","iopub.execute_input":"2024-12-17T17:01:55.348403Z","iopub.status.idle":"2024-12-17T17:01:56.623120Z","shell.execute_reply.started":"2024-12-17T17:01:55.348360Z","shell.execute_reply":"2024-12-17T17:01:56.621318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp /kaggle/input/api-key/kaggle.json /root/.kaggle/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:06:29.791875Z","iopub.execute_input":"2024-12-17T17:06:29.792422Z","iopub.status.idle":"2024-12-17T17:06:31.040488Z","shell.execute_reply.started":"2024-12-17T17:06:29.792380Z","shell.execute_reply":"2024-12-17T17:06:31.038913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!chmod 600 /root/.kaggle/kaggle.json","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:06:46.464370Z","iopub.execute_input":"2024-12-17T17:06:46.464845Z","iopub.status.idle":"2024-12-17T17:06:47.704031Z","shell.execute_reply.started":"2024-12-17T17:06:46.464800Z","shell.execute_reply":"2024-12-17T17:06:47.702466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p kaggle_dataset\n!echo '{\"title\": \"train_data_cleaned\", \"id\": \"charansv/train-data-cleaned\", \"licenses\": [{\"name\": \"CC0-1.0\"}]}' > kaggle_dataset/dataset-metadata.json","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:09:47.249806Z","iopub.execute_input":"2024-12-17T17:09:47.251984Z","iopub.status.idle":"2024-12-17T17:09:49.708709Z","shell.execute_reply.started":"2024-12-17T17:09:47.251901Z","shell.execute_reply":"2024-12-17T17:09:49.706831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mv train_data_cleaned.parquet kaggle_dataset/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:10:05.613102Z","iopub.execute_input":"2024-12-17T17:10:05.613789Z","iopub.status.idle":"2024-12-17T17:10:06.871421Z","shell.execute_reply.started":"2024-12-17T17:10:05.613737Z","shell.execute_reply":"2024-12-17T17:10:06.869761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Search for the file in the entire instance\n!find / -name \"train_data_cleaned.*\" 2>/dev/null\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:12:51.374833Z","iopub.execute_input":"2024-12-17T17:12:51.375352Z","iopub.status.idle":"2024-12-17T17:13:49.647853Z","shell.execute_reply.started":"2024-12-17T17:12:51.375313Z","shell.execute_reply":"2024-12-17T17:13:49.646220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle datasets create -p kaggle_dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:24:24.603126Z","iopub.execute_input":"2024-12-17T17:24:24.603580Z","iopub.status.idle":"2024-12-17T17:24:41.248639Z","shell.execute_reply.started":"2024-12-17T17:24:24.603540Z","shell.execute_reply":"2024-12-17T17:24:41.247279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T11:54:56.108888Z","iopub.status.idle":"2024-12-25T11:54:56.109290Z","shell.execute_reply.started":"2024-12-25T11:54:56.109109Z","shell.execute_reply":"2024-12-25T11:54:56.109129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}