{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/czii-cryo-et-object-identification'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:44:36.219743Z","iopub.execute_input":"2024-11-08T07:44:36.220506Z","iopub.status.idle":"2024-11-08T07:44:38.658975Z","shell.execute_reply.started":"2024-11-08T07:44:36.220460Z","shell.execute_reply":"2024-11-08T07:44:38.657863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nclass CryoETDataset:\n    def __init__(self, data_path):\n        self.data_path = data_path\n        self.files = self.load_data()  # Load and list all files in the directory\n\n    def load_data(self):\n        files = []\n        # Iterate over files in the specified directory\n        for file_name in os.listdir(self.data_path):\n            file_path = os.path.join(self.data_path, file_name)\n            # Check if it is a file (not a directory)\n            if os.path.isfile(file_path):\n                files.append(file_path)\n                print(f\"Loaded file: {file_path}\")\n        return files\n\n# Instantiate the dataset\ndataset = CryoETDataset('/kaggle/input/czii-cryo-et-object-identification')\nprint(\"Files loaded:\", dataset.files)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:44:53.858536Z","iopub.execute_input":"2024-11-08T07:44:53.858976Z","iopub.status.idle":"2024-11-08T07:44:53.867390Z","shell.execute_reply.started":"2024-11-08T07:44:53.858935Z","shell.execute_reply":"2024-11-08T07:44:53.866377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# List all files in the train directory\ntrain_dir = '/kaggle/input/czii-cryo-et-object-identification/train'\nfor filename in os.listdir(train_dir):\n    file_path = os.path.join(train_dir, filename)\n    print(file_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:44:58.917693Z","iopub.execute_input":"2024-11-08T07:44:58.918153Z","iopub.status.idle":"2024-11-08T07:44:58.925175Z","shell.execute_reply.started":"2024-11-08T07:44:58.918113Z","shell.execute_reply":"2024-11-08T07:44:58.924101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Function to generate the submission file\ndef generate_submission(predictions, output_filename='submission.csv'):\n    \"\"\"\n    Generate the submission file based on the model's predictions.\n    \n    Parameters:\n    predictions (list): A list of dictionaries containing the prediction results.\n    output_filename (str): Name of the output CSV file.\n    \n    Returns:\n    None\n    \"\"\"\n    # Convert the list of predictions into a DataFrame\n    df = pd.DataFrame(predictions)\n\n    # Add an 'id' column with unique values\n    df['id'] = range(len(df))\n\n    # Reorder columns to match the required format\n    df = df[['id', 'experiment', 'particle_type', 'x', 'y', 'z']]\n\n    # Save the predictions to a CSV file\n    df.to_csv(output_filename, index=False)\n\n    print(f\"Submission file '{output_filename}' created successfully!\")\n\n\n# Example list of model predictions (for illustration purposes)\npredictions = [\n    {'experiment': 'TS_5_4', 'particle_type': 'beta-amylase', 'x': 2983.596, 'y': 3154.13, 'z': 764.124},\n    {'experiment': 'TS_5_4', 'particle_type': 'beta-amylase', 'x': 2983.596, 'y': 3154.13, 'z': 764.124},\n    {'experiment': 'TS_5_4', 'particle_type': 'beta-galactosidase', 'x': 2983.596, 'y': 3154.13, 'z': 764.124},\n    {'experiment': 'TS_5_4', 'particle_type': 'thyroglobulin', 'x': 3201.124, 'y': 3210.56, 'z': 780.21},\n    {'experiment': 'TS_5_4', 'particle_type': 'virus-like particles', 'x': 3102.8, 'y': 3001.1, 'z': 799.2},\n    {'experiment': 'TS_5_4', 'particle_type': 'apo-ferritin', 'x': 2995.5, 'y': 3150.3, 'z': 760.1}\n]\n\n# Generate the submission file\ngenerate_submission(predictions, 'submission.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:45:03.514531Z","iopub.execute_input":"2024-11-08T07:45:03.515319Z","iopub.status.idle":"2024-11-08T07:45:03.544987Z","shell.execute_reply.started":"2024-11-08T07:45:03.515274Z","shell.execute_reply":"2024-11-08T07:45:03.543849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nimport seaborn as sns\n\n# Function to generate the submission file\ndef generate_submission(predictions, output_filename='submission.csv'):\n    \"\"\"\n    Generate the submission file based on the model's predictions.\n    \"\"\"\n    df = pd.DataFrame(predictions)\n    df['id'] = range(len(df))\n    df = df[['id', 'experiment', 'particle_type', 'x', 'y', 'z']]\n    df.to_csv(output_filename, index=False)\n    print(f\"Submission file '{output_filename}' created successfully!\")\n\n# Example list of model predictions (for illustration purposes)\npredictions = [\n    {'experiment': 'TS_5_4', 'particle_type': 'beta-amylase', 'x': 2983.596, 'y': 3154.13, 'z': 764.124},\n    {'experiment': 'TS_5_4', 'particle_type': 'beta-galactosidase', 'x': 2983.596, 'y': 3154.13, 'z': 764.124},\n    {'experiment': 'TS_5_4', 'particle_type': 'thyroglobulin', 'x': 3201.124, 'y': 3210.56, 'z': 780.21},\n    {'experiment': 'TS_5_4', 'particle_type': 'virus-like particles', 'x': 3102.8, 'y': 3001.1, 'z': 799.2},\n    {'experiment': 'TS_5_4', 'particle_type': 'apo-ferritin', 'x': 2995.5, 'y': 3150.3, 'z': 760.1}\n]\n\n# Generate the submission file\ngenerate_submission(predictions, 'submission.csv')\n\n# Convert predictions to DataFrame for visualization\ndf = pd.DataFrame(predictions)\n\n# 3D Scatter Plot for Particle Positions\ndef plot_3d_particles(data):\n    fig = plt.figure(figsize=(10, 7))\n    ax = fig.add_subplot(111, projection='3d')\n    for particle_type in data['particle_type'].unique():\n        subset = data[data['particle_type'] == particle_type]\n        ax.scatter(subset['x'], subset['y'], subset['z'], label=particle_type, s=50)\n    ax.set_xlabel('X Coordinate')\n    ax.set_ylabel('Y Coordinate')\n    ax.set_zlabel('Z Coordinate')\n    plt.title('3D Scatter Plot of Protein Particles')\n    plt.legend()\n    plt.show()\n\n# 2D Projections for Particle Positions\ndef plot_2d_projections(data):\n    fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n    sns.scatterplot(data=data, x='x', y='y', hue='particle_type', ax=axes[0], s=50)\n    axes[0].set_title('XY Projection')\n    sns.scatterplot(data=data, x='x', y='z', hue='particle_type', ax=axes[1], s=50)\n    axes[1].set_title('XZ Projection')\n    sns.scatterplot(data=data, x='y', y='z', hue='particle_type', ax=axes[2], s=50)\n    axes[2].set_title('YZ Projection')\n    plt.suptitle('2D Projections of Protein Particles')\n    plt.show()\n\n# Visualize the 3D and 2D plots\nplot_3d_particles(df)\nplot_2d_projections(df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:45:08.996779Z","iopub.execute_input":"2024-11-08T07:45:08.997171Z","iopub.status.idle":"2024-11-08T07:45:12.002559Z","shell.execute_reply.started":"2024-11-08T07:45:08.997135Z","shell.execute_reply":"2024-11-08T07:45:12.001310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models\n\n# Sample data (replace this with your actual data loading method)\ndata = pd.DataFrame(predictions)  # Adjust to your data loading method\n\n# Data Preprocessing\nscaler = StandardScaler()\ndata[['x', 'y', 'z']] = scaler.fit_transform(data[['x', 'y', 'z']])\n\nle = LabelEncoder()\ndata['particle_type_encoded'] = le.fit_transform(data['particle_type'])\n\nX = data[['x', 'y', 'z', 'particle_type_encoded']]\ny = data['particle_type_encoded']\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define the model\ndef build_dense_model(input_shape):\n    model = models.Sequential([\n        layers.Input(shape=input_shape),\n        layers.Dense(64, activation='relu'),\n        layers.Dense(32, activation='relu'),\n        layers.Dense(16, activation='relu'),\n        layers.Dense(len(le.classes_), activation='softmax')\n    ])\n    model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    return model\n\ninput_shape = (X_train.shape[1],)\nmodel = build_dense_model(input_shape)\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_split=0.1)\n\n# Generate predictions\ndef make_predictions(model, X_data):\n    preds = model.predict(X_data)\n    preds_labels = le.inverse_transform(np.argmax(preds, axis=1))\n    return preds_labels\n\n# Predictions for test set\ntest_predictions = make_predictions(model, X_test)\n\n# Create submission DataFrame\nsubmission_data = X_test.copy()\nsubmission_data['particle_type'] = test_predictions\nsubmission_data[['x', 'y', 'z']] = scaler.inverse_transform(submission_data[['x', 'y', 'z']])\n\n# Add missing columns\nsubmission_data.reset_index(inplace=True)\nsubmission_data.rename(columns={'index': 'id'}, inplace=True)\n\n# Check if 'experiment' column exists; if not, add a default value\nif 'experiment' not in submission_data.columns:\n    submission_data['experiment'] = 'default_experiment'\n\n# Finalize and save submission\nsubmission = submission_data[['id', 'experiment', 'particle_type', 'x', 'y', 'z']]\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:45:22.292971Z","iopub.execute_input":"2024-11-08T07:45:22.293368Z","iopub.status.idle":"2024-11-08T07:45:38.279541Z","shell.execute_reply.started":"2024-11-08T07:45:22.293331Z","shell.execute_reply":"2024-11-08T07:45:38.278412Z"}},"outputs":[],"execution_count":null}]}