{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7488883,"sourceType":"datasetVersion","datasetId":4354742}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"ab788790-c3d4-4c65-b635-2581c1be8ae2","_cell_guid":"2caaaa88-8e0f-4e7b-ba55-83f5d58b6341","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-27T00:10:34.748912Z","iopub.execute_input":"2024-01-27T00:10:34.749264Z","iopub.status.idle":"2024-01-27T00:10:35.021526Z","shell.execute_reply.started":"2024-01-27T00:10:34.749234Z","shell.execute_reply":"2024-01-27T00:10:35.019660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"########################################################################\n# Create required input setting-files \n########################################################################\nSETTINGS_json = {\"RAW_DATA_DIR\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \n \"TRAIN_DATA_CLEAN_PATH\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \"TEST_DATA_CLEAN_PATH\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \n \"MODEL_CHECKPOINT_DIR\": \"./models/\", \"LOGS_DIR\": \"./logs/\", \n \"SUBMISSION_DIR\": \"./submissions/\",\n \"VERBOSE\": \"100\"}\ndisplay(SETTINGS_json )\nimport json\njson_file_path = 'SETTINGS.json'\nwith open(json_file_path, 'w') as file:\n    # Write the dictionary to the JSON file\n    json.dump(SETTINGS_json, file, indent=0)\n\n!cp /kaggle/input/u900-open-problem-single-cell-perturbations/train_inference_pyboost_catboost.py train_inference_pyboost_catboost.py    ","metadata":{"execution":{"iopub.status.busy":"2024-01-27T00:10:35.023360Z","iopub.execute_input":"2024-01-27T00:10:35.023818Z","iopub.status.idle":"2024-01-27T00:10:35.352042Z","shell.execute_reply.started":"2024-01-27T00:10:35.023788Z","shell.execute_reply":"2024-01-27T00:10:35.350977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The main script it launches particular submission generation and blends them\n# Require 'SETTINGS.json' for paths descriptions \n\n##############################################################################\n# Imports. \n##############################################################################\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport json\nimport time\nimport pickle \nimport subprocess\nt0start = time.time() \n\njson_file_path = 'SETTINGS.json'\nwith open(json_file_path, 'r') as file:\n    dict_SETTINGS = json.load(file)\nverbose =  int(dict_SETTINGS.get('VERBOSE', 0 ) )\nif verbose > 0:\n    print('Verbose = ', verbose)\n\n    \n##############################################################################\n# Prepare and launch CatBoost Model. \n##############################################################################\n\nSETTINGS_MODEL_json = {\"str_model_id\":\"CatBoost\"} #  # \"Pyboost\", 'Ridge' \n# SETTINGS_MODEL_json = {\"str_model_id\":\"Pyboost\"} # \"CatBoost\"} #  # 'Pyboost', 'Ridge' \nprint(SETTINGS_MODEL_json)\njson_file_path = 'SETTINGS_MODEL.json'\nwith open(json_file_path, 'w') as file:\n    # Write the dictionary to the JSON file\n    json.dump(SETTINGS_MODEL_json, file, indent=0)\n    \ncommand = \"python train_inference_pyboost_catboost.py\"\nresult = subprocess.run(command, shell=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)\nif result.returncode == 0:\n    if verbose >= 1:\n        print(\"Command  \"+command +  \" executed successfully.\")\n        print(\"Output:\")\n        print(result.stdout)\nelse:\n    print(\"Error executing the command:  \"+command )\n    print(\"Error message:\")\n    print(result.stderr)\n\n\n##############################################################################\n# Prepare and launch PyBoost Model. \n##############################################################################\n\nSETTINGS_MODEL_json = {\"str_model_id\":\"Pyboost\"} #  # \"CatBoost\"  \"Pyboost\", 'Ridge' \n# SETTINGS_MODEL_json = {\"str_model_id\":\"Pyboost\"} # \"CatBoost\"} #  # 'Pyboost', 'Ridge' \nprint(SETTINGS_MODEL_json)\njson_file_path = 'SETTINGS_MODEL.json'\nwith open(json_file_path, 'w') as file:\n    # Write the dictionary to the JSON file\n    json.dump(SETTINGS_MODEL_json, file, indent=0)\n    \ncommand = \"python train_inference_pyboost_catboost.py\"\nresult = subprocess.run(command, shell=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)\nif result.returncode == 0:\n    if verbose >= 1:\n        print(\"Command  \"+command +  \" executed successfully.\")\n        print(\"Output:\")\n        print(result.stdout)\nelse:\n    print(\"Error executing the command:  \"+command )\n    print(\"Error message:\")\n    print(result.stderr)\n\n##############################################################################\n# Prepare and launch PyBoost Model. \n##############################################################################\n\n\ndirectory_path  =  dict_SETTINGS[\"SUBMISSION_DIR\"]\nl = os.listdir(directory_path)\nif verbose >= 1:\n    print(l)\nfor i,fn in enumerate(l):\n    fn =  os.path.join(directory_path, fn)\n    if verbose >= 1:\n        print(fn)\n    if i == 0:\n        df_sub = pd.read_csv(fn,index_col = 0)\n    else:\n        df_sub += pd.read_csv(fn,index_col = 0)\n        \ndf_sub /= i \nfn =  os.path.join(directory_path, 'submission_blend.csv' )\ndf_sub.to_csv(fn )\n\n##############################################################################\n# Timing. \n##############################################################################\n\n    \nif verbose >= 1:\n    print('%.1f seconds passed total '%(time.time()-t0start) )\n    print('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\n    print('%.2f hours passed total '%( (time.time()-t0start)/3600)  )    \n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-27T00:10:35.353178Z","iopub.execute_input":"2024-01-27T00:10:35.353570Z","iopub.status.idle":"2024-01-27T00:11:59.311974Z","shell.execute_reply.started":"2024-01-27T00:10:35.353543Z","shell.execute_reply":"2024-01-27T00:11:59.310830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nif 0:\n    # check average of Pyboost and Catboost gives score around 0.573-0.574 (public) 0.764 (private)\n    fn = '/kaggle/input/u900-open-problem-single-cell-perturbations/submissionNB1V8_Pyboost_max_depth10_ntrees5000_lr001_subsample1_colsample035_n_components50.csv'\n    df1 = pd.read_csv(fn , index_col = 0 )\n    fn = '/kaggle/input/u900-open-problem-single-cell-perturbations/submissionNB1V13_CatBoost_max_depth6_ntrees250_lr003_subsample1_colsample05_n_components30.csv'\n    df2 = pd.read_csv(fn , index_col = 0 )\n    df = (df1+df2)/2\n    df.to_csv('submit_average_Pyboost_and_Catboost.csv')\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-01-27T00:12:05.362678Z","iopub.execute_input":"2024-01-27T00:12:05.362947Z","iopub.status.idle":"2024-01-27T00:12:19.420457Z","shell.execute_reply.started":"2024-01-27T00:12:05.362924Z","shell.execute_reply":"2024-01-27T00:12:19.418894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}