{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8023315,"sourceType":"datasetVersion","datasetId":4728096}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview","metadata":{}},{"cell_type":"markdown","source":"From the data definition, we can see that there are multiple column names that may describe sentences, such as \"Reason for rejection on the most recent rejected application.\"  \nThese String data could be converted using Count Encoding or other methods, but if there are too many types of data, they will be rejected by filtering, and memory will be overflowed to begin with. (I am sure you are also having trouble with Out of Memory.)  \nThe goal of this notebook is to get to the point where String data can be used as a 1D Vector as a feature using Word2Vec, one of the methods of Natural Language Processing.  \nI am sure there are many improvements to be made, so please feel free to comment.  \nThanks for reading.  \n","metadata":{}},{"cell_type":"markdown","source":"# What is BERT ? \nThe BERT model was proposed in BERT: Pre-training of Deep Bidirectional Transformers for Language   Understanding by Jacob Devlin, Ming-Wei Chang, Kenton Lee and Kristina Toutanova. It’s a   bidirectional transformer pretrained using a combination of masked language modeling objective and   next sentence prediction on a large corpus comprising the Toronto Book Corpus and Wikipedia.  \n\nhttps://huggingface.co/docs/transformers/v4.39.3/en/model_doc/bert","metadata":{}},{"cell_type":"code","source":"!pip install -q transformers\n!pip install -q silence_tensorflow","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import packages","metadata":{}},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport torch\nimport transformers\nfrom transformers import BertTokenizer\nfrom tqdm import tqdm\nfrom sklearn.decomposition import TruncatedSVD\n\n# The logs of tensorflow are too lot....\n# To filter them importing silence_tensorflow.\nfrom silence_tensorflow import silence_tensorflow\nsilence_tensorflow()\n\n# tensorflow\nimport tensorflow as tf\nimport tensorflow.keras.layers as kl\n\n# transformers\nimport transformers\n\n# Output log is error level\nfrom transformers import logging\nlogging.set_verbosity_error()\n\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:57:55.344332Z","iopub.execute_input":"2024-04-07T07:57:55.344745Z","iopub.status.idle":"2024-04-07T07:58:20.660072Z","shell.execute_reply.started":"2024-04-07T07:57:55.344701Z","shell.execute_reply":"2024-04-07T07:58:20.658784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.7:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.661769Z","iopub.execute_input":"2024-04-07T07:58:20.664090Z","iopub.status.idle":"2024-04-07T07:58:20.680683Z","shell.execute_reply.started":"2024-04-07T07:58:20.664043Z","shell.execute_reply":"2024-04-07T07:58:20.679428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass Aggregator:\n    #Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n\n        return expr_max +expr_last+expr_mean+expr_var\n    \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return  expr_max +expr_last+expr_mean\n    \n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        #expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        return  expr_last#+expr_count\n    \n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max +expr_last\n    \n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] \n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max +expr_last\n    \n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.683578Z","iopub.execute_input":"2024-04-07T07:58:20.683956Z","iopub.status.idle":"2024-04-07T07:58:20.717832Z","shell.execute_reply.started":"2024-04-07T07:58:20.683928Z","shell.execute_reply":"2024-04-07T07:58:20.716575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.str_expr(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.str_expr(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.719834Z","iopub.execute_input":"2024-04-07T07:58:20.720256Z","iopub.status.idle":"2024-04-07T07:58:20.734716Z","shell.execute_reply.started":"2024-04-07T07:58:20.720224Z","shell.execute_reply":"2024-04-07T07:58:20.733382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.736359Z","iopub.execute_input":"2024-04-07T07:58:20.737449Z","iopub.status.idle":"2024-04-07T07:58:20.747698Z","shell.execute_reply.started":"2024-04-07T07:58:20.737413Z","shell.execute_reply":"2024-04-07T07:58:20.746317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.749636Z","iopub.execute_input":"2024-04-07T07:58:20.750026Z","iopub.status.idle":"2024-04-07T07:58:20.759830Z","shell.execute_reply.started":"2024-04-07T07:58:20.749994Z","shell.execute_reply":"2024-04-07T07:58:20.758650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.761460Z","iopub.execute_input":"2024-04-07T07:58:20.761949Z","iopub.status.idle":"2024-04-07T07:58:20.780162Z","shell.execute_reply.started":"2024-04-07T07:58:20.761908Z","shell.execute_reply":"2024-04-07T07:58:20.778908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BertSequenceVectorizer:\n    def __init__(self):\n        self.device = 'cuda' if torch.cuda.is_available() else 'cpu'\n        self.model_name = 'bert-base-uncased'\n        self.tokenizer = BertTokenizer.from_pretrained(self.model_name)\n        self.bert_model = transformers.BertModel.from_pretrained(self.model_name)\n        self.bert_model = self.bert_model.to(self.device)\n        self.max_len = 128\n\n\n    def vectorize(self, sentence : str) -> np.array:\n        inp = self.tokenizer.encode(sentence)\n        len_inp = len(inp)\n\n        if len_inp >= self.max_len:\n            inputs = inp[:self.max_len]\n            masks = [1] * self.max_len\n        else:\n            inputs = inp + [0] * (self.max_len - len_inp)\n            masks = [1] * len_inp + [0] * (self.max_len - len_inp)\n\n        inputs_tensor = torch.tensor([inputs], dtype=torch.long).to(self.device)\n        masks_tensor = torch.tensor([masks], dtype=torch.long).to(self.device)\n\n        bert_out = self.bert_model(inputs_tensor, masks_tensor)\n        seq_out, pooled_out = bert_out['last_hidden_state'], bert_out['pooler_output']\n\n        if torch.cuda.is_available():    \n            return seq_out[0][0].cpu().detach().numpy() # 0番目は [CLS] token, 768 dim の文章特徴量\n        else:\n            return seq_out[0][0].detach().numpy()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.781513Z","iopub.execute_input":"2024-04-07T07:58:20.782351Z","iopub.status.idle":"2024-04-07T07:58:20.794374Z","shell.execute_reply.started":"2024-04-07T07:58:20.782317Z","shell.execute_reply":"2024-04-07T07:58:20.793069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Loading Data","metadata":{}},{"cell_type":"code","source":"# ROOT            = Path(\"home-credit-credit-risk-model-stability\")\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.797789Z","iopub.execute_input":"2024-04-07T07:58:20.798761Z","iopub.status.idle":"2024-04-07T07:58:20.812502Z","shell.execute_reply.started":"2024-04-07T07:58:20.798724Z","shell.execute_reply":"2024-04-07T07:58:20.811508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-07T07:58:20.813800Z","iopub.execute_input":"2024-04-07T07:58:20.814389Z","iopub.status.idle":"2024-04-07T08:01:09.756849Z","shell.execute_reply.started":"2024-04-07T07:58:20.814359Z","shell.execute_reply":"2024-04-07T08:01:09.753990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ndf_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)\ndel data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:01:09.760551Z","iopub.execute_input":"2024-04-07T08:01:09.761117Z","iopub.status.idle":"2024-04-07T08:01:19.961590Z","shell.execute_reply.started":"2024-04-07T08:01:09.761075Z","shell.execute_reply":"2024-04-07T08:01:19.960200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stringDF = df_train.select(pl.col(pl.String))\ndel df_train\ndisplay(stringDF.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:01:19.962902Z","iopub.execute_input":"2024-04-07T08:01:19.963649Z","iopub.status.idle":"2024-04-07T08:01:20.217988Z","shell.execute_reply.started":"2024-04-07T08:01:19.963604Z","shell.execute_reply":"2024-04-07T08:01:20.216651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Pretreating String data\n\nSince BERT is implemented including preprocessing, there is basically no need for preprocessing.  \nHowever, in this case, we were only concerned about the symbols, so we excluded them in advance as extra information.","metadata":{}},{"cell_type":"code","source":"str_cols = stringDF.columns\nprint(len(str_cols))\n\nfor col in str_cols:\n    stringDF= stringDF.with_columns(pl.col(col).str.replace(\"[^a-zA-Z0-9]+\", \"\"))","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:24:44.447660Z","iopub.execute_input":"2024-04-07T08:24:44.448156Z","iopub.status.idle":"2024-04-07T08:24:58.890544Z","shell.execute_reply.started":"2024-04-07T08:24:44.448121Z","shell.execute_reply":"2024-04-07T08:24:58.889324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"and fill null value as \"Missing\"","metadata":{}},{"cell_type":"code","source":"stringDF = stringDF.fill_null(\"Missing\")\nstringDF.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:01:30.833215Z","iopub.execute_input":"2024-04-07T08:01:30.833753Z","iopub.status.idle":"2024-04-07T08:01:32.427456Z","shell.execute_reply.started":"2024-04-07T08:01:30.833710Z","shell.execute_reply":"2024-04-07T08:01:32.426225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Converting String to Vector using BERT","metadata":{}},{"cell_type":"code","source":"BSV = BertSequenceVectorizer() # instance","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:01:32.429070Z","iopub.execute_input":"2024-04-07T08:01:32.429508Z","iopub.status.idle":"2024-04-07T08:01:38.128820Z","shell.execute_reply.started":"2024-04-07T08:01:32.429476Z","shell.execute_reply":"2024-04-07T08:01:38.127361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We use 'lastrejectreasonclient_4145040M' as sample column.\nData in this notebook has 75 columns.\nSo Converting all String columns into Vector costs about 5 minutes **minimum**","metadata":{}},{"cell_type":"code","source":"print(stringDF['lastrejectreasonclient_4145040M'].len())","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:01:38.131290Z","iopub.execute_input":"2024-04-07T08:01:38.131868Z","iopub.status.idle":"2024-04-07T08:01:38.142028Z","shell.execute_reply.started":"2024-04-07T08:01:38.131820Z","shell.execute_reply":"2024-04-07T08:01:38.140936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import trange\nsample_col = 'lastrejectreasonclient_4145040M'\nsampleString = stringDF[sample_col].unique()\nsample_num = len(sampleString)\n\nvectorizedString = np.ndarray((sample_num, 768))\nprint(sampleString)\n\nfor i in trange(sample_num):\n    vectorizedString[i] = BSV.vectorize(sampleString[i])\n\nvectorizedString.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:19:49.039250Z","iopub.execute_input":"2024-04-07T08:19:49.039799Z","iopub.status.idle":"2024-04-07T08:19:52.960535Z","shell.execute_reply.started":"2024-04-07T08:19:49.039764Z","shell.execute_reply":"2024-04-07T08:19:52.959349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"768 dimensions are too large.  \nTo reduce columns (and memory usage), use aggrigate function. (in this notebook mean)","metadata":{}},{"cell_type":"code","source":"vectorizedString = vectorizedString.mean(axis = 1)\nvectorizedString","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:20:21.526147Z","iopub.execute_input":"2024-04-07T08:20:21.526740Z","iopub.status.idle":"2024-04-07T08:20:21.535129Z","shell.execute_reply.started":"2024-04-07T08:20:21.526707Z","shell.execute_reply":"2024-04-07T08:20:21.533546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Treat as dict format to save as pickle or json file","metadata":{}},{"cell_type":"code","source":"sampleDF = \\\npd.DataFrame([sampleString, vectorizedString], index=[sample_col, sample_col + '_MeanVect']).T\nsampleDF","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:20:22.444572Z","iopub.execute_input":"2024-04-07T08:20:22.445067Z","iopub.status.idle":"2024-04-07T08:20:22.463761Z","shell.execute_reply.started":"2024-04-07T08:20:22.445028Z","shell.execute_reply":"2024-04-07T08:20:22.462020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ResultDict = {}\nResultDict[sample_col] = sampleDF","metadata":{"execution":{"iopub.status.busy":"2024-04-07T08:23:08.232983Z","iopub.execute_input":"2024-04-07T08:23:08.233553Z","iopub.status.idle":"2024-04-07T08:23:08.241447Z","shell.execute_reply.started":"2024-04-07T08:23:08.233521Z","shell.execute_reply":"2024-04-07T08:23:08.239269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}