{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-08T06:59:44.103411Z","iopub.execute_input":"2022-11-08T06:59:44.103786Z","iopub.status.idle":"2022-11-08T06:59:44.129762Z","shell.execute_reply.started":"2022-11-08T06:59:44.103758Z","shell.execute_reply":"2022-11-08T06:59:44.129112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_list = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == \"test.csv\":\n            continue\n        csv_list.append(pd.read_csv(os.path.join(dirname, filename)))\ndf = pd.concat(csv_list)\n# print(\"Concat complete\")","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:59:47.457645Z","iopub.execute_input":"2022-11-08T06:59:47.457980Z","iopub.status.idle":"2022-11-08T07:00:59.612151Z","shell.execute_reply.started":"2022-11-08T06:59:47.457953Z","shell.execute_reply":"2022-11-08T07:00:59.611299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nimport json\n\nnan_filler_text = \"newtext\"\nthreshold = 0.8\ncolumn_threshold = 0.8\n# skip_files = [\"disk_capacity\", \"Unnamed: 0\", \"__null_dask_index__\", \"machine_id\", \"malware_status\", \"oem_model_id\",\n#               \"av_sig_version\", \"firmware_version_id\", \"system_volume_capacity\"]\n\n\ndf = df.fillna(nan_filler_text)\n\nskip_files = []\nfor column in df:\n    if len(pd.unique(df[column])) > column_threshold* (len(df[column])):\n        skip_files.append(column)\n    if column == \"malware_status\":\n        skip_files.append(column)\nprint(f\"skipping {skip_files}\")","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:01:33.613464Z","iopub.execute_input":"2022-11-08T07:01:33.613775Z","iopub.status.idle":"2022-11-08T07:02:22.440771Z","shell.execute_reply.started":"2022-11-08T07:01:33.613752Z","shell.execute_reply":"2022-11-08T07:02:22.439071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Memory saving function credit to https://www.kaggle.com/gemartin/load-data-reduce-memory-usage\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    #start_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    #end_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    #print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\ndf = reduce_mem_usage(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:02:32.348674Z","iopub.execute_input":"2022-11-08T07:02:32.349030Z","iopub.status.idle":"2022-11-08T07:02:32.633721Z","shell.execute_reply.started":"2022-11-08T07:02:32.349004Z","shell.execute_reply":"2022-11-08T07:02:32.632802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# start of step one\ncorr_matrix = df.corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\nto_drop = [column for column in upper.columns if any(upper[column] > 0.8)]\nprint(f\"dropping the following : {to_drop}\")\ndf.drop(to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:02:36.662334Z","iopub.execute_input":"2022-11-08T07:02:36.662653Z","iopub.status.idle":"2022-11-08T07:02:54.794355Z","shell.execute_reply.started":"2022-11-08T07:02:36.662630Z","shell.execute_reply":"2022-11-08T07:02:54.793310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# start of step 2\ncolumn_headers = [i for i in df if i not in skip_files]\nfor column in column_headers:\n    if type(column) == np.int64 or type(column) == np.int8 or type(column) == np.int16 or type(column) == np.int32:\n        column = int(column)\n\ncondition_dict = {}\nfor column in column_headers:\n    occurrences = df.groupby([column]).size().to_dict()\n    condition_dict[column] = occurrences\n\nsmol_df = df.loc[df['malware_status'] == 1]\ncond_true_dict = {}\nfor column in column_headers:\n    occurrences = smol_df.groupby([column]).size().to_dict()\n    cond_true_dict[column] = occurrences\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:03:02.978507Z","iopub.execute_input":"2022-11-08T07:03:02.978831Z","iopub.status.idle":"2022-11-08T07:03:50.704699Z","shell.execute_reply.started":"2022-11-08T07:03:02.978806Z","shell.execute_reply":"2022-11-08T07:03:50.703589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# start of step 3\nweightage_dict = {}\nfor column in column_headers:\n    value_weights = {}\n    uniques = pd.unique(df[column])\n    for unique in uniques:\n        try:\n            value_intersection_malware = cond_true_dict[column][unique]\n            weightage = value_intersection_malware / condition_dict[column][unique]\n        except KeyError:\n            weightage = 0\n\n        if unique == nan_filler_text:\n            weightage = 0\n\n        if type(unique) == np.int64 or type(unique) == np.int8 or type(unique) == np.int16 or type(unique) == np.int32:\n            unique = int(unique)\n        value_weights[unique] = weightage\n    weightage_dict[column] = value_weights","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:03:57.923300Z","iopub.execute_input":"2022-11-08T07:03:57.923640Z","iopub.status.idle":"2022-11-08T07:04:20.004313Z","shell.execute_reply.started":"2022-11-08T07:03:57.923614Z","shell.execute_reply":"2022-11-08T07:04:20.003354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# step 4\nreliability_dict = {}\nfor column in column_headers:\n    occurrences = smol_df.groupby([column]).size().to_dict()\n    occurrences_total = df.groupby([column]).size().to_dict()\n    corrects = 0\n    total_total = 0\n    for unique in pd.unique(df[column]):\n        try:\n            times_in_smol = occurrences[unique]\n        except KeyError:\n            times_in_smol = 0\n\n        times_total = occurrences_total[unique]\n\n        if weightage_dict[column][unique] >= 0.5:\n            corrects += times_in_smol\n        else:\n            corrects += times_total - times_in_smol\n        total_total += times_total\n\n    reliability = corrects/total_total\n    reliability_dict[column] = reliability","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:04:22.867634Z","iopub.execute_input":"2022-11-08T07:04:22.868005Z","iopub.status.idle":"2022-11-08T07:05:29.409478Z","shell.execute_reply.started":"2022-11-08T07:04:22.867975Z","shell.execute_reply":"2022-11-08T07:05:29.408497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Storing values that we got in a json\ncombined_dict = {\"weightage_dict\": weightage_dict,\"reliability_dict\": reliability_dict}\nwith open(\"/kaggle/working/thing.json\", \"w\") as f:\n    json.dump(combined_dict, f)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:05:56.308953Z","iopub.execute_input":"2022-11-08T07:05:56.309322Z","iopub.status.idle":"2022-11-08T07:05:58.771491Z","shell.execute_reply.started":"2022-11-08T07:05:56.309293Z","shell.execute_reply":"2022-11-08T07:05:58.770552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import codecs\nimport csv\n\nfile_location = \"/kaggle/input/im-somewhat-of-a-cybersecurity-analyst-myself/test.csv/test.csv\"\n\ndf = pd.read_csv(file_location)\nnan_filler_text = \"newtext\"\ndf = df.fillna(nan_filler_text)\n\nwith codecs.open(\"thing.json\", \"r\", encoding='iso-8859-1') as json_file:\n    data = json.load(json_file)\nweight_data = data[\"weightage_dict\"]\n\ndef cls():\n    os.system('cls' if os.name=='nt' else 'clear')\n\nadi_counter = 0    \nrows = []\nfor ind in df.index:\n    total_weight = 0\n    params = 0\n    for column in column_headers:\n        thing = df[column][ind]\n        if thing == nan_filler_text:\n            continue\n        try:\n            weight = weight_data[column][thing]\n        except KeyError:\n            thing = str(thing)\n            try:\n                weight = weight_data[column][thing]\n            except KeyError:\n                continue\n        total_weight += weight\n\n        params += 1\n    probability = total_weight/params\n    guess = 0\n    if probability >= 0.5:\n        guess = 1\n    row = {\"machine_id\": df[\"machine_id\"][ind], \"malware_status\": guess}\n    rows.append(row)\n    adi_counter += 1\n    if adi_counter % 10000 == 0:\n        print(adi_counter,len(rows), row)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T07:34:52.180048Z","iopub.execute_input":"2022-11-08T07:34:52.180421Z","iopub.status.idle":"2022-11-08T08:16:11.937419Z","shell.execute_reply.started":"2022-11-08T07:34:52.180392Z","shell.execute_reply":"2022-11-08T08:16:11.936474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = pd.DataFrame(rows)\nfinal_df.to_csv(\"/kaggle/working/solution.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-08T08:16:36.457297Z","iopub.execute_input":"2022-11-08T08:16:36.457818Z","iopub.status.idle":"2022-11-08T08:16:45.287954Z","shell.execute_reply.started":"2022-11-08T08:16:36.457790Z","shell.execute_reply":"2022-11-08T08:16:45.287045Z"},"trusted":true},"execution_count":null,"outputs":[]}]}