{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Malicious Wallet Visualization Graph\n\nThis notebook visualizes transaction flows between nearly 1 million wallets with 80GB of data, and marks malicious wallets (i.e. hacks, phishings, etc.) as red.\n\nMade with ❤️ by Soptq","metadata":{}},{"cell_type":"markdown","source":"First we need to install some dependencies for later analysis.","metadata":{}},{"cell_type":"code","source":"!pip install tqdm networkx matplotlib\n!apt install graphviz -y","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-16T06:05:05.292573Z","iopub.execute_input":"2022-09-16T06:05:05.293005Z","iopub.status.idle":"2022-09-16T06:05:21.070695Z","shell.execute_reply.started":"2022-09-16T06:05:05.292966Z","shell.execute_reply":"2022-09-16T06:05:21.069016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nfrom tqdm import tqdm\nimport networkx as nx\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:05:25.996933Z","iopub.execute_input":"2022-09-16T06:05:25.997395Z","iopub.status.idle":"2022-09-16T06:05:26.238988Z","shell.execute_reply.started":"2022-09-16T06:05:25.997354Z","shell.execute_reply":"2022-09-16T06:05:26.237689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset dir\nbase_dir = '/kaggle/input/forta-protect-web3'\n\n# read wallet labels from dataset\n_labels = pd.read_csv(os.path.join(base_dir, 'train.csv'), usecols=['address', 'target'])\nlabels = {}\n\nfor index, row in tqdm(_labels.iterrows()):\n    labels[row['address']] = row['target'] == 1\n","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:05:33.639188Z","iopub.execute_input":"2022-09-16T06:05:33.640027Z","iopub.status.idle":"2022-09-16T06:06:19.167542Z","shell.execute_reply.started":"2022-09-16T06:05:33.639977Z","shell.execute_reply":"2022-09-16T06:06:19.166439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we need to process the transaction data to filter invalide transactions out. Also we need to group transactions with the same sender and reciver to calculate the frequency of the transcation.","metadata":{}},{"cell_type":"code","source":"eoa_tx = None\n\nfor dirname, _, filenames in os.walk(os.path.join(base_dir, 'eoa_tx_train')):\n    for filename in tqdm(filenames):\n        _eoa_tx = pd.read_csv(os.path.join(dirname, filename), usecols=['from_address', 'to_address', 'receipt_status'])\n        if eoa_tx is None:\n            eoa_tx = _eoa_tx.dropna()\n        else:\n            eoa_tx = pd.concat([eoa_tx, _eoa_tx.dropna()])\n        del _eoa_tx\n        \neoa_tx = eoa_tx[eoa_tx.receipt_status == 0.0]\ngroup_tx = eoa_tx.groupby(['from_address', 'to_address']).count().reset_index()\n\ngroup_tx","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:31:43.563011Z","iopub.execute_input":"2022-09-16T06:31:43.563454Z","iopub.status.idle":"2022-09-16T06:34:07.936859Z","shell.execute_reply.started":"2022-09-16T06:31:43.563420Z","shell.execute_reply":"2022-09-16T06:34:07.935704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting 1 million points on a graph is not very ideal as most wallets are useless for pattern discovery. Here we will do some filtering based on each wallet's transaction frequency and label.","metadata":{}},{"cell_type":"code","source":"filtered_indexes = []\n\nfor index, row in tqdm(group_tx.iterrows()):\n    from_address_in = row['from_address'] in labels and labels[row['from_address']]\n    to_address_in = row['to_address'] in labels and labels[row['to_address']]\n    if not from_address_in and not to_address_in and row['receipt_status'] < 10:\n        filtered_indexes.append(index)\n\nfiltered_group_tx = group_tx.drop(filtered_indexes, axis=0).reset_index()\n\nfiltered_group_tx","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:35:04.975712Z","iopub.execute_input":"2022-09-16T06:35:04.976401Z","iopub.status.idle":"2022-09-16T06:35:12.014246Z","shell.execute_reply.started":"2022-09-16T06:35:04.976353Z","shell.execute_reply":"2022-09-16T06:35:12.013103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, we will convert the graph data into `dot` file, which can be latter plotted using `graphviz`.","metadata":{}},{"cell_type":"code","source":"G = nx.DiGraph()\nedges = []\nfor index, row in tqdm(filtered_group_tx.iterrows()):\n    edges.append(tuple([row['from_address'], row['to_address'], int(row['receipt_status'])]))\n\nadded_nodes = {}\nfor (u, v, w) in tqdm(edges):\n    if u not in added_nodes:\n        added_nodes[u] = True\n        fillcolor = \"red\" if (u in labels and labels[u]) else \"yellow\"\n        G.add_node(u, label=f\"{u[:6]}...{u[-4:]}\", fillcolor=fillcolor, style=\"filled\")\n    \n    if v not in added_nodes:\n        added_nodes[v] = True\n        fillcolor = \"red\" if (v in labels and labels[v]) else \"yellow\"\n        G.add_node(v, label=f\"{v[:6]}...{v[-4:]}\", fillcolor=fillcolor, style=\"filled\")\n    \nfor (u, v, w) in tqdm(edges):\n    G.add_edge(u, v, weight=w)\n\nnx.nx_pydot.write_dot(G, '/kaggle/working/graph.dot')","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:36:06.599963Z","iopub.execute_input":"2022-09-16T06:36:06.600437Z","iopub.status.idle":"2022-09-16T06:36:11.505190Z","shell.execute_reply.started":"2022-09-16T06:36:06.600399Z","shell.execute_reply":"2022-09-16T06:36:11.504301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`sfdp` is a `graphviz` processor to generate large-scale directed graphs. After running this program, you can check and download the two generated graph in the `Data` section. (or `/kaggle/working/` directory if you are running this program inside a kaggle active notebook)","metadata":{}},{"cell_type":"code","source":"!sfdp -v -x -Goverlap=scale -Tpng -Nshape=rect /kaggle/working/graph.dot > /kaggle/working/sfdp-graph.png","metadata":{"execution":{"iopub.status.busy":"2022-09-16T06:36:38.325270Z","iopub.execute_input":"2022-09-16T06:36:38.325706Z","iopub.status.idle":"2022-09-16T06:37:03.962525Z","shell.execute_reply.started":"2022-09-16T06:36:38.325667Z","shell.execute_reply":"2022-09-16T06:37:03.961262Z"},"trusted":true},"execution_count":null,"outputs":[]}]}