{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is for probing the LB for statistical information to ensure that it has the same distribution as the \ntrain set.  Please leave tests you'd like to see performed in the comments.\n\nAlso, check back every day for new versions.  Here's the todo list\n- generate enough unique scores so we can have better than binary output from thise notebook\n- test all batches for nan, dupes, index orders \n- test dist props for counts, std,mean, min/max values\n- ?\n\nThe saved version should always succeed so the conditions below are the invariants of the test set.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\ntests = pd.read_parquet('/kaggle/input/icecube-neutrinos-in-deep-ice/test_meta.parquet')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T06:44:44.959985Z","iopub.execute_input":"2023-01-31T06:44:44.960522Z","iopub.status.idle":"2023-01-31T06:44:44.973313Z","shell.execute_reply.started":"2023-01-31T06:44:44.960476Z","shell.execute_reply":"2023-01-31T06:44:44.972167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    train = pd.read_parquet('/kaggle/input/icecube-neutrinos-in-deep-ice/train_meta.parquet')\n    fails = []\n    t = train.sample(frac = 0.1)\n    print('meta len', len(t))\n    t['pl'] = t.last_pulse_index - t.first_pulse_index\n    t = t.sort_values([\"batch_id\", \"first_pulse_index\"])\n    t['next_first_pulse_shift'] = t.groupby(\"batch_id\").shift(-1)['first_pulse_index']\n    display(t)\n    if len(t[(~np.isnan(t.next_first_pulse_shift)) & (t.next_first_pulse_shift != t.last_pulse_index+1)]) > 0:\n        fails.append(\"pulses are interleaved\")\n    print(fails)\n    print('pulse len mean', np.mean(t['pl']))\n    print('min pulse len', np.min(t['pl']))\n    print('max pulse len', np.max(t['pl']))\n    print('std pulse len', np.std(t['pl']))\n    print(\"avg events per batch\", t.groupby(\"batch_id\").count()['event_id'].mean())\n    if False:\n        import tqdm\n        fset = \"train\"\n        tev = t.set_index(\"batch_id\")\n        scs = []\n        for i in t.batch_id.unique()[0:1]:\n            events = tev.loc[i]\n            b = pd.read_parquet(f'/kaggle/input/icecube-neutrinos-in-deep-ice/{fset}/batch_{str(i)}.parquet')\n            for i,e in tqdm.tqdm(events.iterrows()):\n                sm = np.sum(b.loc[e.event_id].charge)\n                pl = e.pl\n                scs.append([sm,pl, e.zenith])\n        import scipy.stats\n        scipy.stats.pearsonr([x[0]+x[1] for x in scs], [x[2] for x in scs])","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:44.975268Z","iopub.execute_input":"2023-01-31T06:44:44.975922Z","iopub.status.idle":"2023-01-31T06:44:44.990046Z","shell.execute_reply.started":"2023-01-31T06:44:44.975860Z","shell.execute_reply":"2023-01-31T06:44:44.988797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(pd.read_csv(\"/kaggle/input/icecube-neutrinos-in-deep-ice/sensor_geometry.csv\"))","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:44.991322Z","iopub.execute_input":"2023-01-31T06:44:44.992171Z","iopub.status.idle":"2023-01-31T06:44:45.013894Z","shell.execute_reply.started":"2023-01-31T06:44:44.992132Z","shell.execute_reply":"2023-01-31T06:44:45.012432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    bis = train.batch_id.unique()[0:10]\n    for i in bis:\n        b = pd.read_parquet(f'/kaggle/input/icecube-neutrinos-in-deep-ice/train/batch_{str(i)}.parquet')\n        \n        display(b.describe())","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.016382Z","iopub.execute_input":"2023-01-31T06:44:45.016742Z","iopub.status.idle":"2023-01-31T06:44:45.022578Z","shell.execute_reply.started":"2023-01-31T06:44:45.016689Z","shell.execute_reply":"2023-01-31T06:44:45.021417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(tests.head())","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.024172Z","iopub.execute_input":"2023-01-31T06:44:45.024541Z","iopub.status.idle":"2023-01-31T06:44:45.040692Z","shell.execute_reply.started":"2023-01-31T06:44:45.024507Z","shell.execute_reply":"2023-01-31T06:44:45.039793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fails = []\nt = tests\nif len(t) < 900000 or len(t) > 1100000:  \n    fails.append(f\"test len out of range, {len(t)}\")\n    \nt['pl'] = t.last_pulse_index - t.first_pulse_index\nif np.mean(t['pl']) < 161 or np.mean(t['pl']) > 165:  #train is around 163.5\n    fails.append(f\"mean pulse count per event out of range, {np.mean(t['pl'])}\")\n   \n#Note that pulse index are duped across batch_ids\nif (t.duplicated([\"event_id\"]).any() or t.duplicated([\"batch_id\",\"first_pulse_index\"]).any() \n    or t.duplicated([\"batch_id\", \"last_pulse_index\"]).any()):\n    fails.append(\"One of event_id, first_pulse_index, last_pulse_index was duped\")\n\nif (t.isnull().values.any()):\n    fails.append(\"null value in tests\")    \n    \n#Added jan25\n\n#Make sure that events are never interleaved by pulse index\nt = t.sort_values([\"batch_id\", \"first_pulse_index\"])\nt['next_first_pulse_shift'] = t.groupby(\"batch_id\").shift(-1)['first_pulse_index']\nif len(t[(~np.isnan(t.next_first_pulse_shift)) & (t.next_first_pulse_shift != t.last_pulse_index+1)]) > 0:\n    fails.append(\"pulses are interleaved\")\n    \nif np.std(t['pl']) < 1520 or np.std(t['pl']) > 1600:  #train is around 1560\n    fails.append(f\"std pulse count per event out of range, {np.std(t['pl'])}\")\n\navg_events_per_batch = t.groupby(\"batch_id\").count()['event_id'].mean()\nif avg_events_per_batch < 1990 or np.std(t['pl']) > 2010:  #train is around  2000\n    fails.append(f\"avg_events_per_batch out of range, {avg_events_per_batch}\")\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.041856Z","iopub.execute_input":"2023-01-31T06:44:45.042300Z","iopub.status.idle":"2023-01-31T06:44:45.068744Z","shell.execute_reply.started":"2023-01-31T06:44:45.042254Z","shell.execute_reply":"2023-01-31T06:44:45.067660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(f'/kaggle/input/icecube-neutrinos-in-deep-ice/test/batch_{661}.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.070307Z","iopub.execute_input":"2023-01-31T06:44:45.070827Z","iopub.status.idle":"2023-01-31T06:44:45.091887Z","shell.execute_reply.started":"2023-01-31T06:44:45.070777Z","shell.execute_reply":"2023-01-31T06:44:45.090688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#These will provide the buckets.  \n# When I get to five, that should be enough plus I can use an error conditions for 9 buckets \n# Knowing more about the data, we might be able to make a continous set of buckets, but with only 3 decimal points it's not \n# trivially easy\n#0,0 is 1.533\n#1,6 is 1.534\n#6,1 is 1.551\n#2,2 is 1.586\n#4,4 is 1.595\n#5,5 is 1.561\n#5,3 is 1.607\ndef get_bucket(val, minv, maxv):\n    print(\"bucket for val\", val)\n    buckets = [(0,0,1.533), (1,6,1.534), (6,1,1.551), (2,2,1.586), (4,4,1.595), (5,5,1.561), (5,3,1.607)]\n    blen = len(buckets)\n    bbins = np.linspace(minv, maxv, blen-1)\n    print(bbins)\n    print(\"val <\", bbins[0], buckets[0])\n    for i in range(0, len(bbins)-1):\n        print(bbins[i],\">= val <\",bbins[i+1], buckets[i+1])\n    print(\"val >=\", bbins[blen-2], buckets[-1])\n    if val < bbins[0]:\n        b = 0\n    elif val >= bbins[blen-2]:\n        b = blen-1\n    else:\n        for i in range(0, len(bbins)):\n            if val >= bbins[i]:\n                b = i+1\n    return buckets[b]\nget_bucket(0.25, minv=0.26,maxv=0.3)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.094748Z","iopub.execute_input":"2023-01-31T06:44:45.095704Z","iopub.status.idle":"2023-01-31T06:44:45.111795Z","shell.execute_reply.started":"2023-01-31T06:44:45.095665Z","shell.execute_reply":"2023-01-31T06:44:45.110799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Added jan25\nimport gc\nbis = t.batch_id.unique()\nevents = 0\nevent_count = 0\ncharge_total = 0\naux_true_total = 0\naux_false_total = 0\nmax_charge = -999\nmin_charge = 999\n#bisids, fset = bis[15:20], \"train\"\nbisids, fset = bis, \"test\"\nfor i in bisids:\n    b = pd.read_parquet(f'/kaggle/input/icecube-neutrinos-in-deep-ice/{fset}/batch_{str(i)}.parquet')\n    if np.max(b.sensor_id) > 5159 or np.max(b.sensor_id) < 0:\n        print(\"Sensor id out of range\")\n    \n    if (b.isnull().values.any()):\n        fails.append(\"null value in batch\") \n    \n    events = len(b)\n    event_count = event_count + events\n    charge_total = charge_total + np.sum(b['charge'])\n    aux_trues = len(b[b['auxiliary']])\n    aux_true_total = aux_true_total + aux_trues  \n    aux_false_total = aux_false_total + events - aux_trues  \n    mx = np.max(b['charge']) \n    mi = np.min(b['charge'])\n    \n    if mx > max_charge:\n        max_charge = mx\n    if mi < min_charge:\n        min_charge = mi\n    print(\"np.mean(b['charge'])\",np.mean(b['charge']))\n    print(\"aux_mean\", np.mean(b['auxiliary']))\n    print(\"std charge\", np.std(b['charge']))\n    gc.collect()\n    \n    \ncharge_mean = charge_total/event_count\naux_true_mean = aux_true_total/event_count\naux_false_mean = aux_false_total/event_count\n\ncharge_var = 0\n\nfor i in bisids:\n    b = pd.read_parquet(f'/kaggle/input/icecube-neutrinos-in-deep-ice/{fset}/batch_{str(i)}.parquet')\n    charge_var = charge_var + np.sum((b['charge'] - charge_mean)**2)\n    gc.collect()\n    \ncharge_std = np.sqrt(charge_var / event_count)\n\nprint(\"event_count\", event_count)\nprint(\"charge_total\", charge_total, \"charge_mean\", charge_mean)\nprint(\"aux_true_total\", aux_true_total, \"aux_true_mean\", aux_true_mean)\nprint(\"aux_false_total\", aux_false_total, \"aux_false_mean\", aux_false_mean)\nprint(\"min_charge\",min_charge) #train is 0.02500000037252903\nprint(\"max_charge\",max_charge) #train is around 4261.625\nprint(\"charge_std\", charge_std) #train is 16.5\n\nif aux_true_mean < 0.26 or aux_true_mean > 0.30:  #train is around 0.28, failed at < 0.27 / > 0.29\n    fails.append(f\"aux_true_mean out of range, {aux_true_mean}\")\n\nif aux_false_mean < 0.7184 or aux_false_mean > 0.7216:  #train is around 0.72\n    fails.append(f\"aux_false_mean out of range, {aux_false_mean}\")\n    \nif charge_std < 17.312 or charge_std > 17.336:\n    fails.append(f\"charge_std out of range, {charge_std}\")\n    \nif charge_mean < 4.2112 or charge_mean > 4.2136:\n    fails.append(f\"charge_mean out of range, {charge_mean}\")\n    \nif max_charge < 4236 or max_charge > 4240:\n    fails.append(f\"max_charge out of range, {max_charge}\")\n    \nif min_charge < 0.024 or min_charge > 0.026: #Probably 0.025, same as train\n    fails.append(f\"min_charge out of range, {min_charge}\")\n    \nif len(t) != 1000151:\n    fails.append(f\"len(t) out of range, {len(t)}\")\n    \n#narrow down aux_false_mean\nb = get_bucket(aux_false_mean, minv=0.68,maxv=0.76) #returns 1.586, which is 0.7120 >= val < 0.728\nb = get_bucket(aux_false_mean, minv=0.7120, maxv=0.728) #returns 1.586, which is 0.7184 >= val < 0.7216\nb = get_bucket(aux_false_mean, minv=0.7184,maxv=0.7216) #returns 1.51 which is 0.7152 >= val < 0.7184\n\n\n#narrow down charge mean\nb = get_bucket(charge_mean, minv=4, maxv=4.3) #train is about 4.15, was 1.595 so 4.18 >= val < 4.24\nb = get_bucket(charge_mean, minv=4.18, maxv=4.24) #was 1.586 so 4.204 >= val < 4.216\nb = get_bucket(charge_mean, minv=4.204, maxv=4.216) #was 1.595 so 4.2112 >= val < 4.2136 (4, 4, 1.595)\n\n#narrow down std\nb = get_bucket(charge_std, minv=16, maxv=17) #was (5, 3, 1.607), so greater than 17\nb = get_bucket(charge_std, minv=16, maxv=20) #was 1.551, so 16.8 to 17.6\nb = get_bucket(charge_std, minv=17, maxv=17.6) #was 1.586, so 17.24 >= val < 17.36\nb = get_bucket(charge_std, minv=17.24, maxv=17.36) #was 1.595 so 17.312 >= val < 17.336 (4, 4, 1.595)\n\nb = get_bucket(max_charge, minv=4000, maxv=4500) #train is about 4261, was 1.586 so 4200.0 >= val < 4300.0 \nb = get_bucket(max_charge, minv=4200, maxv=4300) #was 1.551, so 4220.0 >= val < 4240.0 (6, 1, 1.551)\nb = get_bucket(max_charge, minv=4220.0, maxv=4240.0) #was 1.561, so 4236.0 >= val < 4240.0 (5, 5, 1.561)\n\nb = get_bucket(min_charge, minv=0.02, maxv=0.03) #train is 0.025, was 1.586\n\nb = get_bucket(len(t), minv=900000, maxv=1100000) #was 1.586\nb = get_bucket(len(t), minv=980000, maxv=1020000) #was 1.586, 996000.0 >= val < 1004000.0\nb = get_bucket(len(t), minv=996000, maxv=1004000) #was 1.586 999200.0 >= val < 1000800.0 (2, 2, 1.586)\nb = get_bucket(len(t), minv=999200.0, maxv=1000800) #was 1.586, 999840.0 >= val < 1000160.0 (2, 2, 1.586)\nb = get_bucket(len(t), minv=1000096.0, maxv=1000160.0) # was 1.561 1000147.2 >= val < 1000160.0 (5, 5, 1.561)\nb = get_bucket(len(t), minv=1000147.2, maxv=1000160.0) # 1.551 1000149.76 >= val < 1000152.32 (6, 1, 1.551)\nb = get_bucket(len(t), minv=1000149.76, maxv=1000152.32) #1.586,so 1000151 as 1000150.784 >= val < 1000151.296 (2, 2, 1.586)\n\n\n\n\n\n\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:59:23.529369Z","iopub.execute_input":"2023-01-31T06:59:23.529785Z","iopub.status.idle":"2023-01-31T06:59:23.792499Z","shell.execute_reply.started":"2023-01-31T06:59:23.529749Z","shell.execute_reply":"2023-01-31T06:59:23.791187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[print(x) for x in fails]\nif len(fails) > 0:\n    raise Exception","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.398786Z","iopub.execute_input":"2023-01-31T06:44:45.399153Z","iopub.status.idle":"2023-01-31T06:44:45.469356Z","shell.execute_reply.started":"2023-01-31T06:44:45.399119Z","shell.execute_reply":"2023-01-31T06:44:45.467813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(b)\nimport tqdm\nevents = tests.event_id\nres = []\nwith open(\"submission.csv\",\"a\") as f:\n    f.write(\"event_id,azimuth,zenith\\n\")\n    for i in tqdm.tqdm(events):\n        f.write(f\"{i},{b[0]},{b[1]}\\n\") \n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.470352Z","iopub.status.idle":"2023-01-31T06:44:45.470800Z","shell.execute_reply.started":"2023-01-31T06:44:45.470573Z","shell.execute_reply":"2023-01-31T06:44:45.470592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-01-31T06:44:45.472259Z","iopub.status.idle":"2023-01-31T06:44:45.472746Z","shell.execute_reply.started":"2023-01-31T06:44:45.472516Z","shell.execute_reply":"2023-01-31T06:44:45.472537Z"},"trusted":true},"execution_count":null,"outputs":[]}]}