{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Here i construct a model for integer variables by mixing approximately 50 Poisson(1) variables with plus and minus signs.","metadata":{}},{"cell_type":"code","source":"import random\nimport pandas as pd\nimport numpy as np\n\n\n\n#clusters = [0,1,2,3,4,5,6] # all clusters\nclusters = [1] # test - only cluster 1\n\n\n\n# read data\ndf = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nsub0 = pd.read_csv('../input/tps722sub/sub4.csv') # read a good submission\nx0 = np.array(df.drop('id', axis=1), dtype=np.float32)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:06:53.086901Z","iopub.execute_input":"2022-07-23T12:06:53.087408Z","iopub.status.idle":"2022-07-23T12:06:54.709616Z","shell.execute_reply.started":"2022-07-23T12:06:53.087305Z","shell.execute_reply":"2022-07-23T12:06:54.708377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I find mixing covariance matrix by starting with appropriate number of +1 and -1 values to match mean and variance of each data column, and then permuting values to improve match on correlation matrix","metadata":{}},{"cell_type":"code","source":"# find covariance matrix\n# final integer covariance matrix\ncov_m_full = np.zeros([120, 7, 7], dtype=np.int32) # input_var(start with high value, reduce size later), output_var(7), cluster(7)\nfor cl in clusters:# loop over clusters\n    # select current cluster integers only\n    x = x0[sub0['Predicted'] == cl, 7:14]\n\n    # data mean/var/corr\n    xm = np.round(x.mean(axis=0), 0).astype(np.int32)     # mean\n    xv = np.round(x.var(axis=0), 0).astype(np.int32)      # variance\n    xc = np.corrcoef(x.T)   # correlation matrix\n    print(cl, xm, xm.sum(), xv, xv.sum())\n    n_plus = np.round((xv + xm) / 2, 0).astype(np.int32)\n    n_minus =np.round((xv - xm) / 2, 0).astype(np.int32)\n\n    # populate covariance matrix with +-1 - that sets mean and variance\n    j = 0\n    cov_m = np.zeros([(n_plus+n_minus).sum(), 7], dtype=np.int32)\n    for i in range(7): # loop over 7 integer variables\n        # plus\n        cov_m[j:j + n_plus[i], i] = 1\n        j += n_plus[i]\n        # minus\n        cov_m[j:j + n_minus[i], i] = -1\n        j += n_minus[i]\n\n\n    # get difference between correlation maxrices\n    def get_diff(cov_m):\n        c1 = np.zeros([7, 7], dtype=np.float32)\n        for i in range(7):\n            for j in range(7):\n                c1[i, j] = (cov_m[:, i] * cov_m[:, j]).sum() / np.sqrt(xv[i] * xv[j])\n        d = ((c1 - xc)**2).sum()\n        return d\n    \n    # swap 2 elements in a random column of cov_m if it improves match on correlation matrix   \n    d0 = get_diff(cov_m)\n    s0 = cov_m.shape[0]\n    for ii in range(100000):\n        col  = random.randint(0, 6)\n        row1 = random.randint(0, cov_m.shape[0] - 1)\n        row2 = random.randint(0, cov_m.shape[0] - 1)\n        if cov_m[row1,col] != cov_m[row2,col]:\n            # swap values\n            j = cov_m[row1,col]\n            cov_m[row1,col] = cov_m[row2,col]\n            cov_m[row2,col] = j\n            # check if match to corr improved\n            d = get_diff(cov_m)\n            s = np.minimum(np.abs(cov_m).sum(axis=1), 1).sum()\n            if d < d0 or (d == d0 and s < s0): # better, keep. Also accept same score and smaller matrix\n                d0 = d\n                s0 = s\n            else: # worse, undo\n                # swap values back\n                j = cov_m[row1,col]\n                cov_m[row1,col] = cov_m[row2,col]\n                cov_m[row2,col] = j\n        # progress report\n        if ii % 10000 == 0:\n            print(cl, ii, np.round(d0, 3), s0)\n    # select only non-zero rows\n    idx = np.abs(cov_m).sum(axis=1) > 0\n    cov_m = cov_m[idx,:]\n    # save results\n    cov_m_full[:cov_m.shape[0], :, cl] = cov_m\n# select only non-zero rows\nidx = np.abs(cov_m_full).sum(axis=1).sum(axis=1) > 0\ncov_m_full = cov_m_full[idx, :, :]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:06:54.712043Z","iopub.execute_input":"2022-07-23T12:06:54.713139Z","iopub.status.idle":"2022-07-23T12:07:10.786264Z","shell.execute_reply.started":"2022-07-23T12:06:54.713093Z","shell.execute_reply":"2022-07-23T12:07:10.784992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now i confirm that this approach works by generating random numbers, mixing them with my covariance matrix and comparing sample mean/variance/correlation to data ","metadata":{}},{"cell_type":"code","source":"# validate covariance matrix\nfor cl in clusters:# loop over clusters\n    # select current cluster integers only\n    x = x0[sub0['Predicted'] == cl, 7:14]\n\n    # data mean/var/corr\n    xm = np.round(x.mean(axis=0), 1)     # mean\n    xv = np.round(x.var(axis=0), 1)      # variance\n    xc = np.corrcoef(x.T)   # correlation matrix\n    print(cl, xm, np.round(xm.sum(), 1), xv, np.round(xv.sum(), 1))\n    print(np.round(xc, 2))\n\n    # cov matrix\n    cov_m = cov_m_full[:, :, cl]\n    # select only non-zero rows\n    idx = np.abs(cov_m).sum(axis=1) > 0\n    cov_m = cov_m[idx, :]\n\n    # generate Poisson\n    y = np.random.poisson(lam=1, size=[1000000, cov_m.shape[0]]) # 1 Mil - big enough\n    y = y.dot(cov_m) # (size, 7)\n\n    # randoms mean/var/corr\n    ym = np.round(y.mean(axis=0), 1)     # mean\n    yv = np.round(y.var(axis=0), 1)      # variance\n    yc = np.corrcoef(y.T)   # correlation matrix\n    print(cl, ym, np.round(ym.sum(), 1), yv, np.round(yv.sum(), 1))\n    print(np.round(yc, 2))\n\n    # differences(L1)\n    print(cl, np.round(np.abs(xm - ym).mean(), 2), np.round(np.abs(xv - yv).mean(), 2), np.round(np.abs(xc - yc).mean(), 2))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:07:10.787821Z","iopub.execute_input":"2022-07-23T12:07:10.788182Z","iopub.status.idle":"2022-07-23T12:07:14.452278Z","shell.execute_reply.started":"2022-07-23T12:07:10.788150Z","shell.execute_reply":"2022-07-23T12:07:14.450998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Results are close. Main differences are due to data mean/variance not being integers. But using this approach may not be possible, since i see no way to construct probability densisty function for this approach.","metadata":{}}]}