{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#做未來一週的推薦 next week 12 recommendation\n1.按照探索性分析發現年齡可以分2大群（>39 & <=39)\n2.推薦他們上週已購買的產品在加上各自年齡區間的上週12款熱銷產品\n3.對於年紀是Nan的人，則直接推薦上週全年紀的12款熱銷產品\n#Generate 12 recommendations for the next week\n1.According to exploratory analysis, age can be divided into 2 groups (>39 & <=39)\n2.Recommend last week's purchased products plus the top 12 hot-selling products for each age group\n3.For people with missing age information, recommend the top 12 hot-selling products for all age groups in the previous week.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles=pd.read_csv('articles.csv')\narticles.head(2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers=pd.read_csv('customers.csv')\ncustomers","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv('transactions_train.csv')\ntrain.head(2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'customer的重複值:{customers.duplicated().sum()}')\nprint(f'articles的重複值:{articles.duplicated().sum()}')\nprint(f'train的重複值:{train.duplicated().sum()}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'customer的缺失值:{customers.isna().sum().sum()}')\nprint(f'articles的缺失值:{articles.isna().sum().sum()}')\nprint(f'train的缺失值:{train.isna().sum().sum()}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#去重複值\ntrain.drop_duplicates(keep='first', inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#轉換成時間格式\ntrain['t_dat'] = pd.to_datetime(train['t_dat'])\ntrain_1w = train.loc[train['t_dat'] >= pd.to_datetime('2020-09-16')]\ntrain_1w","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_group = train_1w.groupby('customer_id')['article_id'].unique().reset_index()\ndf1=df_group.merge(customers,on='customer_id')\ndf1=df1[['customer_id','article_id','age']]","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#找上週最熱門的前十二名\ntrain_1w['article_id'].value_counts()[:12]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.countplot(x='age',data=df1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.groupby('age').size().sort_index().head(30)\n#39歲為分界","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_group=train_1w.merge(customers,on='customer_id')\ndf_group\ndf_low39=df_group[df_group['age']<39]\ndf_high39=df_group[df_group['age']>=39]\ndf_low39['article_id'].value_counts()\n#看39歲以下的熱門商品","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_high39['article_id'].value_counts()\n#39歲以上的熱門商品","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#找到前12名在39歲以下最熱門的產品\ntop12_low39 = ' 0' + ' 0'.join(df_low39['article_id'].value_counts().sort_values(ascending=False).index.astype('str')[:12])\ntop12_low39","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#找到前12名在39歲以上最熱門的產品\ntop12_high39 = ' 0' + ' 0'.join(df_low39['article_id'].value_counts().sort_values(ascending=False).index.astype('str')[:12])\ntop12_high39","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#arpirori\nfrom mlxtend.frequent_patterns import apriori\nfrom mlxtend.preprocessing import TransactionEncoder\nfrom mlxtend.frequent_patterns import association_rules","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_apri_low39 = df_low39.groupby('customer_id')['article_id'].unique().reset_index()\ntrain_apri_low39\n#train_apri.info","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mlxtend.preprocessing import TransactionEncoder\nte = TransactionEncoder()\ntransactions = df_low39.groupby('customer_id')['article_id'].apply(list).values.tolist()\nte_ary = te.fit(transactions).transform(transactions)\ndf_apri_low39 = pd.DataFrame(te_ary, columns=te.columns_)\nprint(df_apri_low39)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mlxtend.frequent_patterns import apriori\nis_ap = apriori(df_apri_low39, min_support=0.002, max_len=4, use_colnames=True)\nis_ap.sort_values(by=['support'],ascending=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rules = association_rules(is_ap, metric=\"lift\")\nrules.head()\n\n#可以發現923758001跟924243001為頻繁模式 (923758001 <----> 924243001)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mlxtend.preprocessing import TransactionEncoder\nte = TransactionEncoder()\ntransactions = df_high39.groupby('customer_id')['article_id'].apply(list).values.tolist()\nte_ary = te.fit(transactions).transform(transactions)\ndf_apri_high39 = pd.DataFrame(te_ary, columns=te.columns_)\nprint(df_apri_high39)\n","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mlxtend.frequent_patterns import apriori\nis_ap = apriori(df_apri_low39, min_support=0.002, max_len=4, use_colnames=True)\nis_ap.sort_values(by=['support'],ascending=False)\nrules = association_rules(is_ap, metric=\"lift\")\nprint(rules.head())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#923758001跟924243001最常一起搭配\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#先推薦他們在一週內已買過的東西，然後再推薦他們買923758001就接924243001\n#（剛好這兩款都有在<39和>=39的熱門十二名之內） 接著推兩個年齡層前12個熱門的商品\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub=pd.read_csv('sample_submission.csv')\nsub","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_low39 = df_low39['article_id'].value_counts().index.tolist()\nlist_low39[:12]\nlist_high39 = df_low39['article_id'].value_counts().index.tolist()\nlist_high39[:12]\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1=sub.merge(df1,on='customer_id',how='left')\nsub1[sub1['article_id'].notna()]\n","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1['article_id'].notna()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#把article_id的項目丟到prediction裡（推薦他們上週已經買過的東西）\nsub1['prediction'] = sub1['article_id'].apply(lambda x: '0' + str(x).strip('[]') )\nsub1[sub1['article_id'].notna()]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top12_high39","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#對年紀至少39歲的第一類人且上週有購買過的，推薦他們上週自己購買過的商品以及至少39歲區間的上週top12熱銷\nsub1.loc[(sub1['age']>=39) & (sub1['prediction'].notna()),'prediction']=sub1['prediction']+top12_high39\nsub1[sub1['article_id'].notna()]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#對年紀小於39歲的第一類人且上週有購買過的，推薦他們上週自己購買過的商品以及小於39歲區間的上週top12熱銷\nsub1.loc[(sub1['age']<39) & (sub1['prediction'].notna()),'prediction']=sub1['prediction']+top12_low39\nsub1[sub1['article_id'].notna()]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#上週沒買過的至少39歲的人 直接推薦上週39歲以上的12款熱門\nsub1.loc[(sub1['age'] >= 39) & (sub1['prediction']=='0nan'), 'prediction'] = top12_high39\n#上週沒買過的小於39歲的人 直接推薦上週39歲以下的12款熱門\nsub1.loc[(sub1['age']<39) & (sub1['prediction']=='0nan'),'prediction']= top12_low39","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1[sub1['prediction']=='0nan']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#找出不分年齡的上周熱銷12款\ntrain_1w['article_id'].value_counts().index.tolist()\ntop12_allage= ' 0' + ' 0'.join(df_low39['article_id'].value_counts().sort_values(ascending=False).index.astype('str')[:12])\ntop12_allage","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#對那些本來年紀是NaN的人直接推薦上週熱們的全年紀熱銷12款\nsub1.loc[(sub1['age'].isna()) & (sub1['prediction']=='0nan'), 'prediction'] = top12_allage","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1.prediction.isna().sum()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1=sub1[['customer_id','prediction']]\nsub1['prediction'].nunique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1['prediction']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1.to_csv('new_sub_0430.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##結論 private score有0.0125 public socre有0.01327 比起之前做的進步很多(之前在0.006~0.007之間）！","metadata":{},"execution_count":null,"outputs":[]}]}