{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \npd.set_option('display.max_columns', None)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import f1_score\nfrom sklearn import preprocessing","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-23T21:04:20.460017Z","iopub.execute_input":"2022-03-23T21:04:20.460278Z","iopub.status.idle":"2022-03-23T21:04:20.465621Z","shell.execute_reply.started":"2022-03-23T21:04:20.460250Z","shell.execute_reply":"2022-03-23T21:04:20.464076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/music-classification-data-pog-series2/train_features_all.csv')\ntrain_df.drop('Unnamed: 0',axis=1,inplace=True)\ntest_df = pd.read_csv('../input/music-classification-data-pog-series2/test_features (2).csv')\ntest_df.drop('Unnamed: 0',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:20.534779Z","iopub.execute_input":"2022-03-23T21:04:20.535276Z","iopub.status.idle":"2022-03-23T21:04:20.901584Z","shell.execute_reply.started":"2022-03-23T21:04:20.535244Z","shell.execute_reply":"2022-03-23T21:04:20.900823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['IsTrain']= True\ntest_df['IsTrain'] = False\n\nfull = pd.concat([train_df,test_df])","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:20.903203Z","iopub.execute_input":"2022-03-23T21:04:20.903549Z","iopub.status.idle":"2022-03-23T21:04:20.922869Z","shell.execute_reply.started":"2022-03-23T21:04:20.903483Z","shell.execute_reply":"2022-03-23T21:04:20.922211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I extracted features from the audio files on this [notebook](https://www.kaggle.com/code/satoshiss/audio-feature-extraction). <br>\nLet's explore the data and make a model to classify music genre.","metadata":{}},{"cell_type":"markdown","source":"# EDA \n","metadata":{}},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:20.924012Z","iopub.execute_input":"2022-03-23T21:04:20.924384Z","iopub.status.idle":"2022-03-23T21:04:20.974755Z","shell.execute_reply.started":"2022-03-23T21:04:20.924345Z","shell.execute_reply":"2022-03-23T21:04:20.974029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Length\n\n15 audio file was missing when I extracted the features. We do not need to worry about them.","metadata":{}},{"cell_type":"code","source":"full.length.plot(kind='hist')\nfull.length.min(),full.length.max()\n\nprint(f'Number of music less than 20 seconds: {len(full.query(\"length<20\"))}')","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:20.976626Z","iopub.execute_input":"2022-03-23T21:04:20.976871Z","iopub.status.idle":"2022-03-23T21:04:21.189626Z","shell.execute_reply.started":"2022-03-23T21:04:20.976838Z","shell.execute_reply":"2022-03-23T21:04:21.188411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full.query('length!=0 and length<27')","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.190811Z","iopub.execute_input":"2022-03-23T21:04:21.192445Z","iopub.status.idle":"2022-03-23T21:04:21.249491Z","shell.execute_reply.started":"2022-03-23T21:04:21.192405Z","shell.execute_reply":"2022-03-23T21:04:21.248712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.query('length<27')","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.250895Z","iopub.execute_input":"2022-03-23T21:04:21.251142Z","iopub.status.idle":"2022-03-23T21:04:21.307871Z","shell.execute_reply.started":"2022-03-23T21:04:21.251108Z","shell.execute_reply":"2022-03-23T21:04:21.307119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tempo","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(16,23));\nsns.boxplot(x='genre',y='tempo',data=full.query('tempo!=0')[['genre','tempo']]);\nplt.xticks(rotation=45);\nplt.title('Tempo by Genre',fontsize=20);","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.310006Z","iopub.execute_input":"2022-03-23T21:04:21.310266Z","iopub.status.idle":"2022-03-23T21:04:21.862444Z","shell.execute_reply.started":"2022-03-23T21:04:21.310229Z","shell.execute_reply":"2022-03-23T21:04:21.861812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fastes song\nfastest = full.tempo.max()\nfull.query('tempo==@fastest')","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.863729Z","iopub.execute_input":"2022-03-23T21:04:21.864092Z","iopub.status.idle":"2022-03-23T21:04:21.915883Z","shell.execute_reply.started":"2022-03-23T21:04:21.864048Z","shell.execute_reply":"2022-03-23T21:04:21.915067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import IPython.display as idp\n\nidp.display(idp.Audio('../input/kaggle-pog-series-s01e02/train/004760.ogg'))","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.917124Z","iopub.execute_input":"2022-03-23T21:04:21.917450Z","iopub.status.idle":"2022-03-23T21:04:21.934354Z","shell.execute_reply.started":"2022-03-23T21:04:21.917413Z","shell.execute_reply":"2022-03-23T21:04:21.933581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fastes song\nslowest = full.query('tempo!=0').tempo.min()\nfull.query('tempo==@slowest')","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.935685Z","iopub.execute_input":"2022-03-23T21:04:21.935911Z","iopub.status.idle":"2022-03-23T21:04:21.995579Z","shell.execute_reply.started":"2022-03-23T21:04:21.935882Z","shell.execute_reply":"2022-03-23T21:04:21.994926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this song's tempo might be 45*2\n\nidp.display(idp.Audio('../input/kaggle-pog-series-s01e02/train/009655.ogg'))","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:21.996984Z","iopub.execute_input":"2022-03-23T21:04:21.997242Z","iopub.status.idle":"2022-03-23T21:04:22.012400Z","shell.execute_reply.started":"2022-03-23T21:04:21.997207Z","shell.execute_reply":"2022-03-23T21:04:22.011640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Key and Scale\nIt might not be accurate, but we can check it anyways.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(train_df.key_name.unique())\norder = ['C','C#','D','D#','E','F','F#','G','G#','A','A#','B']","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:22.015023Z","iopub.execute_input":"2022-03-23T21:04:22.015320Z","iopub.status.idle":"2022-03-23T21:04:22.021768Z","shell.execute_reply.started":"2022-03-23T21:04:22.015286Z","shell.execute_reply":"2022-03-23T21:04:22.021095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(20,32))\nfig.subplots_adjust(hspace=0.25)\n\nfor  i,g in enumerate(train_df.genre.unique()):\n    df = train_df.query('genre==@g')\n    ax = fig.add_subplot(7,3,i+1)\n    \n    img = sns.countplot(df.key_name,order = order )\n    ax.set_title(f'Key - {g}', fontsize=10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:22.023001Z","iopub.execute_input":"2022-03-23T21:04:22.023792Z","iopub.status.idle":"2022-03-23T21:04:24.675794Z","shell.execute_reply.started":"2022-03-23T21:04:22.023756Z","shell.execute_reply":"2022-03-23T21:04:24.675137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>\nEach instrument has key preference to some degree. Guitar music uses key of E and A most. </font>","metadata":{}},{"cell_type":"code","source":"df.scale","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:24.677059Z","iopub.execute_input":"2022-03-23T21:04:24.677395Z","iopub.status.idle":"2022-03-23T21:04:24.687289Z","shell.execute_reply.started":"2022-03-23T21:04:24.677355Z","shell.execute_reply":"2022-03-23T21:04:24.686652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(20,32))\nfig.subplots_adjust(hspace=0.25)\n\nfor  i,g in enumerate(train_df.genre.unique()):\n    df = train_df.query('genre==@g')\n    ax = fig.add_subplot(7,3,i+1)\n    \n    img = sns.countplot(df.scale,order = ['Major','minor'] )\n    ax.set_title(f'Scale - {g}', fontsize=10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:24.688426Z","iopub.execute_input":"2022-03-23T21:04:24.688617Z","iopub.status.idle":"2022-03-23T21:04:26.848006Z","shell.execute_reply.started":"2022-03-23T21:04:24.688595Z","shell.execute_reply":"2022-03-23T21:04:26.847347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4> Soul-RnB and Hip_Hop have significantly more minor scale music. ","metadata":{}},{"cell_type":"code","source":"train_df['key'] = train_df.key.astype(int).astype('category')\ntest_df['key'] = test_df.key.astype(int).astype('category')\n","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:16:06.736053Z","iopub.execute_input":"2022-03-23T21:16:06.736330Z","iopub.status.idle":"2022-03-23T21:16:06.746088Z","shell.execute_reply.started":"2022-03-23T21:16:06.736300Z","shell.execute_reply":"2022-03-23T21:16:06.745324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['IsMajor'] = [1 if i =='Major' else 0 for i in train_df.scale]\ntest_df['IsMajor'] = [1 if i =='Major' else 0 for i in test_df.scale]","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:26.861444Z","iopub.execute_input":"2022-03-23T21:04:26.861725Z","iopub.status.idle":"2022-03-23T21:04:26.882445Z","shell.execute_reply.started":"2022-03-23T21:04:26.861687Z","shell.execute_reply":"2022-03-23T21:04:26.881821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a model","metadata":{}},{"cell_type":"code","source":"FEATURES = [\n       'mean_stft', 'var_stft','tempo', 'rms_mean', 'rms_var',\n       'centroid_mean', 'centroid_var', 'bandwidth_mean', 'bandwidth_var',\n       'rolloff_mean', 'rolloff_var', 'crossing_mean', 'crossing_var',\n       'harmonic_mean', 'harmonic_var', 'contrast_mean', 'contrast_var',\n       'mfcc1_mean', 'mfcc1_var', 'mfcc2_mean', 'mfcc2_var', 'mfcc3_mean',\n       'mfcc3_var', 'mfcc4_mean', 'mfcc4_var', 'mfcc5_mean', 'mfcc5_var',\n       'mfcc6_mean', 'mfcc6_var', 'mfcc7_mean', 'mfcc7_var', 'mfcc8_mean',\n       'mfcc8_var', 'mfcc9_mean', 'mfcc9_var', 'mfcc10_mean', 'mfcc10_var',\n       'mfcc11_mean', 'mfcc11_var', 'mfcc12_mean', 'mfcc12_var', 'mfcc13_mean',\n       'mfcc13_var', 'mfcc14_mean', 'mfcc14_var', 'mfcc15_mean', 'mfcc15_var',\n       'mfcc16_mean', 'mfcc16_var', 'mfcc17_mean', 'mfcc17_var', 'mfcc18_mean',\n       'mfcc18_var', 'mfcc19_mean', 'mfcc19_var', 'mfcc20_mean', 'mfcc20_var',\n       ]\n\nTARGET = 'genre_id'","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:51:16.019283Z","iopub.execute_input":"2022-03-23T21:51:16.019765Z","iopub.status.idle":"2022-03-23T21:51:16.027063Z","shell.execute_reply.started":"2022-03-23T21:51:16.019725Z","shell.execute_reply":"2022-03-23T21:51:16.026306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_df[TARGET]\nX = train_df[FEATURES]\n\ncols = X.columns\nmin_max_scaler = preprocessing.MinMaxScaler()\nnp_scaled = min_max_scaler.fit_transform(X)\n\nX = pd.DataFrame(np_scaled,columns=cols)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:26.891968Z","iopub.execute_input":"2022-03-23T21:04:26.892234Z","iopub.status.idle":"2022-03-23T21:04:26.921122Z","shell.execute_reply.started":"2022-03-23T21:04:26.892199Z","shell.execute_reply":"2022-03-23T21:04:26.920441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test_df[FEATURES]\ncols = X.columns\nscaled_test = min_max_scaler.transform(test)\ntest = pd.DataFrame(scaled_test,columns=cols)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:26.924000Z","iopub.execute_input":"2022-03-23T21:04:26.924234Z","iopub.status.idle":"2022-03-23T21:04:26.936887Z","shell.execute_reply.started":"2022-03-23T21:04:26.924204Z","shell.execute_reply":"2022-03-23T21:04:26.936187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X.values, y.values, test_size=0.20, random_state=40)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:04:26.938307Z","iopub.execute_input":"2022-03-23T21:04:26.938571Z","iopub.status.idle":"2022-03-23T21:04:26.952875Z","shell.execute_reply.started":"2022-03-23T21:04:26.938535Z","shell.execute_reply":"2022-03-23T21:04:26.952275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold,StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom xgboost import XGBClassifier\n\ncv = StratifiedKFold(n_splits=5, random_state=1, shuffle=True)\npreds=[]\nscores = []\nfold= 1\nfor tr_ix,ts_ix in cv.split(train_df[FEATURES],train_df[TARGET]):\n    print(f'-------------------------------Fold {fold}------------------------------------------------------')\n\n    train_X,test_X = train_df[FEATURES].iloc[tr_ix], train_df[FEATURES].iloc[ts_ix]\n    train_y,test_y = train_df[TARGET].iloc[tr_ix], train_df[TARGET].iloc[ts_ix]\n    \n    model = XGBClassifier(objective= 'multi:softmax',tree_method='gpu_hist',n_estimators=1000,learning_rate=0.05)\n    model.fit(train_X,train_y,eval_set = [(test_X,test_y)],early_stopping_rounds =100,verbose =False)\n    \n    score = f1_score(test_y,model.predict(test_X),average='micro')\n    print(f'Fold {fold}:', score)\n    scores.append(f1_score(test_y,model.predict(test_X),average='micro'))\n    \n    preds.append(model.predict(test_df[FEATURES]))\n    \n    fold+=1\nprint('Average Score: ', np.mean(scores),np.std(scores))","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:51:39.977945Z","iopub.execute_input":"2022-03-23T21:51:39.980336Z","iopub.status.idle":"2022-03-23T21:56:13.135357Z","shell.execute_reply.started":"2022-03-23T21:51:39.980285Z","shell.execute_reply":"2022-03-23T21:56:13.134729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\n\ncv = StratifiedKFold(n_splits=5, random_state=1, shuffle=True)\nscores = []\n\nfold= 1\nfor tr_ix,ts_ix in cv.split(train_df[FEATURES],train_df[TARGET]):\n    print(f'-------------------------------Fold {fold}------------------------------------------------------')\n    \n    train_X,test_X = train_df[FEATURES].iloc[tr_ix], train_df[FEATURES].iloc[ts_ix]\n    train_y,test_y = train_df[TARGET].iloc[tr_ix], train_df[TARGET].iloc[ts_ix]\n    \n    model = CatBoostClassifier(task_type=\"GPU\",n_estimators=1000,learning_rate=0.05)\n    model.fit(train_X,train_y,eval_set = [(test_X,test_y)],early_stopping_rounds =100,verbose =False)\n    score = f1_score(test_y,model.predict(test_X),average='micro')\n    \n    \n    print(f'Fold {fold}:', score)\n    scores.append(score)\n    \n    preds.append([a[0] for a in model.predict(test_df[FEATURES])])\n    \n    fold+=1\nprint('Average Score: ', np.mean(scores),np.std(scores))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:59:20.717908Z","iopub.execute_input":"2022-03-23T21:59:20.718153Z","iopub.status.idle":"2022-03-23T22:01:29.064161Z","shell.execute_reply.started":"2022-03-23T21:59:20.718125Z","shell.execute_reply":"2022-03-23T22:01:29.063374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\npredictions = stats.mode(preds)[0][0]","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:35:12.988221Z","iopub.execute_input":"2022-03-23T21:35:12.988497Z","iopub.status.idle":"2022-03-23T21:35:13.141345Z","shell.execute_reply.started":"2022-03-23T21:35:12.988467Z","shell.execute_reply":"2022-03-23T21:35:13.140615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a submission File","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('../input/kaggle-pog-series-s01e02/sample_submission.csv')\nsub['genre_id'] = predictions","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:06:54.333927Z","iopub.status.idle":"2022-03-23T21:06:54.334749Z","shell.execute_reply.started":"2022-03-23T21:06:54.334499Z","shell.execute_reply":"2022-03-23T21:06:54.334529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:06:54.335857Z","iopub.status.idle":"2022-03-23T21:06:54.336635Z","shell.execute_reply.started":"2022-03-23T21:06:54.336394Z","shell.execute_reply":"2022-03-23T21:06:54.336420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(sub.genre_id)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T21:06:54.337797Z","iopub.status.idle":"2022-03-23T21:06:54.338590Z","shell.execute_reply.started":"2022-03-23T21:06:54.338342Z","shell.execute_reply":"2022-03-23T21:06:54.338368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References\n\nAndrada made a really great notebook along with her music dataset, which I referred to build my dataset. \n\nhttps://www.kaggle.com/code/andradaolteanu/work-w-audio-data-visualise-classify-recommend\n\n\n","metadata":{}}]}