You are on page 1of 26
3127128, 1026 PM In [3]: In [4]: In [27]: In (28): out (28): localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook import numpy as np # Linear algebra import pandas as pd # data processing, CSV file 1/0 (e.g. pd.read_csv) #import tensorflow as tf import matplotlib.pyplot as plt import seaborn as sns from sklearn.manifold import TSNE from sklearn.decomposition import PCA, TruncatedSvo import matplotlib.patches as mpatches import time # Classifier Libraries from sklearn.linear_model import LogisticRegression from sklearn.svm import SVC from sklearn.neighbors import kKNeighborsClassifier from sklearn.tree import DecisionTreeClassifier from sklearn.ensemble import RandonForestClassifier import collections # Other Libraries from sklearn.nodel_selection import train_test_split from sklearn.pipeline import make_pipeline fron imblearn.pipeline import make_pipeline as imbalanced_make_pipeline from imblearn.over_sampling import SMOTE from imblearn.under_sampling import NearMiss from imblearn.metrics import classification_report_imbalanced from sklearn.metrics import precision_score, recall_score, f1_score, roc_auc_s from collections import Counter from sklearn.nodel_selection import KFold, StratifiedKFold import warnings df =pd.read_csv( 'creditcard.csv') df.head() Time vi v2 vs va vs vs vr ve © 0.0 1.359807 -0.072781 2.536347 1.378155 -0.338321 0.462388 0.239599 0.098698 0 1 00 1.191857 0.266151 0.166480 0.448154 0.060018 0.082361 -0.078803 0.085102 -0 2 1.0 1.358954 -1.340163 1.773209 0.379780 -0.503198 1.800499 0.791461 0.247676 3 1.0 0.966272 -0.185226 1.702993 0.863201 -0.010309 1.247203 0.237609 0.377436 4 20 4.158233 0.877737 1.548718 0.403034 -0,407193 0.095921 0.592941 -0,270533 0 5 rows * 31 columns clonleredi card raud detection bivariate classification model yn 128 3127128, 1026 PM In [29]: In [30]: out [30]: In [34]: out (31): In [32]: out [32]: localhost 8888inotebooks/Documents/Credk card fraud de df.info() credit card aus detection bivariate classification mode - Jupyter Notebook Rangelndex: 284807 entries, @ to 284806 Data columns (total 31 columns): # Column Non-Null Count Dtype @ Time 284807 non-null floatea 1 ov 284807 non-null floate4 2 v2 284807 non-null floate4 3 VB 284807 non-null floated 4 va 284807 non-null floated 5 vs 28487 non-null floated 6 v6 28487 non-null floated 77 28487 non-null floated a ve 28487 non-null floate4 9 v9 28487 non-null floated 10 V10 284807 non-null floated 11 V11 284807 non-null floatea 12 V12 284807 non-null floated 13 V13 284807 non-null floated > df.describe() Time vi v2 vs va vs ‘count 284807.000000 2.848070e+05 2.8480700+05 | 2.8480700+05 2.8480700+05 2.8480700+05 mean 94813.859575 1.168375e-15 3.416908e-16 -1.379537e-15 2.074095e-15 9.6040666-16 std 47488.145955 1.9586960+00 1.6513090+00 1.516255+00 1.4158690+00 1.3802470+00 min 0.000000 -5.640751e+01 -7.271873e+01 -4.892559e+01 -6.683171e+00 -1.1374336+02 25% 54201.500000 -9.2037340-01 -5.985499e-01 -8.9036480-01 -8486401e-01 -6.9159710-01 80% 84692.000000 1.8108800-02 6.5485560-02 1.7984830-01 -1.9846530-02 -§.4335830-02 75% 139320.500000 1.315642e+00 8.037239e-01 1.027196e+00 7.433413e-01 6.119264e-01 max 172792,000000 2.4549300+00 2.205773e+01 9.382558e+00 1.6875340+01 3.480167e+01 8 rows x 31 columns df. shape (284807, 31) # Checching if any null values are present or not df. isnul1().sum().max() e clonleredi card raud detection bivariate classification model yn 226 3127123, 1026 PM ‘credit ard fraud detection bivariate classification model -Jupyter Notebook In (33 print("percentage of non froud is", round(df[‘Class'].value_counts()[0]/len(df print("percentage of froud is", round(d#['Class’].value_counts()[1]/len(df)* 1 percentage of non froud is 99.83 % percentage of froud is 0.17 % In [34]: # Plot the Class Distribution colors = [“#o1@10F", "#0F0101"] sns.countplot(x = df["Class"}) plt.title(‘Class Distributions \n (@: No Fraud || 1: Fraud)", fontsize=14) out(34]: Text(@.5, 1.0, ‘Class Distributions \n (@: No Fraud || 1: Fraud)') Class Distributions : No Fraud || 1: Fraud) 250000 200000 150000 ‘count 100000 50000 Class locahost 8888inotebooks/Documents/Credk card fraud delection/redl card fraud detection bivariate classieation model ipynb 3127128, 1026 PM In [35]: In [36]: In [38]: out [38]: localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook fix, ax = plt.subplots(1,2, figsize 18,4)) val = df['Anount'].values tim = df[' Time" ]. values sns.histplot(val, ax = ax(0], ax[0].set_title(' Distribution color ='r') of Transaction Amount") sns.histplot(tim, ax = ax[1], ax[0].set_title( ‘Distribution color = 'r') of Transaction Time’) plt.show() outliers(data): quartile, quartile_3 igr = quartile3 - quartile 1 np.percentile(data, [25, 75]) lower_bound = quartile_a - (igr * 1.5) upper_bound = quartile 3 + (igr * 1.5) return [x for x in data if x < lower_bound or x > upper_bound] data = [1, 2, 3, 4, 5, 6, 7, 8, 9, 18, 100] outliers = detect_outliers(data) print(outliers) # prints [100] [100] df.columns Index(['Time', "Vi", 'V2', 'V3', 'V4', 'V5', 'V6', 'V7", ‘VB', ‘VS", ‘Vi0", tvai', ‘va2", ‘vi3", 'via", vis", "Vie", ‘vi7", ‘vi8", ‘vi9", "V20", "vai", ‘v22", 'v23", 'v24", ‘v2s", 'v26", 'v27", 'V28", ‘Amount', *class'], dtype='object') .clonleredi card raud detection bivariate classification model yn 4126 3127128, 1026 PM credit card aus detection bivariate classification mode - Jupyter Notebook In [39]: # Since most of our data has already been scaled we should scale the columns t from sklearn. preprocessing import StandardScaler, RobustScaler df["time_rob_scales"] df['amount_rob_scaled ] In [40]: df.drop(['Time’, ‘Amount’ ], axis = 1, inplace = True) In [41]: df-head() RobustScaler().fit_transform(df[ 'Time'].values.reshape RobustScaler().fit_transform(df[ ‘Amount’ ].values.res In [43]: df.head() Out [41]: vt v2 v3 va ve ve vw ve ve 0 1.959807 -D.O7ZT61 2596047 1878155 0396021 0452980 0.299599 0.098598 0.96578 4 1191857 0.266151 0.166460 0.448154 0.080018 0.082361 0.078803 0.085102 0.25542: 2 1.958954 1.940169. 1.779209 0379760 0509198 1.800499 o.791481 0.247676. 1.51485. 3 0.966272 0.185225 1.792883 0.863281 0.010309 1.247203 0.257609 0.377436 1.98702. 4 -4158233 0.877737 1548718 0.403084 0407199 0.095921 0.582941 0.270539 0.81773! 5 rows « 31 columns > In [42]: scaled_tine = df[‘tine_rob_scales*] scaled_anount = df[‘anount_rob_scaled"] Gf renane(colunns={"time rob_scales': ‘scaled_tine’, ‘amount_rob_scaled’: ‘sca out 43): en 0 1.950807 0.072761 2530347 1.878185 0398921 0.462380 0.259609 0.098098 0.96578 1 1191857 0.266151 0.66480 0.448154 0.050018 0.082961 0.078803 0.085102 0.25542: 2 1.958954 1.340163. 1.779209 0.379760 0.509198 1.800499 o.791461 0.247676 1.51405. 3 0966272 0.185226 1.792903 0.863291 0.010900 1.247209 0.297609 0.377436 1.98702. 41158239 0.877737 1.548718 0403084 0407189 0.095821 0.582841 0.270539 0.81773! 5 rows x 31 columns In [44]: df.shape out[44): (2848@7, 32) localhost 8888inotebooks/Documents/Credk card fraud de .clonleredi card raud detection bivariate classification model yn 526 3127128, 1026 PM In [48]: In [54]: out [54]: In [57]: localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook from sklearn.model_selection import train_test_split from sklearn.model_selection import StratifiedShutfleSplit print( print( y df.drop( "Class df['Class'] axis =1) percentage of non froud is 99.83 % percentage of froud is @.17 % ercentage of non froud is", round(df['Class'].value_counts()[@]/len(df ercentage of froud is”, round(d#['Class’].value_counts()[1]/len(df)* 1 x vw ve vs v4 vs ve wv ve © -1.859807 -0.072761 2.536847 1.378155 -0.936921 0.462988 0.230509 0.098698 | 4 1.191857 0.266151 0.166480 0.448154 0.060018 -0.082361 0.078803 0.085102 + 2 -1:358354 -1,340163 1.773209 0.379780 -0.503198 1.800499 0.791461 0.247676 - 3 0.986272 0.185225 1.792983 0.865291 0.010309 1.247203 0.237608 0.37436 - 4 4.188283 0.877737 1.548718 0.403034 0.407193 0.095921 0.592041 -0.270533 | 2e4e02 -11.881118 10.071785 -9.834783 -2.066656 5.364473 -2.606837 4.918215 7.305334 284803 0.732789 0.056080 2.035030 -0.738689 0.868229 1.058415 0.024330 0.294869. 1 284804 1.919565 0301254 -3.249880 -0.557828 2.630515 3.031260 -0.206827 0.708417 | 284805 0.240440 0.530483 0.702510 0.889799 -0.37961 0.623708 0.686180 0.879145 | 284808 0.533413 0.189733 0.703387 -0.506271 -0.012546 -0.649617 1.577006 -0.414850 | 284807 rows x 30 columns sss = StratifiedkFold(n_splits=5, random_state= None, shuffle= False) for train_index, test_index in sss.split(x,y): print("Train: ", train_index, "Test: ", test_index) original_xtrain, original_xtest = x. loc [train_index] ,x.iloc[test_index] original_ytrain, original_ytest = y.iloc[train_index],y.iloc[test_index] Train: [ 30473 30496 31002 ... 284804 284805 284806] Test: [ @ 1 2... 57017 57018 57019] Train: [ @ 1 2... 284804 284805 284806] Test: [ 30473 30496 31002 ... 113964 113965 113966] Train: [ @ 1 2... 284804 284805 284806] Test: [ 81609 82400 83053 ... 170946 170947 170948] Train: [ @ 1 2 ... 284804 284805 284806] Test: [150654 150660 150661 ... 227866 227867 227868] Train: [ @ 1 2... 227866 227867 227868] Test: [212516 212644 213092 ... 284804 284805 284806] .clonleredi card raud detection bivariate classification model yn 626 3127128, 1026 PM In [61]: In [66]: out (66): In [68]: In [69]: In [70]: out[70]: In [71]: out [71]: In [77]: out (77): localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook print(original_ytrain.value_counts()) print(original_ytest.value_counts()) @ 227452 1 394 Name: Class, dtype: inte @ 56863 1 98 Name: Class, dtype: inte Len(original_ytrain) 227846 print(1en(original_ytrain)/ len(original_ytest)) print(1en(original_xtrain)/ len(original_xtest)) 4,000035111743123 4.000035111743123 # See if both the train and test Label distribution are similarly distributed train_unique_label, train_counts_label = np.unique(original_ytrain, return_cou test_Unique_label, test_counts_label = np.unique(original_ytest, return_counts print('-' * 100) print('Label Distributions: \n') print(train_counts_label/ len(original_ytrain)) print(test_counts_label/ len(original_ytest)) Label Distributions: [2.99827076 0.00172924) [8.99827952 @.60172048) np-unique(df[ ‘Class’ ].values, return_counts= True) (array([@, 1], dtype=inte4), array([284315, 492], dtype=int64)) np. unique(original_ytrain, return_counts= True) (array([@, 1], dtype=int64), array([227452, 394], dtype=int6a)) df.shape (284807, 31) clonleredi card fraud detection biv classification model Jpynb 128 3127128, 1026 PM In [135]: out [135] In [87]: out [87]: In [88]: out (88): In [89]: out (89): localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook # Since our classes are highly skewed we should make them equivalent in order # Lets shuffle the data before creating the subsamples df = d#.sample(frac=1) # amount of fraud classes 492 rows. fraud_df = df.loc[df['Class'] == 1] non_fraud_df = df.loc[df['Class'] ][:492] normal_distributed_df = pd.concat([fraud_éf, non_fraud_df]) # Shuffle dataframe rows new_df = normal_distributed_df.sample(frac=1, random_state=42) new_df.head() vi v2 vs va vs ve wr v 91406 0.797003 0.683645 1.572596 0.348139 0036879 0.353904 0.936242 0.15508 15508 -21.885434 12,950505 -24,008872 6.203314 -16.466009 -4.459047 -16,519836 14,59556 268151 0.794100 0.073718 | -0.794628 -1.134463 0.512720 -0.253948 -0.000552 0.34069 42674 -7.896885 5.381020 8.451162 7.969928 7.862419 2.376820 -11.940723 5.05135 156988 0.745153 2.809209 5.825406 5.835566 0.512320 -0.615622 -2.916576 0.77671 5 rows x 31 columns new_d#. shape (984, 31) df[ ‘Class’ ].value_counts() @ 284315 1 492 Name: Class, dtype: intoa new_df['Class"] 281215 251881 260023 42007 92777 1208371 6329 1 116645 8 6832001 644601 Name: Class, Lengtl 1 984, dtype: intea .clonleredi card raud detection bivariate classification model yn 826 3127128, 1026 PM credit card aus detection bivariate classification mode - Jupyter Notebook In [98]: print("Distribution of class in the new df") print(new_df[ ‘Class’ ].value_counts()) print(new_df[ ‘Class’ ].value_counts()/len(new_d¥)) sns.countplot(x= ‘Class",data= new_df, palette=colors) plt.title( ‘Equally Distributed Classes’, fontsize=14) plt.show() Distribution of class in the new df @ 492 1 492 Name: Class, dtype: intea @ 85 1 0.5 Name: Class, dtype: floatea Equally Distributed Classes 500 400 300 count 200 100 Class locathost 8888inotebooks/Documentst st card fraud detectioncredit card fraud detecton bv classification model Jpynb 926 3127128, 1026 PM credit card aus detection bivariate classification mode - Jupyter Notebook In [108]: f , (axl, ax2) = plt.subplots(1,2, figsize = (15,10) # Correlation of entire datasetbbbb corr_i = df.corr() k= 10 cols = corr_t.nlargest(k, 'Class')['Class'].index cm = np.corrcoef(df[cols].values.T) sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'siz corr = new_df.corr() k = 10 #nunber of variables for heatmap cols = corr.nlargest(k, ‘Class*)[ Class" ] andex cn = np.corrcoef (new_df{cols).values.T) fm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt: plt.shon() 2F", annot_kws= In [32]: # Find variables which are highly correlated with the class object corr[corr[ ‘Class’ ]<@]["Class' ].sort_values() .head(3) out[32]: va7 -@.326481 via -0.302544 vi2_ -@.260593 Name: Class, dtype: floated clonleredi card fraud detection biv localhost 8888inotebooks/Documents/Credk card fraud de lassiieation model ipynb 10126 3127128, 1026 PM In [122]: In [125]: In [139]: In [142]: In [146]: out(146) localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook # Boxplot for variables which are highly correlater for new_df corr_1 = new_df.corr() k= 10 cols = corr_i.nlargest(k, 'Class’)['Class'].index.tolist() cols.remove('Class') print (cols) print(len(cols)) ['va', "van", ‘v2", "vag", ‘v2e", ‘vaa", ‘v2e", ‘'v26", 'V27"] 8 for i in cols: print(i) va van v2 vis v2e vai v28 v26 v27 x = pd.DataFrane(new_d#) x.reset_index(inplace = True) x.head() indox vi ve vs va vs ve vr © 97408 0.797093 0.683645 1.572505 0.348139 0.036879 0.353094 0.936242 -0.15 1 18506 -21.885434 12.990505 -24.098672 6.203514 -16.466009 4.459642 -16.519836 14.53 2 268151 0.794100 0.073718 -0.794628 -1.134463 0.512720 0.233048 -0.000552 0.34 3 42674 7.896886 5.381020 -8.451162 7.963028 -7.862419 -2.376820 -11.949723 5.05 4 156988 0.745153 2.809299 5.825406 5.835565 0.512320 -0.615622 -2.916578 0.77 5 rows x 32 columns clonleredi card raud detection bivariate classification model yn 26 3127128, 1026 PM credit card aus detection bivariate classification mode - Jupyter Notebook In [148]: sns.boxplot(x="Class", y='V4", data=x, palette=colors) set_title(‘V17 vs Class Negative Correlation’) NameError Traceback (most recent call last) Cell In[148], line 2 1 sns.boxplot(x="Class", y='V4", data=x, palette=colors) -> 2 set_title('V17 vs Class Negative Correlation’) NameError: name ‘set_title’ is not defined 12.5 —___— 10.0 15 ‘ 5.0 s } 25 ~~ T -25 ¢ -5.0 ° 1 Gass In [171]: i, j = divmod(o,3) localhost 8888inotebooks/Documents/Credk card fraud de .clonleredi card raud detection bivariate classification model yn 1226 3127128, 1026 PM ‘credit ard fraud detection bivariate classification model -Jupyter Notebook In [186]: y= @ , axes = plt.subplots(round(len(cols)/3), 3, figsize =(15,20)) for i in range(len(cols)): 2, j= divmod(y, 3) #f, ax = plt.subplots(figsize=(5, 3)) Fig = sns.boxplot(x="Class", y=cols[i], data-new_df, palette=colors, ax=ax yeyel plt.show() ms aaa zr ; i “| 4 EB = t cae ’ es o ~~. ' 4 = oe di 7 3 : mee Ce =e - a c =. s ‘ f { 1926 3127128, 1026 PM ‘credit ard fraud detection bivariate classification model -Jupyter Notebook In [33]: f, axes = plt.subplots(ncols=4, figsize=(20,4)) sns.boxplot(x="Class", y="Vi7", data=new_df, palette-colors, ax=axes[@]) axes[0].set_title('V17 vs Class Negative Correlation’) sns.boxplot(x="Class", y="Vi4", datasnew_df, palette=colors, ax=axes[1]) axes[1].set_title('Vi4 vs Class Negative Correlation’) sns.boxplot(x="Class", y="V12", data=new_df, palette=colors, ax=axes[2]) axes[2].set_title('v12 vs Class Negative Correlation’) sns.boxplot(x="Class", y="Vie", data=new_df, palette-colors, ax=axes[3]) axes[3].set_title('V10 vs Class Negative Correlation’) pltshow() In [158]: # Find variables which are highly correlated with the class object corr[corr[ ‘Class']>@]['Class'].sort_values().tail() Out[158]: v19 8.255667 v2 0.505503 vin 0.688873 Ve. 7oaasa Class 1.900000 Name: Class, dtype: loates In [159]: . | L 4 oe = locas: 8888inotebooks/Documents/Credi card fraud delection/redi card fraud detection bivariate classification mod ‘ipynb 14126 3127128, 1026 PM In [203]: In [218]: out (218) localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook from scipy.stats import norm y=e f, axes = plt.subplots(round(len(cols)/3), 3, figsize =(20,15)) for i in range(len(cols)): 2,j = divmod(y, 3) x = new_df[new_df['Class"]=+1][cols[i]].values sns.distplot(x, fitsnorm, color='#F88861', ax = axes[z,j]) axes[z,j].set_title('{} Distribution \n (Fraud Transactions)’ .format(cols[ ye=1 Users\Ankit \AppData\Local \Temp\ipykernel_1064\3313427548. py ing: UserWarn “distplot” is a deprecated function and will be removed in seaborn v.14. 2. Please adapt your code to use either “displot’ (a figure-level function wi th similar flexibility) or “histplot’ (an axes-level function for histogram s). For a guide to updating your code to use the new functions, please see https: //gist. github. com/mwaskom/de44147ed2974457ad637275@bbe5751 (https: // gist. github.com/mwaskom/ded4147ed2974457ad637275@bbeS751) sns.distplot(x, fit=norm, color="#FB8861', ax = axes[z,j]) C:\Users \Ankit\AppData\Local \Tenp \ipykernel_1064\3313427548.py:9: Userwarn ing: new_d#. columns Index(['V1', "V2", 'V3', ‘V4', "V5", 'V6', 'V7', ‘VB, ‘'v9', ‘V1e', ‘Vi1', ‘vi2', 'va3", "via", 'V15', 'V16", 'Vi7', ‘vi8', 'V19", "v2e", "V21", "v2", 'v23', ‘v2a", 'V25", 'v26", 'V27', 'V28', ‘Class’, ‘scaled tim *scaled_amount'], dtype="'object') ctlonleredi card fraud detection biv classification model Jpynb 19126 3127128, 1026 PM In [226]: out [226] localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook x = new_df[new_df[ 'Class']==1]['V14"].values sns.distplot(x, fit=norm, color='#F88861') Userwarnin C:\Users \Ankit \AppData\Local \Temp \ipykernel_1064\3492287139. py: B “distplot’ is a deprecated function and will be removed in seaborn vd.14.0. Please adapt your code to use either “displot” (a figure-level function with similar flexibility) or “histplot” (an axes-level function for histograms). For a guide to updating your code to use the new functions, please see https: //gist. github. com/mwaskon/de44147ed2974457ad637275@bbe5751 (https: //gis +t. github. com/mwaskom/de44147ed2974457ad6372750bbe5751) sns.distplot(x, fitsnorm, color="#F68861") V14 Removing Outliers (Highest Negative Correlated with Labels) vi4_fraud = new_df{new_df['Class’] == 1]['Vi4"].values 425, q75 = np.percentile(vi4_fraud,25), np.percentile(vi4_fraud, 75) print( ‘Quartile 25: (} | Quartile 75: (}'.format(q25, q75)) vid_igr = q75 - q25 print('iqr: {)".format(vi4_iqr)) v14_cut_off = vid_igr * 1.5 vi4_lower, vi4_upper = q25 - v14_cut_off, q75 + v14_cut_off print('Cut off: (}'.format(vi4_cut_off)) print('Vi4 Lower: {}'.format(vi14_lower)) print('Vi4 Upper: {}'.format(v14_upper)) outliers = [x for x in v14_fraud if x v14_upper] print('Feature V14 Outliers for Fraud Cases: {}'.format(len(outliers))) print('Vl@ outliers:{}'.format (outliers) new_df = new_df.drop(new_df[(new_df['Vi4"] > vi4_upper) | (new_df["vi4"] < vi4 print(* * 44) Quartile 25: -9.692722964972386 | Quartile 75: -4.282820849486865 igr: 5.409902115485521 Cut off: 8.124853173228282 V14 Lower: -17.807576138200666 V4 Upper: 3.8320323237414167 Feature V14 Outliers for Fraud Cases: & v10 outliers: [] clonleredi card fraud detection biv classification model Jpynb 19126 3127128, 1026 PM In [242]: In[]: localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook #- -> V12 removing outliers from fraud transactions vi2_fraud = new_df{new_df['Class"] ==1]['Vi2"].values 425, q75 = np.percentile(vi2_fraud,25), np-percentile(v12_fraud,75) viz ign = 475-425 print ("TQ for V12 is {)".format(v12_iqr)) v_12_cutt_off = vi2_igr*1.5 v_12 lower, v_12_upper = q25- v_12_cutt_off, q75+v_12_cutt_off outliers = [x for x in v12_fraud if x < v_12 lower or x > v_12_upper] print("Outliers for vi2 is {}".format(outliers)) new_df = new_df.drop(new_df[(new_df['V12"]< v_12_lower) | (new_df['V12"]> v_a2 # Removing outliers V10 Feature v1e_fraud = new_df[new_df['Class’] ==1]['V10"].values 25, 475 = np.percentile(vie_fraud, 25), np.percentile(vie_fraud, 75) v1e_igr = q75 - q25 v1e_cut_off = v1@_igr * 1.5 v1@_lower, vi@_upper = q25 - v1@_cut_off, q75 + v1e_cut_otf print('Vi@ Lower: {}'.format(v1@_lower)) print('Vi@ Upper: {}'.format(v10_upper)) outliers = [x for x in vi@_fraud if x < v1@_lower or x > v1@_upper] print('vi@ outliers: {}'.format(outliers)) print(‘Feature V1@ Outliers for Fraud Cases: {}'.format(len(outliers))) new_df = new_df.drop(new_d#[(new_d#['V10"] > v1@_upper) | (new_df['Vie"] < vie print(‘Number of Instances after outliers removal: {}'.format(len(new_df))) IQR for V12 is 5.780002635714835 Outliers for vi2 is [-18.5536970096458, -18.6837146333443, -18.4311310279993, -18.0475965708216] V19 Lower: -14.89885463232024 V0 Upper: 4.92033495834214 v0 outliers: [-22.1870885620007, -15.1241628144947, -16.7460441053944, -15.5 637913387301, -15.2399619587112, -15.1237521803455, -17.1415136412892, -20.94 91915543611, -16.6496281595399, -15.2399619587112, -14.9246547735487, -15.346 0988468775, -18.2711681738888, -24.5882624372475, -18.9132433348732, -22.1870 885620007, -14.9246547735487, -15.2318333653018, -16.3035376590131, -15.56379 13387301, -16.2556117491401, -24.4031849699728, -16.6011969664137, -19.836148 851696, -23.2282548357516, -22.1870885620007, -22.1870885620007] Feature V10 Outliers for Fraud Cases: 27 Number of Instances after outliers removal: 945 Why to use tSNE + SNE algorithm can pretty accurately cluster the cases that were fraud and non-fraud in our dataset. + Although the subsample is pretty small, the t-SNE algorithm is able to detect clusters pretty accurately in every scenario (| shuffle the dataset before running t-SNE) .clonleredi card raud detection bivariate classification model yn 20126 3127128, 1026 PM credit card aus detection bivariate classification mode - Jupyter Notebook + This gives us an indication that further predictive models will perform pretty well in In [262]: # New_df is from the random undersample data (fewer instances) import time x y new_df.drop('Class', axis=1) new_df['Class"] # T-SNE ImpLementation te = time.time() X_reduced_tsne = TSNE(n_components=2, random_state=42).fit_transform(X.values) t1 = time.time() print("T-SNE took {:.2) s".format(t1 - t@)) # PCA Implementation to = time.time() X_reduced_pca = PCA(n_components=2, random_state=42) .fit_transform(X.values) 1 = time.time() print ("PCA took {:.2} s".format(t1 - t@)) # Truncatedsvp t@ = time.time() X_reduced_svd = TruncatedSVD(n_components=2, algorith tl = time.time() print("Truncated SVD took {:.2} s".format(t1 - t@)) randomized’, random_st T-SNE took 1.1e+01 s PCA took 0.031 s Truncated SVD took 2.016 s In [267]: X_reduced_tsne[:0] out(267]: array(E], shape=(@, 2), dtype-float32) localhost 8888inotebooks/Documents/Credk card fraud de .clonleredi card raud detection bivariate classification model yn 21128 3127128, 1026 PM In [270]: In [271]: In [273]: localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook f, (axl, ax2, ax3) = plt-subplots(1,3, figsize= (24,6)) #.suptitle(‘Clusters using Dimensionality Reduction’, fontsize=14) blue_patch red_patch mpatches .Patch(colo mpatches Patch (color: ‘WOA@AFF', label='No Fraud’) "¥AF@000', label='Fraud") # t-SNE scatter plot axl.scatter(X_reduced_tsne[:,@], X_reduced_tsne[:,1], c=(y = axl.scatter(X_reduced_tsne[:,@], X_reduced_tsne[:,1], c=(y axl.set_title('t-SNE', fontsize=14) # PCA scatter plot ax2.scatter(X_reduced_pea[:,0], x_reduced_pea[:,1], coolwara ax2.scatter(Xx_reduced_pea[:,2], X_reduced_pea[:,1], coolwara ax2.set_title("Pca’, Fontsize=14) ax2.grid(True) ax2. legend(handles=[blue_patch, red_patch]) # Truncatedsvo scatter plot ax3.scatter(X_reduced_svd[:,0], X_reduced_svd coolwara ax3.scatter(X_reduced_svd[:,0], X_reduced_svd[: coolwarn ax3.set_title(‘Truncated SVD", fontsiz ) ax3.grid(True) ax3.legend(handles=[blue_patch, red_patch]) plt.show() / / hae ff y = new_df['Class'] # Our data is already scaled we should split our training and test sets from sklearn.model_selection import train_test_split # This is explicitly used for undersampLing. X_train,x_test, y train, y test = train_test_split(X, y, test_siz 2, randon clonleredi card raud detection bivariate classification model yn 2226 3127128, 1026 PM In [274]: In [275]: In [281]: localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook # Turn the values into an array for feeding the classification algorithms. X_train = X_train.values X_test = X_test.values y_train = y_train.values yitest = y test.values # Let's implement simple classifiers classifiers = { “LogisiticRegression": LogisticRegression(), "kNearest": KNeighborsClassifier(), “Support Vector Classifier": SVC(), “DecisionTreeClassifier": DecisionTreeClassifier() # Wow our scores are getting even high scores even when applying cross vatidat from sklearn.model_selection import cross_val_score for key, classifier in classifiers. items(): classifier.fit(x_train, y_train) training_score = cross_val_score(classifier, X_train, y_train, cv=5) print("Classifiers: ", classifier._class_._name_, "Has a training scor > Classifiers: LogisticRegression Has a training score of 94.@ % accuracy scor Classifiers: KkNeighborsClassifier Has a training score of 94.0 % accuracy sc Classifiers Classifiers: score SVC Has a training score of 93.0 % accuracy score DecisionTreeClassifier Has a training score of 91.0 % accuracy clonleredi card raud detection bivariate classification model yn 2226 3127128, 1026 PM In [280]: In [ localhost 8888inotebooks/Documents/Credk card fraud de credit card aus detection bivariate classification mode - Jupyter Notebook # our scores are getting even high scores even when applying cross validation. from sklearn.model_selection import cross_val_score for key, classifier in classifiers. items(): classifier.fit(x_train, y_train) r2_test = round(classifier. score(x_test,y_test),2) r2_train = round(classifier.score(x_train,y_train),2) training_score = cross_val_score(classifier, X_train, y_train, cv=5) print("Classifier print("Classifier print("Classifier: » Classifier. class_._name_, “Has a r2 of {} for tr » Classifier, class _._name_, "Has a r2 of {} for te ", classifier. class. _name_, "Has a training scor > Classifier: LogisticRegression Has a r2 of @.96 for train Classifier: LogisticRegression Has a r2 of @.94 for test Classifiers: LogisticRegression Has a training score of 94.@ % accuracy scor Classifier: KNeighborsClassifier Has a r2 of 0.95 for train Classifier: KNeighborsClassifier Has a r2 of 0.93 for test Classifiers: kNeighborsClassifier Has a training score of 94.0 % accuracy sc Classifier: SVC Has a r2 of 0.95 for train Classifier: SVC Has a r2 of 0.92 for test Classifiers: SVC Has a training score of 93.0 % accuracy score Classifier: DecisionTreeClassifier Has a r2 of 1.@ for train Classifier: DecisionTreeClassifier Has a r2 of @.87 for test Classifiers: DecisionTreeClassifier Has a training score of 90.0 % accuracy score LogisticRegression().fit() .clonleredi card raud detection bivariate classification model yn 24128 3127128, 1026 PM ‘credit ard fraud detection bivariate classification model -Jupyter Notebook In [282]: # Use GridSearchcV to find the best parameters. from sklearn.model_selection import GridSearchCv # Logistic Regression log_reg_params = {"penalty’ By, * grid_log_reg = GridSearchCv(LogisticRegression(), log_reg_parans) grid_log_reg.fit(X_train, y_train) # We automatically get the Logistic regression with the best parameters. log_reg = grid_log_reg.best_estimator_ knears_params = {"n_neighbors": list(range(2,5,1)), ‘algorithm’: [‘auto’, ‘bal grid_knears = GridSearchCV(KNeighborsClassifier(), knears_parans) grid_knears.fit(X_train, y train) # Kiears best estimator knears_neighbors = grid_knears.best_estimator_ # Support Vector Classifier svc_params = {'C': [0.5, @.7, 0.9, 1], ‘kernel’: ['rbf', ‘poly’, ‘sigmoid’, grid_sve = GridSearchcv(svc(), svc_params) grid_svc.fit(X_train, y_train) # SVC best estimator svc = grid_svc.best_estinator_ # DecisionTree Classifier ‘tree_params = {"criterion": ["gini", “entropy"], "max_depth” “min_samples_leaf": List(range(5,7,1))} grid_tree = GridSearchCv(DecisiontreeClassifier(), tree_params) grid_tree.fit(x_train, y_train) # tree best estimator tree_clf = grid_tree.best_estimator_ locahost 8888inotebooks/Documents/Credi card fraud delection/redi card fraud detection bivariate classiction model pynb (0.001, 0.01, @.1, 1, 18, 100, List (range(2,4,1 28126 3127128, 1026 PM In [283]: ‘credit card fraud detection bivariate classification model -Jupyter Notebook ansccacel Hut gy steuse/ muuuses/ pi epi eEsaanig Hume) Please also refer to the documentation for alternative solver options: https: //scikit-learn.org/stable/modules/linear_model .html#logistic-reg ression (https: //scikit-learn.org/stable/modules/linear_model .html#logisti c-regression) n_iter_i = _check_optimize_result( :\Users (Ankit \Appbata \Local \Prograns\Python\Python311\Lib\site-packages\s klearn\linear_model\_logistic.py:458: ConvergenceWarning: Ibfgs failed to converge (status=1): STOP: TOTAL NO. of ITERATIONS REACHED LIMIT. Increase the number of iterations (max_iter) or scale the data as shown i ns https: //scikit-learn.org/stable/modules/preprocessing.html (https: //sc ikit-learn.org/stable/modules/preprocessing-htm1) Please also refer to the documentation for alternative solver options: https: //scikit-learn.org/stable/modules/linear_model .html#logistic-reg ression (https: //scikit-learn.org/stable/modules/linear_model .html#logisti c-regression) n_iter_i = _check_optimize_result( ¥ # Overfitting Case log_reg_score = cross_val_score(log_reg, X_train, y_train, cv=5) print('Logistic Regression Cross Validation Score: ", round(log_reg_score.mear knears_score = cross_val_score(knears_neighbors, X_train, y_train, cv=5) print("Knears Neighbors Cross Validation Score’, round(knears_score.mean() * 1 svc_score = cross_val_score(svc, X_train, y_train, cv=5) print('Support Vector Classifier Cross Validation Score’, round(svc_score.mear tree_score = cross_val_score(tree_clf, X_train, y_train, cv=5) print('DecisionTree Classifier Cross Validation Score’, round(tree_score.mean( Logistic Regression Cross Validation Score: 94.44% Knears Neighbors Cross Validation Score 93.65% Support Vector Classifier Cross Validation Score 94.05% DecisionTree Classifier Cross Validation Score 92.99% locahost 8888inotebooks/Documents/Credi card fraud delection/redi card fraud detection bivariate classiction model pynb 26126

You might also like