forked from PacktPublishing/Machine-Learning-and-Data-Science-with-Python-A-Complete-Beginners-Guide
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathfeature_selection_univariate.py
More file actions
41 lines (26 loc) · 910 Bytes
/
Copy pathfeature_selection_univariate.py
File metadata and controls
41 lines (26 loc) · 910 Bytes
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
# -*- coding: utf-8 -*-
"""
@author: abhilash
"""
#load the csv file using read_csv function of pandas library
from pandas import read_csv
from numpy import set_printoptions
from sklearn.feature_selection import SelectKBest
from sklearn.feature_selection import chi2
filename = 'pima-indians-diabetes.csv'
#url = 'https://myfilecsv.com/test.csv'
names = ['preg', 'plas', 'pres', 'skin', 'test', 'mass', 'pedi', 'age', 'class']
dataframe = read_csv(filename, names=names)
array = dataframe.values
#splitting the array to input and output
X = array[:,0:8]
Y = array[:,8]
#feature selection
test = SelectKBest(score_func=chi2, k=4)
fit = test.fit(X, Y)
#print the scores for the features
set_printoptions(precision=3)
print(fit.scores_)
#print the first five rows of the best 4 features (Columns) selected
features = fit.transform(X)
print(features[0:5,:])