Dimensionality reduction of IRIS data using PCA and JSAT
Code: "jsat_pca.py". Programming language: Python DMelt Version 2.2. Last modified: 03/03/2021. License: Pro
https://datamelt.org/code/cache/jsat_pca_2953.py
To run this script using the DMelt IDE, copy the above URL link to the menu [File]→[Read script from URL] of the DMelt IDE.


from java.io import File
from jsat.classifiers import DataPoint,ClassificationDataSet
from jsat.datatransform import PCA,DataTransform,ZeroMeanTransform
from jsat import ARFFLoader,DataSet

print "Download iris_org.arff"
from jhplot import *
print Web.get("https://datamelt.org/examples/data/iris_org.arff")
fi=File("iris_org.arff")
dataSet = ARFFLoader.loadArffFile(fi)
# We specify '0' as the class we would like to make the target class. 
cData = ClassificationDataSet(dataSet, 0)

# The IRIS data set has 4 numerical attributes, unfortunately humans are not good at visualizing 4 dimensional things.
# Instead, we can reduce the dimensionality down to two. 
# PCA needs the data samples to have a mean of ZERO, so we need a transform to ensue this property as well
zeroMean = ZeroMeanTransform(cData);
cData.applyTransform(zeroMean);

# PCA is a transform that attempts to reduce the dimensionality while maintaining all the variance in the data. 
# PCA also allows us to specify the exact number of dimensions we would like 
pca = PCA(cData, 2, 1e-9);
        
# We can now apply the transformations to our data set
cData.applyTransform(pca);

c1 = SPlot()
c1.visible()
c1.setAutoRange()
c1.setMarksStyle('various')
#c1.setConnected(1, 0)
c1.setNameX('X')
c1.setNameY('Y')
# output
for i in range(cData.getSampleSize()):
  dataPoint = cData.getDataPoint(i)  
  category = cData.getDataPointCategory(i) # get category 
  vec=dataPoint.getNumericalValues();
  c1.addPoint(category,vec.get(0),vec.get(1),1)
  if (i%10==0): 
                c1.update()
                print i,dataPoint,category


Ads help maintain this website.