score:-1

Accepted answer

I got the solution I was able to do so.

 import scala.beans.BeanInfo
 import org.apache.spark.{SparkConf, SparkContext}
 import org.apache.spark.ml.Pipeline
 import org.apache.spark.ml.classification.LogisticRegression
 import org.apache.spark.ml.feature.{HashingTF, Tokenizer}
 import org.apache.spark.mllib.linalg.Vector
 import org.apache.spark.sql.{Row, SQLContext}
 import org.apache.spark.mllib.linalg.Vectors
 import org.apache.spark.ml.attribute.NominalAttribute
 import org.apache.spark.sql.Row
 import org.apache.spark.sql.types.{StructType,StructField,StringType}
 case class LabeledDocument(Userid: Double, Date: String, label: Double)
 val trainingData = spark.read.option("inferSchema", true).csv("/root/Predictiondata10.csv").toDF("Userid","Date","label").toDF().as[LabeledDocument]
 import org.apache.spark.ml.feature.StringIndexer
 import org.apache.spark.ml.feature.VectorAssembler
 val DateIndexer = new StringIndexer().setInputCol("Date").setOutputCol("DateCat")
 val indexed = DateIndexer.fit(trainingData).transform(trainingData)
 val assembler = new VectorAssembler().setInputCols(Array("DateCat", "Userid")).setOutputCol("rawfeatures")
 val output = assembler.transform(indexed)
 val rows = output.select("Userid","Date","label","DateCat","rawfeatures").collect()
 val asTuple=rows.map(a=>(a.getInt(0),a.getString(1),a.getDouble(2),a.getDouble(3),a(4).toString()))
 val r2 = sc.parallelize(asTuple).toDF("Userid","Date","label","DateCat","rawfeatures")
 val Array(training, testData) = r2.randomSplit(Array(0.7, 0.3))
 import org.apache.spark.ml.feature.{HashingTF, Tokenizer}
 val tokenizer = new Tokenizer().setInputCol("rawfeatures").setOutputCol("words")
 val hashingTF = new HashingTF().setNumFeatures(1000).setInputCol(tokenizer.getOutputCol).setOutputCol("features")
 import org.apache.spark.ml.regression.LinearRegression
 val lr = new LinearRegression().setMaxIter(100).setRegParam(0.001).setElasticNetParam(0.0001)
 val pipeline = new Pipeline().setStages(Array(tokenizer, hashingTF, lr))
 val model = pipeline.fit(training.toDF())
 model.transform(testData.toDF()).show()

Related Query

More Query from same tag