[SPARK-10028][MLLIB][PYTHON] Add Python API for PrefixSpan

Author: Yu ISHIKAWA <yuu.ishikawa@gmail.com> Closes #9469 from yu-iskw/SPARK-10028.
author: Yu ISHIKAWA <yuu.ishikawa@gmail.com> 2015-11-04 15:28:19 -0800
committer: Xiangrui Meng <meng@databricks.com> 2015-11-04 15:28:19 -0800
commit: 411ff6afb485c9d8cfc667c9346f836f2529ea9f (patch)
tree: dd950d6387ddb6dc24980d4536e4db08d06d4456 /mllib
parent: 1b6a5d4af9691c3f7f3ebee3146dc13d12a0e047 (diff)
download: spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.tar.gz
spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.tar.bz2
spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.zip
2 files changed, 54 insertions, 1 deletions
diff --git a/mllib/src/main/scala/org/apache/spark/mllib/api/python/PrefixSpanModelWrapper.scala b/mllib/src/main/scala/org/apache/spark/mllib/api/python/PrefixSpanModelWrapper.scala
new file mode 100644
index 0000000000..0027602a04
--- /dev/null
+++ b/mllib/src/main/scala/org/apache/spark/mllib/api/python/PrefixSpanModelWrapper.scala
@@ -0,0 +1,32 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *    http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.spark.mllib.api.python
+
+import org.apache.spark.mllib.fpm.PrefixSpanModel
+import org.apache.spark.rdd.RDD
+
+/**
+ * A Wrapper of PrefixSpanModel to provide helper method for Python
+ */
+private[python] class PrefixSpanModelWrapper(model: PrefixSpanModel[Any])
+  extends PrefixSpanModel(model.freqSequences) {
+
+  def getFreqSequences: RDD[Array[Any]] = {
+    SerDe.fromTuple2RDD(model.freqSequences.map(x => (x.javaSequence, x.freq)))
+  }
+}
diff --git a/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala b/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala
index 21e55938fa..40c41806cd 100644
--- a/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala
+++ b/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala
@@ -35,7 +35,7 @@ import org.apache.spark.mllib.classification._
 import org.apache.spark.mllib.clustering._
 import org.apache.spark.mllib.evaluation.RankingMetrics
 import org.apache.spark.mllib.feature._
-import org.apache.spark.mllib.fpm.{FPGrowth, FPGrowthModel}
+import org.apache.spark.mllib.fpm.{FPGrowth, FPGrowthModel, PrefixSpan}
 import org.apache.spark.mllib.linalg._
 import org.apache.spark.mllib.linalg.distributed._
 import org.apache.spark.mllib.optimization._
@@ -558,6 +558,27 @@ private[python] class PythonMLLibAPI extends Serializable {
   }
 
   /**
+   * Java stub for Python mllib PrefixSpan.train().  This stub returns a handle
+   * to the Java object instead of the content of the Java object.  Extra care
+   * needs to be taken in the Python code to ensure it gets freed on exit; see
+   * the Py4J documentation.
+   */
+  def trainPrefixSpanModel(
+      data: JavaRDD[java.util.ArrayList[java.util.ArrayList[Any]]],
+      minSupport: Double,
+      maxPatternLength: Int,
+      localProjDBSize: Int ): PrefixSpanModelWrapper = {
+    val prefixSpan = new PrefixSpan()
+      .setMinSupport(minSupport)
+      .setMaxPatternLength(maxPatternLength)
+      .setMaxLocalProjDBSize(localProjDBSize)
+
+    val trainData = data.rdd.map(_.asScala.toArray.map(_.asScala.toArray))
+    val model = prefixSpan.run(trainData)
+    new PrefixSpanModelWrapper(model)
+  }
+
+  /**
    * Java stub for Normalizer.transform()
    */
   def normalizeVector(p: Double, vector: Vector): Vector = {
author	Yu ISHIKAWA <yuu.ishikawa@gmail.com>	2015-11-04 15:28:19 -0800
committer	Xiangrui Meng <meng@databricks.com>	2015-11-04 15:28:19 -0800
commit	411ff6afb485c9d8cfc667c9346f836f2529ea9f (patch)
tree	dd950d6387ddb6dc24980d4536e4db08d06d4456 /mllib
parent	1b6a5d4af9691c3f7f3ebee3146dc13d12a0e047 (diff)
download	spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.tar.gz spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.tar.bz2 spark-411ff6afb485c9d8cfc667c9346f836f2529ea9f.zip