[SPARK-8467][MLlib][PySpark] Add LDAModel.describeTopics() in Python

yu-iskw · yu-iskw · commit 7237c36b5a33 · 2015-11-02T09:40:51.000-08:00
diff --git a/mllib/src/main/scala/org/apache/spark/mllib/api/python/LDAModelWrapper.scala b/mllib/src/main/scala/org/apache/spark/mllib/api/python/LDAModelWrapper.scala
@@ -0,0 +1,46 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *    http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.spark.mllib.api.python
+
+import org.apache.spark.SparkContext
+import org.apache.spark.mllib.clustering.LDAModel
+import org.apache.spark.mllib.linalg.Matrix
+
+/**
+ * Wrapper around LDAModel to provide helper methods in Python
+ */
+private[python] class LDAModelWrapper(model: LDAModel) {
+
+  def topicsMatrix(): Matrix = model.topicsMatrix
+
+  def vocabSize(): Int = model.vocabSize
+
+  def describeTopics(): java.util.List[Array[Any]] = describeTopics(this.model.vocabSize)
+
+  def describeTopics(maxTermsPerTopic: Int): java.util.List[Array[Any]] = {
+    import scala.collection.JavaConversions._
+
+    val javaList: java.util.List[Array[Any]] =
+      model.describeTopics(maxTermsPerTopic).map { case (terms, termWeights) =>
+        var array = Array.empty[Any]
+        Array.empty[Any] ++ terms ++ termWeights
+      }.toList
+    javaList
+  }
+
+  def save(sc: SparkContext, path: String): Unit = model.save(sc, path)
+}
diff --git a/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala b/mllib/src/main/scala/org/apache/spark/mllib/api/python/PythonMLLibAPI.scala
@@ -517,7 +517,7 @@ private[python] class PythonMLLibAPI extends Serializable {
       topicConcentration: Double,
       seed: java.lang.Long,
       checkpointInterval: Int,
-      optimizer: String): LDAModel = {
+      optimizer: String): LDAModelWrapper = {
     val algo = new LDA()
       .setK(k)
       .setMaxIterations(maxIterations)
@@ -535,7 +535,16 @@ private[python] class PythonMLLibAPI extends Serializable {
         case _ => throw new IllegalArgumentException("input values contains invalid type value.")
       }
     }
-    algo.run(documents)
+    val model = algo.run(documents)
+    new LDAModelWrapper(model)
+  }
+
+  /**
+   * Load a LDA model
+   */
+  def loadLDAModel(jsc: JavaSparkContext, path: String): LDAModelWrapper = {
+    val model = DistributedLDAModel.load(jsc.sc, path)
+    new LDAModelWrapper(model)
   }
 
 
diff --git a/python/pyspark/mllib/clustering.py b/python/pyspark/mllib/clustering.py
@@ -667,7 +667,7 @@ def predictOnValues(self, dstream):
         return dstream.mapValues(lambda x: self._model.predict(x))
 
 
-class LDAModel(JavaModelWrapper):
+class LDAModel(JavaModelWrapper, JavaSaveable, Loader):
 
     """ A clustering model derived from the LDA method.
 
@@ -690,6 +690,21 @@ class LDAModel(JavaModelWrapper):
     >>> model = LDA.train(rdd, k=2)
     >>> model.vocabSize()
     2
+    >>> topics = model.describeTopics()
+    >>> len(topics)
+    2
+    >>> len(list(topics[0])[0])
+    2
+    >>> len(list(topics[0])[1])
+    2
+    >>> topics = model.describeTopics(1)
+    >>> len(topics)
+    2
+    >>> len(list(topics[0])[0])
+    1
+    >>> len(list(topics[0])[1])
+    1
+
     >>> topics = model.topicsMatrix()
     >>> topics_expect = array([[0.5,  0.5], [0.5, 0.5]])
     >>> assert_almost_equal(topics, topics_expect, 1)
@@ -720,18 +735,27 @@ def vocabSize(self):
         """Vocabulary size (number of terms or terms in the vocabulary)"""
         return self.call("vocabSize")
 
-    @since('1.5.0')
-    def save(self, sc, path):
-        """Save the LDAModel on to disk.
+    def describeTopics(self, maxTermsPerTopic=None):
+        """Return the topics described by weighted terms.
 
-        :param sc: SparkContext
-        :param path: str, path to where the model needs to be stored.
+        WARNING: If vocabSize and k are large, this can return a large object!
         """
-        if not isinstance(sc, SparkContext):
-            raise TypeError("sc should be a SparkContext, got type %s" % type(sc))
-        if not isinstance(path, basestring):
-            raise TypeError("path should be a basestring, got type %s" % type(path))
-        self._java_model.save(sc._jsc.sc(), path)
+        if maxTermsPerTopic is None:
+            topics = self.call("describeTopics")
+        else:
+            topics = self.call("describeTopics", maxTermsPerTopic)
+
+        # Converts the result to make the format similar to Scala.
+        # The returned value is mixed up with topics and topi weights.
+        converted = []
+        for elms in [list(elms) for elms in topics]:
+            half_len = int(len(elms) / 2)
+            topics = elms[:half_len]
+            topicWeights = elms[(-1 * half_len):]
+            if len(topics) != len(topicWeights):
+                raise TypeError("Something wrong with a return value: %s" % (topics))
+            converted.append((topics, topicWeights))
+        return converted
 
     @classmethod
     @since('1.5.0')
@@ -745,9 +769,8 @@ def load(cls, sc, path):
             raise TypeError("sc should be a SparkContext, got type %s" % type(sc))
         if not isinstance(path, basestring):
             raise TypeError("path should be a basestring, got type %s" % type(path))
-        java_model = sc._jvm.org.apache.spark.mllib.clustering.DistributedLDAModel.load(
-            sc._jsc.sc(), path)
-        return cls(java_model)
+        wrapper_model = callMLlibFunc("loadLDAModel", sc, path)
+        return LDAModel(wrapper_model)
 
 
 class LDA(object):
@@ -773,10 +796,10 @@ def train(cls, rdd, k=10, maxIterations=20, docConcentration=-1.0,
         :param optimizer:           LDAOptimizer used to perform the actual calculation.
             Currently "em", "online" are supported. Default to "em".
         """
-        model = callMLlibFunc("trainLDAModel", rdd, k, maxIterations,
-                              docConcentration, topicConcentration, seed,
-                              checkpointInterval, optimizer)
-        return LDAModel(model)
+        wrapper_model = callMLlibFunc("trainLDAModel", rdd, k, maxIterations,
+                                      docConcentration, topicConcentration, seed,
+                                      checkpointInterval, optimizer)
+        return LDAModel(wrapper_model)
 
 
 def _test():