Github user jkbradley commented on a diff in the pull request:
https://github.com/apache/spark/pull/11581#discussion_r56082025
--- Diff: mllib/src/main/scala/org/apache/spark/ml/tree/treeModels.scala ---
@@ -101,3 +109,127 @@ private[ml] trait TreeEnsembleModel {
/** Total number of nodes, summed over all trees in the ensemble. */
lazy val totalNumNodes: Int = trees.map(_.numNodes).sum
}
+
+/** Helper classes for tree model persistence */
+private[ml] object DecisionTreeModelReadWrite {
+
+ /**
+ * Info for a [[org.apache.spark.ml.tree.Split]]
+ *
+ * @param featureIndex Index of feature split on
+ * @param leftCategoriesOrThreshold For categorical feature, set of
leftCategories.
+ * For continuous feature, threshold.
+ * @param numCategories For categorical feature, number of categories.
+ * For continuous feature, -1.
+ */
+ case class SplitData(
+ featureIndex: Int,
+ leftCategoriesOrThreshold: Array[Double],
+ numCategories: Int) {
+
+ def getSplit: Split = {
+ if (numCategories != -1) {
+ new CategoricalSplit(featureIndex, leftCategoriesOrThreshold,
numCategories)
+ } else {
+ assert(leftCategoriesOrThreshold.length == 1, s"DecisionTree split
data expected" +
+ s" 1 threshold for ContinuousSplit, but found thresholds: " +
+ leftCategoriesOrThreshold.mkString(", "))
+ new ContinuousSplit(featureIndex, leftCategoriesOrThreshold(0))
+ }
+ }
+ }
+
+ object SplitData {
+ def apply(split: Split): SplitData = split match {
+ case s: CategoricalSplit =>
+ SplitData(s.featureIndex, s.leftCategories, s.numCategories)
+ case s: ContinuousSplit =>
+ SplitData(s.featureIndex, Array(s.threshold), -1)
+ }
+ }
+
+ /**
+ * Info for a [[Node]]
+ *
+ * @param id Index used for tree reconstruction. Indices follow an
in-order traversal.
+ * @param impurityStats Stats array. Impurity type is stored in
metadata.
+ * @param gain Gain, or arbitrary value if leaf node.
--- End diff --
I like having this weaker public guarantee: people should not assume
anything about the values
---
If your project is set up for it, you can reply to this email and have your
reply appear on GitHub as well. If your project does not have this feature
enabled and wishes so, or if the feature is enabled but not working, please
contact infrastructure at [email protected] or file a JIRA ticket
with INFRA.
---
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]