Decision Tree Action Set: Syntax

Provides actions for modeling and scoring with decision trees, forests, and gradient boosting

dtreeSplit Action

Splits decision tree nodes.

decisionTree.dtreeSplit <result=results> <status=rc> /
alpha=double,
applyRowOrder=TRUE | FALSE,
attributes
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
binOrder=TRUE | FALSE,
bonferroni=TRUE | FALSE,
casOut
={
caslib="string",
compress=TRUE | FALSE,
indexVars={"variable-name-1" <, "variable-name-2", ...>},
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=TRUE | FALSE,
promote=TRUE | FALSE,
replace=TRUE | FALSE,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where={"string-1" <, "string-2", ...>}
},
cfLev=double,
code
={
casOut
={
caslib="string"
compress=TRUE | FALSE
indexVars={"variable-name-1" <, "variable-name-2", ...>}
label="string"
lifetime=64-bit-integer
maxMemSize=64-bit-integer
memoryFormat="DVR" | "INHERIT" | "STANDARD"
name="table-name"
onDemand=TRUE | FALSE
promote=TRUE | FALSE
replace=TRUE | FALSE
replication=integer
threadBlockSize=64-bit-integer
timeStamp="string"
where={"string-1" <, "string-2", ...>}
},
comment=TRUE | FALSE,
fmtWdth=integer,
indentSize=integer,
labelId=integer,
lineSize=integer,
noTrim=TRUE | FALSE,
tabForm=TRUE | FALSE
},
crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE",
encodeName=TRUE | FALSE,
freq="variable-name",
greedy=TRUE | FALSE,
includeMissing=TRUE | FALSE,
required parameter inputs
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
leafSize=integer,
maxBranch=integer,
maxLevel=integer,
mergeBin=TRUE | FALSE,
minGain=double,
modelId="string",
required parameter modelTable
={
caslib="string",
computedOnDemand=TRUE | FALSE,
computedVars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>},
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter name="table-name",
singlePass=TRUE | FALSE,
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
where="where-expression",
whereTable
={
casLib="string"
dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter name="table-name"
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}}
where="where-expression"
}
},
nBins=integer,
nodeId={integer-1 <, integer-2, ...>},
nominals
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
nominalSearch
={
handling="CLASSIC" | "ENHANCED",
maxCategories=64-bit-integer,
shrinkage=double,
sort=64-bit-integer,
sortBy="COUNT" | "TARGET"
},
noSplit=TRUE | FALSE,
prune=TRUE | FALSE,
pVal=double,
quantileBin=TRUE | FALSE,
saveState
={
caslib="string",
compress=TRUE | FALSE,
indexVars={"variable-name-1" <, "variable-name-2", ...>},
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=TRUE | FALSE,
promote=TRUE | FALSE,
replace=TRUE | FALSE,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where={"string-1" <, "string-2", ...>}
},
splitOnce=TRUE | FALSE,
stat=TRUE | FALSE,
required parameter table
={
caslib="string",
computedOnDemand=TRUE | FALSE,
computedVars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>},
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter name="table-name",
singlePass=TRUE | FALSE,
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
where="where-expression",
whereTable
={
casLib="string"
dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter name="table-name"
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}}
where="where-expression"
}
},
required parameter target="variable-name",
updateBin=TRUE | FALSE,
userDefinedSplit
={
intervalSplit={double-1 <, double-2, ...>},
missingBranch=64-bit-integer,
nominalSplit={any-list-or-data-type-1 <, any-list-or-data-type-2, ...>},
splitVar="variable-name",
unseenBranch=64-bit-integer
},
varImp=TRUE | FALSE
;
indicates a required parameter

Summary: Input and Output Tables

If a row includes a subparameter, you can specify the name, caslib, and so on in the subparameter. Otherwise, you can specify the name, caslib, and so on in the parameter.

Parameters for Reading Input Tables

Parameter

Subparameter

Description

required parametermodelTable

—

specifies the table containing the model.

required parametertable

—

specifies the settings for an input table.

Parameters for Creating Output Tables

Parameter

Subparameter

Description

 casOut

—

specifies the table to store the decision tree model in. When not specified, a random name is generated.

 code

casOut

requests that the action produce SAS score code. Specify additional parameters.

 saveState

—

specifies the table to store the generated aStore model.

Parameter Descriptions

alpha=double

specifies the value to use for minimal cost-complexity pruning for regression trees.

Minimum value0

applyRowOrder=TRUE | FALSE

Specifies that you wish the action use a prespecified row ordering. This requires using the orderby and groupby parameters on a preliminary table.partition action call.

AliasreproducibleRowOrder
DefaultFALSE

attributes={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies temporary attributes, such as a format, to apply to input variables.

For more information about specifying the attributes parameter, see the common casinvardesc parameter.

Aliasesattribute
attrs
attr
varAttrs

binOrder=TRUE | FALSE

by default, the bin order is preserved for numeric variables. When set to False, the bin order is ignored for numeric variables.

DefaultTRUE

bonferroni=TRUE | FALSE

when set to True, specifies to perform a Bonferroni correction when the split criterion uses a chi-square statistic or CHAID.

DefaultFALSE

casOut={casouttable}

specifies the table to store the decision tree model in. When not specified, a random name is generated.

For more information about specifying the casOut parameter, see the common casouttable parameter.

cfLev=double

specifies the aggressiveness of tree pruning according to the C4.5 algorithm.

Default0.25

code={codegen}

requests that the action produce SAS score code. Specify additional parameters.

For more information about specifying the code parameter, see the common codegen parameter.

crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE"

specifies the split criterion for each tree node.

encodeName=TRUE | FALSE

specifies whether to encode the variable names such as predicted probabilities of a binary or nominal target in the generated casout table. The predicted probabilities are named with the prefix P_ instead of _DT_P_.

DefaultFALSE

freq="variable-name"

specifies a numeric variable that contains the frequency of occurrence of each observation.

greedy=TRUE | FALSE

by default, a greedy search or exhaustive search is used to determine the best split for each variable of each tree node. When set to False, a fast and efficient algorithm that is based on clustering is applied. Setting this parameter to False is recommended for variables with high cardinality.

DefaultTRUE

includeMissing=TRUE | FALSE

by default, observations with missing values are included. When set to False, observations with missing values for the analysis variables are excluded.

DefaultTRUE

* inputs={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the input variables to use in the analysis.

For more information about specifying the inputs parameter, see the common casinvardesc parameter.

Aliasinput

leafSize=integer

specifies the minimum number of observations on each node.

Default5
Minimum value1

maxBranch=integer

specifies the maximum number of children (branches) allowed for each level of the tree.

Default2
Minimum value1

maxLevel=integer

specifies the maximum number of the tree level.

Default6
Minimum value1

mergeBin=TRUE | FALSE

by default, when the largest value in one bin matches the lowest value in a neighboring bin, the values are merged into the lower bin. When set to False, the action does not try to merge bins.

DefaultTRUE

minGain=double

specifies the minimum value to use to validate a splitting point when the criteria is not chi-square or CHAID.

Minimum value0

minUseInSearch=integer

specifies a threshold for utilizing missing values in the split search when the missing parameter is set to USEINSEARCH. If the number of observations in which the splitting variable has missing values in a node is greater than or equal to the specified value, then the action initiates the USEINSEARCH policy. Otherwise, the missing values are assigned to a popular branch.

Default1

missing="BRANCH" | "MACSMALL" | "POPULAR" | "SIMILAR" | "USEINSEARCH"

specifies the missing policy to handle missing values.

DefaultUSEINSEARCH
BRANCH

specifies to assign the missing values to a separate branch for missing values.

MACSMALL

specifies to treat the missing values for numeric variables as the smallest machine value and to treat missing values for nominal variables as a separate level.

POPULAR

specifies to assign the missing values to the most popular (largest) child.

SIMILAR

specifies to assign missing values to the most similar node. Similarity is calculated using a chi-square test for categorical response variables or an F-test for continuous response variables.

USEINSEARCH

specifies to incorporate missing values in the calculation of the worth of a splitting rule, and consequently to produce a splitting rule that associates missing values with a branch that maximizes the worth of the split.

modelId="string"

specifies the model ID variable name to use when generating SAS score code. By default, _DT_ is prefixed to the target variable name and _ is appended.

* modelTable={castable}

specifies the table containing the model.

Long formmodelTable={name="table-name"}
Shortcut formmodelTable="table-name"
Aliasmodel

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=TRUE | FALSE

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFALSE
computedVars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>}

specifies data source options.

Aliasesoptions
dataSource
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=TRUE | FALSE

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFALSE
vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable={groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

nBins=integer

specifies the number of bins to use for numeric variables in the calculation of the decision tree.

Default50
Minimum value1

nodeId={integer-1 <, integer-2, ...>}

specifies the leaf node IDs to split. When not specified or the list is not valid then all leaf nodes in the tree model are split.

nominals={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the nominal input variables to use in the analysis.

For more information about specifying the nominals parameter, see the common casinvardesc parameter.

Aliasnominal

nominalSearch={tkcasdt_nomSearchOpts}

specifies the method for finding a split on a nominal input.

AliasnomSearch

The tkcasdt_nomSearchOpts value can be one or more of the following:

handling="CLASSIC" | "ENHANCED"
maxCategories=64-bit-integer

specifies the maximum number of levels for a splitting rule to include.

AliasesmaxCats
maxLevels
maxValues
cluster
minCardCluster
Default128
Minimum value0
shrinkage=double

specifies how much weight to give the category average in the sort method.

Default10
Minimum value0
sort=64-bit-integer

specifies the minimum cardinality of an input to use the sort method.

AliasminCardSort
Default10
Minimum value0
sortBy="COUNT" | "TARGET"

noSplit=TRUE | FALSE

runs the action without splitting nodes or growing the tree. If you specify this parameter, then new columns appear in the output model table that you specify in the casOut parameter. These columns contain statistics about the data specified in the table parameter with respect to the tree model that you specify in the modelTable parameter. You should use this parameter if you want statistics about validation data.

DefaultFALSE

prune=TRUE | FALSE

specify true to use a C4.5 pruning method for classification trees or minimal cost-complexity pruning for regression trees.

DefaultFALSE

pVal=double

specifies the maximum P-value to use in chi-square and CHAID split criteria to validate a splitting point.

Default1
Range0–1

quantileBin=TRUE | FALSE

specifies bin boundaries at quantiles of numerical inputs instead of bins of equal width.

Aliasesqbin
qtbin
DefaultTRUE

saveState={casouttable}

specifies the table to store the generated aStore model.

For more information about specifying the saveState parameter, see the common casouttable parameter.

splitOnce=TRUE | FALSE

when set to True, the analysis variable only appears once for each path from the tree root node to the leaf node.

DefaultFALSE

stat=TRUE | FALSE

specifies whether to include the variable splitting information for each node in the decision tree.

DefaultFALSE

* table={castable}

specifies the settings for an input table.

Long formtable={name="table-name"}
Shortcut formtable="table-name"

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=TRUE | FALSE

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFALSE
computedVars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>}

specifies data source options.

Aliasesoptions
dataSource
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=TRUE | FALSE

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFALSE
vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable={groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

* target="variable-name"

specifies the target or response variable for training. If the variable is numeric, but not specified in the nominal= parameter and nbinstarget= is not specified, then a regression tree is trained.

updateBin=TRUE | FALSE

when set to True, the bin information for all the numerical analysis variables is computed on each node that is used for a split. All the splits make use of the updated bin information. Note that this changes the bin information significantly compared to the bin information that is used in the input model. As a result, this might bias the splits that occur later.

DefaultFALSE

userDefinedSplit={userDefinedSplit}

indicates that you will manually define the split of a single node as specified in the nodeId parameter. The userDefinedSplit parameter contains sub-parameters needed for defining this split.

Long formuserDefinedSplit={splitVar="variable-name"}
Shortcut formuserDefinedSplit="variable-name"

The userDefinedSplit value can be one or more of the following:

intervalSplit={double-1 <, double-2, ...>}

is a list of interval split points. The number of branches is the number of numbers listed plus one. The branches are numbered with the 0 branch being all values less than the least value. Each other branch is numbered in increasing order.

missingBranch=64-bit-integer

indicates to which branch missing values should be assigned. By default, missing values are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0
nominalSplit={any-list-or-data-type-1 <, any-list-or-data-type-2, ...>}

is a list of string lists that contain nominal split values. The number of branches is the number of string lists. The branches are ordered in which they are given to the action, with the first branch being numbered as branch 0.

splitVar="variable-name"

indicates which variable the decision tree is to use for splitting.

unseenBranch=64-bit-integer

indicates to which branch unseen levels should be assigned. By default, unseen levels are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0

varImp=TRUE | FALSE

specifies whether the variable importance information is generated. The importance value is determined by the total Gini reduction.

DefaultFALSE

dtreeSplit Action

Splits decision tree nodes.

results, info = s:decisionTree_dtreeSplit{
alpha=double,
applyRowOrder=true | false,
attributes
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
binOrder=true | false,
bonferroni=true | false,
casOut
={
caslib="string",
compress=true | false,
indexVars={"variable-name-1" <, "variable-name-2", ...>},
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=true | false,
promote=true | false,
replace=true | false,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where={"string-1" <, "string-2", ...>}
},
cfLev=double,
code
={
casOut
={
caslib="string"
compress=true | false
indexVars={"variable-name-1" <, "variable-name-2", ...>}
label="string"
lifetime=64-bit-integer
maxMemSize=64-bit-integer
memoryFormat="DVR" | "INHERIT" | "STANDARD"
name="table-name"
onDemand=true | false
promote=true | false
replace=true | false
replication=integer
threadBlockSize=64-bit-integer
timeStamp="string"
where={"string-1" <, "string-2", ...>}
},
comment=true | false,
fmtWdth=integer,
indentSize=integer,
labelId=integer,
lineSize=integer,
noTrim=true | false,
tabForm=true | false
},
crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE",
encodeName=true | false,
freq="variable-name",
greedy=true | false,
includeMissing=true | false,
required parameter inputs
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
leafSize=integer,
maxBranch=integer,
maxLevel=integer,
mergeBin=true | false,
minGain=double,
modelId="string",
required parameter modelTable
={
caslib="string",
computedOnDemand=true | false,
computedVars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>},
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter name="table-name",
singlePass=true | false,
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
where="where-expression",
whereTable
={
casLib="string"
dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter name="table-name"
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}}
where="where-expression"
}
},
nBins=integer,
nodeId={integer-1 <, integer-2, ...>},
nominals
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
nominalSearch
={
handling="CLASSIC" | "ENHANCED",
maxCategories=64-bit-integer,
shrinkage=double,
sort=64-bit-integer,
sortBy="COUNT" | "TARGET"
},
noSplit=true | false,
prune=true | false,
pVal=double,
quantileBin=true | false,
saveState
={
caslib="string",
compress=true | false,
indexVars={"variable-name-1" <, "variable-name-2", ...>},
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=true | false,
promote=true | false,
replace=true | false,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where={"string-1" <, "string-2", ...>}
},
splitOnce=true | false,
stat=true | false,
required parameter table
={
caslib="string",
computedOnDemand=true | false,
computedVars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>},
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter name="table-name",
singlePass=true | false,
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}},
where="where-expression",
whereTable
={
casLib="string"
dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter name="table-name"
vars
={{
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
}, {...}}
where="where-expression"
}
},
required parameter target="variable-name",
updateBin=true | false,
userDefinedSplit
={
intervalSplit={double-1 <, double-2, ...>},
missingBranch=64-bit-integer,
nominalSplit={any-list-or-data-type-1 <, any-list-or-data-type-2, ...>},
splitVar="variable-name",
unseenBranch=64-bit-integer
},
varImp=true | false
}
indicates a required parameter

Summary: Input and Output Tables

If a row includes a subparameter, you can specify the name, caslib, and so on in the subparameter. Otherwise, you can specify the name, caslib, and so on in the parameter.

Parameters for Reading Input Tables

Parameter

Subparameter

Description

required parametermodelTable

—

specifies the table containing the model.

required parametertable

—

specifies the settings for an input table.

Parameters for Creating Output Tables

Parameter

Subparameter

Description

 casOut

—

specifies the table to store the decision tree model in. When not specified, a random name is generated.

 code

casOut

requests that the action produce SAS score code. Specify additional parameters.

 saveState

—

specifies the table to store the generated aStore model.

Parameter Descriptions

alpha=double

specifies the value to use for minimal cost-complexity pruning for regression trees.

Minimum value0

applyRowOrder=true | false

Specifies that you wish the action use a prespecified row ordering. This requires using the orderby and groupby parameters on a preliminary table.partition action call.

AliasreproducibleRowOrder
Defaultfalse

attributes={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies temporary attributes, such as a format, to apply to input variables.

For more information about specifying the attributes parameter, see the common casinvardesc parameter.

Aliasesattribute
attrs
attr
varAttrs

binOrder=true | false

by default, the bin order is preserved for numeric variables. When set to False, the bin order is ignored for numeric variables.

Defaulttrue

bonferroni=true | false

when set to True, specifies to perform a Bonferroni correction when the split criterion uses a chi-square statistic or CHAID.

Defaultfalse

casOut={casouttable}

specifies the table to store the decision tree model in. When not specified, a random name is generated.

For more information about specifying the casOut parameter, see the common casouttable parameter.

cfLev=double

specifies the aggressiveness of tree pruning according to the C4.5 algorithm.

Default0.25

code={codegen}

requests that the action produce SAS score code. Specify additional parameters.

For more information about specifying the code parameter, see the common codegen parameter.

crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE"

specifies the split criterion for each tree node.

encodeName=true | false

specifies whether to encode the variable names such as predicted probabilities of a binary or nominal target in the generated casout table. The predicted probabilities are named with the prefix P_ instead of _DT_P_.

Defaultfalse

freq="variable-name"

specifies a numeric variable that contains the frequency of occurrence of each observation.

greedy=true | false

by default, a greedy search or exhaustive search is used to determine the best split for each variable of each tree node. When set to False, a fast and efficient algorithm that is based on clustering is applied. Setting this parameter to False is recommended for variables with high cardinality.

Defaulttrue

includeMissing=true | false

by default, observations with missing values are included. When set to False, observations with missing values for the analysis variables are excluded.

Defaulttrue

* inputs={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the input variables to use in the analysis.

For more information about specifying the inputs parameter, see the common casinvardesc parameter.

Aliasinput

leafSize=integer

specifies the minimum number of observations on each node.

Default5
Minimum value1

maxBranch=integer

specifies the maximum number of children (branches) allowed for each level of the tree.

Default2
Minimum value1

maxLevel=integer

specifies the maximum number of the tree level.

Default6
Minimum value1

mergeBin=true | false

by default, when the largest value in one bin matches the lowest value in a neighboring bin, the values are merged into the lower bin. When set to False, the action does not try to merge bins.

Defaulttrue

minGain=double

specifies the minimum value to use to validate a splitting point when the criteria is not chi-square or CHAID.

Minimum value0

minUseInSearch=integer

specifies a threshold for utilizing missing values in the split search when the missing parameter is set to USEINSEARCH. If the number of observations in which the splitting variable has missing values in a node is greater than or equal to the specified value, then the action initiates the USEINSEARCH policy. Otherwise, the missing values are assigned to a popular branch.

Default1

missing="BRANCH" | "MACSMALL" | "POPULAR" | "SIMILAR" | "USEINSEARCH"

specifies the missing policy to handle missing values.

DefaultUSEINSEARCH
BRANCH

specifies to assign the missing values to a separate branch for missing values.

MACSMALL

specifies to treat the missing values for numeric variables as the smallest machine value and to treat missing values for nominal variables as a separate level.

POPULAR

specifies to assign the missing values to the most popular (largest) child.

SIMILAR

specifies to assign missing values to the most similar node. Similarity is calculated using a chi-square test for categorical response variables or an F-test for continuous response variables.

USEINSEARCH

specifies to incorporate missing values in the calculation of the worth of a splitting rule, and consequently to produce a splitting rule that associates missing values with a branch that maximizes the worth of the split.

modelId="string"

specifies the model ID variable name to use when generating SAS score code. By default, _DT_ is prefixed to the target variable name and _ is appended.

* modelTable={castable}

specifies the table containing the model.

Long formmodelTable={name="table-name"}
Shortcut formmodelTable="table-name"
Aliasmodel

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=true | false

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
Defaultfalse
computedVars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>}

specifies data source options.

Aliasesoptions
dataSource
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=true | false

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

Defaultfalse
vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable={groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

nBins=integer

specifies the number of bins to use for numeric variables in the calculation of the decision tree.

Default50
Minimum value1

nodeId={integer-1 <, integer-2, ...>}

specifies the leaf node IDs to split. When not specified or the list is not valid then all leaf nodes in the tree model are split.

nominals={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the nominal input variables to use in the analysis.

For more information about specifying the nominals parameter, see the common casinvardesc parameter.

Aliasnominal

nominalSearch={tkcasdt_nomSearchOpts}

specifies the method for finding a split on a nominal input.

AliasnomSearch

The tkcasdt_nomSearchOpts value can be one or more of the following:

handling="CLASSIC" | "ENHANCED"
maxCategories=64-bit-integer

specifies the maximum number of levels for a splitting rule to include.

AliasesmaxCats
maxLevels
maxValues
cluster
minCardCluster
Default128
Minimum value0
shrinkage=double

specifies how much weight to give the category average in the sort method.

Default10
Minimum value0
sort=64-bit-integer

specifies the minimum cardinality of an input to use the sort method.

AliasminCardSort
Default10
Minimum value0
sortBy="COUNT" | "TARGET"

noSplit=true | false

runs the action without splitting nodes or growing the tree. If you specify this parameter, then new columns appear in the output model table that you specify in the casOut parameter. These columns contain statistics about the data specified in the table parameter with respect to the tree model that you specify in the modelTable parameter. You should use this parameter if you want statistics about validation data.

Defaultfalse

prune=true | false

specify true to use a C4.5 pruning method for classification trees or minimal cost-complexity pruning for regression trees.

Defaultfalse

pVal=double

specifies the maximum P-value to use in chi-square and CHAID split criteria to validate a splitting point.

Default1
Range0–1

quantileBin=true | false

specifies bin boundaries at quantiles of numerical inputs instead of bins of equal width.

Aliasesqbin
qtbin
Defaulttrue

saveState={casouttable}

specifies the table to store the generated aStore model.

For more information about specifying the saveState parameter, see the common casouttable parameter.

splitOnce=true | false

when set to True, the analysis variable only appears once for each path from the tree root node to the leaf node.

Defaultfalse

stat=true | false

specifies whether to include the variable splitting information for each node in the decision tree.

Defaultfalse

* table={castable}

specifies the settings for an input table.

Long formtable={name="table-name"}
Shortcut formtable="table-name"

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=true | false

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
Defaultfalse
computedVars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions={key-1=any-list-or-data-type-1 <, key-2=any-list-or-data-type-2, ...>}

specifies data source options.

Aliasesoptions
dataSource
importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=true | false

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

Defaultfalse
vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable={groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions={adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions={fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars={{casinvardesc-1} <, {casinvardesc-2}, ...>}

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

* target="variable-name"

specifies the target or response variable for training. If the variable is numeric, but not specified in the nominal= parameter and nbinstarget= is not specified, then a regression tree is trained.

updateBin=true | false

when set to True, the bin information for all the numerical analysis variables is computed on each node that is used for a split. All the splits make use of the updated bin information. Note that this changes the bin information significantly compared to the bin information that is used in the input model. As a result, this might bias the splits that occur later.

Defaultfalse

userDefinedSplit={userDefinedSplit}

indicates that you will manually define the split of a single node as specified in the nodeId parameter. The userDefinedSplit parameter contains sub-parameters needed for defining this split.

Long formuserDefinedSplit={splitVar="variable-name"}
Shortcut formuserDefinedSplit="variable-name"

The userDefinedSplit value can be one or more of the following:

intervalSplit={double-1 <, double-2, ...>}

is a list of interval split points. The number of branches is the number of numbers listed plus one. The branches are numbered with the 0 branch being all values less than the least value. Each other branch is numbered in increasing order.

missingBranch=64-bit-integer

indicates to which branch missing values should be assigned. By default, missing values are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0
nominalSplit={any-list-or-data-type-1 <, any-list-or-data-type-2, ...>}

is a list of string lists that contain nominal split values. The number of branches is the number of string lists. The branches are ordered in which they are given to the action, with the first branch being numbered as branch 0.

splitVar="variable-name"

indicates which variable the decision tree is to use for splitting.

unseenBranch=64-bit-integer

indicates to which branch unseen levels should be assigned. By default, unseen levels are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0

varImp=true | false

specifies whether the variable importance information is generated. The importance value is determined by the total Gini reduction.

Defaultfalse

dtreeSplit Action

Splits decision tree nodes.

results=s.decisionTree.dtreeSplit(
alpha=double,
applyRowOrder=True | False,
attributes
=[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
binOrder=True | False,
bonferroni=True | False,
casOut
={
"caslib":"string",
"compress":True | False,
"indexVars":["variable-name-1" <, "variable-name-2", ...>],
"label":"string",
"lifetime":64-bit-integer,
"maxMemSize":64-bit-integer,
"memoryFormat":"DVR" | "INHERIT" | "STANDARD",
"name":"table-name",
"onDemand":True | False,
"promote":True | False,
"replace":True | False,
"replication":integer,
"threadBlockSize":64-bit-integer,
"timeStamp":"string",
"where":["string-1" <, "string-2", ...>]
},
cfLev=double,
code
={
"casOut"
:{
"caslib":"string"
"compress":True | False
"indexVars":["variable-name-1" <, "variable-name-2", ...>]
"label":"string"
"lifetime":64-bit-integer
"maxMemSize":64-bit-integer
"memoryFormat":"DVR" | "INHERIT" | "STANDARD"
"name":"table-name"
"onDemand":True | False
"promote":True | False
"replace":True | False
"replication":integer
"threadBlockSize":64-bit-integer
"timeStamp":"string"
"where":["string-1" <, "string-2", ...>]
},
"comment":True | False,
"fmtWdth":integer,
"indentSize":integer,
"labelId":integer,
"lineSize":integer,
"noTrim":True | False,
"tabForm":True | False
},
crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE",
encodeName=True | False,
freq="variable-name",
greedy=True | False,
includeMissing=True | False,
required parameter inputs
=[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
leafSize=integer,
maxBranch=integer,
maxLevel=integer,
mergeBin=True | False,
minGain=double,
modelId="string",
required parameter modelTable
={
"caslib":"string",
"computedOnDemand":True | False,
"computedVars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
"dataSourceOptions":{"key-1":{any-list-or-data-type-1} <, "key-2":{any-list-or-data-type-2}, ...>},
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter "name":"table-name",
"singlePass":True | False,
"vars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
"where":"where-expression",
"whereTable"
:{
"casLib":"string"
"dataSourceOptions":{adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter "name":"table-name"
"vars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>]
"where":"where-expression"
}
},
nBins=integer,
nodeId=[integer-1 <, integer-2, ...>],
nominals
=[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
nominalSearch
={
"handling":"CLASSIC" | "ENHANCED",
"maxCategories":64-bit-integer,
"shrinkage":double,
"sort":64-bit-integer,
"sortBy":"COUNT" | "TARGET"
},
noSplit=True | False,
prune=True | False,
pVal=double,
quantileBin=True | False,
saveState
={
"caslib":"string",
"compress":True | False,
"indexVars":["variable-name-1" <, "variable-name-2", ...>],
"label":"string",
"lifetime":64-bit-integer,
"maxMemSize":64-bit-integer,
"memoryFormat":"DVR" | "INHERIT" | "STANDARD",
"name":"table-name",
"onDemand":True | False,
"promote":True | False,
"replace":True | False,
"replication":integer,
"threadBlockSize":64-bit-integer,
"timeStamp":"string",
"where":["string-1" <, "string-2", ...>]
},
splitOnce=True | False,
stat=True | False,
required parameter table
={
"caslib":"string",
"computedOnDemand":True | False,
"computedVars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
"dataSourceOptions":{"key-1":{any-list-or-data-type-1} <, "key-2":{any-list-or-data-type-2}, ...>},
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters},
required parameter "name":"table-name",
"singlePass":True | False,
"vars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>],
"where":"where-expression",
"whereTable"
:{
"casLib":"string"
"dataSourceOptions":{adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}
required parameter "name":"table-name"
"vars"
:[{
"format":"string",
"formattedLength":integer,
"label":"string",
required parameter "name":"variable-name",
"nfd":integer,
"nfl":integer
}<, {...}>]
"where":"where-expression"
}
},
required parameter target="variable-name",
updateBin=True | False,
userDefinedSplit
={
"intervalSplit":[double-1 <, double-2, ...>],
"missingBranch":64-bit-integer,
"nominalSplit":{any-list-or-data-type-1 <, any-list-or-data-type-2, ...>},
"splitVar":"variable-name",
"unseenBranch":64-bit-integer
},
varImp=True | False
)
indicates a required parameter

Summary: Input and Output Tables

If a row includes a subparameter, you can specify the name, caslib, and so on in the subparameter. Otherwise, you can specify the name, caslib, and so on in the parameter.

Parameters for Reading Input Tables

Parameter

Subparameter

Description

required parametermodelTable

—

specifies the table containing the model.

required parametertable

—

specifies the settings for an input table.

Parameters for Creating Output Tables

Parameter

Subparameter

Description

 casOut

—

specifies the table to store the decision tree model in. When not specified, a random name is generated.

 code

casOut

requests that the action produce SAS score code. Specify additional parameters.

 saveState

—

specifies the table to store the generated aStore model.

Parameter Descriptions

alpha=double

specifies the value to use for minimal cost-complexity pruning for regression trees.

Minimum value0

applyRowOrder=True | False

Specifies that you wish the action use a prespecified row ordering. This requires using the orderby and groupby parameters on a preliminary table.partition action call.

AliasreproducibleRowOrder
DefaultFalse

attributes=[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies temporary attributes, such as a format, to apply to input variables.

For more information about specifying the attributes parameter, see the common casinvardesc parameter.

Aliasesattribute
attrs
attr
varAttrs

binOrder=True | False

by default, the bin order is preserved for numeric variables. When set to False, the bin order is ignored for numeric variables.

DefaultTrue

bonferroni=True | False

when set to True, specifies to perform a Bonferroni correction when the split criterion uses a chi-square statistic or CHAID.

DefaultFalse

casOut={casouttable}

specifies the table to store the decision tree model in. When not specified, a random name is generated.

For more information about specifying the casOut parameter, see the common casouttable parameter.

cfLev=double

specifies the aggressiveness of tree pruning according to the C4.5 algorithm.

Default0.25

code={codegen}

requests that the action produce SAS score code. Specify additional parameters.

For more information about specifying the code parameter, see the common codegen parameter.

crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE"

specifies the split criterion for each tree node.

encodeName=True | False

specifies whether to encode the variable names such as predicted probabilities of a binary or nominal target in the generated casout table. The predicted probabilities are named with the prefix P_ instead of _DT_P_.

DefaultFalse

freq="variable-name"

specifies a numeric variable that contains the frequency of occurrence of each observation.

greedy=True | False

by default, a greedy search or exhaustive search is used to determine the best split for each variable of each tree node. When set to False, a fast and efficient algorithm that is based on clustering is applied. Setting this parameter to False is recommended for variables with high cardinality.

DefaultTrue

includeMissing=True | False

by default, observations with missing values are included. When set to False, observations with missing values for the analysis variables are excluded.

DefaultTrue

* inputs=[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the input variables to use in the analysis.

For more information about specifying the inputs parameter, see the common casinvardesc parameter.

Aliasinput

leafSize=integer

specifies the minimum number of observations on each node.

Default5
Minimum value1

maxBranch=integer

specifies the maximum number of children (branches) allowed for each level of the tree.

Default2
Minimum value1

maxLevel=integer

specifies the maximum number of the tree level.

Default6
Minimum value1

mergeBin=True | False

by default, when the largest value in one bin matches the lowest value in a neighboring bin, the values are merged into the lower bin. When set to False, the action does not try to merge bins.

DefaultTrue

minGain=double

specifies the minimum value to use to validate a splitting point when the criteria is not chi-square or CHAID.

Minimum value0

minUseInSearch=integer

specifies a threshold for utilizing missing values in the split search when the missing parameter is set to USEINSEARCH. If the number of observations in which the splitting variable has missing values in a node is greater than or equal to the specified value, then the action initiates the USEINSEARCH policy. Otherwise, the missing values are assigned to a popular branch.

Default1

missing="BRANCH" | "MACSMALL" | "POPULAR" | "SIMILAR" | "USEINSEARCH"

specifies the missing policy to handle missing values.

DefaultUSEINSEARCH
BRANCH

specifies to assign the missing values to a separate branch for missing values.

MACSMALL

specifies to treat the missing values for numeric variables as the smallest machine value and to treat missing values for nominal variables as a separate level.

POPULAR

specifies to assign the missing values to the most popular (largest) child.

SIMILAR

specifies to assign missing values to the most similar node. Similarity is calculated using a chi-square test for categorical response variables or an F-test for continuous response variables.

USEINSEARCH

specifies to incorporate missing values in the calculation of the worth of a splitting rule, and consequently to produce a splitting rule that associates missing values with a branch that maximizes the worth of the split.

modelId="string"

specifies the model ID variable name to use when generating SAS score code. By default, _DT_ is prefixed to the target variable name and _ is appended.

* modelTable={castable}

specifies the table containing the model.

Long formmodelTable={"name":"table-name"}
Shortcut formmodelTable="table-name"
Aliasmodel

The castable value can be one or more of the following:

"caslib":"string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

"computedOnDemand":True | False

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFalse
"computedVars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"computedVarsProgram":"string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
"dataSourceOptions":{"key-1":{any-list-or-data-type-1} <, "key-2":{any-list-or-data-type-2}, ...>}

specifies data source options.

Aliasesoptions
dataSource
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport_

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* "name":"table-name"

specifies the name of the input table.

"singlePass":True | False

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFalse
"vars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"where":"where-expression"

specifies an expression for subsetting the input data.

"whereTable":{groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

"casLib":"string"

specifies the caslib for the filter table. By default, the active caslib is used.

"dataSourceOptions":{adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport_

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* "name":"table-name"

specifies the name of the filter table.

"vars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"where":"where-expression"

specifies an expression for subsetting the data from the filter table.

nBins=integer

specifies the number of bins to use for numeric variables in the calculation of the decision tree.

Default50
Minimum value1

nodeId=[integer-1 <, integer-2, ...>]

specifies the leaf node IDs to split. When not specified or the list is not valid then all leaf nodes in the tree model are split.

nominals=[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the nominal input variables to use in the analysis.

For more information about specifying the nominals parameter, see the common casinvardesc parameter.

Aliasnominal

nominalSearch={tkcasdt_nomSearchOpts}

specifies the method for finding a split on a nominal input.

AliasnomSearch

The tkcasdt_nomSearchOpts value can be one or more of the following:

"handling":"CLASSIC" | "ENHANCED"
"maxCategories":64-bit-integer

specifies the maximum number of levels for a splitting rule to include.

AliasesmaxCats
maxLevels
maxValues
cluster
minCardCluster
Default128
Minimum value0
"shrinkage":double

specifies how much weight to give the category average in the sort method.

Default10
Minimum value0
"sort":64-bit-integer

specifies the minimum cardinality of an input to use the sort method.

AliasminCardSort
Default10
Minimum value0
"sortBy":"COUNT" | "TARGET"

noSplit=True | False

runs the action without splitting nodes or growing the tree. If you specify this parameter, then new columns appear in the output model table that you specify in the casOut parameter. These columns contain statistics about the data specified in the table parameter with respect to the tree model that you specify in the modelTable parameter. You should use this parameter if you want statistics about validation data.

DefaultFalse

prune=True | False

specify true to use a C4.5 pruning method for classification trees or minimal cost-complexity pruning for regression trees.

DefaultFalse

pVal=double

specifies the maximum P-value to use in chi-square and CHAID split criteria to validate a splitting point.

Default1
Range0–1

quantileBin=True | False

specifies bin boundaries at quantiles of numerical inputs instead of bins of equal width.

Aliasesqbin
qtbin
DefaultTrue

saveState={casouttable}

specifies the table to store the generated aStore model.

For more information about specifying the saveState parameter, see the common casouttable parameter.

splitOnce=True | False

when set to True, the analysis variable only appears once for each path from the tree root node to the leaf node.

DefaultFalse

stat=True | False

specifies whether to include the variable splitting information for each node in the decision tree.

DefaultFalse

* table={castable}

specifies the settings for an input table.

Long formtable={"name":"table-name"}
Shortcut formtable="table-name"

The castable value can be one or more of the following:

"caslib":"string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

"computedOnDemand":True | False

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFalse
"computedVars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"computedVarsProgram":"string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
"dataSourceOptions":{"key-1":{any-list-or-data-type-1} <, "key-2":{any-list-or-data-type-2}, ...>}

specifies data source options.

Aliasesoptions
dataSource
"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport_

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* "name":"table-name"

specifies the name of the input table.

"singlePass":True | False

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFalse
"vars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"where":"where-expression"

specifies an expression for subsetting the input data.

"whereTable":{groupbytable}

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

"casLib":"string"

specifies the caslib for the filter table. By default, the active caslib is used.

"dataSourceOptions":{adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters}

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

"importOptions":{"fileType":"ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters}

specifies the settings for reading a table from a data source.

Aliasimport_

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* "name":"table-name"

specifies the name of the filter table.

"vars":[{casinvardesc-1} <, {casinvardesc-2}, ...>]

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

"format":"string"

specifies the format to apply to the variable.

"formattedLength":integer

specifies the length of format field plus the length of the format precision.

"label":"string"

specifies the descriptive label for the variable.

* "name":"variable-name"

specifies the name for the variable.

"nfd":integer

specifies the length of the format precision.

"nfl":integer

specifies the length of the format field.

"where":"where-expression"

specifies an expression for subsetting the data from the filter table.

* target="variable-name"

specifies the target or response variable for training. If the variable is numeric, but not specified in the nominal= parameter and nbinstarget= is not specified, then a regression tree is trained.

updateBin=True | False

when set to True, the bin information for all the numerical analysis variables is computed on each node that is used for a split. All the splits make use of the updated bin information. Note that this changes the bin information significantly compared to the bin information that is used in the input model. As a result, this might bias the splits that occur later.

DefaultFalse

userDefinedSplit={userDefinedSplit}

indicates that you will manually define the split of a single node as specified in the nodeId parameter. The userDefinedSplit parameter contains sub-parameters needed for defining this split.

Long formuserDefinedSplit={"splitVar":"variable-name"}
Shortcut formuserDefinedSplit="variable-name"

The userDefinedSplit value can be one or more of the following:

"intervalSplit":[double-1 <, double-2, ...>]

is a list of interval split points. The number of branches is the number of numbers listed plus one. The branches are numbered with the 0 branch being all values less than the least value. Each other branch is numbered in increasing order.

"missingBranch":64-bit-integer

indicates to which branch missing values should be assigned. By default, missing values are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0
"nominalSplit":{any-list-or-data-type-1 <, any-list-or-data-type-2, ...>}

is a list of string lists that contain nominal split values. The number of branches is the number of string lists. The branches are ordered in which they are given to the action, with the first branch being numbered as branch 0.

"splitVar":"variable-name"

indicates which variable the decision tree is to use for splitting.

"unseenBranch":64-bit-integer

indicates to which branch unseen levels should be assigned. By default, unseen levels are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0

varImp=True | False

specifies whether the variable importance information is generated. The importance value is determined by the total Gini reduction.

DefaultFalse

dtreeSplit Action

Splits decision tree nodes.

results <– cas.decisionTree.dtreeSplit(s,
alpha=double,
applyRowOrder=TRUE | FALSE,
attributes
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
binOrder=TRUE | FALSE,
bonferroni=TRUE | FALSE,
casOut
=list(
caslib="string",
compress=TRUE | FALSE,
indexVars=list("variable-name-1" <, "variable-name-2", ...>),
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=TRUE | FALSE,
promote=TRUE | FALSE,
replace=TRUE | FALSE,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where=list("string-1" <, "string-2", ...>)
),
cfLev=double,
code
=list(
casOut
=list(
caslib="string"
compress=TRUE | FALSE
indexVars=list("variable-name-1" <, "variable-name-2", ...>)
label="string"
lifetime=64-bit-integer
maxMemSize=64-bit-integer
memoryFormat="DVR" | "INHERIT" | "STANDARD"
name="table-name"
onDemand=TRUE | FALSE
promote=TRUE | FALSE
replace=TRUE | FALSE
replication=integer
threadBlockSize=64-bit-integer
timeStamp="string"
where=list("string-1" <, "string-2", ...>)
),
comment=TRUE | FALSE,
fmtWdth=integer,
indentSize=integer,
labelId=integer,
lineSize=integer,
noTrim=TRUE | FALSE,
tabForm=TRUE | FALSE
),
crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE",
encodeName=TRUE | FALSE,
freq="variable-name",
greedy=TRUE | FALSE,
includeMissing=TRUE | FALSE,
required parameter inputs
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
leafSize=integer,
maxBranch=integer,
maxLevel=integer,
mergeBin=TRUE | FALSE,
minGain=double,
modelId="string",
required parameter modelTable
=list(
caslib="string",
computedOnDemand=TRUE | FALSE,
computedVars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
dataSourceOptions=list(key-1=list(any-list-or-data-type-1) <, key-2=list(any-list-or-data-type-2), ...>),
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters),
required parameter name="table-name",
singlePass=TRUE | FALSE,
vars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
where="where-expression",
whereTable
=list(
casLib="string"
dataSourceOptions=list(adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters)
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)
required parameter name="table-name"
vars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>)
where="where-expression"
)
),
nBins=integer,
nodeId=list(integer-1 <, integer-2, ...>),
nominals
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
nominalSearch
=list(
handling="CLASSIC" | "ENHANCED",
maxCategories=64-bit-integer,
shrinkage=double,
sort=64-bit-integer,
sortBy="COUNT" | "TARGET"
),
noSplit=TRUE | FALSE,
prune=TRUE | FALSE,
pVal=double,
quantileBin=TRUE | FALSE,
saveState
=list(
caslib="string",
compress=TRUE | FALSE,
indexVars=list("variable-name-1" <, "variable-name-2", ...>),
label="string",
lifetime=64-bit-integer,
maxMemSize=64-bit-integer,
memoryFormat="DVR" | "INHERIT" | "STANDARD",
name="table-name",
onDemand=TRUE | FALSE,
promote=TRUE | FALSE,
replace=TRUE | FALSE,
replication=integer,
threadBlockSize=64-bit-integer,
timeStamp="string",
where=list("string-1" <, "string-2", ...>)
),
splitOnce=TRUE | FALSE,
stat=TRUE | FALSE,
required parameter table
=list(
caslib="string",
computedOnDemand=TRUE | FALSE,
computedVars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
dataSourceOptions=list(key-1=list(any-list-or-data-type-1) <, key-2=list(any-list-or-data-type-2), ...>),
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters),
required parameter name="table-name",
singlePass=TRUE | FALSE,
vars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>),
where="where-expression",
whereTable
=list(
casLib="string"
dataSourceOptions=list(adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters)
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DELIMITED" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SOUND" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)
required parameter name="table-name"
vars
=list( list(
format="string",
formattedLength=integer,
label="string",
required parameter name="variable-name",
nfd=integer,
nfl=integer
) <, list(...)>)
where="where-expression"
)
),
required parameter target="variable-name",
updateBin=TRUE | FALSE,
userDefinedSplit
=list(
intervalSplit=list(double-1 <, double-2, ...>),
missingBranch=64-bit-integer,
nominalSplit=list(any-list-or-data-type-1 <, any-list-or-data-type-2, ...>),
splitVar="variable-name",
unseenBranch=64-bit-integer
),
varImp=TRUE | FALSE
)
indicates a required parameter

Summary: Input and Output Tables

If a row includes a subparameter, you can specify the name, caslib, and so on in the subparameter. Otherwise, you can specify the name, caslib, and so on in the parameter.

Parameters for Reading Input Tables

Parameter

Subparameter

Description

required parametermodelTable

—

specifies the table containing the model.

required parametertable

—

specifies the settings for an input table.

Parameters for Creating Output Tables

Parameter

Subparameter

Description

 casOut

—

specifies the table to store the decision tree model in. When not specified, a random name is generated.

 code

casOut

requests that the action produce SAS score code. Specify additional parameters.

 saveState

—

specifies the table to store the generated aStore model.

Parameter Descriptions

alpha=double

specifies the value to use for minimal cost-complexity pruning for regression trees.

Minimum value0

applyRowOrder=TRUE | FALSE

Specifies that you wish the action use a prespecified row ordering. This requires using the orderby and groupby parameters on a preliminary table.partition action call.

AliasreproducibleRowOrder
DefaultFALSE

attributes=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies temporary attributes, such as a format, to apply to input variables.

For more information about specifying the attributes parameter, see the common casinvardesc parameter.

Aliasesattribute
attrs
attr
varAttrs

binOrder=TRUE | FALSE

by default, the bin order is preserved for numeric variables. When set to False, the bin order is ignored for numeric variables.

DefaultTRUE

bonferroni=TRUE | FALSE

when set to True, specifies to perform a Bonferroni correction when the split criterion uses a chi-square statistic or CHAID.

DefaultFALSE

casOut=list(casouttable)

specifies the table to store the decision tree model in. When not specified, a random name is generated.

For more information about specifying the casOut parameter, see the common casouttable parameter.

cfLev=double

specifies the aggressiveness of tree pruning according to the C4.5 algorithm.

Default0.25

code=list(codegen)

requests that the action produce SAS score code. Specify additional parameters.

For more information about specifying the code parameter, see the common codegen parameter.

crit="CHAID" | "CHISQUARE" | "FTEST" | "GAIN" | "GAINRATIO" | "GINI" | "VARIANCE"

specifies the split criterion for each tree node.

encodeName=TRUE | FALSE

specifies whether to encode the variable names such as predicted probabilities of a binary or nominal target in the generated casout table. The predicted probabilities are named with the prefix P_ instead of _DT_P_.

DefaultFALSE

freq="variable-name"

specifies a numeric variable that contains the frequency of occurrence of each observation.

greedy=TRUE | FALSE

by default, a greedy search or exhaustive search is used to determine the best split for each variable of each tree node. When set to False, a fast and efficient algorithm that is based on clustering is applied. Setting this parameter to False is recommended for variables with high cardinality.

DefaultTRUE

includeMissing=TRUE | FALSE

by default, observations with missing values are included. When set to False, observations with missing values for the analysis variables are excluded.

DefaultTRUE

* inputs=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the input variables to use in the analysis.

For more information about specifying the inputs parameter, see the common casinvardesc parameter.

Aliasinput

leafSize=integer

specifies the minimum number of observations on each node.

Default5
Minimum value1

maxBranch=integer

specifies the maximum number of children (branches) allowed for each level of the tree.

Default2
Minimum value1

maxLevel=integer

specifies the maximum number of the tree level.

Default6
Minimum value1

mergeBin=TRUE | FALSE

by default, when the largest value in one bin matches the lowest value in a neighboring bin, the values are merged into the lower bin. When set to False, the action does not try to merge bins.

DefaultTRUE

minGain=double

specifies the minimum value to use to validate a splitting point when the criteria is not chi-square or CHAID.

Minimum value0

minUseInSearch=integer

specifies a threshold for utilizing missing values in the split search when the missing parameter is set to USEINSEARCH. If the number of observations in which the splitting variable has missing values in a node is greater than or equal to the specified value, then the action initiates the USEINSEARCH policy. Otherwise, the missing values are assigned to a popular branch.

Default1

missing="BRANCH" | "MACSMALL" | "POPULAR" | "SIMILAR" | "USEINSEARCH"

specifies the missing policy to handle missing values.

DefaultUSEINSEARCH
BRANCH

specifies to assign the missing values to a separate branch for missing values.

MACSMALL

specifies to treat the missing values for numeric variables as the smallest machine value and to treat missing values for nominal variables as a separate level.

POPULAR

specifies to assign the missing values to the most popular (largest) child.

SIMILAR

specifies to assign missing values to the most similar node. Similarity is calculated using a chi-square test for categorical response variables or an F-test for continuous response variables.

USEINSEARCH

specifies to incorporate missing values in the calculation of the worth of a splitting rule, and consequently to produce a splitting rule that associates missing values with a branch that maximizes the worth of the split.

modelId="string"

specifies the model ID variable name to use when generating SAS score code. By default, _DT_ is prefixed to the target variable name and _ is appended.

* modelTable=list(castable)

specifies the table containing the model.

Long formmodelTable=list(name="table-name")
Shortcut formmodelTable="table-name"
Aliasmodel

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=TRUE | FALSE

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFALSE
computedVars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions=list(key-1=list(any-list-or-data-type-1) <, key-2=list(any-list-or-data-type-2), ...>)

specifies data source options.

Aliasesoptions
dataSource
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=TRUE | FALSE

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFALSE
vars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable=list(groupbytable)

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions=list(adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters)

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

nBins=integer

specifies the number of bins to use for numeric variables in the calculation of the decision tree.

Default50
Minimum value1

nodeId=list(integer-1 <, integer-2, ...>)

specifies the leaf node IDs to split. When not specified or the list is not valid then all leaf nodes in the tree model are split.

nominals=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the nominal input variables to use in the analysis.

For more information about specifying the nominals parameter, see the common casinvardesc parameter.

Aliasnominal

nominalSearch=list(tkcasdt_nomSearchOpts)

specifies the method for finding a split on a nominal input.

AliasnomSearch

The tkcasdt_nomSearchOpts value can be one or more of the following:

handling="CLASSIC" | "ENHANCED"
maxCategories=64-bit-integer

specifies the maximum number of levels for a splitting rule to include.

AliasesmaxCats
maxLevels
maxValues
cluster
minCardCluster
Default128
Minimum value0
shrinkage=double

specifies how much weight to give the category average in the sort method.

Default10
Minimum value0
sort=64-bit-integer

specifies the minimum cardinality of an input to use the sort method.

AliasminCardSort
Default10
Minimum value0
sortBy="COUNT" | "TARGET"

noSplit=TRUE | FALSE

runs the action without splitting nodes or growing the tree. If you specify this parameter, then new columns appear in the output model table that you specify in the casOut parameter. These columns contain statistics about the data specified in the table parameter with respect to the tree model that you specify in the modelTable parameter. You should use this parameter if you want statistics about validation data.

DefaultFALSE

prune=TRUE | FALSE

specify true to use a C4.5 pruning method for classification trees or minimal cost-complexity pruning for regression trees.

DefaultFALSE

pVal=double

specifies the maximum P-value to use in chi-square and CHAID split criteria to validate a splitting point.

Default1
Range0–1

quantileBin=TRUE | FALSE

specifies bin boundaries at quantiles of numerical inputs instead of bins of equal width.

Aliasesqbin
qtbin
DefaultTRUE

saveState=list(casouttable)

specifies the table to store the generated aStore model.

For more information about specifying the saveState parameter, see the common casouttable parameter.

splitOnce=TRUE | FALSE

when set to True, the analysis variable only appears once for each path from the tree root node to the leaf node.

DefaultFALSE

stat=TRUE | FALSE

specifies whether to include the variable splitting information for each node in the decision tree.

DefaultFALSE

* table=list(castable)

specifies the settings for an input table.

Long formtable=list(name="table-name")
Shortcut formtable="table-name"

The castable value can be one or more of the following:

caslib="string"

specifies the caslib for the input table that you want to use with the action. By default, the active caslib is used. Specify a value only if you need to access a table from a different caslib.

computedOnDemand=TRUE | FALSE

when set to True, creates the computed variables when the table is loaded instead of when the action begins.

AliascompOnDemand
DefaultFALSE
computedVars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the names of the computed variables to create. Specify an expression for each variable in the computedVarsProgram parameter. If you do not specify this parameter, then all variables from computedVarsProgram are automatically included.

AliascompVars

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

computedVarsProgram="string"

specifies an expression for each computed variable that you include in the computedVars parameter.

AliascompPgm
dataSourceOptions=list(key-1=list(any-list-or-data-type-1) <, key-2=list(any-list-or-data-type-2), ...>)

specifies data source options.

Aliasesoptions
dataSource
importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the input table.

singlePass=TRUE | FALSE

when set to True, does not create a transient table on the server. Setting this parameter to True can be efficient, but the data might not have stable ordering upon repeated runs.

DefaultFALSE
vars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the variables to use in the action.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the input data.

whereTable=list(groupbytable)

specifies an input table that contains rows to use as a WHERE filter. If the vars parameter is not specified, then all the variable names that are common to the input table and the filtering table are used to find matching rows. If the where parameter for the input table and this parameter are specified, then this filtering table is applied first.

The groupbytable value can be one or more of the following:

casLib="string"

specifies the caslib for the filter table. By default, the active caslib is used.

dataSourceOptions=list(adls_noreq-parameters | bigquery-parameters | cas_noreq-parameters | clouddex-parameters | db2-parameters | dnfs-parameters | esp-parameters | fedsvr-parameters | gcs_noreq-parameters | hadoop-parameters | hana-parameters | impala-parameters | jdbc-parameters | mongodb-parameters | mysql-parameters | odbc-parameters | oracle-parameters | path-parameters | postgres-parameters | redshift-parameters | s3-parameters | sapiq-parameters | sforce-parameters | singlestore_standard-parameters | snowflake-parameters | spark-parameters | spde-parameters | sqlserver-parameters | ss_noreq-parameters | teradata-parameters | vertica-parameters | yellowbrick-parameters)

specifies data source options.

Aliasesoptions
dataSource

For more information about specifying the dataSourceOptions parameter, see the common dataSourceOptions parameter.

importOptions=list(fileType="ANY" | "AUDIO" | "AUTO" | "BASESAS" | "CSV" | "DOCUMENT" | "DTA" | "ESP" | "EXCEL" | "FMT" | "HDAT" | "IMAGE" | "JMP" | "LASR" | "PARQUET" | "SPSS" | "VIDEO" | "XLS", fileType-specific-parameters)

specifies the settings for reading a table from a data source.

Aliasimport

For more information about specifying the importOptions parameter, see the common importOptions parameter.

* name="table-name"

specifies the name of the filter table.

vars=list( list(casinvardesc-1) <, list(casinvardesc-2), ...>)

specifies the variable names to use from the filter table.

The casinvardesc value can be one or more of the following:

format="string"

specifies the format to apply to the variable.

formattedLength=integer

specifies the length of format field plus the length of the format precision.

label="string"

specifies the descriptive label for the variable.

* name="variable-name"

specifies the name for the variable.

nfd=integer

specifies the length of the format precision.

nfl=integer

specifies the length of the format field.

where="where-expression"

specifies an expression for subsetting the data from the filter table.

* target="variable-name"

specifies the target or response variable for training. If the variable is numeric, but not specified in the nominal= parameter and nbinstarget= is not specified, then a regression tree is trained.

updateBin=TRUE | FALSE

when set to True, the bin information for all the numerical analysis variables is computed on each node that is used for a split. All the splits make use of the updated bin information. Note that this changes the bin information significantly compared to the bin information that is used in the input model. As a result, this might bias the splits that occur later.

DefaultFALSE

userDefinedSplit=list(userDefinedSplit)

indicates that you will manually define the split of a single node as specified in the nodeId parameter. The userDefinedSplit parameter contains sub-parameters needed for defining this split.

Long formuserDefinedSplit=list(splitVar="variable-name")
Shortcut formuserDefinedSplit="variable-name"

The userDefinedSplit value can be one or more of the following:

intervalSplit=list(double-1 <, double-2, ...>)

is a list of interval split points. The number of branches is the number of numbers listed plus one. The branches are numbered with the 0 branch being all values less than the least value. Each other branch is numbered in increasing order.

missingBranch=64-bit-integer

indicates to which branch missing values should be assigned. By default, missing values are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0
nominalSplit=list(any-list-or-data-type-1 <, any-list-or-data-type-2, ...>)

is a list of string lists that contain nominal split values. The number of branches is the number of string lists. The branches are ordered in which they are given to the action, with the first branch being numbered as branch 0.

splitVar="variable-name"

indicates which variable the decision tree is to use for splitting.

unseenBranch=64-bit-integer

indicates to which branch unseen levels should be assigned. By default, unseen levels are assigned to the left-most branch. For continuous variable splits, this corresponds to the branch with the smallest values.

Minimum value0

varImp=TRUE | FALSE

specifies whether the variable importance information is generated. The importance value is determined by the total Gini reduction.

DefaultFALSE
Last updated: February 15, 2023