Last updated on 2026-08-01 00:51:05 CEST.
| Flavor | Version | Tinstall | Tcheck | Ttotal | Status | Flags |
|---|---|---|---|---|---|---|
| r-devel-linux-x86_64-debian-clang | 1.2-29 | 15.47 | 253.21 | 268.68 | ERROR | |
| r-devel-linux-x86_64-debian-gcc | 1.2-29 | 10.38 | 192.05 | 202.43 | ERROR | |
| r-devel-linux-x86_64-fedora-clang | 1.2-29 | 24.00 | 399.49 | 423.49 | ERROR | |
| r-devel-linux-x86_64-fedora-gcc | 1.2-29 | 10.00 | 185.24 | 195.24 | ERROR | |
| r-devel-windows-x86_64 | 1.2-29 | 18.00 | 314.00 | 332.00 | ERROR | |
| r-patched-linux-x86_64 | 1.2-29 | 15.11 | 241.06 | 256.17 | ERROR | |
| r-release-linux-x86_64 | 1.2-29 | 14.35 | 246.32 | 260.67 | ERROR | |
| r-release-macos-arm64 | 1.2-29 | 4.00 | 78.00 | 82.00 | OK | |
| r-release-macos-x86_64 | 1.2-29 | 12.00 | 392.00 | 404.00 | OK | |
| r-release-windows-x86_64 | 1.2-29 | 18.00 | 304.00 | 322.00 | ERROR | |
| r-oldrel-macos-arm64 | 1.2-29 | 3.00 | 81.00 | 84.00 | OK | |
| r-oldrel-macos-x86_64 | 1.2-29 | 12.00 | 597.00 | 609.00 | OK | |
| r-oldrel-windows-x86_64 | 1.2-29 | 24.00 | 381.00 | 405.00 | ERROR |
Version: 1.2-29
Check: examples
Result: ERROR
Running examples in ‘partykit-Ex.R’ failed
The error most likely occurred in:
> base::assign(".ptime", proc.time(), pos = "CheckExEnv")
> ### Name: glmtree
> ### Title: Generalized Linear Model Trees
> ### Aliases: glmtree plot.glmtree predict.glmtree print.glmtree
> ### Keywords: tree
>
> ### ** Examples
>
> if(require("mlbench") && require("vcd")) {
+
+ ## Pima Indians diabetes data
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ ## recursive partitioning of a logistic regression model
+ pid_tree2 <- glmtree(diabetes ~ glucose | pregnant +
+ pressure + triceps + insulin + mass + pedigree + age,
+ data = PimaIndiansDiabetes, family = binomial)
+
+ ## printing whole tree or individual nodes
+ print(pid_tree2)
+ print(pid_tree2, node = 1)
+
+ ## visualization
+ plot(pid_tree2)
+ plot(pid_tree2, tp_args = list(cdplot = TRUE))
+ plot(pid_tree2, terminal_panel = NULL)
+
+ ## estimated parameters
+ coef(pid_tree2)
+ coef(pid_tree2, node = 5)
+ summary(pid_tree2, node = 5)
+
+ ## deviance, log-likelihood and information criteria
+ deviance(pid_tree2)
+ logLik(pid_tree2)
+ AIC(pid_tree2)
+ BIC(pid_tree2)
+
+ ## different types of predictions
+ pid <- head(PimaIndiansDiabetes)
+ predict(pid_tree2, newdata = pid, type = "node")
+ predict(pid_tree2, newdata = pid, type = "response")
+ predict(pid_tree2, newdata = pid, type = "link")
+
+ }
Loading required package: mlbench
Loading required package: vcd
Warning in data("PimaIndiansDiabetes", package = "mlbench") :
data set ‘PimaIndiansDiabetes’ not found
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: glmtree ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
Execution halted
Flavors: r-devel-linux-x86_64-debian-clang, r-devel-linux-x86_64-debian-gcc, r-patched-linux-x86_64, r-release-linux-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’ [5s/7s]
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’ [5s/6s]
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’ [2s/2s]
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [14s/18s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’ [2s/3s]
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [34s/42s]
Running ‘regtest-honesty.R’ [2s/2s]
Running ‘regtest-lmtree.R’ [3s/3s]
Running ‘regtest-nmax.R’ [2s/3s]
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’ [2s/2s]
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’ [2s/3s]
Running ‘regtest-party.R’ [5s/5s]
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’ [2s/2s]
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’ [2s/3s]
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-devel-linux-x86_64-debian-clang
Version: 1.2-29
Check: re-building of vignette outputs
Result: ERROR
Error(s) in re-building vignettes:
...
--- re-building ‘constparty.Rnw’ using knitr
--- finished re-building ‘constparty.Rnw’
--- re-building ‘ctree.Rnw’ using knitr
--- finished re-building ‘ctree.Rnw’
--- re-building ‘mob.Rnw’ using knitr
Quitting from mob.Rnw:443-445 [PimaIndiansDiabetes-mob]
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
<error/rlang_error>
Error:
! object 'PimaIndiansDiabetes' not found
---
Backtrace:
x
1. +-stats::model.frame(...)
2. \-Formula:::model.frame.Formula(...)
3. +-stats::model.frame(...)
4. +-stats::terms(formula, lhs = lhs, rhs = rhs, data = data, dot = dot)
5. \-Formula:::terms.Formula(...)
6. +-stats::terms(form, ...)
7. \-stats::terms.formula(form, ...)
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Error: processing vignette 'mob.Rnw' failed with diagnostics:
object 'PimaIndiansDiabetes' not found
--- failed re-building ‘mob.Rnw’
--- re-building ‘partykit.Rnw’ using knitr
--- finished re-building ‘partykit.Rnw’
SUMMARY: processing the following file failed:
‘mob.Rnw’
Error: Vignette re-building failed.
Execution halted
Flavors: r-devel-linux-x86_64-debian-clang, r-devel-linux-x86_64-debian-gcc, r-patched-linux-x86_64, r-release-linux-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’ [5s/6s]
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’ [4s/5s]
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’ [2s/2s]
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [9s/11s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’ [2s/2s]
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [31s/43s]
Running ‘regtest-honesty.R’ [2s/2s]
Running ‘regtest-lmtree.R’ [2s/2s]
Running ‘regtest-nmax.R’ [1s/2s]
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’ [1s/2s]
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’ [2s/2s]
Running ‘regtest-party.R’ [3s/4s]
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’ [1s/2s]
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’ [2s/2s]
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-devel-linux-x86_64-debian-gcc
Version: 1.2-29
Check: for new files in some other directories
Result: NOTE
Found the following files/directories:
‘~/tmp/scratch/Rtmp0tT8rz’ ‘~/tmp/scratch/Rtmp12KXDF’
‘~/tmp/scratch/Rtmp159K4k’ ‘~/tmp/scratch/Rtmp1BWSju’
‘~/tmp/scratch/Rtmp1IsWgT’ ‘~/tmp/scratch/Rtmp21262E’
‘~/tmp/scratch/Rtmp2ICQMD’ ‘~/tmp/scratch/Rtmp2IVDpn’
‘~/tmp/scratch/Rtmp2eGoTf’ ‘~/tmp/scratch/Rtmp2pRASJ’
‘~/tmp/scratch/Rtmp2uYtAV’ ‘~/tmp/scratch/Rtmp3NhI0W’
‘~/tmp/scratch/Rtmp3XcnB2’ ‘~/tmp/scratch/Rtmp3vRCGx’
‘~/tmp/scratch/Rtmp4ZMzwG’ ‘~/tmp/scratch/Rtmp4hD7PF’
‘~/tmp/scratch/Rtmp4iN4ge’ ‘~/tmp/scratch/Rtmp4m8Nqb’
‘~/tmp/scratch/Rtmp5NFeJR’ ‘~/tmp/scratch/Rtmp5aLUH5’
‘~/tmp/scratch/Rtmp5i66h4’ ‘~/tmp/scratch/Rtmp63piWt’
‘~/tmp/scratch/Rtmp66npL6’ ‘~/tmp/scratch/Rtmp6oWY7n’
‘~/tmp/scratch/Rtmp7APpP8’ ‘~/tmp/scratch/Rtmp7qEKNQ’
‘~/tmp/scratch/Rtmp8QwCTZ’ ‘~/tmp/scratch/Rtmp9AIR0L’
‘~/tmp/scratch/Rtmp9Qz1JH’ ‘~/tmp/scratch/Rtmp9yoT1z’
‘~/tmp/scratch/Rtmp9yuNrr’ ‘~/tmp/scratch/RtmpBIuMuV’
‘~/tmp/scratch/RtmpBooUw8’ ‘~/tmp/scratch/RtmpBr2u6I’
‘~/tmp/scratch/RtmpBsnze6’ ‘~/tmp/scratch/RtmpC1wxEs’
‘~/tmp/scratch/RtmpE6mTKr’ ‘~/tmp/scratch/RtmpEHsPMR’
‘~/tmp/scratch/RtmpEgPrPT’ ‘~/tmp/scratch/RtmpFXn8lN’
‘~/tmp/scratch/RtmpGFzGAz’ ‘~/tmp/scratch/RtmpGSdX8M’
‘~/tmp/scratch/RtmpH11Qq7’ ‘~/tmp/scratch/RtmpHIJbxi’
‘~/tmp/scratch/RtmpHN54WN’ ‘~/tmp/scratch/RtmpHW7Qsu’
‘~/tmp/scratch/RtmpHi3veA’ ‘~/tmp/scratch/RtmpHnug8r’
‘~/tmp/scratch/RtmpI6fGsw’ ‘~/tmp/scratch/RtmpIPLKL5’
‘~/tmp/scratch/RtmpJITjTz’ ‘~/tmp/scratch/RtmpJboFiS’
‘~/tmp/scratch/RtmpJi71jH’ ‘~/tmp/scratch/RtmpJm06ts’
‘~/tmp/scratch/RtmpJnrSQH’ ‘~/tmp/scratch/RtmpKIjO9f’
‘~/tmp/scratch/RtmpKvOkrq’ ‘~/tmp/scratch/RtmpLYl53M’
‘~/tmp/scratch/RtmpLtHBBZ’ ‘~/tmp/scratch/RtmpMqEGNJ’
‘~/tmp/scratch/RtmpMvUvV4’ ‘~/tmp/scratch/RtmpMwkUcw’
‘~/tmp/scratch/RtmpNZkbyx’ ‘~/tmp/scratch/RtmpO4nY3G’
‘~/tmp/scratch/RtmpOFqYEI’ ‘~/tmp/scratch/RtmpOG3rZ1’
‘~/tmp/scratch/RtmpOT8qTN’ ‘~/tmp/scratch/RtmpOTUFFB’
‘~/tmp/scratch/RtmpOaUYYN’ ‘~/tmp/scratch/RtmpOkcSWg’
‘~/tmp/scratch/RtmpPiLeAc’ ‘~/tmp/scratch/RtmpPjmMuF’
‘~/tmp/scratch/RtmpQRwHFd’ ‘~/tmp/scratch/RtmpRLF7iE’
‘~/tmp/scratch/RtmpRUjEKe’ ‘~/tmp/scratch/RtmpRz0sL1’
‘~/tmp/scratch/RtmpS37exP’ ‘~/tmp/scratch/RtmpT09UZi’
‘~/tmp/scratch/RtmpTZJMuy’ ‘~/tmp/scratch/RtmpTc6qJu’
‘~/tmp/scratch/RtmpThsMfv’ ‘~/tmp/scratch/RtmpUBQTFR’
‘~/tmp/scratch/RtmpUqFtDx’ ‘~/tmp/scratch/RtmpVIt5Ui’
‘~/tmp/scratch/RtmpVKnZuc’ ‘~/tmp/scratch/RtmpVZvKTG’
‘~/tmp/scratch/RtmpVa2aif’ ‘~/tmp/scratch/RtmpW0I3xF’
‘~/tmp/scratch/RtmpWnjwRq’ ‘~/tmp/scratch/RtmpX7Mlfm’
‘~/tmp/scratch/RtmpXO7MR7’ ‘~/tmp/scratch/RtmpXV1oJe’
‘~/tmp/scratch/RtmpXnoBsH’ ‘~/tmp/scratch/RtmpY7TeiT’
‘~/tmp/scratch/RtmpYAoYIO’ ‘~/tmp/scratch/RtmpYPWYwM’
‘~/tmp/scratch/RtmpZPNzLQ’ ‘~/tmp/scratch/RtmpZVv6Rz’
‘~/tmp/scratch/Rtmpa2axtw’ ‘~/tmp/scratch/Rtmpa2rUdd’
‘~/tmp/scratch/RtmpaKxM8e’ ‘~/tmp/scratch/Rtmpb3ieq8’
‘~/tmp/scratch/RtmpbKiE15’ ‘~/tmp/scratch/RtmpbhSlnQ’
‘~/tmp/scratch/RtmpchwMBJ’ ‘~/tmp/scratch/RtmpcstfFB’
‘~/tmp/scratch/Rtmpd4YDa1’ ‘~/tmp/scratch/RtmpdT7FOd’
‘~/tmp/scratch/RtmpdbqbQz’ ‘~/tmp/scratch/RtmpeAPfav’
‘~/tmp/scratch/Rtmpex7zPA’ ‘~/tmp/scratch/RtmpfkFUsa’
‘~/tmp/scratch/RtmpgKYaXb’ ‘~/tmp/scratch/RtmpgWD1PC’
‘~/tmp/scratch/RtmpgYDgaH’ ‘~/tmp/scratch/RtmpgcIuzy’
‘~/tmp/scratch/RtmphTJf1q’ ‘~/tmp/scratch/RtmpjDkzud’
‘~/tmp/scratch/RtmpjNQSPW’ ‘~/tmp/scratch/Rtmpjc3uWS’
‘~/tmp/scratch/Rtmpjua8gk’ ‘~/tmp/scratch/RtmpkKtFzJ’
‘~/tmp/scratch/Rtmpko3SkP’ ‘~/tmp/scratch/RtmpkqGeGw’
‘~/tmp/scratch/RtmplaPbxX’ ‘~/tmp/scratch/RtmpleJeIU’
‘~/tmp/scratch/RtmpmIEZoQ’ ‘~/tmp/scratch/RtmpmPGDJT’
‘~/tmp/scratch/RtmpmPzNs3’ ‘~/tmp/scratch/Rtmpmz8qOO’
‘~/tmp/scratch/RtmpnCBUuJ’ ‘~/tmp/scratch/RtmpnUBJIf’
‘~/tmp/scratch/RtmpnY653A’ ‘~/tmp/scratch/Rtmpng7iEz’
‘~/tmp/scratch/Rtmpny2Xp1’ ‘~/tmp/scratch/RtmpnybCli’
‘~/tmp/scratch/RtmpoDBcrD’ ‘~/tmp/scratch/RtmpoIxXb7’
‘~/tmp/scratch/RtmpoKZJ20’ ‘~/tmp/scratch/RtmpomnfPg’
‘~/tmp/scratch/Rtmpp3XFHn’ ‘~/tmp/scratch/RtmppBWiSO’
‘~/tmp/scratch/RtmppFL6dZ’ ‘~/tmp/scratch/RtmppICu3f’
‘~/tmp/scratch/RtmppSv4PR’ ‘~/tmp/scratch/RtmppW5rHN’
‘~/tmp/scratch/RtmppiluJC’ ‘~/tmp/scratch/RtmpqMieem’
‘~/tmp/scratch/RtmpqiAk2k’ ‘~/tmp/scratch/RtmpqmzKzu’
‘~/tmp/scratch/RtmpqtUCiD’ ‘~/tmp/scratch/RtmprasTCO’
‘~/tmp/scratch/RtmprkR24D’ ‘~/tmp/scratch/RtmprwQVXT’
‘~/tmp/scratch/RtmpsCpb72’ ‘~/tmp/scratch/RtmpsIaDen’
‘~/tmp/scratch/RtmpsZZOjB’ ‘~/tmp/scratch/RtmpscVSxS’
‘~/tmp/scratch/RtmpsiiGHo’ ‘~/tmp/scratch/RtmpsppD1B’
‘~/tmp/scratch/Rtmpst7JYU’ ‘~/tmp/scratch/RtmptIZOAK’
‘~/tmp/scratch/RtmptQdMjD’ ‘~/tmp/scratch/RtmpuJZJ5X’
‘~/tmp/scratch/RtmpuNAhMQ’ ‘~/tmp/scratch/RtmpuPZ7bq’
‘~/tmp/scratch/Rtmpv7RVgl’ ‘~/tmp/scratch/RtmpwgFKmP’
‘~/tmp/scratch/Rtmpwtkxvx’ ‘~/tmp/scratch/RtmpxfXp0M’
‘~/tmp/scratch/RtmpxgnNCt’ ‘~/tmp/scratch/RtmpxpED2g’
‘~/tmp/scratch/RtmpxwHLNG’ ‘~/tmp/scratch/Rtmpy3mNIZ’
‘~/tmp/scratch/RtmpyFDnVf’ ‘~/tmp/scratch/RtmpyGJ0i3’
‘~/tmp/scratch/RtmpyVpIV5’ ‘~/tmp/scratch/RtmpyiOtVK’
‘~/tmp/scratch/Rtmpymj4GA’ ‘~/tmp/scratch/Rtmpzizpfp’
‘~/tmp/scratch/RtmpzpfDg6’ ‘~/tmp/scratch/xvfb-run.0Ze5Vd’
‘~/tmp/scratch/xvfb-run.1Jh5yr’ ‘~/tmp/scratch/xvfb-run.1VKI24’
‘~/tmp/scratch/xvfb-run.2BF4Lp’ ‘~/tmp/scratch/xvfb-run.3ffO3A’
‘~/tmp/scratch/xvfb-run.40PLAQ’ ‘~/tmp/scratch/xvfb-run.72QtBg’
‘~/tmp/scratch/xvfb-run.7m1pEs’ ‘~/tmp/scratch/xvfb-run.81tuRY’
‘~/tmp/scratch/xvfb-run.9gnkWQ’ ‘~/tmp/scratch/xvfb-run.9pjlfK’
‘~/tmp/scratch/xvfb-run.9q5tNE’ ‘~/tmp/scratch/xvfb-run.AHwiOF’
‘~/tmp/scratch/xvfb-run.Aw2DLX’ ‘~/tmp/scratch/xvfb-run.BIKWLH’
‘~/tmp/scratch/xvfb-run.BTBczy’ ‘~/tmp/scratch/xvfb-run.BiUPF9’
‘~/tmp/scratch/xvfb-run.CX2lH0’ ‘~/tmp/scratch/xvfb-run.EofsZN’
‘~/tmp/scratch/xvfb-run.FlRQuZ’ ‘~/tmp/scratch/xvfb-run.Hu9cBT’
‘~/tmp/scratch/xvfb-run.Hw5OxD’ ‘~/tmp/scratch/xvfb-run.IT3Eam’
‘~/tmp/scratch/xvfb-run.J4OHt7’ ‘~/tmp/scratch/xvfb-run.JmfyWy’
‘~/tmp/scratch/xvfb-run.JrViix’ ‘~/tmp/scratch/xvfb-run.JwvZIx’
‘~/tmp/scratch/xvfb-run.MQsQcE’ ‘~/tmp/scratch/xvfb-run.NCHyMB’
‘~/tmp/scratch/xvfb-run.NQMDuc’ ‘~/tmp/scratch/xvfb-run.NRGLag’
‘~/tmp/scratch/xvfb-run.QTq6Tk’ ‘~/tmp/scratch/xvfb-run.QwBUaO’
‘~/tmp/scratch/xvfb-run.SsD3eH’ ‘~/tmp/scratch/xvfb-run.UQEcmC’
‘~/tmp/scratch/xvfb-run.V8ZRzR’ ‘~/tmp/scratch/xvfb-run.VvsSky’
‘~/tmp/scratch/xvfb-run.X5XHBF’ ‘~/tmp/scratch/xvfb-run.X7RnHh’
‘~/tmp/scratch/xvfb-run.XPGA6c’ ‘~/tmp/scratch/xvfb-run.XujblN’
‘~/tmp/scratch/xvfb-run.Yi4h1e’ ‘~/tmp/scratch/xvfb-run.ZUwowr’
‘~/tmp/scratch/xvfb-run.ZYV0TD’ ‘~/tmp/scratch/xvfb-run.cBdHQu’
‘~/tmp/scratch/xvfb-run.cHJnbn’ ‘~/tmp/scratch/xvfb-run.ce7ku8’
‘~/tmp/scratch/xvfb-run.dk9XJX’ ‘~/tmp/scratch/xvfb-run.dpazio’
‘~/tmp/scratch/xvfb-run.eVxN1m’ ‘~/tmp/scratch/xvfb-run.fd5Kqc’
‘~/tmp/scratch/xvfb-run.g3gZD0’ ‘~/tmp/scratch/xvfb-run.gfFNiE’
‘~/tmp/scratch/xvfb-run.hPqpsY’ ‘~/tmp/scratch/xvfb-run.jchors’
‘~/tmp/scratch/xvfb-run.jyPfKh’ ‘~/tmp/scratch/xvfb-run.kd1rR3’
‘~/tmp/scratch/xvfb-run.lfOzcP’ ‘~/tmp/scratch/xvfb-run.mrrs8C’
‘~/tmp/scratch/xvfb-run.o6k9T8’ ‘~/tmp/scratch/xvfb-run.pBUDs4’
‘~/tmp/scratch/xvfb-run.qApWe4’ ‘~/tmp/scratch/xvfb-run.qCUbDx’
‘~/tmp/scratch/xvfb-run.rF7P3W’ ‘~/tmp/scratch/xvfb-run.rPifQy’
‘~/tmp/scratch/xvfb-run.smA6yg’ ‘~/tmp/scratch/xvfb-run.tGPgSH’
‘~/tmp/scratch/xvfb-run.tLNmqv’ ‘~/tmp/scratch/xvfb-run.tTsELm’
‘~/tmp/scratch/xvfb-run.unqaur’ ‘~/tmp/scratch/xvfb-run.vcZ8zo’
‘~/tmp/scratch/xvfb-run.vy4EXM’ ‘~/tmp/scratch/xvfb-run.xU6QZw’
‘~/tmp/scratch/xvfb-run.xmnTui’ ‘~/tmp/scratch/xvfb-run.xsvKjL’
‘~/tmp/scratch/xvfb-run.yHdkLO’ ‘~/tmp/scratch/xvfb-run.zW9rze’
‘~/tmp/scratch/xvfb-run.zl5sdJ’
‘/dev/shm/sm_segment.gimli1.1001.19120000.0’
‘/dev/shm/sm_segment.gimli1.1001.1a2a0000.0’
‘/dev/shm/sm_segment.gimli1.1001.1bb70000.0’
‘/dev/shm/sm_segment.gimli1.1001.1e960000.0’
‘/dev/shm/sm_segment.gimli1.1001.24340000.0’
‘/dev/shm/sm_segment.gimli1.1001.26970000.0’
‘/dev/shm/sm_segment.gimli1.1001.32d90000.0’
‘/dev/shm/sm_segment.gimli1.1001.3560000.0’
‘/dev/shm/sm_segment.gimli1.1001.3da00000.0’
‘/dev/shm/sm_segment.gimli1.1001.3e800000.0’
‘/dev/shm/sm_segment.gimli1.1001.49c60000.0’
‘/dev/shm/sm_segment.gimli1.1001.4c070000.0’
‘/dev/shm/sm_segment.gimli1.1001.4f430000.0’
‘/dev/shm/sm_segment.gimli1.1001.51d60000.0’
‘/dev/shm/sm_segment.gimli1.1001.5acb0000.0’
‘/dev/shm/sm_segment.gimli1.1001.5d0000.0’
‘/dev/shm/sm_segment.gimli1.1001.5f8a0000.0’
‘/dev/shm/sm_segment.gimli1.1001.67f30000.0’
‘/dev/shm/sm_segment.gimli1.1001.77bf0000.0’
‘/dev/shm/sm_segment.gimli1.1001.7bbd0000.0’
‘/dev/shm/sm_segment.gimli1.1001.7e70000.0’
‘/dev/shm/sm_segment.gimli1.1001.81700000.0’
‘/dev/shm/sm_segment.gimli1.1001.82550000.0’
‘/dev/shm/sm_segment.gimli1.1001.82d60000.0’
‘/dev/shm/sm_segment.gimli1.1001.96900000.0’
‘/dev/shm/sm_segment.gimli1.1001.9d940000.0’
‘/dev/shm/sm_segment.gimli1.1001.9e390000.0’
‘/dev/shm/sm_segment.gimli1.1001.a2ce0000.0’
‘/dev/shm/sm_segment.gimli1.1001.a63a0000.0’
‘/dev/shm/sm_segment.gimli1.1001.a6cf0000.0’
‘/dev/shm/sm_segment.gimli1.1001.adfd0000.0’
‘/dev/shm/sm_segment.gimli1.1001.b8180000.0’
‘/dev/shm/sm_segment.gimli1.1001.bad60000.0’
‘/dev/shm/sm_segment.gimli1.1001.bda10000.0’
‘/dev/shm/sm_segment.gimli1.1001.be240000.0’
‘/dev/shm/sm_segment.gimli1.1001.c5b70000.0’
‘/dev/shm/sm_segment.gimli1.1001.c6f50000.0’
‘/dev/shm/sm_segment.gimli1.1001.c7640000.0’
‘/dev/shm/sm_segment.gimli1.1001.cd7c0000.0’
‘/dev/shm/sm_segment.gimli1.1001.d0b10000.0’
‘/dev/shm/sm_segment.gimli1.1001.d4f00000.0’
‘/dev/shm/sm_segment.gimli1.1001.e9660000.0’
‘/dev/shm/sm_segment.gimli1.1001.eaf70000.0’
‘/dev/shm/sm_segment.gimli1.1001.f5010000.0’
‘/dev/shm/sm_segment.gimli1.1001.fc3b0000.0’
‘/dev/shm/sm_segment.gimli1.1001.fefd0000.0’
‘~/.cache/pocl/uncached/tempfile_0K4HG5’
‘~/.cache/pocl/uncached/tempfile_1dxtRQ’
‘~/.cache/pocl/uncached/tempfile_4z01PT’
‘~/.cache/pocl/uncached/tempfile_6Y7Y8X’
‘~/.cache/pocl/uncached/tempfile_6Z0tul’
‘~/.cache/pocl/uncached/tempfile_8AGRz8’
‘~/.cache/pocl/uncached/tempfile_8jTv7B’
‘~/.cache/pocl/uncached/tempfile_Bm2USW’
‘~/.cache/pocl/uncached/tempfile_CzJAvf’
‘~/.cache/pocl/uncached/tempfile_DOg881’
‘~/.cache/pocl/uncached/tempfile_DtjIQg’
‘~/.cache/pocl/uncached/tempfile_Hw2oVw’
‘~/.cache/pocl/uncached/tempfile_IOaU8k’
‘~/.cache/pocl/uncached/tempfile_IZii7X’
‘~/.cache/pocl/uncached/tempfile_M7WY3Y’
‘~/.cache/pocl/uncached/tempfile_OsoLYa’
‘~/.cache/pocl/uncached/tempfile_QUgulZ’
‘~/.cache/pocl/uncached/tempfile_VNdMhz’
‘~/.cache/pocl/uncached/tempfile_VYzhbe’
‘~/.cache/pocl/uncached/tempfile_YZRLyH’
‘~/.cache/pocl/uncached/tempfile_bGZZTL’
‘~/.cache/pocl/uncached/tempfile_cusjCX’
‘~/.cache/pocl/uncached/tempfile_fjRk9u’
‘~/.cache/pocl/uncached/tempfile_iAJOV2’
‘~/.cache/pocl/uncached/tempfile_iUH4ab’
‘~/.cache/pocl/uncached/tempfile_ilGCbv’
‘~/.cache/pocl/uncached/tempfile_jx4Js9’
‘~/.cache/pocl/uncached/tempfile_kGKpxC’
‘~/.cache/pocl/uncached/tempfile_l1gNlI’
‘~/.cache/pocl/uncached/tempfile_lI1iSz’
‘~/.cache/pocl/uncached/tempfile_lIpYwx’
‘~/.cache/pocl/uncached/tempfile_nflEHp’
‘~/.cache/pocl/uncached/tempfile_on5eHU’
‘~/.cache/pocl/uncached/tempfile_ouPzDz’
‘~/.cache/pocl/uncached/tempfile_pfudYn’
‘~/.cache/pocl/uncached/tempfile_qGSJeq’
‘~/.cache/pocl/uncached/tempfile_qW1ctd’
‘~/.cache/pocl/uncached/tempfile_t2aLjs’
‘~/.cache/pocl/uncached/tempfile_t5BL1N’
‘~/.cache/pocl/uncached/tempfile_vDXNSI’
‘~/.cache/pocl/uncached/tempfile_wb3OZf’
‘~/.cache/pocl/uncached/tempfile_xIHZFq’
‘~/.cache/pocl/uncached/tempfile_xn8JqS’
‘~/.cache/pocl/uncached/tempfile_zBnyip’
‘~/.cache/pocl/uncached/tempfile_zJzzTQ’
‘~/.cache/pocl/uncached/tempfile_zmbiId’
Flavor: r-devel-linux-x86_64-debian-gcc
Version: 1.2-29
Check: examples
Result: ERROR
Running examples in ‘partykit-Ex.R’ failed
The error most likely occurred in:
> ### Name: glmtree
> ### Title: Generalized Linear Model Trees
> ### Aliases: glmtree plot.glmtree predict.glmtree print.glmtree
> ### Keywords: tree
>
> ### ** Examples
>
> if(require("mlbench") && require("vcd")) {
+
+ ## Pima Indians diabetes data
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ ## recursive partitioning of a logistic regression model
+ pid_tree2 <- glmtree(diabetes ~ glucose | pregnant +
+ pressure + triceps + insulin + mass + pedigree + age,
+ data = PimaIndiansDiabetes, family = binomial)
+
+ ## printing whole tree or individual nodes
+ print(pid_tree2)
+ print(pid_tree2, node = 1)
+
+ ## visualization
+ plot(pid_tree2)
+ plot(pid_tree2, tp_args = list(cdplot = TRUE))
+ plot(pid_tree2, terminal_panel = NULL)
+
+ ## estimated parameters
+ coef(pid_tree2)
+ coef(pid_tree2, node = 5)
+ summary(pid_tree2, node = 5)
+
+ ## deviance, log-likelihood and information criteria
+ deviance(pid_tree2)
+ logLik(pid_tree2)
+ AIC(pid_tree2)
+ BIC(pid_tree2)
+
+ ## different types of predictions
+ pid <- head(PimaIndiansDiabetes)
+ predict(pid_tree2, newdata = pid, type = "node")
+ predict(pid_tree2, newdata = pid, type = "response")
+ predict(pid_tree2, newdata = pid, type = "link")
+
+ }
Loading required package: mlbench
Loading required package: vcd
Warning in data("PimaIndiansDiabetes", package = "mlbench") :
data set ‘PimaIndiansDiabetes’ not found
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: glmtree ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
Execution halted
Flavors: r-devel-linux-x86_64-fedora-clang, r-devel-linux-x86_64-fedora-gcc, r-devel-windows-x86_64, r-release-windows-x86_64, r-oldrel-windows-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’ [9s/12s]
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’ [9s/12s]
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [21s/28s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [52s/58s]
Running ‘regtest-honesty.R’
Running ‘regtest-lmtree.R’
Running ‘regtest-nmax.R’
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’
Running ‘regtest-party.R’
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-devel-linux-x86_64-fedora-clang
Version: 1.2-29
Check: re-building of vignette outputs
Result: ERROR
Error(s) in re-building vignettes:
--- re-building ‘constparty.Rnw’ using knitr
--- finished re-building ‘constparty.Rnw’
--- re-building ‘ctree.Rnw’ using knitr
--- finished re-building ‘ctree.Rnw’
--- re-building ‘mob.Rnw’ using knitr
Quitting from mob.Rnw:443-445 [PimaIndiansDiabetes-mob]
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
<error/rlang_error>
Error:
! object 'PimaIndiansDiabetes' not found
---
Backtrace:
x
1. +-stats::model.frame(...)
2. \-Formula:::model.frame.Formula(...)
3. +-stats::model.frame(...)
4. +-stats::terms(formula, lhs = lhs, rhs = rhs, data = data, dot = dot)
5. \-Formula:::terms.Formula(...)
6. +-stats::terms(form, ...)
7. \-stats::terms.formula(form, ...)
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Error: processing vignette 'mob.Rnw' failed with diagnostics:
object 'PimaIndiansDiabetes' not found
--- failed re-building ‘mob.Rnw’
--- re-building ‘partykit.Rnw’ using knitr
--- finished re-building ‘partykit.Rnw’
SUMMARY: processing the following file failed:
‘mob.Rnw’
Error: Vignette re-building failed.
Execution halted
Flavors: r-devel-linux-x86_64-fedora-clang, r-devel-linux-x86_64-fedora-gcc, r-devel-windows-x86_64, r-release-windows-x86_64, r-oldrel-windows-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [8s/15s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [27s/51s]
Running ‘regtest-honesty.R’
Running ‘regtest-lmtree.R’
Running ‘regtest-nmax.R’
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’
Running ‘regtest-party.R’
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-devel-linux-x86_64-fedora-gcc
Version: 1.2-29
Check: tests
Result: ERROR
Running 'bugfixes.R' [5s]
Comparing 'bugfixes.Rout' to 'bugfixes.Rout.save' ... OK
Running 'constparty.R' [5s]
Comparing 'constparty.Rout' to 'constparty.Rout.save' ... OK
Running 'regtest-MIA.R' [2s]
Comparing 'regtest-MIA.Rout' to 'regtest-MIA.Rout.save' ... OK
Running 'regtest-cforest.R' [9s]
Comparing 'regtest-cforest.Rout' to 'regtest-cforest.Rout.save' ... OK
Running 'regtest-ctree.R' [2s]
Comparing 'regtest-ctree.Rout' to 'regtest-ctree.Rout.save' ... OK
Running 'regtest-glmtree.R' [45s]
Running 'regtest-honesty.R' [2s]
Running 'regtest-lmtree.R' [3s]
Running 'regtest-nmax.R' [2s]
Comparing 'regtest-nmax.Rout' to 'regtest-nmax.Rout.save' ... OK
Running 'regtest-node.R' [1s]
Comparing 'regtest-node.Rout' to 'regtest-node.Rout.save' ... OK
Running 'regtest-party-random.R' [2s]
Running 'regtest-party.R' [5s]
Comparing 'regtest-party.Rout' to 'regtest-party.Rout.save' ... OK
Running 'regtest-split.R' [2s]
Comparing 'regtest-split.Rout' to 'regtest-split.Rout.save' ... OK
Running 'regtest-weights.R' [2s]
Comparing 'regtest-weights.Rout' to 'regtest-weights.Rout.save' ... OK
Running the tests in 'tests/regtest-glmtree.R' failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-devel-windows-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’ [5s/7s]
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’ [5s/7s]
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’ [2s/4s]
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [12s/15s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’ [2s/3s]
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [33s/40s]
Running ‘regtest-honesty.R’ [2s/2s]
Running ‘regtest-lmtree.R’ [3s/3s]
Running ‘regtest-nmax.R’ [2s/2s]
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’ [2s/3s]
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’ [2s/3s]
Running ‘regtest-party.R’ [5s/5s]
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’ [2s/2s]
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’ [2s/3s]
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-patched-linux-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running ‘bugfixes.R’ [5s/7s]
Comparing ‘bugfixes.Rout’ to ‘bugfixes.Rout.save’ ... OK
Running ‘constparty.R’ [5s/6s]
Comparing ‘constparty.Rout’ to ‘constparty.Rout.save’ ... OK
Running ‘regtest-MIA.R’ [2s/3s]
Comparing ‘regtest-MIA.Rout’ to ‘regtest-MIA.Rout.save’ ... OK
Running ‘regtest-cforest.R’ [12s/15s]
Comparing ‘regtest-cforest.Rout’ to ‘regtest-cforest.Rout.save’ ... OK
Running ‘regtest-ctree.R’ [2s/2s]
Comparing ‘regtest-ctree.Rout’ to ‘regtest-ctree.Rout.save’ ... OK
Running ‘regtest-glmtree.R’ [35s/45s]
Running ‘regtest-honesty.R’ [2s/3s]
Running ‘regtest-lmtree.R’ [3s/3s]
Running ‘regtest-nmax.R’ [2s/2s]
Comparing ‘regtest-nmax.Rout’ to ‘regtest-nmax.Rout.save’ ... OK
Running ‘regtest-node.R’ [2s/2s]
Comparing ‘regtest-node.Rout’ to ‘regtest-node.Rout.save’ ... OK
Running ‘regtest-party-random.R’ [2s/3s]
Running ‘regtest-party.R’ [5s/6s]
Comparing ‘regtest-party.Rout’ to ‘regtest-party.Rout.save’ ... OK
Running ‘regtest-split.R’ [2s/2s]
Comparing ‘regtest-split.Rout’ to ‘regtest-split.Rout.save’ ... OK
Running ‘regtest-weights.R’ [2s/2s]
Comparing ‘regtest-weights.Rout’ to ‘regtest-weights.Rout.save’ ... OK
Running the tests in ‘tests/regtest-glmtree.R’ failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-release-linux-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running 'bugfixes.R' [5s]
Comparing 'bugfixes.Rout' to 'bugfixes.Rout.save' ... OK
Running 'constparty.R' [4s]
Comparing 'constparty.Rout' to 'constparty.Rout.save' ... OK
Running 'regtest-MIA.R' [2s]
Comparing 'regtest-MIA.Rout' to 'regtest-MIA.Rout.save' ... OK
Running 'regtest-cforest.R' [8s]
Comparing 'regtest-cforest.Rout' to 'regtest-cforest.Rout.save' ... OK
Running 'regtest-ctree.R' [2s]
Comparing 'regtest-ctree.Rout' to 'regtest-ctree.Rout.save' ... OK
Running 'regtest-glmtree.R' [41s]
Running 'regtest-honesty.R' [2s]
Running 'regtest-lmtree.R' [2s]
Running 'regtest-nmax.R' [2s]
Comparing 'regtest-nmax.Rout' to 'regtest-nmax.Rout.save' ... OK
Running 'regtest-node.R' [2s]
Comparing 'regtest-node.Rout' to 'regtest-node.Rout.save' ... OK
Running 'regtest-party-random.R' [3s]
Running 'regtest-party.R' [4s]
Comparing 'regtest-party.Rout' to 'regtest-party.Rout.save' ... OK
Running 'regtest-split.R' [1s]
Comparing 'regtest-split.Rout' to 'regtest-split.Rout.save' ... OK
Running 'regtest-weights.R' [2s]
Comparing 'regtest-weights.Rout' to 'regtest-weights.Rout.save' ... OK
Running the tests in 'tests/regtest-glmtree.R' failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-release-windows-x86_64
Version: 1.2-29
Check: tests
Result: ERROR
Running 'bugfixes.R' [8s]
Comparing 'bugfixes.Rout' to 'bugfixes.Rout.save' ... OK
Running 'constparty.R' [7s]
Comparing 'constparty.Rout' to 'constparty.Rout.save' ... OK
Running 'regtest-MIA.R' [3s]
Comparing 'regtest-MIA.Rout' to 'regtest-MIA.Rout.save' ... OK
Running 'regtest-cforest.R' [13s]
Comparing 'regtest-cforest.Rout' to 'regtest-cforest.Rout.save' ... OK
Running 'regtest-ctree.R' [3s]
Comparing 'regtest-ctree.Rout' to 'regtest-ctree.Rout.save' ... OK
Running 'regtest-glmtree.R' [57s]
Running 'regtest-honesty.R' [3s]
Running 'regtest-lmtree.R' [4s]
Running 'regtest-nmax.R' [3s]
Comparing 'regtest-nmax.Rout' to 'regtest-nmax.Rout.save' ... OK
Running 'regtest-node.R' [2s]
Comparing 'regtest-node.Rout' to 'regtest-node.Rout.save' ... OK
Running 'regtest-party-random.R' [4s]
Running 'regtest-party.R' [6s]
Comparing 'regtest-party.Rout' to 'regtest-party.Rout.save' ... OK
Running 'regtest-split.R' [2s]
Comparing 'regtest-split.Rout' to 'regtest-split.Rout.save' ... OK
Running 'regtest-weights.R' [3s]
Comparing 'regtest-weights.Rout' to 'regtest-weights.Rout.save' ... OK
Running the tests in 'tests/regtest-glmtree.R' failed.
Complete output:
> suppressWarnings(RNGversion("3.5.2"))
>
> library("partykit")
Loading required package: grid
Loading required package: libcoin
Loading required package: mvtnorm
>
> set.seed(29)
> n <- 1000
> x <- runif(n)
> z <- runif(n)
> y <- rnorm(n, mean = x * c(-1, 1)[(z > 0.7) + 1], sd = 3)
> z_noise <- factor(sample(1:3, size = n, replace = TRUE))
> d <- data.frame(y = y, x = x, z = z, z_noise = z_noise)
>
>
> fmla <- as.formula("y ~ x | z + z_noise")
> fmly <- gaussian()
> fit <- partykit:::glmfit
>
> # versions of the data
> d1 <- d
> d1$z <- signif(d1$z, digits = 1)
>
> k <- 20
> zs_noise <- matrix(rnorm(n*k), nrow = n)
> colnames(zs_noise) <- paste0("z_noise_", 1:k)
> d2 <- cbind(d, zs_noise)
> fmla2 <- as.formula(paste("y ~ x | z + z_noise +",
+ paste0("z_noise_", 1:k, collapse = " + ")))
>
>
> d3 <- d2
> d3$z <- factor(sample(1:3, size = n, replace = TRUE, prob = c(0.1, 0.5, 0.4)))
> d3$y <- rnorm(n, mean = x * c(-1, 1)[(d3$z == 2) + 1], sd = 3)
>
> ## check weights
> w <- rep(1, n)
> w[1:10] <- 2
> (mw1 <- glmtree(formula = fmla, data = d, weights = w))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 706
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 304
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
> (mw2 <- glmtree(formula = fmla, data = d, weights = w, caseweights = FALSE))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1447422 -0.8138701
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.07006626 0.73278593
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.48
>
>
>
> ## check dfsplit
> (mmfluc2 <- mob(formula = fmla, data = d, fit = partykit:::glmfit))
Model-based recursive partitioning (partykit:::glmfit)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function: 2551.673
> (mmfluc3 <- glmtree(formula = fmla, data = d))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
> (mmfluc3_dfsplit <- glmtree(formula = fmla, data = d, dfsplit = 10))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
>
> ## check tests
> if (require("strucchange"))
+ print(sctest(mmfluc3, node = 1)) # does not yet work
Loading required package: strucchange
Loading required package: zoo
Attaching package: 'zoo'
The following objects are masked from 'package:base':
as.Date, as.Date.numeric
Loading required package: sandwich
z z_noise
statistic 2.292499e+01 0.6165335
p.value 7.780038e-04 0.9984952
>
> x <- mmfluc3
> (tst3 <- nodeapply(x, ids = nodeids(x), function(n) n$info$criterion))
$`1`
NULL
$`2`
NULL
$`3`
NULL
>
>
>
>
> ## check logLik and AIC
> logLik(mmfluc2)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3)
'log Lik.' -2551.673 (df=7)
> logLik(mmfluc3_dfsplit)
'log Lik.' -2551.673 (df=16)
> logLik(glm(y ~ x, data = d))
'log Lik.' -2563.694 (df=3)
>
> AIC(mmfluc3)
[1] 5117.347
> AIC(mmfluc3_dfsplit)
[1] 5135.347
>
> ## check pruning
> pr2 <- prune.modelparty(mmfluc2)
> AIC(mmfluc2)
[1] 5117.347
> AIC(pr2)
[1] 5117.347
>
> mmfluc_dfsplit3 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 3)
> mmfluc_dfsplit4 <- glmtree(formula = fmla, data = d, alpha = 0.5, dfsplit = 4)
> pr_dfsplit3 <- prune.modelparty(mmfluc_dfsplit3)
> pr_dfsplit4 <- prune.modelparty(mmfluc_dfsplit4)
> AIC(mmfluc_dfsplit3)
[1] 5142.774
> AIC(mmfluc_dfsplit4)
[1] 5156.774
> AIC(pr_dfsplit3)
[1] 5142.774
> AIC(pr_dfsplit4)
[1] 5124.456
>
> width(mmfluc_dfsplit3)
[1] 8
> width(mmfluc_dfsplit4)
[1] 8
> width(pr_dfsplit3)
[1] 8
> width(pr_dfsplit4)
[1] 3
>
> ## check inner and terminal
> options <- list(NULL,
+ "object",
+ "estfun",
+ c("object", "estfun"))
>
> arguments <- list("inner",
+ "terminal",
+ c("inner", "terminal"))
>
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, inner = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
>
> for (o in options) {
+ print(o)
+ x <- glmtree(formula = fmla, data = d, terminal = o)
+ str(nodeapply(x, ids = nodeids(x), function(n) n$info[c("object", "estfun")]), 2)
+ }
NULL
List of 3
$ 1:List of 2
..$ NA: NULL
..$ NA: NULL
$ 2:List of 2
..$ NA: NULL
..$ NA: NULL
$ 3:List of 2
..$ NA: NULL
..$ NA: NULL
[1] "object"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ NA : NULL
[1] "estfun"
List of 3
$ 1:List of 2
..$ NA : NULL
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ NA : NULL
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ NA : NULL
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
[1] "object" "estfun"
List of 3
$ 1:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:1000, 1:2] -0.1375 -0.0583 -0.0553 0.1043 -0.0744 ...
.. ..- attr(*, "dimnames")=List of 2
$ 2:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:704, 1:2] -0.1291 0.5104 -0.0603 -0.1868 -0.0981 ...
.. ..- attr(*, "dimnames")=List of 2
$ 3:List of 2
..$ object:List of 24
.. ..- attr(*, "class")= chr [1:2] "glm" "lm"
..$ estfun: num [1:296, 1:2] -0.1053 -0.0877 0.0544 -0.1581 0.43 ...
.. ..- attr(*, "dimnames")=List of 2
>
>
> ## check model
> m_mt <- glmtree(formula = fmla, data = d, model = TRUE)
> m_mf <- glmtree(formula = fmla, data = d, model = FALSE)
>
> dim(m_mt$data)
[1] 1000 4
> dim(m_mf$data)
[1] 0 4
>
>
> ## check multiway
> (m_mult <- glmtree(formula = fmla2, data = d3, catsplit = "multiway", minsize = 80))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise + z_noise_1 + z_noise_2 + z_noise_3 + z_noise_4 +
z_noise_5 + z_noise_6 + z_noise_7 + z_noise_8 + z_noise_9 +
z_noise_10 + z_noise_11 + z_noise_12 + z_noise_13 + z_noise_14 +
z_noise_15 + z_noise_16 + z_noise_17 + z_noise_18 + z_noise_19 +
z_noise_20
Fitted party:
[1] root
| [2] z in 1: n = 76
| (Intercept) x
| 0.9859847 -3.2600047
| [3] z in 2: n = 537
| (Intercept) x
| -0.06970187 1.12305074
| [4] z in 3: n = 387
| (Intercept) x
| 0.3824392 -1.8337151
Number of inner nodes: 1
Number of terminal nodes: 3
Number of parameters per node: 2
Objective function (negative log-likelihood): 2511.927
>
>
> ## check parm
> fmla_p <- as.formula("y ~ x + z_noise + z_noise_1 | z + z_noise_2")
> (m_interc <- glmtree(formula = fmla_p, data = d2, parm = 1))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root
| [2] z <= 0.65035: n = 644
| (Intercept) x z_noise2 z_noise3 z_noise_1
| -0.05585503 -1.01257554 0.34044520 -0.16384987 0.24197601
| [3] z > 0.65035: n = 356
| (Intercept) x z_noise2 z_noise3 z_noise_1
| 0.06411865 0.78733976 -0.67811149 -0.14240432 -0.01239154
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 5
Objective function (negative log-likelihood): 2548.32
>
> (m_p3 <- glmtree(formula = fmla_p, data = d2, parm = 3))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x + z_noise + z_noise_1 | z + z_noise_2
Fitted party:
[1] root: n = 1000
(Intercept) x z_noise2 z_noise3 z_noise_1
-0.058855295 -0.340314311 -0.008404682 -0.109839080 0.154798281
Number of inner nodes: 0
Number of terminal nodes: 1
Number of parameters per node: 5
Objective function (negative log-likelihood): 2562.32
>
>
> ## check trim
> (m_tt <- glmtree(formula = fmla, data = d, trim = 0.2))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.70311: n = 704
| (Intercept) x
| -0.1619978 -0.7896293
| [3] z > 0.70311: n = 296
| (Intercept) x
| 0.08683535 0.65598287
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2551.673
>
> (m_tf <- glmtree(formula = fmla, data = d, trim = 300, minsize = 300))
Generalized linear model tree (family: gaussian)
Model formula:
y ~ x | z + z_noise
Fitted party:
[1] root
| [2] z <= 0.6892: n = 691
| (Intercept) x
| -0.1778199 -0.7692901
| [3] z > 0.6892: n = 309
| (Intercept) x
| 0.1065746 0.5562243
Number of inner nodes: 1
Number of terminal nodes: 2
Number of parameters per node: 2
Objective function (negative log-likelihood): 2552.12
>
>
>
> ## check breakties
> m_bt <- glmtree(formula = fmla, data = d1, breakties = TRUE)
> m_df <- glmtree(formula = fmla, data = d1, breakties = FALSE)
>
> all.equal(m_bt, m_df, check.environment = FALSE)
[1] "Component \"node\": Component \"kids\": Component 1: Component 5: Component 6: Mean relative difference: 0.1237503"
[2] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 5: Mean relative difference: 0.1746109"
[3] "Component \"node\": Component \"kids\": Component 2: Component 5: Component 6: Mean relative difference: 0.0443985"
[4] "Component \"node\": Component \"info\": Component \"p.value\": Mean relative difference: 1.100407"
[5] "Component \"node\": Component \"info\": Component \"test\": Mean relative difference: 0.07721086"
[6] "Component \"info\": Component \"call\": target, current do not match when deparsed"
[7] "Component \"info\": Component \"control\": Component \"breakties\": 1 element mismatch"
>
> unclass(m_bt)$node$info$criterion
NULL
> unclass(m_df)$node$info$criterion
NULL
>
> if (requireNamespace("mlbench")) {
+
+ ### example from mob vignette
+ data("PimaIndiansDiabetes", package = "mlbench")
+
+ logit <- function(y, x, start = NULL, weights = NULL, offset = NULL, ...) {
+ glm(y ~ 0 + x, family = binomial, start = start, ...)
+ }
+
+ pid_formula <- diabetes ~ glucose | pregnant + pressure + triceps +
+ insulin + mass + pedigree + age
+
+ pid_tree <- mob(pid_formula, data = PimaIndiansDiabetes, fit = logit)
+ print(pid_tree)
+ print(nodeapply(pid_tree, ids = nodeids(pid_tree), function(n) n$info$criterion))
+
+ }
Loading required namespace: mlbench
Error in eval(mf, parent.frame()) :
object 'PimaIndiansDiabetes' not found
Calls: mob ... model.frame -> terms -> terms.Formula -> terms -> terms.formula
In addition: Warning message:
In data("PimaIndiansDiabetes", package = "mlbench") :
data set 'PimaIndiansDiabetes' not found
Execution halted
Flavor: r-oldrel-windows-x86_64