为什么要进行另一个基准测试?
- Frank 在他的 comment 中提到,从
.(rn, gear) 切换到 c("rn", "gear") 可能会加快速度,但没有单独进行基准测试。
- 在R yoda's benchmark 中,样本数据的类型为integer,但
LIMIT <- 500 的类型为double。 data.table 偶尔会警告类型转换,所以我想知道在这种情况下类型转换可能会对性能产生什么影响。
以什么为基准?
到目前为止,已经提供了 3 个答案,构成了五个代码变体:
很遗憾,我无法让 row.filter 在 SE 版本中工作。
使用了哪些参数?
- 问题大小(行数):102、103、...、108
-
LIMIT 的不同值:100、500、900
-
LIMIT的类型:integer,double来测试类型转换的效果
重复次数是根据问题大小计算得出的,最少运行 3 次,最多运行 100 次。
结果
类型转换确实会花费大约 4%(中位数)到 9%(平均)的性能。所以确实很重要,如果你写LIMIT <- 500 或LIMIT <- 500L 使用L 来表示一个整数常量。
使用非标准评估的性能损失要高得多:对于这两种方法,NSE 平均需要比 SE 多 50% 以上的时间。
(请注意,下面的图表仅显示类型 integer 的结果)
下面的限制 500 和类型 integer 的图表表明,对于所有问题大小,SE 变体都比 NSE 变体更快。有趣的是,对于高达 5000 行的较小问题,chaining_se 似乎比 which_se 略有优势,而对于超过 5 M 行的问题,which_se 是明显更快。
根据要求,下表显示了上图的时间(以毫秒为单位):
dcast(bm_med[limit == 500L & type == "int"][
, expr := forcats::fct_reorder(factor(expr), -time)],
expr ~ n_rows, fun.aggregate = function(x) max(x/1E6), value.var = "time")
expr 100 1000 10000 1e+05 1e+06 1e+07 1e+08
1: chaining_nse 0.8189745 0.8493695 1.0115405 2.870750 22.34469 441.1621 2671.179
2: row.filter 0.7693225 0.7972635 0.9622665 2.677807 21.30861 247.3984 2677.495
3: which_nse 0.8486145 0.8690035 1.0117295 2.620980 18.39406 219.0794 2341.990
4: chaining_se 0.5299360 0.5582545 0.6454755 1.700626 12.48982 166.0164 2049.904
5: which_se 0.5894045 0.6114935 0.7040005 1.624166 13.00125 130.0718 1289.050
基准代码
library(data.table)
library(microbenchmark)
run_bm <- function(n_rows, limit = 500L, type = "int") {
set.seed(1234L)
DT <- data.table(x = sample(1000, n_rows, replace = TRUE),
y = sample(1000, n_rows, replace = TRUE))
LIMIT <- switch(type,
int = as.integer(limit),
dbl = as.double(limit))
times <- round(scales::squish(sqrt(1E8 / n_rows) , c(3L, 100L)))
cat("Start run:", n_rows, limit, type, times, "\n")
microbenchmark(row.filter = {
row.numbers <- DT[, .I[x > LIMIT]]
DT[row.numbers, .(row.numbers, x, y)]
},
chaining_nse = {
DT[, row.number := .I][x > LIMIT, .(row.number, x, y)]
},
chaining_se = {
DT[, row.number := .I][x > LIMIT, c("row.number", "x", "y")]
},
which_nse = {
row.numbers <- DT[x > LIMIT, which = TRUE ]
DT[row.numbers, .(x, y)][, row.numbers := row.numbers ][]
},
which_se = {
row.numbers <- DT[x > LIMIT, which = TRUE ]
DT[row.numbers, c("x", "y")][, row.numbers := row.numbers][]
},
times = times)
}
# parameter
bm_par <- CJ(n_rows = 10^seq(2L, 8L, 1L),
limit = seq(100L, 900L, 400L),
type = c("int", "dbl"))
# run the benchmarks
bm_raw <- bm_par[, run_bm(n_rows, limit, type), by = .(n_rows, limit, type)]
# aggregate results
bm_med <- bm_raw[, .(time = median(time)), by = .(n_rows, limit, type, expr)]
图形代码
library(ggplot2)
# chart 1
ggplot(
dcast(bm_med, n_rows + limit + expr ~ type, value.var = "time")[
, ratio := dbl / int - 1.0] #[limit == 500L]
) +
aes(n_rows, ratio, colour = expr) +
geom_point() +
geom_line() +
facet_grid(limit ~ expr) +
scale_x_log10(labels = function(x) scales::math_format()(log10(x))) +
scale_y_continuous(labels = scales::percent) +
coord_cartesian(ylim = c(-0.1, 0.5)) +
geom_hline(yintercept = 0) +
theme_bw() +
ggtitle("Performance loss due to type conversion") +
ylab("Relative computing time dbl vs int") +
xlab("Number of rows (log scale)")
ggsave("p2.png")
# chart 2
ggplot(
dcast(bm_med[, c("code", "eval") := tstrsplit(expr, "_")][!is.na(eval)],
n_rows + limit + type + code ~ eval, value.var = "time")[
, ratio := nse / se - 1.0][type == "int"]
) +
aes(n_rows, ratio, colour = code) +
geom_point() +
geom_line() +
facet_grid(limit + type ~ code) +
scale_x_log10(labels = function(x) scales::math_format()(log10(x))) +
scale_y_continuous(labels = scales::percent) +
geom_hline(yintercept = 0) +
theme_bw() +
ggtitle("Performance loss due to non standard evaluation") +
ylab("Relative computing time NSE vs SE") +
xlab("Number of rows (log scale)")
ggsave("p3.png")
# chart 3
ggplot(bm_med[limit == 500L][type == "int"]) +
aes(n_rows, time/1E6, colour = expr) +
geom_point() +
geom_smooth(se = FALSE) +
facet_grid(limit ~ type) +
facet_grid(type ~ limit) +
scale_x_log10(labels = function(x) scales::math_format()(log10(x))) +
scale_y_log10(labels = function(x) scales::math_format()(log10(x))) +
theme_bw() +
ggtitle("Benchmark results (log-log scale)") +
ylab("Computing time in ms (log scale)") +
xlab("Number of rows (log scale)")
ggsave("p1.png")