Code
library(tidyverse)
library(gt)
ccc <- function(x, y) {
mx <- mean(x); my <- mean(y)
vx <- mean((x - mx)^2); vy <- mean((y - my)^2)
cxy <- mean((x - mx) * (y - my))
2 * cxy / (vx + vy + (mx - my)^2)
}Compare item-level ensemble aggregation across the 25 evaluated models to the best single model (Qwen 3 22B-235B), using the indirect scoring approach. Prompted by an editor suggestion during review at JPCS.
library(tidyverse)
library(gt)
ccc <- function(x, y) {
mx <- mean(x); my <- mean(y)
vx <- mean((x - mx)^2); vy <- mean((y - my)^2)
cxy <- mean((x - mx) * (y - my))
2 * cxy / (vx + vy + (mx - my)^2)
}# Sessions used as fewshot examples (excluded from evaluation)
source("load_data.R")
excluded_ids <- load_exemplars()
# Sessions excluded from any item's evaluation (union across items 1-10)
union_excl <- reduce(excluded_ids[2:11], union)# Predictions (public, ../data) joined to human ratings (NDA, ../nda)
main_sheets <- load_prediction_sheets(conditions = "full")
ens_items <-
map(
.x = 1:10,
.f = \(i) {
main_sheets[[i + 1]] |>
filter(session %in% excluded_ids[[i + 1]] == FALSE) |>
transmute(
session,
patient,
model = str_trim(model_name),
item = sprintf("%02d", i),
label = ground_truth,
# Self-ensemble across the three seeds (as in the main evaluation)
pred = rowMeans(cbind(rating_0, rating_1, rating_2), na.rm = TRUE)
)
}
) |>
bind_rows() |>
filter(session %in% union_excl == FALSE, !is.nan(pred))best_model <- str_subset(unique(ens_items$model), "22B-235B")
tiny_models <- str_subset(unique(ens_items$model), "0\\.6B|1\\.7B|\\(4B\\)")
# Post-hoc set: five best models by indirect MAE in the primary benchmarking.
# NOTE: selected on the same evaluation data, so estimates are optimistically
# biased; report only with that caveat.
top5_models <- str_subset(
unique(ens_items$model),
"22B-235B|14B\\): 1M|R1 Qwen 2.5|GPT OSS 120B|R1 Llama 3.3"
)
item_est <-
ens_items |>
summarize(
.by = c(session, item),
label = first(label),
best = pred[model == best_model][1],
ens_mean = mean(pred),
ens_median = median(pred),
ens_median_8b = median(pred[model %in% tiny_models == FALSE]),
ens_mean_top5 = mean(pred[model %in% top5_models])
)
# Complete sessions only: all 10 items scored and every estimator available,
# so all estimators are evaluated on the same session set
total_est <-
item_est |>
summarize(
.by = session,
n_items = n(),
across(c(label, best, ens_mean, ens_median, ens_median_8b, ens_mean_top5), sum)
) |>
filter(n_items == 10, if_all(everything(), \(x) !is.na(x)))total_est |>
pivot_longer(
c(best, ens_mean, ens_median, ens_median_8b, ens_mean_top5),
names_to = "estimator", values_to = "pred"
) |>
summarize(
.by = estimator,
N = n(),
MAE = mean(abs(pred - label)),
CCC = ccc(pred, label)
) |>
mutate(
estimator = recode(estimator,
best = "Best single model (Qwen 3 22B-235B)",
ens_mean = "Ensemble: mean (all 25)",
ens_median = "Ensemble: median (all 25)",
ens_median_8b = "Ensemble: median (models >= 8B)",
ens_mean_top5 = "Ensemble: mean (top 5; post hoc)"
)
) |>
gt() |>
fmt_number(columns = MAE, decimals = 2) |>
fmt_number(columns = CCC, decimals = 3) |>
cols_label(estimator = "Estimator") |>
cols_align(align = "left", columns = estimator) |>
tab_header(
title = "Ensemble vs. Best Single Model",
subtitle = "Indirect MADRS total score prediction"
) |>
tab_source_note(paste(
"Item predictions averaged over three seeds within model, then aggregated",
"across models per item and summed to totals. All estimators evaluated on",
"the same complete sessions. Top-5 selection used the same evaluation data",
"and is optimistically biased."
)) |>
tab_options(table.font.size = 14)| Ensemble vs. Best Single Model | |||
| Indirect MADRS total score prediction | |||
| Estimator | N | MAE | CCC |
|---|---|---|---|
| Best single model (Qwen 3 22B-235B) | 444 | 3.49 | 0.901 |
| Ensemble: mean (all 25) | 444 | 3.86 | 0.895 |
| Ensemble: median (all 25) | 444 | 3.49 | 0.909 |
| Ensemble: median (models >= 8B) | 444 | 3.46 | 0.911 |
| Ensemble: mean (top 5; post hoc) | 444 | 3.28 | 0.914 |
| Item predictions averaged over three seeds within model, then aggregated across models per item and summed to totals. All estimators evaluated on the same complete sessions. Top-5 selection used the same evaluation data and is optimistically biased. | |||