Already know what you need? Search by name, or by the test you know it as.
+
Already know what you need? Search by name, or by the traditional test you know it as.
{FILTERS.map((option) => (
diff --git a/src/pages/SampleSize.test.tsx b/src/pages/SampleSize.test.tsx
index cd3a1f8..1ddfd0f 100644
--- a/src/pages/SampleSize.test.tsx
+++ b/src/pages/SampleSize.test.tsx
@@ -54,11 +54,11 @@ describe('sample size page', () => {
expect(question()).toBe('The sample you need');
// pwr.t.test(d = 0.3, power = 0.8) gives n = 175.4 per group.
expect(report()).toContain('352 participants (176 per group)');
- expect(report()).toContain('d = 0.30 with 80% power in a two-sided independent-samples t-test at α = .05');
+ expect(report()).toContain('d = 0.30 with 80% power in a two-sided test of the group difference (the traditional independent-samples t-test) at α = .05');
// 176 / 0.9 rounds up to 196 per group.
expect(report()).toContain('we will recruit 392 participants');
expect(screen.getByText(/pwr.t.test\(d = 0.3, sig.level = 0.05, power = 0.8/)).toBeTruthy();
- expect(screen.getByText(/is the independent-samples t-test \(Lesson 12-1\)/)).toBeTruthy();
+ expect(screen.getByText(/is the same as the traditional independent-samples t-test \(Lesson 12-1\)/)).toBeTruthy();
expect(trail()).toEqual(['Compare two separate groups', 'Plan a new study', 'My own guess of the scores', 'd = 0.30', '80% power', 'α = .05', 'Two-sided', '10% dropout']);
});
@@ -215,7 +215,7 @@ describe('sample size page', () => {
await pick(/^None/);
// λ = N·f² with df2 = N − 2: 128 people, 64 per group.
expect(report()).toContain('128 participants (64 per group, each measured in both conditions) are needed to detect an effect of f = 0.25 (with a correlation of r = .50 between conditions)');
- expect(report()).toContain('the F test for the interaction in a 2 × 2 mixed ANOVA');
+ expect(report()).toContain('the F test for the interaction in a 2 × 2 mixed design, one factor between and one within people (a traditional mixed ANOVA)');
expect(trail()).toContain('One factor between, one within');
});
diff --git a/src/pages/SampleSize.tsx b/src/pages/SampleSize.tsx
index 85a270e..d417dd5 100644
--- a/src/pages/SampleSize.tsx
+++ b/src/pages/SampleSize.tsx
@@ -67,9 +67,9 @@ const WHICH_TEXT: Record = {
'main-between': 'main effect of the between-person factor',
};
const LAYOUT_TEXT: Record = {
- between: 'a 2 × 2 between-subjects ANOVA',
- within: 'a 2 × 2 within-subjects (repeated-measures) ANOVA',
- mixed: 'a 2 × 2 mixed ANOVA (one factor between, one within people)',
+ between: 'a 2 × 2 between-subjects design (a traditional two-way ANOVA)',
+ within: 'a 2 × 2 within-subjects design (a traditional repeated-measures ANOVA)',
+ mixed: 'a 2 × 2 mixed design, one factor between and one within people (a traditional mixed ANOVA)',
};
function testText(design: Design, sides: 1 | 2, plan?: Plan): string {
@@ -214,8 +214,8 @@ const INPUT_TEXT: Partial> =
n: { question: 'How many people will you have?', help: 'Count the people you expect to have complete data for, not everyone you invite.' },
k: { question: 'How many groups will you compare?', help: 'For example 3 for three training programmes. The plan assumes groups of about equal size.' },
predictors: { question: 'How many predictors will the model have?', help: 'Count every term in the model, control variables included. A numeric predictor counts once; a categorical predictor with g categories counts g − 1 times, because R turns it into g − 1 dummy variables (Lesson 12-2).' },
- added: { question: 'How many of those predictors are you testing?', help: 'Usually 1: the predictor your question is about, with the others as controls. With one tested predictor, this is the same test as that predictor\'s t-test in summary().' },
- table: { question: 'How big is your table of counts?', help: 'Rows are the categories of one variable, columns those of the other: department (4) by remote work (2) is a 4 × 2 table. For a goodness-of-fit test of one variable, enter its number of categories as rows and 2 as columns; that gives the right degrees of freedom.' },
+ added: { question: 'How many of those predictors are you testing?', help: 'Usually 1: the predictor your question is about, with the others as controls. With one tested predictor, this is the same test as that predictor\'s p-value in summary().' },
+ table: { question: 'How big is your table of counts?', help: 'Rows are the categories of one variable, columns those of the other: department (4) by remote work (2) is a 4 × 2 table. For a traditional goodness-of-fit test of one variable, enter its number of categories as rows and 2 as columns; that gives the right degrees of freedom.' },
p1: { question: 'What percentage says "yes" in the first group?', help: 'The comparison or control group. Take it from records, earlier studies or national figures. The same difference in percentage points is harder to detect near 50% than near 0% or 100%.' },
nstim: { question: 'How many stimuli per condition?', help: 'For example 4 pictures in every condition. If every condition uses the same stimuli, count them once.' },
tests: { question: 'How many tests will you correct for?', help: 'Bonferroni divides α by the number of tests you correct for together, so each test needs a smaller p-value, and the study needs more people.' },
@@ -398,7 +398,7 @@ export default function SampleSize() {
const input = step === 'result' ? null : inputStep(step, answers, design);
if (input) {
const text = step === 'k' && design.id === 'repeated'
- ? { question: 'How many conditions will each person do?', help: 'Count the conditions every participant goes through, for example 4. With 2 conditions and one score each, this is the paired t-test.' }
+ ? { question: 'How many conditions will each person do?', help: 'Count the conditions every participant goes through, for example 4. With 2 conditions and one score each, this is the traditional paired t-test.' }
: INPUT_TEXT[step as StepId]!;
question = text.question;
help = text.help;
diff --git a/src/pages/modelTree.ts b/src/pages/modelTree.ts
index ab47716..005419f 100644
--- a/src/pages/modelTree.ts
+++ b/src/pages/modelTree.ts
@@ -132,7 +132,7 @@ export const TREE: Question = {
{
label: 'Nothing: compare the average with a fixed number',
example: 'Is the average exam score different from 70?',
- tech: 'One-sample t-test',
+ tech: 'Traditional one-sample t-test',
next: {
kind: 'answer',
id: 'mean-vs-value',
@@ -144,8 +144,8 @@ d <- read.csv("data/wellbeing-population.csv", stringsAsFactors = TRUE)
model <- lm(I(exam_score - 70) ~ 1, data = d)
model %>% tidy(conf.int = TRUE)`,
check:
- 'Independent scores and no extreme outliers; roughly normal scores, which matters mainly in small samples. For a small, clearly skewed sample, the one-sample Wilcoxon signed-rank test: wilcox.test(d$exam_score, mu = 70).',
- traditional: 'The one-sample t-test: t.test(d$exam_score, mu = 70). Same t, same p.',
+ 'Independent scores and no extreme outliers; roughly normal scores, which matters mainly in small samples. For a small, clearly skewed sample, the traditional one-sample Wilcoxon signed-rank test: wilcox.test(d$exam_score, mu = 70).',
+ traditional: 'The traditional one-sample t-test: t.test(d$exam_score, mu = 70). Same t, same p.',
note: 'Here the question is whether the mean exam score differs from 70; put your own comparison value in its place. ~ 1 means a model with no predictors, so the intercept is how far the mean lies from that value, with its confidence interval.',
buildsOn: '08-3',
further: 'Van den Berg, Analysing Data Using Linear Models, section 5.15, The intercept only model.',
@@ -175,8 +175,8 @@ model <- lm(wellbeing ~ autonomy, data = d)
model %>% tidy()
model %>% glance()`,
check:
- "A scatterplot with geom_smooth(method = lm) shows a roughly straight-line pattern; no extreme outliers; residuals with similar spread along the whole line; residuals roughly normal, which matters mainly in small samples. For a curved-but-consistent pattern, ranks or outliers, Spearman's correlation: cor.test(d$wellbeing, d$autonomy, method = \"spearman\").",
- traditional: "Pearson correlation: cor.test(d$wellbeing, d$autonomy). Its t and p match the slope's.",
+ "A scatterplot with geom_smooth(method = lm) shows a roughly straight-line pattern; no extreme outliers; residuals with similar spread along the whole line; residuals roughly normal, which matters mainly in small samples. For a curved-but-consistent pattern, ranks or outliers, the traditional Spearman correlation test: cor.test(d$wellbeing, d$autonomy, method = \"spearman\").",
+ traditional: "The traditional Pearson correlation test: cor.test(d$wellbeing, d$autonomy). Its t and p match the slope's.",
note: 'The slope b is the difference in predicted outcome (here wellbeing) between people who differ by one unit on the predictor (autonomy).',
lessonId: '10-2',
},
@@ -259,7 +259,7 @@ emtrends(model, ~ autonomy_c, var = "workload_c", at = list(autonomy_c = c(-s, 0
{
label: 'Groups',
example: 'Do wellbeing scores differ between departments?',
- tech: 'Group comparison: t-test, ANOVA',
+ tech: 'Group comparison (traditional t-test, ANOVA)',
group: 'Numeric outcome: comparing groups',
next: {
kind: 'question',
@@ -268,7 +268,7 @@ emtrends(model, ~ autonomy_c, var = "workload_c", at = list(autonomy_c = c(-s, 0
{
label: 'Two groups',
example: 'Do remote workers report higher wellbeing than office workers?',
- tech: 'Independent t-test',
+ tech: 'Traditional independent t-test',
next: {
kind: 'answer',
id: 'two-groups',
@@ -281,9 +281,9 @@ model <- lm(wellbeing ~ remote, data = d)
model %>% tidy()
d %>% group_by(remote) %>% summarise(mean = mean(wellbeing), sd = sd(wellbeing))`,
check:
- 'Similar spread in each group, which matters especially when group sizes differ. Residuals roughly normal, which matters mainly in small samples; with large groups the Central Limit Theorem covers moderate skew. For a small, clearly skewed sample or extreme outliers, the Mann-Whitney test: wilcox.test(wellbeing ~ remote, data = d).',
+ 'Similar spread in each group, which matters especially when group sizes differ. Residuals roughly normal, which matters mainly in small samples; with large groups the Central Limit Theorem covers moderate skew. For a small, clearly skewed sample or extreme outliers, the traditional Mann-Whitney test: wilcox.test(wellbeing ~ remote, data = d).',
traditional:
- 'The independent-samples t-test: t.test(wellbeing ~ remote, data = d, var.equal = TRUE). Same t with the sign reversed, because t.test subtracts the groups the other way round, and the same p. When the spreads differ, leave out var.equal = TRUE to get Welch\'s test.',
+ 'The traditional independent-samples t-test: t.test(wellbeing ~ remote, data = d, var.equal = TRUE). Same t with the sign reversed, because t.test subtracts the groups the other way round, and the same p. When the spreads differ, leave out var.equal = TRUE to get the traditional Welch test.',
note: 'The slope is the difference between the two group means. Always look at the means: the sign of b depends on which group R took as the reference.',
lessonId: '12-1',
},
@@ -291,7 +291,7 @@ d %>% group_by(remote) %>% summarise(mean = mean(wellbeing), sd = sd(wellbeing))
{
label: 'Three or more groups',
example: 'Do wellbeing scores differ between departments?',
- tech: 'One-way ANOVA',
+ tech: 'Traditional one-way ANOVA',
next: {
kind: 'answer',
id: 'several-groups',
@@ -306,8 +306,8 @@ model %>% glance()
model %>% tidy()
emmeans(model, pairwise ~ department, adjust = "tukey")`,
check:
- 'Similar spread in each group, which matters especially when group sizes differ. Residuals roughly normal, which matters mainly in small samples; with large groups the Central Limit Theorem covers moderate skew. For a small, clearly skewed sample or extreme outliers, the Kruskal-Wallis test: kruskal.test(wellbeing ~ department, data = d).',
- traditional: 'One-way ANOVA: summary(aov(wellbeing ~ department, data = d)). Same F, same p.',
+ 'Similar spread in each group, which matters especially when group sizes differ. Residuals roughly normal, which matters mainly in small samples; with large groups the Central Limit Theorem covers moderate skew. For a small, clearly skewed sample or extreme outliers, the traditional Kruskal-Wallis test: kruskal.test(wellbeing ~ department, data = d).',
+ traditional: 'The traditional one-way ANOVA: summary(aov(wellbeing ~ department, data = d)). Same F, same p.',
note: 'Each b compares one group with the reference group. glance() gives the overall F; emmeans gives every pairwise comparison, corrected for multiple testing.',
lessonId: '12-2',
},
@@ -315,7 +315,7 @@ emmeans(model, pairwise ~ department, adjust = "tukey")`,
{
label: 'Groups, adjusting for a number',
example: 'Are trained employees more engaged, adjusting for how engaged they were before?',
- tech: 'ANCOVA',
+ tech: 'Traditional ANCOVA',
next: {
kind: 'answer',
id: 'groups-with-covariate',
@@ -330,7 +330,7 @@ model %>% tidy(conf.int = TRUE)
emmeans(model, pairwise ~ training)`,
check:
'The covariate relates to the outcome in a roughly straight line with a similar slope in every group: fit lm(engagement_t2 ~ engagement_t1 * training, data = d) and check that the interaction is small. The covariate is measured before the groups could have changed it. If the slopes clearly differ, that difference is your finding: see the moderation model under numeric predictors.',
- traditional: 'ANCOVA: anova(model) gives the F test for the groups, adjusted for the covariate entered before them.',
+ traditional: 'The traditional ANCOVA: anova(model) gives the F test for the groups, adjusted for the covariate entered before them.',
note: 'The group coefficients are differences between groups at the same value of the covariate: here, trained and untrained employees who started with the same engagement. emmeans gives each group\'s adjusted mean, its predicted mean at the average covariate, and the pairwise differences between them.',
lessonId: '11-2',
},
@@ -338,7 +338,7 @@ emmeans(model, pairwise ~ training)`,
{
label: 'Two kinds of groups together',
example: 'Do training and mentoring work better in combination?',
- tech: 'Factorial ANOVA',
+ tech: 'Traditional factorial ANOVA',
next: {
kind: 'answer',
id: 'factorial',
@@ -354,7 +354,7 @@ Anova(model, type = "III")
emmeans(model, pairwise ~ training:mentoring, adjust = "tukey")`,
check:
'Similar spread in each cell, which matters especially when group sizes differ. Residuals roughly normal, which matters mainly in small samples. Plot the cell means before interpreting main effects.',
- traditional: 'Two-way (factorial) ANOVA. Anova(model, type = "III") gives its F tests for each main effect and the interaction.',
+ traditional: 'The traditional two-way (factorial) ANOVA. Anova(model, type = "III") gives its F tests for each main effect and the interaction.',
// Type III main-effect tests are only meaningful with sum-to-zero contrasts. Under R's
// default treatment contrasts they test each factor at the other's reference level.
note: 'The contrasts = list(...) line matters: type III tests of the main effects are only correct with sum-to-zero contrasts, and R does not use those by default. An interaction means the effect of one factor depends on the level of the other: in an interaction plot, the lines are not parallel.',
@@ -364,11 +364,11 @@ emmeans(model, pairwise ~ training:mentoring, adjust = "tukey")`,
{
label: 'Groups on several outcomes at once',
example: 'Do departments differ in both wellbeing and engagement?',
- tech: 'MANOVA',
+ tech: 'Traditional MANOVA',
next: {
kind: 'answer',
id: 'several-outcomes',
- model: 'Multivariate linear model (MANOVA)',
+ model: 'Multivariate linear model (traditional MANOVA)',
when: 'Several related numeric outcomes, such as three subscales of one questionnaire, and the question is whether the groups differ on them taken together.',
rCode: `d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
model <- manova(cbind(wellbeing, engagement_t2, performance) ~ department, data = d)
@@ -376,10 +376,10 @@ summary(model, test = "Pillai")
summary.aov(model)`,
check:
"The outcomes correlate moderately with each other; if they do not, analyse them one at a time. No extreme outliers, and more cases in every group than there are outcomes. Pillai's trace is the most robust of the four test statistics.",
- traditional: 'One-way MANOVA.',
- note: 'The overall test asks whether the groups differ on the combination of outcomes. summary.aov() then gives one ANOVA per outcome; correct those for multiple testing, for example with p.adjust(p, method = "holm").',
+ traditional: 'The traditional one-way MANOVA.',
+ note: 'The overall test asks whether the groups differ on the combination of outcomes. summary.aov() then gives one traditional ANOVA per outcome; correct those for multiple testing, for example with p.adjust(p, method = "holm").',
buildsOn: '12-2',
- further: 'Field, Miles and Field, Discovering Statistics Using R, has a chapter on MANOVA and the discriminant analysis that often follows it.',
+ further: 'Field, Miles and Field, Discovering Statistics Using R, has a chapter on the traditional MANOVA and the discriminant analysis that often follows it.',
},
},
],
@@ -401,7 +401,7 @@ summary.aov(model)`,
{
label: 'Twice, such as before and after',
example: 'Did engagement rise from the first to the second measurement?',
- tech: 'Paired t-test',
+ tech: 'Traditional paired t-test',
next: {
kind: 'answer',
id: 'before-after',
@@ -416,16 +416,16 @@ long_d <- d %>% pivot_longer(cols = c(engagement_t1, engagement_t2), names_to =
model <- lmer(engagement ~ time + (1 | employee_id), data = long_d)
summary(model)`,
check:
- 'Data in long format: one row per person per measurement. Residuals roughly normal, which matters mainly in small samples. With skewed differences in a small sample, the Wilcoxon signed-rank test: wilcox.test(d$engagement_t2, d$engagement_t1, paired = TRUE).',
- traditional: 'The paired-samples t-test: t.test(d$engagement_t2, d$engagement_t1, paired = TRUE). Same t, same p, when nobody is missing a measurement.',
- note: '(1 | employee_id) gives every employee their own starting level, so the model knows which scores belong together. engagement_t1 sorts first, so it is the reference and the time coefficient is the change from the first measurement to the second. Unlike the paired t-test, it keeps people who missed a measurement.',
+ 'Data in long format: one row per person per measurement. Residuals roughly normal, which matters mainly in small samples. With skewed differences in a small sample, the traditional Wilcoxon signed-rank test: wilcox.test(d$engagement_t2, d$engagement_t1, paired = TRUE).',
+ traditional: 'The traditional paired-samples t-test: t.test(d$engagement_t2, d$engagement_t1, paired = TRUE). Same t, same p, when nobody is missing a measurement.',
+ note: '(1 | employee_id) gives every employee their own starting level, so the model knows which scores belong together. engagement_t1 sorts first, so it is the reference and the time coefficient is the change from the first measurement to the second. Unlike the traditional paired t-test, it keeps people who missed a measurement.',
lessonId: '14-3',
},
},
{
label: 'Three or more times',
example: 'Does stress change across three exam weeks?',
- tech: 'Repeated-measures ANOVA',
+ tech: 'Traditional repeated-measures ANOVA',
next: {
kind: 'answer',
id: 'repeated-measures',
@@ -440,16 +440,16 @@ model <- lmer(weight ~ time + (1 | Chick), data = long_d)
anova(model)
emmeans(model, pairwise ~ time, adjust = "holm")`,
check:
- 'Data in long format: one row per case per measurement, which ChickWeight already is. Residuals roughly normal, which matters mainly in small samples. The model assumes the same spread at every time point and the same correlation between every pair of them. Compare the SDs per time point first: in ChickWeight they grow from about 1 g at hatching to about 72 g at day 21, so read its p-values with care. For a growth process like this, a growth-curve model with random slopes fits better. For a small, clearly skewed sample, the Friedman test, which needs every chick at every time point, so keep only the complete ones first: complete <- long_d %>% group_by(Chick) %>% filter(n() == 3) %>% ungroup() %>% droplevels(), then friedman.test(weight ~ time | Chick, data = complete).',
- traditional: 'Repeated-measures ANOVA. With complete data the F matches its uncorrected F: both assume the time points are equally correlated.',
- note: 'anova() tests whether weight differs across the time points at all; emmeans then compares each pair of time points. (1 | Chick) lets each chick have its own level, as (1 | id) would for people. Unlike repeated-measures ANOVA, the model keeps the chicks that missed a weighing.',
+ 'Data in long format: one row per case per measurement, which ChickWeight already is. Residuals roughly normal, which matters mainly in small samples. The model assumes the same spread at every time point and the same correlation between every pair of them. Compare the SDs per time point first: in ChickWeight they grow from about 1 g at hatching to about 72 g at day 21, so read its p-values with care. For a growth process like this, a growth-curve model with random slopes fits better. For a small, clearly skewed sample, the traditional Friedman test, which needs every chick at every time point, so keep only the complete ones first: complete <- long_d %>% group_by(Chick) %>% filter(n() == 3) %>% ungroup() %>% droplevels(), then friedman.test(weight ~ time | Chick, data = complete).',
+ traditional: 'The traditional repeated-measures ANOVA. With complete data the model\'s F matches its uncorrected F: both assume the time points are equally correlated.',
+ note: 'anova() tests whether weight differs across the time points at all; emmeans then compares each pair of time points. (1 | Chick) lets each chick have its own level, as (1 | id) would for people. Unlike the traditional repeated-measures ANOVA, the model keeps the chicks that missed a weighing.',
lessonId: '14-4',
},
},
{
label: 'Over time, in different groups',
example: 'Did engagement rise more for trained employees than for others?',
- tech: 'Mixed ANOVA',
+ tech: 'Traditional mixed ANOVA',
next: {
kind: 'answer',
id: 'time-by-group',
@@ -467,7 +467,7 @@ anova(model)
emmeans(model, pairwise ~ training | time)`,
check:
'Data in long format, with time and group (here training) as factors. Residuals with similar spread in each group, and roughly normal in small samples.',
- traditional: 'Mixed (split-plot) ANOVA.',
+ traditional: 'The traditional mixed (split-plot) ANOVA.',
note: 'The time:training interaction is usually the question: did the trained employees change differently from the others? In a randomised trial it is the treatment effect. emmeans compares the groups at each time point.',
lessonId: '14-4',
},
@@ -509,7 +509,7 @@ model <- lmer(y ~ service + (1 | student) + (1 | lecturer), data = ratings)
summary(model)`,
check:
'One row per participant per item (here, per rating). Enough participants and enough items, since both are samples. Add random slopes, such as (service | student), when the design supports them.',
- traditional: 'Separate by-participant and by-item ANOVAs (F1 and F2), which this model replaces.',
+ traditional: 'Separate by-participant and by-item traditional ANOVAs (F1 and F2), which this model replaces.',
note: 'Both participants and items vary, here students and lecturers, so both get their own random intercept, and the effect of service (a course taught for another department) then generalises across both. Averaging over items first, as F1 analyses did, treats the items as if they were the only ones possible. The first 5000 ratings keep the example quick.',
buildsOn: '14-3',
further: 'Baayen, Davidson and Bates (2008), Mixed-effects modeling with crossed random effects for subjects and items, Journal of Memory and Language.',
@@ -574,7 +574,7 @@ exp(cbind(OR = coef(model), confint(model)))`,
check:
'Independent observations, and enough cases of the rarer outcome: a common rule of thumb is at least 10 per estimated coefficient (a factor with k levels uses k − 1). If R warns that fitted probabilities of 0 or 1 occurred, a predictor separates the outcomes perfectly; Firth\'s method, logistf::logistf(), handles that in RStudio.',
traditional:
- 'With one categorical predictor, such as remote, the chi-square test of independence: chisq.test(table(d$left_company, d$remote), correct = FALSE), which matches anova(glm(left_company ~ remote, data = d, family = binomial), test = "Rao").',
+ 'With one categorical predictor, such as remote, the traditional chi-square test of independence: chisq.test(table(d$left_company, d$remote), correct = FALSE), which matches anova(glm(left_company ~ remote, data = d, family = binomial), test = "Rao").',
note: 'The coefficients are in log odds. exp() turns them into odds ratios: above 1, the outcome becomes more likely; below 1, less likely.',
lessonId: '15-2',
},
@@ -582,23 +582,23 @@ exp(cbind(OR = coef(model), confint(model)))`,
{
label: 'One other category (a cross-table)',
example: 'Does the share who leave differ between departments?',
- tech: 'Chi-square test',
+ tech: 'Traditional chi-square test',
next: crossTable(),
},
{
label: 'Nothing: compare one percentage with a fixed value',
example: 'Do more than half of the students pass?',
- tech: 'Binomial test',
+ tech: 'Traditional binomial test',
next: {
kind: 'answer',
id: 'proportion-vs-value',
- model: 'Exact binomial test',
+ model: 'Traditional exact binomial test',
when: 'One yes-or-no variable, and the question is whether the proportion of yes differs from a fixed value, such as 50% or a known pass rate.',
rCode: `d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
successes <- sum(d$left_company == 1)
binom.test(successes, nrow(d), p = 0.2)`,
check: 'Independent cases, each counted once, and no missing values in the outcome (nrow() would count them).',
- traditional: 'The large-sample version is prop.test(successes, nrow(d), p = 0.2, correct = FALSE). As a model, an intercept-only logistic regression tests the same value only with an offset: glm(left_company ~ 1, offset = rep(qlogis(0.2), nrow(d)), data = d, family = binomial), whose intercept is the difference in log odds from 20%. Without the offset it tests 50%.',
+ traditional: 'The traditional large-sample version is prop.test(successes, nrow(d), p = 0.2, correct = FALSE). As a model, an intercept-only logistic regression tests the same value only with an offset: glm(left_company ~ 1, offset = rep(qlogis(0.2), nrow(d)), data = d, family = binomial), whose intercept is the difference in log odds from 20%. Without the offset it tests 50%.',
note: 'Here the question is whether the share of employees who left differs from 20%. For your own data, count your own level in place of left_company == 1 and use your own proportion in place of 0.2. The output gives the observed proportion with its exact 95% confidence interval.',
lessonId: '09-1',
},
@@ -622,7 +622,7 @@ summary(model)
exp(fixef(model))`,
check:
'For repeated measurements, long format: one row per person per measurement. Clustered data, such as employees in sites, is already in shape. With few clusters or a rare outcome the model may fail to converge; simplify the random part first.',
- traditional: "For one yes-or-no answer at two time points, McNemar's test: mcnemar.test(table(before, after)).",
+ traditional: "For one yes-or-no answer at two time points, the traditional McNemar test: mcnemar.test(table(before, after)).",
note: 'exp() turns the coefficients into odds ratios for people in the same cluster, here the same site, holding its random intercept fixed. These are usually further from 1 than the population-average odds ratios an ordinary logistic regression gives.',
buildsOn: '14-2',
further: 'Gelman and Hill, Data Analysis Using Regression and Multilevel/Hierarchical Models.',
@@ -666,20 +666,20 @@ exp(cbind(OR = coef(model), confint(model)))`,
{
label: 'Do groups differ, or do two things go together?',
example: 'Do the departments differ in their satisfaction rating?',
- tech: 'Rank tests: Mann-Whitney, Kruskal-Wallis, Spearman',
+ tech: 'Traditional rank tests: Mann-Whitney, Kruskal-Wallis, Spearman',
next: {
kind: 'answer',
id: 'rank-tests',
- model: 'Rank-based tests',
+ model: 'Traditional rank-based tests',
when: 'An ordered outcome, or a numeric one that is clearly skewed or has extreme outliers, and the question is whether groups differ or whether two variables go together.',
rCode: `d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
wilcox.test(wellbeing ~ remote, data = d)
kruskal.test(wellbeing ~ department, data = d)
cor.test(~ autonomy + wellbeing, data = d, method = "spearman", exact = FALSE)`,
check:
- 'Independent cases, each counted once. The group tests compare whole distributions; read them as a difference in medians only when the groups have a similar shape and spread. When the same people are measured more than once, use a test built on within-person comparisons instead: the Wilcoxon signed-rank test (which ranks each person\'s difference) for two measurements, wilcox.test(d$engagement_t2, d$engagement_t1, paired = TRUE), and the Friedman test (which ranks within each person) for three or more.',
+ 'Independent cases, each counted once. The group tests compare whole distributions; read them as a difference in medians only when the groups have a similar shape and spread. When the same people are measured more than once, use a test built on within-person comparisons instead: the traditional Wilcoxon signed-rank test (which ranks each person\'s difference) for two measurements, wilcox.test(d$engagement_t2, d$engagement_t1, paired = TRUE), and the traditional Friedman test (which ranks within each person) for three or more.',
traditional:
- "The Mann-Whitney test, the Kruskal-Wallis test and Spearman's rank correlation. After a significant Kruskal-Wallis test, pairwise.wilcox.test(d$wellbeing, d$department, p.adjust.method = \"holm\", exact = FALSE) compares every pair of groups.",
+ "The traditional Mann-Whitney test, Kruskal-Wallis test and Spearman rank correlation test. After a significant traditional Kruskal-Wallis test, pairwise.wilcox.test(d$wellbeing, d$department, p.adjust.method = \"holm\", exact = FALSE) compares every pair of groups.",
note: 'wilcox.test() compares two groups, kruskal.test() three or more, and cor.test() with method = "spearman" gives rho, the correlation between the ranks; method = "kendall" gives Kendall\'s tau instead. These tests work on ranks, so report each group\'s median alongside them. To add predictors or covariates, use ordinal regression.',
lessonId: '12-4',
},
@@ -742,17 +742,17 @@ tidy(model, conf.int = TRUE, exponentiate = TRUE)`,
{
label: 'Linked to one other category',
example: 'Is the way people travel to work linked to their department?',
- tech: 'Chi-square test of independence',
+ tech: 'Traditional chi-square test of independence',
next: crossTable(),
},
{
label: 'Do the shares match what you expect?',
example: 'Are the four entrances of the building used equally often?',
- tech: 'Chi-square goodness of fit',
+ tech: 'Traditional chi-square goodness of fit',
next: {
kind: 'answer',
id: 'goodness-of-fit',
- model: 'Chi-square goodness-of-fit test',
+ model: 'Traditional chi-square goodness-of-fit test',
when: 'One categorical variable, and the question is whether its categories occur in the proportions you expected, such as equal shares or last year\'s figures.',
rCode: `d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
counts <- table(d$department)
@@ -850,16 +850,16 @@ summary(model)`,
next: {
kind: 'question',
text: 'What do you want to know?',
- help: 'Some people have not had the event yet when the study ends. These models use that information; a t-test on the times would not.',
+ help: 'Some people have not had the event yet when the study ends. These models use that information; a traditional t-test on the times would not.',
options: [
{
label: 'Do groups differ in how long it takes?',
example: 'Do remote workers stay longer than office workers?',
- tech: 'Kaplan-Meier, log-rank test',
+ tech: 'Kaplan-Meier, traditional log-rank test',
next: {
kind: 'answer',
id: 'survival-curves',
- model: 'Kaplan-Meier estimate and log-rank test',
+ model: 'Kaplan-Meier estimate and traditional log-rank test',
when: 'Time to an event, such as leaving a job or relapse, compared between groups, with some cases censored.',
rCode: `library(survival)
d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
@@ -867,9 +867,9 @@ fit <- survfit(Surv(tenure_years, left_company) ~ remote, data = d)
summary(fit)$table
survdiff(Surv(tenure_years, left_company) ~ remote, data = d)`,
check:
- 'Censoring unrelated to the outcome: cases whose follow-up ends early (censored, here still employed when the data were collected) are no more or less likely to have the event than cases followed for longer. The log-rank test has most power when one group\'s risk is a constant multiple of the other\'s; when the curves cross it can miss a real difference.',
- traditional: 'The log-rank test is the score test of a Cox model with the group as its only predictor: summary(coxph(Surv(tenure_years, left_company) ~ remote, data = d, ties = "breslow"))$sctest.',
- note: 'Surv(tenure_years, left_company) pairs each employee\'s years at the company with whether they left (1) or still work there, which makes them censored (0). summary(fit)$table gives each group\'s median time to the event; survdiff() is the log-rank test; plot(fit) draws the curves.',
+ 'Censoring unrelated to the outcome: cases whose follow-up ends early (censored, here still employed when the data were collected) are no more or less likely to have the event than cases followed for longer. The traditional log-rank test has most power when one group\'s risk is a constant multiple of the other\'s; when the curves cross it can miss a real difference.',
+ traditional: 'The traditional log-rank test is the score test of a Cox model with the group as its only predictor: summary(coxph(Surv(tenure_years, left_company) ~ remote, data = d, ties = "breslow"))$sctest.',
+ note: 'Surv(tenure_years, left_company) pairs each employee\'s years at the company with whether they left (1) or still work there, which makes them censored (0). summary(fit)$table gives each group\'s median time to the event; survdiff() is the traditional log-rank test; plot(fit) draws the curves.',
further: 'Kleinbaum and Klein, Survival Analysis: A Self-Learning Text, and the survival package\'s vignettes.',
},
},
@@ -933,7 +933,7 @@ fit <- sem(model, data = d, se = "bootstrap", bootstrap = 1000)
parameterEstimates(fit, boot.ci.type = "perc")`,
check:
'Mediation is a causal claim: the predictor comes before the mediator, and the mediator before the outcome, ideally measured at different times. With data from one moment, say the pattern is consistent with mediation rather than that it shows it.',
- traditional: 'Mediation analysis, as in PROCESS model 4. The older Baron and Kenny steps and the Sobel test are no longer recommended.',
+ traditional: 'Mediation analysis, as in PROCESS model 4. The older Baron and Kenny steps and the traditional Sobel test are no longer recommended.',
note: 'lavaan needs numbers, so trained codes training as 1 for Yes and 0 for No. The indirect effect a × b is the part of the effect that runs through the mediator; it is supported when its bootstrap confidence interval excludes 0. c is the direct effect that remains, and total is their sum.',
lessonId: '17-1',
},
@@ -1229,7 +1229,7 @@ qbeta(c(0.025, 0.5, 0.975), 1 + left, 1 + stayed)
pbeta(0.2, 1 + left, 1 + stayed, lower.tail = FALSE)`,
check:
'Independent cases, each counted once, and no missing values in the outcome. The 1 + in each shape is a flat Beta(1, 1) prior; with fewer than about 30 cases, show that the answer holds under another reasonable prior too.',
- traditional: 'The exact binomial test and its confidence interval: binom.test(left, left + stayed, p = 0.2).',
+ traditional: 'The traditional exact binomial test and its confidence interval: binom.test(left, left + stayed, p = 0.2).',
note: 'Here the proportion is the share of employees who left. The qbeta() line gives the posterior median with a 95% credible interval around it; the pbeta() line is the posterior probability that the share is above 20%. Put the value your own claim is about in place of 0.2.',
lessonId: '16-2',
},
@@ -1257,11 +1257,11 @@ c(BF10 = bf10, BF01 = 1 / bf10)`,
{
label: 'Evidence about two groups',
example: 'Is there evidence that remote and office workers do not differ?',
- tech: 'Bayesian t-test',
+ tech: 'Bayesian version of the traditional t-test',
next: {
kind: 'answer',
id: 'bayes-t-test',
- model: 'Bayesian t-test (default Bayes factor)',
+ model: 'Bayesian two-group comparison (default Bayes factor)',
when: 'Two independent groups and a numeric outcome, and you want the evidence for a difference or for no difference, as psychology journals often ask for.',
rCode: `library(BayesFactor)
d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
@@ -1269,8 +1269,8 @@ bf <- ttestBF(formula = wellbeing ~ remote, data = d)
bf
1 / bf`,
check:
- 'The same as for the t-test: independent scores, roughly normal within each group. The default prior on the standardised difference is a Cauchy with scale 0.707 ("medium"); rerun with rscale = "wide" to show the conclusion does not hang on it.',
- traditional: 'The independent-samples t-test: t.test(wellbeing ~ remote, data = d, var.equal = TRUE).',
+ 'The same as for the traditional t-test: independent scores, roughly normal within each group. The default prior on the standardised difference is a Cauchy with scale 0.707 ("medium"); rerun with rscale = "wide" to show the conclusion does not hang on it.',
+ traditional: 'The traditional independent-samples t-test: t.test(wellbeing ~ remote, data = d, var.equal = TRUE).',
note: 'The printout is BF10, the evidence for a difference against none; 1 / bf gives BF01, the evidence for no difference. posterior(bf, iterations = 10000) draws plausible values of the difference itself.',
buildsOn: '16-3',
further: 'The BayesFactor package manual by Richard Morey (free online), and the free program JASP for the same analysis through menus.',
@@ -1307,7 +1307,7 @@ function crossTable(): Answer {
return {
kind: 'answer',
id: 'cross-table',
- model: 'Chi-square test of independence',
+ model: 'Traditional chi-square test of independence',
when: 'Two categorical variables, such as department and whether people left, and the question is whether they are related.',
rCode: `d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
counts <- table(d$left_company, d$department)
@@ -1315,7 +1315,7 @@ counts
chisq.test(counts, correct = FALSE)
prop.table(counts, margin = 2)`,
check:
- "Each case counted once, in one cell. Expected counts of at least 5 in nearly every cell; chisq.test() warns when they are lower, and then Fisher's exact test is the alternative: fisher.test(counts).",
+ "Each case counted once, in one cell. Expected counts of at least 5 in nearly every cell; chisq.test() warns when they are lower, and then the traditional Fisher exact test is the alternative: fisher.test(counts).",
note: 'A significant result means the outcome is distributed differently across the categories of the predictor. The last line shows how: the proportion of each outcome within each predictor category. For a yes-or-no outcome, logistic regression with this one predictor gives the same chi-square with anova(glm(left_company ~ department, data = d, family = binomial), test = "Rao"), and extends to more predictors.',
lessonId: '09-3',
};
diff --git a/src/pages/questionMatcher.ts b/src/pages/questionMatcher.ts
index 6113e28..9fd7fc2 100644
--- a/src/pages/questionMatcher.ts
+++ b/src/pages/questionMatcher.ts
@@ -79,7 +79,7 @@ export const CUES: Cue[] = [
{
id: 'ranks',
label: 'Ranks instead of means',
- why: 'You ask for a rank-based (nonparametric) test',
+ why: 'You ask for a traditional rank-based (nonparametric) test',
pattern: /\bnon ?parametric\b|\bmann whitney\b|\bwilcoxon\b|\bkruskal\b|\brank based\b|\bskewed\b/,
},
{
diff --git a/src/pages/sampleSizeDesigns.ts b/src/pages/sampleSizeDesigns.ts
index efe7ea1..73cc96b 100644
--- a/src/pages/sampleSizeDesigns.ts
+++ b/src/pages/sampleSizeDesigns.ts
@@ -27,9 +27,9 @@ export const DESIGNS: Design[] = [
id: 'two-groups',
title: 'Compare two separate groups',
example: 'Do remote and office workers differ in wellbeing? Or an A/B test on a score, such as time on a page.',
- test: 'independent-samples t-test',
+ test: 'test of the group difference (the traditional independent-samples t-test)',
model: 'lm(outcome ~ group)',
- modelNote: 'The test of the group coefficient is the independent-samples t-test (Lesson 12-1).',
+ modelNote: 'The test of the group coefficient is the same as the traditional independent-samples t-test (Lesson 12-1).',
lessonId: '12-1',
symbol: 'd',
effectName: "Cohen's d",
@@ -37,15 +37,15 @@ export const DESIGNS: Design[] = [
benchmarks: [0.2, 0.5, 0.8],
defaultEffect: 0.5,
assumes:
- 'Two groups of equal size, and scores that are roughly normal with the same spread in both groups. With equal groups, Welch\'s t-test (R\'s default) has almost exactly this power. Unequal groups need more people in total for the same power.',
+ 'Two groups of equal size, and scores that are roughly normal with the same spread in both groups. With equal groups, the traditional Welch t-test (the default of t.test()) has almost exactly this power. Unequal groups need more people in total for the same power.',
},
{
id: 'paired',
title: 'Measure the same people twice',
example: 'Does engagement change between the first and second measurement?',
- test: 'paired t-test',
+ test: 'test of the change (the traditional paired t-test)',
model: 'lmer(outcome ~ time + (1 | person))',
- modelNote: 'With two time points and nobody missing, the test of time is the paired t-test (Lesson 14-3).',
+ modelNote: 'With two time points and nobody missing, the test of time is the same as the traditional paired t-test (Lesson 14-3).',
lessonId: '14-3',
symbol: 'dz',
effectName: 'Cohen\'s dz',
@@ -59,9 +59,9 @@ export const DESIGNS: Design[] = [
id: 'repeated',
title: 'Measure the same people in several conditions',
example: 'Does reaction time differ between four conditions that every participant does?',
- test: 'repeated-measures ANOVA (the overall F test)',
+ test: 'overall F test for condition (the traditional repeated-measures ANOVA)',
model: 'lmer(outcome ~ condition + (1 | person))',
- modelNote: 'The F test for condition is the repeated-measures ANOVA (Lesson 14-4).',
+ modelNote: 'The F test for condition is the same as the traditional repeated-measures ANOVA (Lesson 14-4).',
lessonId: '14-4',
symbol: 'f',
effectName: "Cohen's f",
@@ -69,13 +69,13 @@ export const DESIGNS: Design[] = [
benchmarks: [0.1, 0.25, 0.4],
defaultEffect: 0.25,
assumes:
- 'Every person does every condition, scores are roughly normal, and sphericity holds: the differences between each pair of conditions vary about equally. This is the formula G*Power uses for "ANOVA: repeated measures, within factors" with ε = 1. When sphericity fails, the Greenhouse-Geisser correction costs power, so plan for more people. The same test is the F test for condition in a mixed model with a random intercept per person (Lesson 14-4). The plan is for the overall F test; follow-up comparisons between pairs of conditions have their own, often lower, power.',
+ 'Every person does every condition, scores are roughly normal, and sphericity holds: the differences between each pair of conditions vary about equally. This is the formula G*Power uses for repeated measures within factors, with ε = 1. When sphericity fails, the Greenhouse-Geisser correction costs power, so plan for more people. The same test is the F test for condition in a mixed model with a random intercept per person (Lesson 14-4). The plan is for the overall F test; follow-up comparisons between pairs of conditions have their own, often lower, power.',
},
{
id: 'factorial',
title: 'Two factors at once, like a 2 × 2 design',
example: 'Does a training raise wellbeing more than a waiting list, measured before and after? Or: do photo and text ads work differently for young and old viewers?',
- test: '2 × 2 ANOVA',
+ test: 'F test for one effect in a 2 × 2 design (a traditional two-way ANOVA)',
model: 'lm(outcome ~ a * b)',
modelNote: 'Each effect is one term of this model: a * b gives both main effects and their interaction (Lesson 13-2). When a factor is within people, add a random intercept per person: lmer(outcome ~ a * b + (1 | person)) (Lesson 14-4).',
lessonId: '13-2',
@@ -85,15 +85,15 @@ export const DESIGNS: Design[] = [
benchmarks: [0.1, 0.25, 0.4],
defaultEffect: 0.25,
assumes:
- 'Two factors with two levels each, the same number of people in every cell or group, and roughly normal scores with the same spread in every cell. For factors within people, the correlation between conditions is the same for every pair. Each effect in a 2 × 2 design has 1 degree of freedom, so its F test is a t-test on one contrast; for a design with both factors between people this is G*Power\'s "ANOVA: fixed effects, special, main effects and interactions". A factor with three or more levels, or several stimuli per condition, needs a simulation or the Superpower package. If each condition repeats the same trial several times, plan on each person\'s average over those trials.',
+ 'Two factors with two levels each, the same number of people in every cell or group, and roughly normal scores with the same spread in every cell. For factors within people, the correlation between conditions is the same for every pair. Each effect in a 2 × 2 design has 1 degree of freedom, so its F is the square of the t for one contrast; for a design with both factors between people this is G*Power\'s option for fixed effects, special, main effects and interactions. A factor with three or more levels, or several stimuli per condition, needs a simulation or the Superpower package. If each condition repeats the same trial several times, plan on each person\'s average over those trials.',
},
{
id: 'one-sample',
title: 'Compare one group with a fixed value',
example: 'Is average wellbeing different from a benchmark score of 50?',
- test: 'one-sample t-test',
+ test: 'test of the mean against a fixed value (the traditional one-sample t-test)',
model: 'lm(outcome - value ~ 1)',
- modelNote: 'The test of the intercept is the one-sample t-test against the fixed value (Lesson 8-2 runs it as t.test()).',
+ modelNote: 'The test of the intercept is the same as the traditional one-sample t-test against the fixed value (Lesson 8-2 runs it as t.test()).',
lessonId: '08-2',
symbol: 'd',
effectName: "Cohen's d",
@@ -106,9 +106,9 @@ export const DESIGNS: Design[] = [
id: 'anova',
title: 'Compare three or more groups',
example: 'Does wellbeing differ between the four departments?',
- test: 'one-way ANOVA (the overall F test)',
+ test: 'overall F test for the groups (the traditional one-way ANOVA)',
model: 'lm(outcome ~ group)',
- modelNote: 'With group a factor of three or more categories, the overall F test of the model is the one-way ANOVA (Lesson 12-2).',
+ modelNote: 'With group a factor of three or more categories, the overall F test of the model is the same as the traditional one-way ANOVA (Lesson 12-2).',
lessonId: '12-2',
symbol: 'f',
effectName: "Cohen's f",
@@ -122,7 +122,7 @@ export const DESIGNS: Design[] = [
id: 'correlation',
title: 'Relate two numeric variables',
example: 'Is workload related to wellbeing?',
- test: 'Pearson correlation test',
+ test: 'test of the correlation (the traditional Pearson correlation test)',
model: 'lm(outcome ~ predictor)',
modelNote: 'The test of the slope gives the same p-value as the test of the correlation (Lesson 10-2).',
lessonId: '10-1',
@@ -153,7 +153,7 @@ export const DESIGNS: Design[] = [
id: 'r2-change',
title: 'Test predictors on top of others',
example: 'Does autonomy predict wellbeing over and above workload and tenure?',
- test: 'F test of the R² change (with one added predictor, the same as its t-test)',
+ test: 'F test of the R² change (with one added predictor, the same as the test of its coefficient)',
model: 'anova(small_model, big_model)',
modelNote: 'The F test comparing lm(outcome ~ controls) with lm(outcome ~ controls + x) is the test of the R² change (Lesson 11-3).',
lessonId: '11-3',
@@ -169,9 +169,9 @@ export const DESIGNS: Design[] = [
id: 'proportions',
title: 'Compare a yes/no outcome between two groups',
example: 'Do fewer mentored employees leave the company? Or an A/B test: do more visitors click with version B than with version A?',
- test: 'test of two proportions (a chi-square test on a 2 × 2 table)',
+ test: 'traditional test of two proportions (a traditional chi-square test on a 2 × 2 table)',
model: 'glm(yes_no ~ group, family = binomial)',
- modelNote: 'The page plans with the test of two proportions. The group coefficient in this logistic regression has almost the same power, at most a few percentage points less, so a few extra people cover it (Module 15).',
+ modelNote: 'The page plans with the traditional test of two proportions. The group coefficient in this logistic regression has almost the same power, at most a few percentage points less, so a few extra people cover it (Module 15).',
lessonId: '09-3',
symbol: 'p₂',
effectName: 'the two proportions',
@@ -184,9 +184,9 @@ export const DESIGNS: Design[] = [
id: 'chi-square',
title: 'Relate two categorical variables',
example: 'Is department related to working remotely?',
- test: 'chi-square test of independence (or goodness of fit)',
+ test: 'traditional chi-square test of independence (or goodness of fit)',
model: 'chisq.test(table(a, b))',
- modelNote: 'As a model: glm(count ~ a + b, family = poisson) on the table of counts gives the same chi-square test (Lesson 15-4).',
+ modelNote: 'As a model: glm(count ~ a + b, family = poisson) on the table of counts gives the same chi-square as the traditional test (Lesson 15-4).',
lessonId: '09-3',
symbol: 'w',
effectName: "Cohen's w",
@@ -194,7 +194,7 @@ export const DESIGNS: Design[] = [
benchmarks: [0.1, 0.3, 0.5],
defaultEffect: 0.3,
assumes:
- 'Independent observations and expected counts of at least about 5 in every cell. The degrees of freedom are (rows − 1) × (columns − 1), or categories − 1 for a goodness-of-fit test.',
+ 'Independent observations and expected counts of at least about 5 in every cell. The degrees of freedom are (rows − 1) × (columns − 1), or categories − 1 for a traditional goodness-of-fit test.',
},
];
diff --git a/src/pages/traditionalNames.test.ts b/src/pages/traditionalNames.test.ts
new file mode 100644
index 0000000..c7be24d
--- /dev/null
+++ b/src/pages/traditionalNames.test.ts
@@ -0,0 +1,44 @@
+import { expect, test } from 'vitest';
+import { GROUPS, TREE, type Node } from './modelTree';
+import { CUES } from './questionMatcher';
+import { DESIGNS } from './sampleSizeDesigns';
+
+// The course teaches models, so the chooser and the sample size page name a
+// traditional test only with "traditional" a few words before it.
+const NAMES =
+ /t-test|\bANOVAs?\b|ANCOVA|MANOVA|chi-square test|chi-square goodness|goodness-of-fit test|Mann-Whitney|Wilcoxon|Kruskal-Wallis|Friedman|Fisher'?s? exact|McNemar|Welch|binomial test|log-rank|Sobel|Spearman|Pearson correlation|test of two proportions/gi;
+
+function unprefixed(text: string): string[] {
+ const bad: string[] = [];
+ for (const match of text.matchAll(NAMES)) {
+ const at = match.index!;
+ const after = text.slice(at + match[0].length);
+ // R code in the prose: anova(), friedman.test(), method = "spearman".
+ if (/^(\(|\.test)/.test(after) || /["._]$/.test(text.slice(0, at))) continue;
+ if (!/traditional\s+(\S+\s+){0,5}$/i.test(text.slice(0, at))) bad.push(`${match[0]} in: ${text.slice(Math.max(0, at - 60), at + 30)}`);
+ }
+ return bad;
+}
+
+function treeTexts(node: Node): string[] {
+ if (node.kind === 'answer') {
+ const { rCode: _code, id: _id, lessonId: _lesson, buildsOn: _builds, kind: _kind, ...texts } = node;
+ return Object.values(texts).filter((value): value is string => typeof value === 'string');
+ }
+ return [node.text, node.help ?? '', ...node.options.flatMap((option) => [option.label, option.example, option.tech, option.group ?? '', ...treeTexts(option.next)])];
+}
+
+test('the model chooser names traditional tests only as traditional', () => {
+ const texts = [...treeTexts(TREE), ...GROUPS.flatMap((group) => [group.name, group.blurb]), ...CUES.flatMap((cue) => [cue.label, cue.why])];
+ expect(texts.flatMap(unprefixed)).toEqual([]);
+});
+
+test('the sample size page names traditional tests only as traditional', () => {
+ const texts = DESIGNS.flatMap(({ id: _id, model: _model, lessonId: _lesson, ...design }) => Object.values(design).filter((value): value is string => typeof value === 'string'));
+ expect(texts.flatMap(unprefixed)).toEqual([]);
+});
+
+test('the check itself catches a bare test name', () => {
+ expect(unprefixed('the paired t-test')).toHaveLength(1);
+ expect(unprefixed('the traditional paired t-test, then anova(model)')).toEqual([]);
+});
diff --git a/src/stats/power.ts b/src/stats/power.ts
index 03896f3..a89dfe7 100644
--- a/src/stats/power.ts
+++ b/src/stats/power.ts
@@ -274,7 +274,7 @@ export function rCode(plan: Plan, solveFor: 'n' | 'effect', target: number, n?:
return `library(pwr)\npwr.anova.test(k = ${plan.k ?? 3}, ${size}, ${a}, ${pw})`;
}
case 'repeated': {
- const head = `# Repeated-measures ANOVA, assuming sphericity (G*Power's formula)\nk <- ${plan.k ?? 3}; r <- ${num(plan.rho ?? 0.5)}; alpha <- ${num(plan.alpha)}; target <- ${num(target)}\npower_at <- function(n, f) {\n df1 <- k - 1; df2 <- (n - 1) * (k - 1)\n 1 - pf(qf(1 - alpha, df1, df2), df1, df2, ncp = n * k * f^2 / (1 - r))\n}\n`;
+ const head = `# Repeated measures, assuming sphericity (G*Power's formula)\nk <- ${plan.k ?? 3}; r <- ${num(plan.rho ?? 0.5)}; alpha <- ${num(plan.alpha)}; target <- ${num(target)}\npower_at <- function(n, f) {\n df1 <- k - 1; df2 <- (n - 1) * (k - 1)\n 1 - pf(qf(1 - alpha, df1, df2), df1, df2, ncp = n * k * f^2 / (1 - r))\n}\n`;
return e
? `${head}n <- 2\nwhile (power_at(n, ${num(plan.effect)}) < target) n <- n + 1\nn`
: `${head}uniroot(function(f) power_at(${n}, f) - target, c(1e-6, 10), tol = 1e-10)$root`;
@@ -282,7 +282,7 @@ export function rCode(plan: Plan, solveFor: 'n' | 'effect', target: number, n?:
case 'factorial': {
const layout = plan.layout ?? 'between';
const per = { between: 'people per cell', mixed: 'people per group', within: 'people' }[layout];
- const head = `# One effect in a 2 × 2 ANOVA (${layout === 'mixed' ? 'one factor between, one within people' : `both factors ${layout} people`}): F test with 1 df\nalpha <- ${num(plan.alpha)}; target <- ${num(target)}\npower_at <- function(n, f) { # n = ${per}\n N <- ${CELLS[layout]} * n; df2 <- N - ${LOST_DF[layout]}\n 1 - pf(qf(1 - alpha, 1, df2), 1, df2, ncp = N * f^2)\n}\n`;
+ const head = `# One effect in a 2 × 2 design (${layout === 'mixed' ? 'one factor between, one within people' : `both factors ${layout} people`}): F test with 1 df\nalpha <- ${num(plan.alpha)}; target <- ${num(target)}\npower_at <- function(n, f) { # n = ${per}\n N <- ${CELLS[layout]} * n; df2 <- N - ${LOST_DF[layout]}\n 1 - pf(qf(1 - alpha, 1, df2), 1, df2, ncp = N * f^2)\n}\n`;
return e
? `${head}n <- 2\nwhile (power_at(n, ${num(plan.effect)}) < target) n <- n + 1\nn # ${per}`
: `${head}uniroot(function(f) power_at(${n}, f) - target, c(1e-6, 10), tol = 1e-10)$root`;
From 483be4a92e6c3df1c07da1d7c0f75599d812a897 Mon Sep 17 00:00:00 2001
From: Peter Slijkhuis
Date: Wed, 30 Sep 2026 21:10:42 +0000
Subject: [PATCH 2/3] Fix statements and question routes the chooser audit
found
---
src/pages/ModelChooser.tsx | 4 ++--
src/pages/modelTree.ts | 26 +++++++++++++-------------
src/pages/questionMatcher.test.ts | 11 +++++++++++
src/pages/questionMatcher.ts | 16 ++++++++--------
4 files changed, 34 insertions(+), 23 deletions(-)
diff --git a/src/pages/ModelChooser.tsx b/src/pages/ModelChooser.tsx
index 29a79b8..51a0fb5 100644
--- a/src/pages/ModelChooser.tsx
+++ b/src/pages/ModelChooser.tsx
@@ -498,8 +498,8 @@ export default function ModelChooser() {
Most snippets run as they are in the R Workspace, and the ones that need RStudio say so.
- Each reads workplace.csv or wellbeing-population.csv, the course data the workspace already has, or a dataset built
- into R when the course data has nothing that fits. For your own data, change the file name in read.csv() and the
+ Each reads workplace.csv or wellbeing-population.csv, the course data the workspace already has, or a dataset that
+ comes with R or with the package the snippet uses, when the course data has nothing that fits. For your own data, change the file name in read.csv() and the
column names.
diff --git a/src/pages/modelTree.ts b/src/pages/modelTree.ts
index 005419f..c1aeea7 100644
--- a/src/pages/modelTree.ts
+++ b/src/pages/modelTree.ts
@@ -375,7 +375,7 @@ model <- manova(cbind(wellbeing, engagement_t2, performance) ~ department, data
summary(model, test = "Pillai")
summary.aov(model)`,
check:
- "The outcomes correlate moderately with each other; if they do not, analyse them one at a time. No extreme outliers, and more cases in every group than there are outcomes. Pillai's trace is the most robust of the four test statistics.",
+ "The outcomes correlate moderately with each other; if they do not, analyse them one at a time. No extreme outliers, more cases in every group than there are outcomes, and similar variances and correlations of the outcomes in every group. Pillai's trace is the most robust of the four test statistics when that last condition fails.",
traditional: 'The traditional one-way MANOVA.',
note: 'The overall test asks whether the groups differ on the combination of outcomes. summary.aov() then gives one traditional ANOVA per outcome; correct those for multiple testing, for example with p.adjust(p, method = "holm").',
buildsOn: '12-2',
@@ -417,7 +417,7 @@ model <- lmer(engagement ~ time + (1 | employee_id), data = long_d)
summary(model)`,
check:
'Data in long format: one row per person per measurement. Residuals roughly normal, which matters mainly in small samples. With skewed differences in a small sample, the traditional Wilcoxon signed-rank test: wilcox.test(d$engagement_t2, d$engagement_t1, paired = TRUE).',
- traditional: 'The traditional paired-samples t-test: t.test(d$engagement_t2, d$engagement_t1, paired = TRUE). Same t, same p, when nobody is missing a measurement.',
+ traditional: 'The traditional paired-samples t-test: t.test(d$engagement_t2, d$engagement_t1, paired = TRUE). Same t, same p, when nobody is missing a measurement and the two measurements correlate positively (no singular fit).',
note: '(1 | employee_id) gives every employee their own starting level, so the model knows which scores belong together. engagement_t1 sorts first, so it is the reference and the time coefficient is the change from the first measurement to the second. Unlike the traditional paired t-test, it keeps people who missed a measurement.',
lessonId: '14-3',
},
@@ -441,7 +441,7 @@ anova(model)
emmeans(model, pairwise ~ time, adjust = "holm")`,
check:
'Data in long format: one row per case per measurement, which ChickWeight already is. Residuals roughly normal, which matters mainly in small samples. The model assumes the same spread at every time point and the same correlation between every pair of them. Compare the SDs per time point first: in ChickWeight they grow from about 1 g at hatching to about 72 g at day 21, so read its p-values with care. For a growth process like this, a growth-curve model with random slopes fits better. For a small, clearly skewed sample, the traditional Friedman test, which needs every chick at every time point, so keep only the complete ones first: complete <- long_d %>% group_by(Chick) %>% filter(n() == 3) %>% ungroup() %>% droplevels(), then friedman.test(weight ~ time | Chick, data = complete).',
- traditional: 'The traditional repeated-measures ANOVA. With complete data the model\'s F matches its uncorrected F: both assume the time points are equally correlated.',
+ traditional: 'The traditional repeated-measures ANOVA. With complete data and no singular fit, the model\'s F matches its uncorrected F. The model assumes equal variances and correlations at every time point (compound symmetry), which is stricter than the sphericity the traditional F needs.',
note: 'anova() tests whether weight differs across the time points at all; emmeans then compares each pair of time points. (1 | Chick) lets each chick have its own level, as (1 | id) would for people. Unlike the traditional repeated-measures ANOVA, the model keeps the chicks that missed a weighing.',
lessonId: '14-4',
},
@@ -532,7 +532,7 @@ summary(model)`,
d <- read.csv("data/workplace.csv", stringsAsFactors = TRUE)
model <- lmer(wellbeing ~ autonomy + workload + (1 | site), data = d)
summary(model)`,
- check: 'Enough groups to estimate how they vary: a handful at the very least, and ideally twenty or more. The 6 sites here are few, which is why Lesson 14-3 finds SD_site near 0. Residuals roughly normal, which matters mainly in small samples.',
+ check: 'Enough groups to estimate how they vary: a handful at the very least, and ideally twenty or more. With only 6 sites, as here, the site SD is estimated imprecisely and may come out as 0 (a singular fit). Residuals roughly normal, which matters mainly in small samples.',
note: 'Employees at the same site are more alike than employees at different sites; (1 | site) accounts for that. If people are also measured repeatedly, nest them: (1 | site/employee_id). If a predictor\'s effect may differ between sites, add a random slope: (autonomy | site).',
lessonId: '14-3',
},
@@ -572,10 +572,10 @@ model <- glm(left_company ~ wellbeing + tenure_years, data = d, family = binomia
summary(model)
exp(cbind(OR = coef(model), confint(model)))`,
check:
- 'Independent observations, and enough cases of the rarer outcome: a common rule of thumb is at least 10 per estimated coefficient (a factor with k levels uses k − 1). If R warns that fitted probabilities of 0 or 1 occurred, a predictor separates the outcomes perfectly; Firth\'s method, logistf::logistf(), handles that in RStudio.',
+ 'Independent observations, numeric predictors that relate in a straight line to the log odds, and enough cases of the rarer outcome: a common rule of thumb is at least 10 per estimated coefficient (a factor with k levels uses k − 1). If R warns that fitted probabilities of 0 or 1 occurred, a predictor separates the outcomes perfectly; Firth\'s method, logistf::logistf(), handles that in RStudio.',
traditional:
'With one categorical predictor, such as remote, the traditional chi-square test of independence: chisq.test(table(d$left_company, d$remote), correct = FALSE), which matches anova(glm(left_company ~ remote, data = d, family = binomial), test = "Rao").',
- note: 'The coefficients are in log odds. exp() turns them into odds ratios: above 1, the outcome becomes more likely; below 1, less likely.',
+ note: 'The intercept is in log odds and each slope is a difference in log odds (a log odds ratio). exp() turns the slopes into odds ratios: above 1, the outcome becomes more likely; below 1, less likely.',
lessonId: '15-2',
},
},
@@ -622,7 +622,7 @@ summary(model)
exp(fixef(model))`,
check:
'For repeated measurements, long format: one row per person per measurement. Clustered data, such as employees in sites, is already in shape. With few clusters or a rare outcome the model may fail to converge; simplify the random part first.',
- traditional: "For one yes-or-no answer at two time points, the traditional McNemar test: mcnemar.test(table(before, after)).",
+ traditional: "For one yes-or-no answer at two time points, the traditional McNemar test: mcnemar.test(table(before, after)). It asks the same question; the mixed model gives a similar but not identical p.",
note: 'exp() turns the coefficients into odds ratios for people in the same cluster, here the same site, holding its random intercept fixed. These are usually further from 1 than the population-average odds ratios an ordinary logistic regression gives.',
buildsOn: '14-2',
further: 'Gelman and Hill, Data Analysis Using Regression and Multilevel/Hierarchical Models.',
@@ -699,7 +699,7 @@ model <- polr(Sat ~ Infl + Type, data = housing, weights = Freq, Hess = TRUE)
summary(model)
exp(cbind(OR = coef(model), confint(model)))`,
check:
- 'Proportional odds: each predictor shifts the odds of being above any cut-off by the same amount. The ordinal package tests it: ordinal::nominal_test(ordinal::clm(Sat ~ Infl + Type, data = housing, weights = Freq)), where a small p flags a predictor that breaks the assumption. A total of many Likert items is usually analysed as a number instead. MASS hides dplyr\'s select(), so attach MASS before dplyr, or write dplyr::select().',
+ 'Proportional odds: each predictor multiplies the odds of being above any cut-off by the same factor (the same odds ratio at every cut-off). The ordinal package tests it: ordinal::nominal_test(ordinal::clm(Sat ~ Infl + Type, data = housing, weights = Freq)), where a small p flags a predictor that breaks the assumption. A total of many Likert items is usually analysed as a number instead. MASS hides dplyr\'s select(), so attach MASS before dplyr, or write dplyr::select().',
note: 'Each row of housing is one combination of answers, and Freq says how many tenants gave it, hence weights = Freq; with one row per person, leave it out. Your own outcome needs to be a factor with its levels in order: factor(x, levels = c("low", "medium", "high"), ordered = TRUE). An odds ratio above 1 means that predictor goes with higher categories of the outcome. The intercepts (the zeta values in the output) are the cut-offs between adjacent categories.',
buildsOn: '15-2',
further: 'Alan Agresti, Analysis of Ordinal Categorical Data, and the ordinal package\'s vignettes.',
@@ -868,7 +868,7 @@ summary(fit)$table
survdiff(Surv(tenure_years, left_company) ~ remote, data = d)`,
check:
'Censoring unrelated to the outcome: cases whose follow-up ends early (censored, here still employed when the data were collected) are no more or less likely to have the event than cases followed for longer. The traditional log-rank test has most power when one group\'s risk is a constant multiple of the other\'s; when the curves cross it can miss a real difference.',
- traditional: 'The traditional log-rank test is the score test of a Cox model with the group as its only predictor: summary(coxph(Surv(tenure_years, left_company) ~ remote, data = d, ties = "breslow"))$sctest.',
+ traditional: 'The traditional log-rank test is the score test of a Cox model with the group as its only predictor: summary(coxph(Surv(tenure_years, left_company) ~ remote, data = d, ties = "exact"))$sctest. With no tied event times, any ties option gives the same value.',
note: 'Surv(tenure_years, left_company) pairs each employee\'s years at the company with whether they left (1) or still work there, which makes them censored (0). summary(fit)$table gives each group\'s median time to the event; survdiff() is the traditional log-rank test; plot(fit) draws the curves.',
further: 'Kleinbaum and Klein, Survival Analysis: A Self-Learning Text, and the survival package\'s vignettes.',
},
@@ -890,7 +890,7 @@ tidy(model, conf.int = TRUE, exponentiate = TRUE)
cox.zph(model)`,
check:
'Proportional hazards: each predictor multiplies the risk by the same factor at every point in time. cox.zph() tests this for each predictor; a small p flags a problem. Roughly 10 events or more per predictor.',
- note: 'The exponentiated coefficients are hazard ratios: 1.5 means a 50% higher risk of the event at any moment for each one-unit increase in the predictor.',
+ note: 'The exponentiated coefficients are hazard ratios: 1.5 means a 50% higher hazard (the rate of the event at any moment, among those who have not had it yet) for each one-unit increase in the predictor.',
further: 'Kleinbaum and Klein, Survival Analysis: A Self-Learning Text, and the survival package\'s vignettes.',
},
},
@@ -932,7 +932,7 @@ set.seed(1)
fit <- sem(model, data = d, se = "bootstrap", bootstrap = 1000)
parameterEstimates(fit, boot.ci.type = "perc")`,
check:
- 'Mediation is a causal claim: the predictor comes before the mediator, and the mediator before the outcome, ideally measured at different times. With data from one moment, say the pattern is consistent with mediation rather than that it shows it.',
+ 'Mediation is a causal claim: the predictor comes before the mediator, and the mediator before the outcome, ideally measured at different times. With data from one moment, say the pattern is consistent with mediation rather than that it shows it. It also assumes no unmeasured variable affects both the mediator and the outcome; randomising the predictor does not guarantee this, and the bootstrap cannot fix it.',
traditional: 'Mediation analysis, as in PROCESS model 4. The older Baron and Kenny steps and the traditional Sobel test are no longer recommended.',
note: 'lavaan needs numbers, so trained codes training as 1 for Yes and 0 for No. The indirect effect a × b is the part of the effect that runs through the mediator; it is supported when its bootstrap confidence interval excludes 0. c is the direct effect that remains, and total is their sum.',
lessonId: '17-1',
@@ -1112,7 +1112,7 @@ model
predict(model, n.ahead = 10, newxreg = 1973:1982 - 1920)`,
check:
'Equally spaced observations with no gaps. A trend or a seasonal pattern needs differencing, a trend regressor (as xreg does here) or a seasonal term; plot(series) and acf(series) show both. In RStudio, forecast::auto.arima(series) chooses the orders for you.',
- note: 'order = c(p, d, q) sets the autoregressive terms, the number of differences and the moving-average terms. xreg adds a straight-line trend in years, centred on 1920; predict() needs the future years in newxreg. predict() gives forecasts with standard errors, which grow the further ahead you look. Compare candidate orders by AIC: lower is better. For your own data, ts(d$value, start = c(2020, 1), frequency = 12) turns a column of monthly values into a series.',
+ note: 'order = c(p, d, q) sets the autoregressive terms, the number of differences and the moving-average terms. xreg adds a straight-line trend in years, centred on 1920; predict() needs the future years in newxreg. predict() gives forecasts with standard errors, which grow the further ahead you look. Compare candidate orders with the same d by AIC: lower is better. For your own data, ts(d$value, start = c(2020, 1), frequency = 12) turns a column of monthly values into a series.',
further: 'Hyndman and Athanasopoulos, Forecasting: Principles and Practice (free online at otexts.com), which uses the fable package.',
},
},
@@ -1173,7 +1173,7 @@ set.seed(1)
cv <- cv.glmnet(x, y, alpha = 1)
coef(cv, s = "lambda.1se")`,
check:
- 'Complete cases only, since model.matrix() drops rows with missing values. glmnet standardises the predictors for you. For a yes-or-no outcome, add family = "binomial". Avoid stepwise selection: its p-values and R² are biased upwards.',
+ 'Complete cases only, since model.matrix() drops rows with missing values. glmnet standardises the predictors for you. For a yes-or-no outcome, add family = "binomial". Avoid stepwise selection: its p-values come out too small and its R² too large.',
note: 'The lasso shrinks coefficients towards zero and sets the weakest to exactly zero, so the predictors left over are the selection. lambda.1se is the simplest model whose cross-validated error is within one standard error of the best. The coefficients are shrunk on purpose, so they come without p-values.',
further: 'An Introduction to Statistical Learning, the chapter on linear model selection and regularisation.',
},
diff --git a/src/pages/questionMatcher.test.ts b/src/pages/questionMatcher.test.ts
index ce43d2a..8762485 100644
--- a/src/pages/questionMatcher.test.ts
+++ b/src/pages/questionMatcher.test.ts
@@ -5,6 +5,17 @@ import { CUES, detectCues, matchQuestion, RULE_IDS } from './questionMatcher';
// Realistic questions as students write them. `top` must rank first; `inTop2`
// is for questions a careful reader could answer two ways.
const QUESTIONS: { q: string; top?: string; inTop2?: string }[] = [
+ // Found in the strict audit of 2026-09-30.
+ { q: 'Does stress differ between the three exam weeks?', top: 'repeated-measures' },
+ { q: 'Do some chicks grow faster than others over 21 days?', top: 'growth-curve' },
+ { q: 'Does exercise reduce the number of doctor visits per year?', top: 'poisson-regression' },
+ { q: 'Does the number of sick days per month differ between departments?', top: 'poisson-regression' },
+ { q: 'Does social class predict income?', top: 'simple-regression' },
+ { q: 'Does therapy reduce depression more than a waiting list, measured before and after?', top: 'time-by-group' },
+ { q: 'Do training and mentoring work better in combination?', top: 'factorial' },
+ { q: 'Did engagement rise more for trained employees than for others?', top: 'time-by-group' },
+ { q: 'Is the way people travel to work linked to their department?', top: 'cross-table' },
+ { q: 'Most people take 0 to 3 sick days, but a few take 40: what predicts them?', top: 'poisson-regression' },
{ q: 'Do remote workers report higher wellbeing than office workers?', top: 'two-groups' },
{ q: 'Is there a difference in exam scores between men and women?', top: 'two-groups' },
{ q: 'Do the four departments differ in wellbeing?', top: 'several-groups' },
diff --git a/src/pages/questionMatcher.ts b/src/pages/questionMatcher.ts
index 9fd7fc2..fe42ac2 100644
--- a/src/pages/questionMatcher.ts
+++ b/src/pages/questionMatcher.ts
@@ -67,7 +67,7 @@ export const CUES: Cue[] = [
id: 'count',
label: 'Outcome is a count',
why: 'The outcome is a count of how often something happens',
- pattern: /\bhow many\b|\bnumber of (?!(items|variables|questions|predictors|factors|components|hours|years)\b)|\bcount of\b|\bhow often\b|\bfrequency\b|\btimes (a|per|each) (day|week|month|year)\b/,
+ pattern: /\bhow many\b|\bnumber of (?!(items|variables|questions|predictors|factors|components|hours|years)\b)|\bcount of\b|\bhow often\b|\bsick days\b|\bfrequency\b|\btimes (a|per|each) (day|week|month|year)\b/,
},
{
id: 'ordinal',
@@ -87,7 +87,7 @@ export const CUES: Cue[] = [
label: 'Outcome is a category',
why: 'The outcome is one of several categories with no order',
pattern:
- /\bwhich (kind|type|category|brand|option|mode|party|product|programme|program|one) (of|do|does|people|they|students|employees)\b|\bchoice of\b|\bchoose (between|among)\b|\bhow (people|they|students|employees|staff) (travel|commute|vote)|\bvote for\b|\bcar,? bike\b/,
+ /\bwhich (kind|type|category|brand|option|mode|party|product|programme|program|one) (of|do|does|people|they|students|employees)\b|\bchoice of\b|\bchoose (between|among)\b|\b(how|the way) (people|they|students|employees|staff) (travel|commute|vote)|\bvote for\b|\bcar,? bike\b/,
},
{
id: 'survival',
@@ -106,7 +106,7 @@ export const CUES: Cue[] = [
label: 'Compares groups',
why: 'You compare groups',
pattern: new RegExp(
- `\\b(differ|differs|differed|difference|differences|different from each other|compare|compared|comparing|comparison|versus|vs|group|condition|treatment|control group|experimental|intervention|gender|men|women|male|female|remote|office workers|training|trained|untrained|mentoring|nationality|department|${GROUP_NOUNS})\\b|\\bper (programme|program|department|group|condition|faculty|country)\\b|(? found.delete(id as CueId));
+ if (/\b(paired|matched)\b/.test(t) || ((found.has('twice') || found.has('waves')) && !groupNamed)) ['groups', 'twoGroups', 'manyGroups'].forEach((id) => found.delete(id as CueId));
// "Differ in how fast they improve" is about rates, not groups.
if (found.has('growth') && !found.has('twoGroups') && !found.has('manyGroups')) found.delete('groups');
// Reducing many variables is not picking predictors.
From a3ffb3cb9d002c9f9a2a49c335dfc489ae518287 Mon Sep 17 00:00:00 2001
From: Peter Slijkhuis
Date: Wed, 30 Sep 2026 21:36:35 +0000
Subject: [PATCH 3/3] Fix the sample size findings: rounding, correlation
direction, G*Power f, partial eta squared, simulation scale and alpha
---
src/pages/SampleSize.test.tsx | 5 +++--
src/pages/SampleSize.tsx | 40 +++++++++++++++++++++-------------
src/pages/sampleSizeDesigns.ts | 19 ++++++++++++----
src/stats/power.itest.ts | 2 +-
src/stats/power.ts | 15 ++++++++-----
5 files changed, 53 insertions(+), 28 deletions(-)
diff --git a/src/pages/SampleSize.test.tsx b/src/pages/SampleSize.test.tsx
index 1ddfd0f..a682bae 100644
--- a/src/pages/SampleSize.test.tsx
+++ b/src/pages/SampleSize.test.tsx
@@ -214,7 +214,7 @@ describe('sample size page', () => {
expect(question()).toBe('How many people will you lose?');
await pick(/^None/);
// λ = N·f² with df2 = N − 2: 128 people, 64 per group.
- expect(report()).toContain('128 participants (64 per group, each measured in both conditions) are needed to detect an effect of f = 0.25 (with a correlation of r = .50 between conditions)');
+ expect(report()).toContain("128 participants (64 per group, each measured in both conditions) are needed to detect an effect of f = 0.25 (an f that already includes the correlation of r = .50 between conditions; in G*Power's repeated-measures procedures, with r = .50, this is f = 0.125)");
expect(report()).toContain('the F test for the interaction in a 2 × 2 mixed design, one factor between and one within people (a traditional mixed ANOVA)');
expect(trail()).toContain('One factor between, one within');
});
@@ -227,7 +227,7 @@ describe('sample size page', () => {
await pick(/Whether one factor's effect depends on the other/);
expect(question()).toBe('How big is the effect you want to be able to find?');
await pick(/Choose a size in plain words/);
- const disappears = screen.getByRole('button', { name: /The effect disappears \(f = 0.13\)/ });
+ const disappears = screen.getByRole('button', { name: /The effect disappears \(f = 0.125\)/ });
expect(disappears.textContent).toContain('Recommended');
// f = 0.5 / 4 = 0.125: λ = N/64 reaches 80% power at 508 people (127 per cell), as in R.
expect(disappears.textContent).toContain('Needs about 508 people');
@@ -236,6 +236,7 @@ describe('sample size page', () => {
await pick(/α = \.05/);
await pick(/^None/);
expect(report()).toContain('508 participants (127 in each of the four cells)');
+ expect(report()).toContain('an effect of f = 0.125');
});
test("the browser's Back button goes back one question, keeping the answers", async () => {
diff --git a/src/pages/SampleSize.tsx b/src/pages/SampleSize.tsx
index d417dd5..cbdce69 100644
--- a/src/pages/SampleSize.tsx
+++ b/src/pages/SampleSize.tsx
@@ -3,7 +3,7 @@ import { Link, useSearchParams } from 'react-router-dom';
import AvatarTip from '../components/AvatarTip';
import { pnorm } from '../stats/distributions';
import { HAS_SIDES, minN, PER_GROUP, powerAt, rCode, solveEffect, solveN, stimulusSimCode, totalN, type DesignId, type Layout, type Plan, type Which } from '../stats/power';
-import { DESIGNS, designById, effectInputs, presets, type Design, type EffectRoute } from './sampleSizeDesigns';
+import { DESIGNS, designById, effectInputs, presets, shortEffect, type Design, type EffectRoute } from './sampleSizeDesigns';
import './ModelChooser.css';
import './SampleSize.css';
@@ -17,11 +17,11 @@ const noZero = (text: string) => text.replace(/^(-?)0\./, '$1.');
function formatEffect(design: Design, value: number): string {
if (design.id === 'correlation') return noZero(value.toFixed(2));
if (design.symbol === 'f²') return value.toFixed(3);
- return value.toFixed(2);
+ return shortEffect(value);
}
/** What an effect of this size means, in words a student can check against their field. */
-function meaning(design: Design, effect: number, p1: number): string {
+function meaning(design: Design, effect: number, p1: number, k = 3, rho = 0.5): string {
switch (design.id) {
case 'two-groups':
return `If scores are normal, about ${pct(pnorm(effect))} of the higher group scores above the other group's mean (it would be 50% with no difference).`;
@@ -30,7 +30,8 @@ function meaning(design: Design, effect: number, p1: number): string {
case 'one-sample':
return `If scores are normal, about ${pct(pnorm(effect))} of people score on the expected side of the fixed value.`;
case 'repeated':
- return `Within one condition, the conditions would explain ${pct(effect ** 2 / (1 + effect ** 2), 1)} of the variance in the scores (η², leaving out the differences between people).`;
+ // Partial η² as a repeated-measures analysis reports it: λ = n(k − 1)·ηp²/(1 − ηp²).
+ return `The conditions would explain ${pct((k * effect ** 2) / (k * effect ** 2 + (k - 1) * (1 - rho)), 1)} of the variance left once the differences between people are taken out (partial η², as a repeated-measures analysis reports it).`;
case 'factorial':
return `This effect explains ${pct(effect ** 2 / (1 + effect ** 2), 1)} of the variance the other effects leave (partial η²).`;
case 'anova':
@@ -79,7 +80,13 @@ function testText(design: Design, sides: 1 | 2, plan?: Plan): string {
function effectText(design: Design, plan: Plan, value: number): string {
if (design.id === 'proportions') return `a difference between ${pct(plan.p1 ?? 0, 1)} and ${pct(value, 1)}`;
- if (design.id === 'repeated' || (design.id === 'factorial' && plan.layout !== 'between')) return `an effect of f = ${formatEffect(design, value)} (with a correlation of r = ${noZero((plan.rho ?? 0.5).toFixed(2))} between conditions)`;
+ const r = plan.rho ?? 0.5;
+ if (design.id === 'repeated') return `an effect of f = ${formatEffect(design, value)} (with a correlation of r = ${noZero(r.toFixed(2))} between conditions)`;
+ if (design.id === 'factorial' && plan.layout !== 'between') {
+ // This page's f already contains the correlation (λ = N·f²). G*Power's repeated-measures f does not.
+ const gpower = plan.layout === 'mixed' ? `; in G*Power's repeated-measures procedures, with r = ${noZero(r.toFixed(2))}, this is f = ${shortEffect(value * Math.sqrt((plan.which === 'main-between' ? 1 + r : 1 - r) / 2))}` : '';
+ return `an effect of f = ${formatEffect(design, value)} (an f that already includes the correlation of r = ${noZero(r.toFixed(2))} between conditions${gpower})`;
+ }
return `an effect of ${design.symbol} = ${formatEffect(design, value)}`;
}
@@ -348,7 +355,7 @@ export default function SampleSize() {
const stimuli = design.id === 'repeated' && answers.stimuli && result && (result.kind === 'n' || result.kind === 'effect');
const simCode = stimuli
- ? stimulusSimCode(result.n, parse(answers.text.k), parse(answers.text.nstim), result.kind === 'n' ? result.plan.effect : result.effect)
+ ? stimulusSimCode(result.n, parse(answers.text.k), parse(answers.text.nstim), result.kind === 'n' ? result.plan.effect : result.effect, result.plan.alpha)
: '';
/** Everyone needed for an effect at the usual settings, to show next to each plain-words option. */
@@ -510,7 +517,7 @@ export default function SampleSize() {
{!missing && !error && (
- {design.id === 'proportions' ? <>That is a difference of {(100 * Math.abs(effect - p1)).toFixed(1)} percentage points.> : <>That is {design.symbol} = {formatEffect(design, effect)}. {meaning(design, effect, 0)}>}
+ {design.id === 'proportions' ? <>That is a difference of {(100 * Math.abs(effect - p1)).toFixed(1)} percentage points.> : <>That is {design.symbol} = {formatEffect(design, effect)}. {meaning(design, effect, 0, parse(answers.text.k), answers.rho)}>}
)}
{error &&
{error}
}
@@ -533,7 +540,7 @@ export default function SampleSize() {
);
} else if (step === 'which') {
question = 'Which effect is your question about?';
- help = 'Plan for the effect your hypothesis is about. An interaction asks whether the effect of one factor depends on the other; it needs far more people than a main effect of the same size, often four times as many or more.';
+ help = 'Plan for the effect your hypothesis is about. An interaction asks whether the effect of one factor depends on the other; it needs far more people than a main effect of a within-person factor of the same size, often four times as many.';
const main: Option[] = answers.layout === 'mixed'
? [
{ value: 'main-within', label: 'The effect of the within-person factor', note: 'Averaged over both groups. For example: do scores change from before to after, whichever group people are in?' },
@@ -552,13 +559,16 @@ export default function SampleSize() {
);
} else if (step === 'rho') {
question = 'How strongly do a person\'s scores in different conditions go together?';
- help = 'People who score high in one condition usually score high in the others too. The stronger that correlation, the fewer people you need, because every person is compared with themselves. Take it from a pilot or an earlier study with the same measure if you can. If people do each condition several times (repeated trials), plan on their average per condition: averages over many trials are more reliable, so they usually correlate more strongly.';
+ const between = answers.which === 'main-between';
+ help = between
+ ? 'People who score high in one condition usually score high in the others too. For the difference between the groups this works the other way round: each person\'s scores are averaged over the conditions, and the stronger they go together, the less that averaging helps. So here a stronger correlation means you need more people. Take it from a pilot or an earlier study with the same measure if you can.'
+ : 'People who score high in one condition usually score high in the others too. The stronger that correlation, the fewer people you need, because every person is compared with themselves. Take it from a pilot or an earlier study with the same measure if you can. If people do each condition several times (repeated trials), plan on their average per condition: averages over many trials are more reliable, so they usually correlate more strongly.';
body = (
answer({ rho: value, text: { ...answers.text, rho: String(value) } })}
@@ -750,7 +760,7 @@ export default function SampleSize() {
With several stimuli per condition, this effect is optimistic: it treats your {answers.text.nstim} stimuli as the only ones that matter. See Your stimuli below.
)}
- {design.id !== 'proportions' &&
@@ -780,8 +790,8 @@ export default function SampleSize() {
<>
Your stimuli
- The formula above averages each person's scores over the stimuli in a condition. That ignores that some stimuli are easier or
- react more strongly to a condition than others. When stimuli really differ, a test on those averages finds effects too often,
+ The formula above averages each person's scores over the stimuli in a condition. That ignores that stimuli differ. When they
+ differ in how the conditions affect them (or, with different stimuli in each condition, in their average score), a test on those averages finds effects too often,
and the true power is lower than the page says (Judd, Westfall & Kenny, 2012).
diff --git a/src/pages/sampleSizeDesigns.ts b/src/pages/sampleSizeDesigns.ts
index 73cc96b..86020a4 100644
--- a/src/pages/sampleSizeDesigns.ts
+++ b/src/pages/sampleSizeDesigns.ts
@@ -178,7 +178,7 @@ export const DESIGNS: Design[] = [
effectMeaning: 'the share of "yes" you expect in each group.',
defaultEffect: 0.45,
assumes:
- 'Two groups of equal size and the normal approximation that base R\'s power.prop.test() uses. prop.test() applies a continuity correction by default, which costs a little power; with small expected counts (under about 5 in a cell), plan by simulation instead.',
+ 'Two groups of equal size and the normal approximation that base R\'s power.prop.test() uses. prop.test() and chisq.test() apply a continuity correction by default, which costs about 3 to 5 percentage points of power: use correct = FALSE, or add about 10 to 15% more people; with small expected counts (under about 5 in a cell), plan by simulation instead.',
},
{
id: 'chi-square',
@@ -329,13 +329,24 @@ export function effectInputs(id: DesignId, route: 'matter' | 'research', factors
}
}
+/**
+ * Two decimals, or three significant digits when two decimals would be more
+ * than 0.5% off (f = 0.0625 would otherwise show as 0.06). The page plans with
+ * the value as shown, so the report and the R code give the same number.
+ */
+export function shortEffect(value: number): string {
+ const two = value.toFixed(2);
+ return Math.abs(Number(two) - value) <= 0.005 * Math.abs(value) ? two : String(Number(value.toPrecision(3)));
+}
+
/** Effect sizes in plain words, for a student with no numbers to go on. */
export type Preset = { value: number; label: string; note: string; recommended?: boolean };
export function presets(design: Design, factors?: { layout: Layout; which: Which; rho?: number }): Preset[] {
if (design.id === 'factorial' && factors) return factorialPresets(factors.layout, factors.which, factors.rho);
const [s, m, l] = design.benchmarks ?? [0, 0, 0];
- const d = design.symbol === 'd' || design.symbol === 'dz';
+ // Cohen's height examples compare separate groups, so they fit d, not dz.
+ const d = design.symbol === 'd';
const cohen: Preset[] = [
{ value: s, label: `Small: subtle (${design.symbol} = ${fmt(design, s)})`, note: `Real, but hard to notice without measuring many people.${d ? ' Cohen\'s example: the height difference between 15- and 16-year-old girls.' : ''}` },
{ value: m, label: `Medium: noticeable (${design.symbol} = ${fmt(design, m)})`, note: `Visible to a careful observer.${d ? ' Cohen\'s example: the height difference between 14- and 18-year-old girls.' : ''} Common in student projects, but real effects are often smaller.` },
@@ -352,8 +363,8 @@ export function presets(design: Design, factors?: { layout: Layout; which: Which
/** 2 × 2 effects in plain words: a main effect as Cohen's d, an interaction as the shape of the two simple effects. */
function factorialPresets(layout: Layout, which: Which, rho = 0.5): Preset[] {
- const f = (delta: number) => factorialF(layout, which, delta, rho);
- const show = (delta: number) => `f = ${f(delta).toFixed(2)}`;
+ const f = (delta: number) => Number(shortEffect(factorialF(layout, which, delta, rho)));
+ const show = (delta: number) => `f = ${shortEffect(f(delta))}`;
if (which === 'interaction') {
return [
{ value: f(0.5), label: `The effect disappears (${show(0.5)})`, note: 'One factor has a medium effect (d = 0.50) at one level of the other factor and none at the other level. For example: the training helps, the waiting list does not.', recommended: true },
diff --git a/src/stats/power.itest.ts b/src/stats/power.itest.ts
index c5e3842..350cdd0 100644
--- a/src/stats/power.itest.ts
+++ b/src/stats/power.itest.ts
@@ -76,7 +76,7 @@ describe('the R code on the sample size page gives the page\'s answer', () => {
test('the stimulus simulation runs and returns a power', async () => {
await ensurePackages(webR, ['lmerTest']);
- const code = stimulusSimCode(12, 3, 3, 0.4, 3);
+ const code = stimulusSimCode(12, 3, 3, 0.4, 0.05, 3);
const power = await webR.evalRNumber(`local({\n${code}\n})`);
expect(power).toBeGreaterThanOrEqual(0);
expect(power).toBeLessThanOrEqual(1);
diff --git a/src/stats/power.ts b/src/stats/power.ts
index a89dfe7..ffd79ae 100644
--- a/src/stats/power.ts
+++ b/src/stats/power.ts
@@ -317,22 +317,25 @@ export function rCode(plan: Plan, solveFor: 'n' | 'effect', target: number, n?:
* condition, where no formula applies. The standard deviations are
* placeholders the student replaces with estimates from a pilot.
*/
-export function stimulusSimCode(nPeople: number, k: number, nStim: number, f: number, nsim = 100): string {
+export function stimulusSimCode(nPeople: number, k: number, nStim: number, f: number, alpha = 0.05, nsim = 100): string {
return `# Power with several stimuli per condition, by simulation (lmerTest).
# The standard deviations below are placeholders. Replace them with
# estimates from a pilot or an earlier study: fit the same model to those
-# data and read them from VarCorr(). Everything is in single-trial units.
+# data and read them from VarCorr(). They are in single-trial units.
library(lmerTest)
n_people <- ${nPeople}; k <- ${k}; n_stim <- ${nStim} # people, conditions, stimuli per condition
same_stimuli <- TRUE # FALSE if each condition has its own stimuli
-f <- ${num(f)}
-means <- seq(-1, 1, length.out = k)
-means <- f * means / sqrt(mean(means^2)) # or type your own condition means
+alpha <- ${num(alpha)}
+f <- ${num(f)} # in SDs of people's average score in one condition, as on the page
sd_person <- 0.6 # people differ in their average score
sd_person_cond <- 0.3 # people differ in how the conditions affect them
sd_stim <- 0.3 # stimuli differ in their average score
sd_stim_cond <- 0.2 # stimuli differ in how the conditions affect them
sd_noise <- 0.7 # everything else, from trial to trial
+sd_average <- sqrt(sd_person^2 + sd_person_cond^2 + sd_noise^2 / n_stim) # SD of people's averages
+sd_person^2 / sd_average^2 # the correlation between conditions these imply; compare with the page's r
+means <- seq(-1, 1, length.out = k)
+means <- f * sd_average * means / sqrt(mean(means^2)) # or type your own condition means
nsim <- ${nsim} # use 1000 in desktop R for a precise answer
one_study <- function() {
@@ -357,5 +360,5 @@ one_study <- function() {
}
p <- replicate(nsim, one_study())
-mean(p < 0.05, na.rm = TRUE) # the estimated power`;
+mean(p < alpha, na.rm = TRUE) # the estimated power`;
}