diff --git a/src/content/exercises/module-03.ts b/src/content/exercises/module-03.ts
index 6ea049d..2334023 100644
--- a/src/content/exercises/module-03.ts
+++ b/src/content/exercises/module-03.ts
@@ -394,7 +394,7 @@ export const module03: ExerciseDef[] = [
{
id: 'm3-4-b',
prompt:
- 'How consistently do the five items measure one thing? Compute Cronbach\'s alpha for q1, q2, q3_r, q4 and q5 and store it in alpha. The formula is in the starter code.',
+ 'How consistently do the five items agree with each other? Compute Cronbach\'s alpha for q1, q2, q3_r, q4 and q5 and store it in alpha. The formula is in the starter code.',
starterCode:
'# survey, with q3_r already added, is in your environment.\nitems <- survey[, c("q1", "q2", "q3_r", "q4", "q5")]\nk <- ncol(items)\n\n# alpha = k / (k - 1) * (1 - sum of the item variances / variance of the total)\nalpha <- ',
setupCode: REVERSED,
@@ -434,7 +434,7 @@ export const module03: ExerciseDef[] = [
} else if (abs(got - target) > 1e-6) {
list(pass = FALSE, message = paste0("alpha is ", round(got, 3), ", but it should be ", round(target, 3), ". The formula uses variances, var(), both for the items and for the total, rowSums()."))
} else {
- list(pass = TRUE, message = paste0("Correct: alpha = ", format(round(target, 2), nsmall = 2), ". Acceptable, and the lesson shows which item is holding it back."))
+ list(pass = TRUE, message = paste0("Correct: alpha = ", sub("^0", "", format(round(target, 2), nsmall = 2)), ". Acceptable, and the lesson shows which item is holding it back."))
}
}
`,
diff --git a/src/content/exercises/module-10.ts b/src/content/exercises/module-10.ts
index b53da99..93cc4a0 100644
--- a/src/content/exercises/module-10.ts
+++ b/src/content/exercises/module-10.ts
@@ -49,7 +49,7 @@ export const module10: ExerciseDef[] = [
} else if (!isTRUE(all.equal(r_w, exp_w, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("r_workload is ", round(r_w, 3), ", but cor(d$workload, d$wellbeing) is ", round(exp_w, 3), ". Check that you correlated workload with wellbeing, and not with autonomy."))
} else {
- list(pass = TRUE, message = paste0("Autonomy r = ", sub("0.", ".", round(exp_a, 3), fixed = TRUE), "; workload r = ", sub("0.", ".", round(exp_w, 3), fixed = TRUE), ". Same outcome, opposite directions. A correlation is only ever a description of these 480 employees - it is not evidence that giving someone autonomy would raise their wellbeing."))
+ list(pass = TRUE, message = paste0("Autonomy r = ", sub("0.", ".", format(round(exp_a, 2), nsmall = 2), fixed = TRUE), "; workload r = ", sub("0.", ".", format(round(exp_w, 2), nsmall = 2), fixed = TRUE), ". Same outcome, opposite directions. A correlation is only ever a description of these 480 employees - it is not evidence that giving someone autonomy would raise their wellbeing."))
}
}
`,
diff --git a/src/content/exercises/module-11.ts b/src/content/exercises/module-11.ts
index 59b15aa..d1f579d 100644
--- a/src/content/exercises/module-11.ts
+++ b/src/content/exercises/module-11.ts
@@ -42,7 +42,7 @@ export const module11: ExerciseDef[] = [
alone <- as.vector(coef(lm(wellbeing ~ workload, data = d))["workload"])
terms_in <- names(coef(model2))
if (!identical(sort(terms_in), sort(c("(Intercept)", "autonomy", "workload")))) {
- list(pass = FALSE, message = paste0("model2 should have exactly two predictors, autonomy and workload. Yours has: ", paste(setdiff(terms_in, "(Intercept)"), collapse = ", "), ". A star between predictors adds an interaction term as well; a plus sign is what you want here."))
+ list(pass = FALSE, message = paste0("model2 should have exactly two predictors, autonomy and workload. Yours has: ", paste(setdiff(terms_in, "(Intercept)"), collapse = ", "), if (any(grepl(":", terms_in))) ". A star between predictors adds an interaction term as well; a plus sign is what you want here." else ". Use exactly those two, joined with a plus sign."))
} else if (!is.numeric(b) || length(b) != 1L) {
list(pass = FALSE, message = "b_workload should be a single number: filter tidy() to the workload row and pull(estimate).")
} else if (isTRUE(all.equal(b, exp_autonomy, tolerance = 1e-6, check.attributes = FALSE))) {
@@ -248,7 +248,7 @@ export const module11: ExerciseDef[] = [
} else if (!isTRUE(all.equal(df_resid, exp_den, tolerance = 1e-9, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("df_resid is ", df_resid, " but should be ", exp_den, "."))
} else {
- list(pass = TRUE, message = paste0("F(", exp_num, ", ", exp_den, ") = ", round(exp_f, 2), ", R-squared = ", sub("0.", ".", round(exp_r2, 3), fixed = TRUE), ". Those are the three numbers the first sentence of an APA regression report needs; the coefficients go in the sentences after it."))
+ list(pass = TRUE, message = paste0("F(", exp_num, ", ", exp_den, ") = ", round(exp_f, 2), ", R-squared = ", sub("0.", ".", round(exp_r2, 3), fixed = TRUE), ". Together with the model's p-value, those numbers make the first sentence of an APA regression report; the coefficients go in the sentences after it."))
}
}
`,
diff --git a/src/content/exercises/module-12.ts b/src/content/exercises/module-12.ts
index 49a2db9..1cce258 100644
--- a/src/content/exercises/module-12.ts
+++ b/src/content/exercises/module-12.ts
@@ -180,11 +180,11 @@ export const module12: ExerciseDef[] = [
} else if (isTRUE(all.equal(b, intercept, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("That is the intercept, ", round(intercept, 2), ". With one factor and nothing else in the model the intercept is the mean of the reference department (", exp_ref, "), not of Support and not of everybody."))
} else if (isTRUE(all.equal(b, as.vector(means[["Support"]]), tolerance = 1e-6, check.attributes = FALSE))) {
- list(pass = FALSE, message = paste0("That is Support's own mean (", round(means[["Support"]], 2), "). The coefficient is a difference: Support's mean minus the reference department's, which is ", round(exp_b, 2), "."))
+ list(pass = FALSE, message = paste0("That is Support's own mean (", round(means[["Support"]], 2), "). The coefficient is a difference: Support's mean minus the reference department's, which is ", sprintf("%.2f", exp_b), "."))
} else if (!isTRUE(all.equal(b, exp_b, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("b_support is ", round(b, 4), " but the departmentSupport coefficient is ", round(exp_b, 4), "."))
} else {
- list(pass = TRUE, message = paste0("The reference is ", exp_ref, ", whose mean is the intercept, ", round(intercept, 2), ". b = ", round(exp_b, 2), " for Support means its mean is ", round(means[["Support"]], 2), ". Every coefficient in this model is a comparison with ", exp_ref, " - so none of them compares Marketing with Sales, which is what Lesson 12-3 is for."))
+ list(pass = TRUE, message = paste0("The reference is ", exp_ref, ", whose mean is the intercept, ", round(intercept, 2), ". b = ", sprintf("%.2f", exp_b), " for Support means its mean is ", round(means[["Support"]], 2), ". Every coefficient in this model is a comparison with ", exp_ref, " - so none of them compares Marketing with Sales, which is what Lesson 12-3 is for."))
}
}
`,
diff --git a/src/content/exercises/module-13.ts b/src/content/exercises/module-13.ts
index ec8761d..f11941c 100644
--- a/src/content/exercises/module-13.ts
+++ b/src/content/exercises/module-13.ts
@@ -115,7 +115,7 @@ export const module13: ExerciseDef[] = [
isTRUE(all.equal(sort(as.vector(value)), sort(as.vector(cells)), tolerance = 1e-6, check.attributes = FALSE))) found <- TRUE
}
if (!found) {
- list(pass = FALSE, message = paste0("No column of cell_means holds the four mean change scores, which are ", paste(round(sort(as.vector(cells)), 2), collapse = ", "), ". Check that you summarised change (engagement_t2 minus engagement_t1) rather than engagement itself."))
+ list(pass = FALSE, message = paste0("No column of cell_means holds the four mean change scores, which are ", paste(sprintf("%.2f", sort(as.vector(cells))), collapse = ", "), ". Check that you summarised change (engagement_t2 minus engagement_t1) rather than engagement itself."))
} else if (!is.numeric(boost) || length(boost) != 1L) {
list(pass = FALSE, message = "boost should be a single number.")
} else if (isTRUE(all.equal(boost, corner, tolerance = 1e-6, check.attributes = FALSE))) {
@@ -189,7 +189,7 @@ export const module13: ExerciseDef[] = [
} else if (isTRUE(all.equal(f_training, exp_int, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("That is the interaction's F (", round(exp_int, 2), "), on the training:mentoring row. The training main effect is on the row called training."))
} else if (isTRUE(all.equal(f_training, as.vector(default3["training", "F value"]), tolerance = 1e-6, check.attributes = FALSE))) {
- list(pass = FALSE, message = paste0("Mentoring is still on R's default treatment contrasts, so that F (", round(as.vector(default3["training", "F value"]), 2), ") is not the main effect of training. With mentoring treatment-coded, a type III test of training asks about training among employees with NO mentoring only: a simple effect wearing a main effect's name. With contrasts = list(training = contr.sum, mentoring = contr.sum) the same row becomes ", round(exp_f, 2), "."))
+ list(pass = FALSE, message = paste0("Mentoring is on R's default treatment contrasts, so that F (", round(as.vector(default3["training", "F value"]), 2), ") is not the main effect of training. With mentoring treatment-coded, a type III test of training asks about training among employees with NO mentoring only: a simple effect wearing a main effect's name. With contrasts = list(training = contr.sum, mentoring = contr.sum) the same row becomes ", round(exp_f, 2), "."))
} else if (isTRUE(all.equal(f_training, as.vector(type2["training", "F value"]), tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("That is the type II F (", round(as.vector(type2["training", "F value"]), 2), "). Type II tests each main effect while ignoring the interaction, which is only defensible when the interaction is negligible. This exercise asks for type III."))
} else if (!isTRUE(all.equal(f_training, exp_f, tolerance = 1e-6, check.attributes = FALSE))) {
@@ -242,12 +242,15 @@ export const module13: ExerciseDef[] = [
exp_no <- as.vector(cells["Yes", "No"] - cells["No", "No"])
exp_yes <- as.vector(cells["Yes", "Yes"] - cells["No", "Yes"])
overall <- mean(d$change[d$training == "Yes"]) - mean(d$change[d$training == "No"])
+ t2cells <- tapply(d$engagement_t2, list(d$training, d$mentoring), mean)
if (!is.numeric(no_m) || length(no_m) != 1L || !is.numeric(yes_m) || length(yes_m) != 1L) {
list(pass = FALSE, message = "Both should be single numbers.")
} else if (isTRUE(all.equal(no_m, yes_m, tolerance = 1e-12, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("Your two simple effects are identical, so you have reported the overall training effect (", round(overall, 2), ") twice. The whole point of a simple effect is that it differs between the levels of the other factor - here they are ", round(exp_no, 2), " and ", round(exp_yes, 2), "."))
} else if (isTRUE(all.equal(no_m, -exp_no, tolerance = 1e-6, check.attributes = FALSE)) && isTRUE(all.equal(yes_m, -exp_yes, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = "Both of your differences run backwards. Each simple effect is the training-Yes cell minus the training-No cell within that level of mentoring.")
+ } else if (isTRUE(all.equal(no_m, as.vector(t2cells["Yes", "No"] - t2cells["No", "No"]), tolerance = 1e-6, check.attributes = FALSE)) && isTRUE(all.equal(yes_m, as.vector(t2cells["Yes", "Yes"] - t2cells["No", "Yes"]), tolerance = 1e-6, check.attributes = FALSE))) {
+ list(pass = FALSE, message = "Those are differences in engagement at time 2, not in the change score. Compute change = engagement_t2 - engagement_t1 first, then take the differences.")
} else if (!isTRUE(all.equal(no_m, exp_no, tolerance = 1e-6, check.attributes = FALSE))) {
list(pass = FALSE, message = paste0("effect_no_mentoring is ", round(no_m, 4), " but the training effect among employees without mentoring is ", round(exp_no, 4), ". Hold mentoring at No and take the difference across training."))
} else if (!isTRUE(all.equal(yes_m, exp_yes, tolerance = 1e-6, check.attributes = FALSE))) {
diff --git a/src/content/exercises/module-15.ts b/src/content/exercises/module-15.ts
index d1eb039..3484dad 100644
--- a/src/content/exercises/module-15.ts
+++ b/src/content/exercises/module-15.ts
@@ -75,7 +75,7 @@ export const module15: ExerciseDef[] = [
}
}
if (!found) {
- list(pass = FALSE, message = paste0("No column of rate_by_third holds the three leaving rates, which are ", paste(round(exp_rates, 3), collapse = ", "), " from the lowest third of wellbeing to the highest. Split on wellbeing, not on left_company."))
+ list(pass = FALSE, message = paste0("No column of rate_by_third holds the three leaving rates, which are ", paste(round(exp_rates_ntile, 3), collapse = ", "), " from the lowest third of wellbeing to the highest. Split on wellbeing, not on left_company."))
} else {
list(pass = TRUE, message = paste0(exp_n, " of the ", nrow(d), " fitted values are impossible probabilities. And the descriptives say the effect is real: ", round(100 * shown[1], 1), "% of the least happy third left, against ", round(100 * shown[3], 1), "% of the happiest. A model that predicts a negative probability for the very employees it should be most confident about is the wrong shape, not the wrong data."))
}
diff --git a/src/content/exercises/module-16.ts b/src/content/exercises/module-16.ts
index a794614..ac2f109 100644
--- a/src/content/exercises/module-16.ts
+++ b/src/content/exercises/module-16.ts
@@ -144,6 +144,8 @@ export const module16: ExerciseDef[] = [
list(pass = FALSE, message = "That is a 90% interval. A 95% interval leaves 2.5% in each tail: qbeta(c(0.025, 0.975), ...).")
} else if (close(ci, binom.test(k, n)$conf.int)) {
list(pass = FALSE, message = "That is the frequentist confidence interval from binom.test(). Here the interval should come from the posterior, with qbeta().")
+ } else if (close(ci, qbeta(c(0.025, 0.975), 1 + sum(d$left_company), 1 + nrow(d) - sum(d$left_company)))) {
+ list(pass = FALSE, message = "That is the interval for the whole company. Use Engineering's counts: left_eng out of n_eng.")
} else if (!close(ci, expected)) {
list(pass = FALSE, message = paste0("ci_engineering is [", paste(round(ci, 3), collapse = ", "), "]. The posterior is Beta(1 + ", k, ", 1 + ", n - k, "): leavers in the first shape, stayers in the second."))
} else {
diff --git a/src/content/exercises/module-17.ts b/src/content/exercises/module-17.ts
index 0d37f0c..194d498 100644
--- a/src/content/exercises/module-17.ts
+++ b/src/content/exercises/module-17.ts
@@ -103,7 +103,7 @@ export const module17: ExerciseDef[] = [
} else if (!identical(as.character(got$call$rotation), "promax")) {
list(pass = FALSE, message = "Two factors, good. Let them correlate with rotation = \\"promax\\": pressure and support are unlikely to be unrelated.")
} else {
- list(pass = TRUE, message = paste0("Correct. The chi-square test of fit gives p = ", format(round(got$PVAL, 2), nsmall = 2), ", so there is no evidence that two factors fall short. Print it with print(efa, cutoff = 0.3) to see which items load where."))
+ list(pass = TRUE, message = paste0("Correct. The chi-square test of fit gives p = ", sub("^0", "", format(round(got$PVAL, 2), nsmall = 2)), ", so there is no evidence that two factors fall short. Print it with print(efa, cutoff = 0.3) to see which items load where."))
}
}
`,
diff --git a/src/content/lessons/00-1-rstudio-and-projects.mdx b/src/content/lessons/00-1-rstudio-and-projects.mdx
index edf0e30..f72784d 100644
--- a/src/content/lessons/00-1-rstudio-and-projects.mdx
+++ b/src/content/lessons/00-1-rstudio-and-projects.mdx
@@ -85,9 +85,10 @@ top right of the window shows the project name, the Files pane shows the new
folder, and the folder contains a file ending in `.Rproj`.
To come back to the project later, double-click that `.Rproj` file, or pick the
-project from **File > Recent Projects**. Opening RStudio on its own and then
-opening a script does not open the project, and the working directory will be
-wrong.
+project from **File > Recent Projects**. Opening RStudio on its own reopens
+whichever project was open last, which may not be this one, and opening a script
+does not switch projects. If the top right does not show this project's name,
+the working directory will be wrong.
## A folder layout that works
diff --git a/src/content/lessons/01-2-functions-and-help.mdx b/src/content/lessons/01-2-functions-and-help.mdx
index 9a825d2..7c60067 100644
--- a/src/content/lessons/01-2-functions-and-help.mdx
+++ b/src/content/lessons/01-2-functions-and-help.mdx
@@ -86,7 +86,8 @@ inside out.
length(seq(from = 0, to = 100, by = 10))`} />
Read the first line inside out: average the scores, ignoring the missing one,
-then round to one decimal. It is compact, and once there are four or five
+then round to one decimal. R prints 64.2, not the 64.3 you may have learned:
+when a number sits exactly halfway, R rounds to the even digit. It is compact, and once there are four or five
functions stacked up it becomes unreadable. The pipe you met in Module 0 is
R's answer to exactly that problem.
diff --git a/src/content/lessons/02-2-pipe-and-verbs.mdx b/src/content/lessons/02-2-pipe-and-verbs.mdx
index 487646a..9c67cea 100644
--- a/src/content/lessons/02-2-pipe-and-verbs.mdx
+++ b/src/content/lessons/02-2-pipe-and-verbs.mdx
@@ -63,7 +63,7 @@ A minus sign drops instead of keeps: `select(-site)` returns everything except
head(4)`} />
`mutate()` works on whole columns at once, so every row gets its own result and
-the new column is as long as the table. Nothing has been saved yet: every block on this page has printed a new
+the new column is as long as the table. Nothing has been saved yet: every pipeline on this page has printed a new
table and thrown it away. Storing it takes the arrow.
diff --git a/src/content/lessons/03-4-scales-and-reliability.mdx b/src/content/lessons/03-4-scales-and-reliability.mdx
index 0d8f160..db8aaf8 100644
--- a/src/content/lessons/03-4-scales-and-reliability.mdx
+++ b/src/content/lessons/03-4-scales-and-reliability.mdx
@@ -85,7 +85,7 @@ total_variance <- var(rowSums(items))
k / (k - 1) * (1 - item_variances / total_variance)`} />
-Alpha is 0.74. Rough conventions call .70 acceptable and .80 good for research
+Alpha is .74. Rough conventions call .70 acceptable and .80 good for research
use. Alpha also rises with the number of items, so a long scale can reach .80
with items that agree only weakly.
@@ -104,7 +104,7 @@ same thing.
sapply(names(items), function(item) alpha_of(items[, names(items) != item]))`} />
-Dropping any of q1 to q4 lowers alpha. Dropping q5 raises it to 0.82, and its
+Dropping any of q1 to q4 lowers alpha. Dropping q5 raises it to .82, and its
correlations with the other items were close to zero. The coffee question does
not belong in a wellbeing scale. With a real questionnaire, you would look at
the wording before you drop anything, and you would report which items you
@@ -118,7 +118,7 @@ you when an item looks as if it still needs reversing.
question="You forget to reverse q3 and compute alpha on the original five items. What happens?"
choices={[
{ text: 'Nothing much, alpha only looks at variances', response: 'The total variance depends on how the items move together, and an unreversed item moves against the rest.' },
- { text: 'Alpha collapses towards zero, because q3 cancels the other items out', correct: true, response: 'Right. Here it drops from 0.74 to 0.02. A suspiciously low alpha is often an item nobody reversed.' },
+ { text: 'Alpha collapses towards zero, because q3 cancels the other items out', correct: true, response: 'Right. Here it drops from .74 to .02. A suspiciously low alpha is often an item nobody reversed.' },
{ text: 'Alpha goes up, because there is more variation', response: 'The items vary as much as before, but the total varies less, because q3 pulls against the others. Alpha goes down.' },
]}
/>
diff --git a/src/content/lessons/04-2-boxplots-and-facets.mdx b/src/content/lessons/04-2-boxplots-and-facets.mdx
index 3572f51..1bf4047 100644
--- a/src/content/lessons/04-2-boxplots-and-facets.mdx
+++ b/src/content/lessons/04-2-boxplots-and-facets.mdx
@@ -21,8 +21,10 @@ computed:
- the box spans the **quartiles** — the middle half of the department, so its
height is the IQR;
- the whiskers reach out to the furthest observation within 1.5 IQRs of the box
- (ggplot2 measures from the quartiles; the textbook describes the distance from
- the median, so its whiskers can end slightly differently);
+ (ggplot2 measures from the quartiles, as does the textbook's worked wage
+ example; the textbook's prose measures from the median instead, which gives
+ shorter whiskers and flags more points: here Sales would show seven points
+ below its whisker instead of none);
- anything beyond that is drawn as a point, and is worth looking at rather than
deleting.
diff --git a/src/content/lessons/05-1-density-and-area.mdx b/src/content/lessons/05-1-density-and-area.mdx
index e6ea8af..5e284b6 100644
--- a/src/content/lessons/05-1-density-and-area.mdx
+++ b/src/content/lessons/05-1-density-and-area.mdx
@@ -12,7 +12,9 @@ population <- read.csv("data/wellbeing-population.csv", stringsAsFactors = TRUE)
population %>% summarise(mu = mean(exam_score), sigma = sd(exam_score), n = n())`} />
-Two numbers: a mean of about 73.3 and a standard deviation of about 12.4. The
+Two numbers: a mean of about 73.3 and a standard deviation of about 12.4. (sd()
+divides by n − 1; the textbook's population σ divides by N and gives 12.35
+instead of 12.36, a difference that changes nothing in this module.) The
claim of this lesson is that those two numbers are almost the whole story.
At 50 students the penalty is 2.01 against 1.96 — about 2.5 per cent wider. At
-five students it is 2.78, and the interval is forty per cent wider. The smaller
+five students it is 2.78, and the interval is about 42 per cent wider. The smaller
your sample, the less you know about how variable your data are, and the more
you pay for that ignorance.
diff --git a/src/content/lessons/07-2-what-95-percent-means.mdx b/src/content/lessons/07-2-what-95-percent-means.mdx
index dde4df6..5f76d33 100644
--- a/src/content/lessons/07-2-what-95-percent-means.mdx
+++ b/src/content/lessons/07-2-what-95-percent-means.mdx
@@ -77,6 +77,10 @@ watch different ones miss.
+The 95 per cent is exact for a normal population. Pick Strongly skewed with a
+small n and you will see fewer than 95 hits on average: the t-interval relies on
+the sampling distribution being close to normal.
+
## What you may and may not say
diff --git a/src/content/lessons/08-2-p-values-and-alpha.mdx b/src/content/lessons/08-2-p-values-and-alpha.mdx
index f31827f..fb7eee1 100644
--- a/src/content/lessons/08-2-p-values-and-alpha.mdx
+++ b/src/content/lessons/08-2-p-values-and-alpha.mdx
@@ -57,7 +57,7 @@ refute:
the subject of the next lesson.
What a p-value *does* say is narrow and useful: **if nothing were going on, data
-this extreme would turn up this often.**
+at least this extreme would turn up this often.**
diff --git a/src/content/lessons/08-3-errors-and-power.mdx b/src/content/lessons/08-3-errors-and-power.mdx
index a2a82bd..19b9aa6 100644
--- a/src/content/lessons/08-3-errors-and-power.mdx
+++ b/src/content/lessons/08-3-errors-and-power.mdx
@@ -2,7 +2,7 @@ A hypothesis test makes a decision, and a decision can be wrong in two
directions. Both have names, and only one of them gets talked about.
- The null is **true** and you **reject** it — a **Type I error**. You announce
- an effect that is not there. Its long-run rate is exactly α, because that is
+ an effect that is not there. When the test's assumptions hold, its long-run rate is α, because that is
what α was defined to be.
- The null is **false** and you **fail to reject** it — a **Type II error**. You
miss a real effect. Its rate is called β.
@@ -105,7 +105,8 @@ shows the Type I side of the trade; the power curve above shows the other side.
## What to do with a non-significant result
Say what it is. "We did not detect an effect" is honest; "there is no effect" is
-not, unless the study had the power to have seen one. A non-significant result
+not. Even a well-powered study can only rule out effects larger than those it
+could detect, and its confidence interval shows which those are. A non-significant result
from an underpowered study is uninformative, and a confidence interval says so
much better than a p-value does — a wide interval that includes zero tells the
reader plainly that the study could not distinguish "nothing" from "quite a lot".
diff --git a/src/content/lessons/08-4-effect-sizes-and-planning.mdx b/src/content/lessons/08-4-effect-sizes-and-planning.mdx
index f660848..9be9b78 100644
--- a/src/content/lessons/08-4-effect-sizes-and-planning.mdx
+++ b/src/content/lessons/08-4-effect-sizes-and-planning.mdx
@@ -1,5 +1,5 @@
A p-value answers one question: how surprising would data like these be if there were no effect? It does not say how big
-the effect is, and it tends to shrink as the sample grows even when the effect stays
+the effect is, and when there is a real effect it tends to shrink as the sample grows even though the effect stays
the same. The size of an effect needs its own number, and that number is also
what you need to plan how large a study should be.
@@ -54,7 +54,7 @@ The true d is 0.50. The study found 0.64, and it was significant.
id="p-winners"
question="Imagine many small studies of this effect, and only the significant ones get published. How do the published effect sizes compare with the true d of 0.50?"
choices={[
- { text: 'They average out at 0.50, because each study is unbiased', response: 'Each study is, but the filter is not. Throwing away the studies that happened to land low leaves the ones that landed high.' },
+ { text: 'They average out at 0.50, because each study is unbiased', response: 'Each study\'s estimate is very nearly unbiased, but the filter is not. Throwing away the studies that happened to land low leaves the ones that landed high.' },
{ text: 'They are too large on average', correct: true, response: 'Right. With a small sample, only an estimate that happens to land high clears the significance bar, so the published record overstates the effect.' },
{ text: 'They are too small on average, because small studies lack power', response: 'Low power means many studies miss the effect. The ones that do not miss it are the lucky, high estimates.' },
]}
diff --git a/src/content/lessons/09-1-one-proportion.mdx b/src/content/lessons/09-1-one-proportion.mdx
index bfdf463..f77ba45 100644
--- a/src/content/lessons/09-1-one-proportion.mdx
+++ b/src/content/lessons/09-1-one-proportion.mdx
@@ -49,7 +49,7 @@ exactly on *p*. Give it the number of yes answers and the number of cases.
The two intervals agree to within half a percentage point. With a sample this size the
-correction hardly matters; with 30 people it would.
+choice of interval hardly matters; with 30 people it would.
diff --git a/src/content/lessons/10-2-fitting-a-line.mdx b/src/content/lessons/10-2-fitting-a-line.mdx
index 170a9e4..475aa3a 100644
--- a/src/content/lessons/10-2-fitting-a-line.mdx
+++ b/src/content/lessons/10-2-fitting-a-line.mdx
@@ -18,8 +18,9 @@ between what the line predicted and what they actually reported is a
Infinitely many lines are available. The one R fits is the one that makes the
**sum of the squared residuals** as small as it can be — the least-squares line.
-Squaring is what stops a line from paying for a large miss above with a large
-miss below, and it is why a single far-out point can pull the line noticeably.
+Squaring stops misses above and below the line from cancelling out, as absolute
+values would too, but it also makes one large miss cost far more than several
+small ones. That is why a single far-out point can pull the line noticeably.
% glance() %>% select(r.squared, adj.r.squared, statistic, df, df.resid
*R*² can never fall when you add a predictor, even a useless one, and in practice it always rises a little: it is the share
of variance the fitted model accounts for, and an extra column always lets the
fit wriggle a little closer. **Adjusted *R*²** subtracts a penalty for each
-predictor, so it can fall. When the two are far apart you have paid for
-predictors that bought you nothing.
+predictor, so it can fall. When the two are far apart, the model has many
+predictors for its sample size, and some of them may be buying you nothing.
diff --git a/src/content/lessons/12-3-pairwise-comparisons.mdx b/src/content/lessons/12-3-pairwise-comparisons.mdx
index 76d2fc9..1f39017 100644
--- a/src/content/lessons/12-3-pairwise-comparisons.mdx
+++ b/src/content/lessons/12-3-pairwise-comparisons.mdx
@@ -24,7 +24,8 @@ emm`} />
With department as the only predictor these match the raw group means exactly.
They usually stop matching as soon as the model contains anything else; then they are
-the predicted means with the other predictors held at their average, which is
+the predicted means with any numeric predictors held at their mean and averaged
+equally over the levels of any other factors, which is
what makes `emmeans` worth learning rather than just averaging by hand.
diff --git a/src/content/lessons/14-2-random-intercepts.mdx b/src/content/lessons/14-2-random-intercepts.mdx
index 8963ace..b52efb9 100644
--- a/src/content/lessons/14-2-random-intercepts.mdx
+++ b/src/content/lessons/14-2-random-intercepts.mdx
@@ -58,7 +58,7 @@ c(
The `employee_id` row is how much employees differ from each other; the
`Residual` row is what is left within an employee once their own level and the
-time effect are accounted for. Their ratio is the **intraclass correlation**: the
+time effect are accounted for. The employee variance divided by the sum of the two is the **intraclass correlation**: the
share of the total variance that is stable differences between people.
A high ICC shows how much is at stake. It says most of the raw
@@ -97,7 +97,7 @@ as such: *t*(479.0) for time here, or *t*(539.8) for the intercept, rounded to o
choices={[
{ text: 'A separate coefficient for every employee, reported in the output', response: 'The deviations are estimated but not reported as fixed coefficients. What is reported is their standard deviation.' },
{ text: 'One extra parameter: the standard deviation of employees\' own starting levels, which lets the model know which rows belong to the same person', correct: true, response: 'Correct - one parameter for 480 employees, which is what makes the approach practical.' },
- { text: 'An interaction between employee and time', response: 'That would be a random slope, written (1 + time | employee_id). This is an intercept only.' },
+ { text: 'An interaction between employee and time', response: 'That would be a random slope, written (1 + time | employee_id), and with only two measurements per employee it cannot be estimated. This is an intercept only.' },
{ text: 'A correction applied to the p-value after fitting', response: 'Nothing is corrected afterwards. The dependence is part of the model from the start.' },
]}
/>
diff --git a/src/content/lessons/14-3-nesting-and-paired-t.mdx b/src/content/lessons/14-3-nesting-and-paired-t.mdx
index d234706..e1d3d13 100644
--- a/src/content/lessons/14-3-nesting-and-paired-t.mdx
+++ b/src/content/lessons/14-3-nesting-and-paired-t.mdx
@@ -50,7 +50,7 @@ the same analysis wearing three names.
{ text: 'It does not; the t-test is simpler and should be preferred', response: 'It is simpler, and it stops working the moment the design grows past two measurements.' },
{ text: 'Because it keeps working with three or more measurements, with missing data, and with other predictors in the model', correct: true, response: 'Exactly. The paired t-test is one special case; the model is the general tool.' },
{ text: 'Because it gives smaller p-values', response: 'It gives the same p-value in this special case, which is the whole point of the comparison above.' },
- { text: 'Because the paired t-test requires normally distributed data and the model does not', response: 'They make the same distributional assumptions - they are the same model.' },
+ { text: 'Because the paired t-test requires normally distributed data and the model does not', response: 'For the effect of time they give the same test, and both rely on normality: the paired t-test on the differences, the mixed model on its random intercepts and residuals. Neither escapes the assumption.' },
]}
/>
diff --git a/src/content/lessons/15-3-odds-ratios-and-reporting.mdx b/src/content/lessons/15-3-odds-ratios-and-reporting.mdx
index cd596db..bb611aa 100644
--- a/src/content/lessons/15-3-odds-ratios-and-reporting.mdx
+++ b/src/content/lessons/15-3-odds-ratios-and-reporting.mdx
@@ -46,7 +46,7 @@ message, not a warning, and the intervals are the better ones.
{ text: 'Yes, because the interval does not contain 1', correct: true, response: 'Right. On the odds-ratio scale, 1 means no effect - the value 0 has been transformed away.' },
{ text: 'No, because the interval contains values below 1', response: 'Every value in the interval is below 1, which is what a protective effect looks like.' },
{ text: 'Yes, because the interval does not contain 0', response: 'True but irrelevant: an odds ratio can never be 0, so that test would call everything significant.' },
- { text: 'Impossible to tell without the p-value', response: 'A 95% interval and a test at .05 carry the same information. The interval also tells you the plausible sizes.' },
+ { text: 'Impossible to tell without the p-value', response: 'A 95% interval and a two-sided test at .05 carry the same information when both come from the same method; a profile interval and the Wald z can disagree in borderline cases. Here the interval is far from 1, so both agree, and it also tells you the plausible sizes.' },
]}
/>
diff --git a/src/content/lessons/15-4-poisson-regression.mdx b/src/content/lessons/15-4-poisson-regression.mdx
index 29321fc..b3fb164 100644
--- a/src/content/lessons/15-4-poisson-regression.mdx
+++ b/src/content/lessons/15-4-poisson-regression.mdx
@@ -22,7 +22,9 @@ d %>% summarise(mean_sick = mean(sick_days), var_sick = var(sick_days), min = mi
Poisson regression models the **logarithm** of the expected count as a straight
line. Transforming back, each coefficient multiplies the expected count instead of
-adding to it.
+adding to it. The book describes the same model as having an exponential link
+function, naming the transformation back from the linear part; R and most texts
+call it the log link.
The peak sits at 0.15, the observed proportion. The posterior mean is a little
higher, about 0.18, because the curve has a long right tail: with only twenty
-people, rates of 30% or more are still quite believable.
+people, rates of 30% or more still carry almost 9% of the posterior.
## More data, narrower posterior
diff --git a/src/content/lessons/17-1-mediation.mdx b/src/content/lessons/17-1-mediation.mdx
index 7914d6d..4fbd282 100644
--- a/src/content/lessons/17-1-mediation.mdx
+++ b/src/content/lessons/17-1-mediation.mdx
@@ -43,7 +43,7 @@ engagement.
id="p-mediation-test"
question="How should you decide whether the indirect effect of 1.78 is more than chance?"
choices={[
- { text: 'Check that a and b are both significant', response: 'That is part of the old Baron and Kenny steps. It never estimates the indirect effect or gives it an interval, so a reader cannot judge how large it is.' },
+ { text: 'Check that a and b are both significant', response: 'Checking that a and b are both significant (the joint significance test) is a reasonable test, but it never estimates the indirect effect or gives it an interval, so a reader cannot judge how large it is.' },
{ text: 'Divide it by its standard error and compare with a normal distribution (the Sobel test)', response: 'The Sobel test assumes a × b is normally distributed, and a product of two estimates usually is not. It is no longer recommended.' },
{ text: 'Bootstrap a confidence interval for a × b', correct: true, response: 'Right. Resampling makes no assumption about the shape of the product, and it is the current standard.' },
]}
@@ -73,11 +73,15 @@ supported.
Mediation is a causal story, and the statistics cannot make it true.
- **Order in time.** The predictor should come before the mediator, and the
- mediator before the outcome. Engagement here was measured after training,
- which helps. Performance was not measured later still.
+ mediator before the outcome. Engagement here was measured at the end of the
+ year, which helps, but performance was not measured later still. Order alone
+ is not enough: trained employees were already about 1.8 points more engaged at
+ the start of the year (engagement_t1), so part of the a path existed before
+ training.
- **No hidden common cause.** If some third thing, such as a good manager, raised
- both engagement and performance, the b path would be inflated, and randomising
- training does not protect against that.
+ both engagement and performance, the b path would be inflated; if it raised
+ both training uptake and engagement, the a path would be. Randomising training
+ protects the a path but not the b path.
- **Consistent with, not proof of.** With data from one study, write that the
pattern is consistent with mediation.
@@ -123,6 +127,6 @@ macro for SPSS, model 4, fits the same model.
{ text: 'Training was associated with higher engagement (a = 4.07), which in turn was associated with higher performance (b = 0.44). The indirect effect was 1.78, 95% bootstrap CI [1.19, 2.40], based on 1,000 resamples; the direct effect was small and non-significant (c′ = -0.16, p = .82). The pattern is consistent with engagement mediating the effect of training.', correct: true, response: 'Correct: every path, the indirect effect with its interval and how it was computed, and a causal claim no stronger than the design allows.' },
{ text: 'Training caused higher performance through engagement (Sobel test, p < .001).', response: 'The Sobel test is no longer recommended, and "caused" overstates what these data can show.' },
{ text: 'Because the direct effect was not significant, training does not affect performance.', response: 'The total effect of training is 1.62 and significant. The point of mediation is that it runs through engagement.' },
- { text: 'Both a and b were significant, so there was mediation.', response: 'That is the old steps approach. Report and test the indirect effect itself, with its interval.' },
+ { text: 'Both a and b were significant, so there was mediation.', response: 'Both significant paths are consistent with mediation, but this reports no indirect effect and no interval. Report the indirect effect itself, with its bootstrap interval.' },
]}
/>
diff --git a/src/content/lessons/17-2-factor-analysis.mdx b/src/content/lessons/17-2-factor-analysis.mdx
index 1c00678..96eb1f4 100644
--- a/src/content/lessons/17-2-factor-analysis.mdx
+++ b/src/content/lessons/17-2-factor-analysis.mdx
@@ -89,7 +89,8 @@ traits a questionnaire measures, use factors.
An exploratory analysis finds a structure in one dataset. To test that
structure, specify it in advance and fit a **confirmatory factor analysis** to
-new data. In RStudio, with the `lavaan` package:
+new data. In RStudio, with the `lavaan` package (here `new_items` stands for a
+fresh sample with the same six items):
```r
library(lavaan)
@@ -97,7 +98,7 @@ model <- "
pressure =~ deadlines + overtime + too_much
support =~ help + listened + feedback
"
-fit <- cfa(model, data = items)
+fit <- cfa(model, data = new_items)
summary(fit, fit.measures = TRUE, standardized = TRUE)
```
diff --git a/src/pages/sampleSizeDesigns.ts b/src/pages/sampleSizeDesigns.ts
index 86020a4..431e214 100644
--- a/src/pages/sampleSizeDesigns.ts
+++ b/src/pages/sampleSizeDesigns.ts
@@ -178,7 +178,7 @@ export const DESIGNS: Design[] = [
effectMeaning: 'the share of "yes" you expect in each group.',
defaultEffect: 0.45,
assumes:
- 'Two groups of equal size and the normal approximation that base R\'s power.prop.test() uses. prop.test() and chisq.test() apply a continuity correction by default, which costs about 3 to 5 percentage points of power: use correct = FALSE, or add about 10 to 15% more people; with small expected counts (under about 5 in a cell), plan by simulation instead.',
+ 'Two groups of equal size and the normal approximation that base R\'s power.prop.test() uses. prop.test() and chisq.test() apply a continuity correction by default, which costs about 2 to 6 percentage points of power, most with small groups: use correct = FALSE, or add about 10 to 20% more people; with small expected counts (under about 5 in a cell), plan by simulation instead.',
},
{
id: 'chi-square',
@@ -356,7 +356,7 @@ export function presets(design: Design, factors?: { layout: Layout; which: Which
return [{ value: 0.36, label: 'Typical (d = 0.36)', note: 'Between small and medium: the median effect found in social psychology (Lovakov & Agadullina, 2021). An honest default when you have nothing else.', recommended: true }, ...cohen];
}
if (design.id === 'correlation') {
- return [{ value: 0.2, label: 'Typical (r = .20)', note: 'Between small and medium: a typical correlation in psychology (Gignac & Szodorai, 2016). An honest default when you have nothing else.', recommended: true }, ...cohen];
+ return [{ value: 0.2, label: 'Typical (r = .20)', note: 'Between small and medium: a typical correlation in individual-differences research (Gignac & Szodorai, 2016). An honest default when you have nothing else.', recommended: true }, ...cohen];
}
return cohen;
}