From 56d548abcd4eaef0d7a5d4911d509f6c0dd37e3c Mon Sep 17 00:00:00 2001 From: Simon Parten Date: Wed, 30 Sep 2026 11:48:52 +0200 Subject: [PATCH 1/3] Add typed join, joinOn, leftJoin and leftJoinOn Joining is the one relational operation that cannot be delegated to the standard library: stdlib can group and fold rows, but it has no way to compute the *type* of two named tuples concatenated on a key. Until now the only cross-table join available was via the scalasql integration, i.e. only if the data was already in a database. The key is written after the right hand table - `orders.join(customers)["custId"]` - because clause interleaving requires a term clause between two type parameter lists, and an extension's receiver does not count. This is the same shape stdlib's own `NamedTuple.++` uses. The left side streams; the right is read into a hash index, lazily, so building a join drains neither side. A column name appearing on both sides is a compile error naming the offender, rather than a silently duplicated column. `leftJoin` optionalises the right hand columns, and does so idempotently - an already-optional column does not nest - with `optionalise` as the runtime counterpart of `Optional`. Only inner and left joins are built in; a right join is a left join with the tables swapped. Composite keys are not supported. Co-Authored-By: Claude Opus 5 (1M context) --- scautable/src/columnExtensions.scala | 259 +++++++++++++++++++++++++++ scautable/src/columnType.scala | 24 +++ scautable/test/src/JoinSuite.scala | 121 +++++++++++++ scautable/test/src/typedy.scala | 41 +++++ site/docs/cheatsheet.md | 15 ++ site/docs/csv.md | 70 ++++++++ 6 files changed, 530 insertions(+) create mode 100644 scautable/test/src/JoinSuite.scala diff --git a/scautable/src/columnExtensions.scala b/scautable/src/columnExtensions.scala index 79fca7196..e6d8f336f 100644 --- a/scautable/src/columnExtensions.scala +++ b/scautable/src/columnExtensions.scala @@ -62,6 +62,91 @@ object NamedTupleIteratorExtensions: Tuple.fromArray(cells) end applyPlan + /** Fail compilation if any name appears on both sides of a join. `Who` is the calling method, as in [[checkSpecNames]]. + * + * `Tuple.Disjoint[Left, Right] =:= true` would also reject these, but it cannot say *which* column is at fault, and that is the whole of the message's value. + */ + private inline def checkNoSharedNames[Left <: Tuple, Right <: Tuple, Who <: String]: Unit = + inline erasedValue[Right] match + case _: EmptyTuple => () + case _: (n *: rest) => + inline erasedValue[NameIn[Left, n]] match + case _: true => error(constValue[Who] + ": column " + constValue[n & String] + " exists on both sides - rename or drop it first") + case _ => checkNoSharedNames[Left, rest, Who] + + /** Drop the cell at `idx`. The `EmptyTuple` case is not redundant - it is what a right hand table consisting of nothing but the join key hits. */ + private def removeAt(t: Tuple, idx: Int): Tuple = + val (head, tail) = t.splitAt(idx) + head match + case _: EmptyTuple => tail.tail + case _ => head ++ tail.tail + end match + end removeAt + + /** Runtime counterpart of [[ColumnTyped.Optional]]: wrap each cell in `Some` unless it already is an `Option`. + * + * The two must agree. Wrapping unconditionally would hand back `Some(None)` where the static type says `None`. + */ + private def optionalise(t: Tuple): Tuple = + Tuple.fromArray(t.toArray.map { + case o: Option[?] => o + case x => Some(x) + }) + + /** Hash join. The left side streams; the right is materialised, but not until the result is first pulled - so building a join does not drain its argument. + * + * `rightArity` is the number of right hand columns *after* the key has been dropped, and is only used to shape the all-`None` row a left join emits for a left row that matched + * nothing. + */ + private def hashJoin( + left: Iterator[Tuple], + lIdx: Int, + right: => IterableOnce[Tuple], + rIdx: Int, + leftOuter: Boolean, + rightArity: Int + ): Iterator[Tuple] = + // groupMap keeps right hand rows in encounter order within a key, so the output order is deterministic. + // `Map` alone would resolve to `NamedTuple.Map`, courtesy of the wildcard import at the top of this file. + lazy val index: scala.collection.immutable.Map[Any, Seq[Tuple]] = + right.iterator + .map { t => + val rest = removeAt(t, rIdx) + t.productElement(rIdx) -> (if leftOuter then optionalise(rest) else rest) + } + .toSeq + .groupMap(_._1)(_._2) + + lazy val nones: Tuple = Tuple.fromArray(Array.fill[Object](rightArity)(None)) + + left.flatMap { lt => + val matches = index.getOrElse(lt.productElement(lIdx), Nil) + if matches.nonEmpty then matches.iterator.map(rt => lt ++ rt) + else if leftOuter then Iterator.single(lt ++ nones) + else Iterator.empty + end if + } + end hashJoin + + /** The shared body of every join: resolve both key positions, join, and re-attach the output names. Callers own the compile time checks, so that errors name *their* method. */ + private inline def joinCore[K <: Tuple, V <: Tuple, K2 <: Tuple, V2 <: Tuple, OutK <: Tuple, OutV <: Tuple]( + itr: Iterator[NamedTuple[K, V]], + that: IterableOnce[NamedTuple[K2, V2]], + leftKey: String, + rightKey: String, + leftOuter: Boolean + ): Iterator[NamedTuple[OutK, OutV]] = + val rightHeaders = constValueTuple[K2].toList.map(_.toString()) + hashJoin( + itr.map(_.toTuple), + constValueTuple[K].toList.map(_.toString()).indexOf(leftKey), + that.iterator.map(_.toTuple), + rightHeaders.indexOf(rightKey), + leftOuter, + rightHeaders.size - 1 + ).map(_.withNames[OutK].asInstanceOf[NamedTuple[OutK, OutV]]) + end joinCore + extension [K <: Tuple, V <: Tuple](itr: Iterator[NamedTuple[K, V]]) def sample(frac: Double, deterministic: Boolean = false): Iterator[NamedTuple[K, V]] = @@ -252,6 +337,84 @@ object NamedTupleIteratorExtensions: end match } end dropColumn + + /** Inner join on a column of the same name in both tables. + * + * The key is given *after* the right hand table, because the compiler infers that table's shape from the argument and only the key needs writing out: + * {{{ + * orders.join(customers)["custId"] + * }}} + * The output is every column of the left table, then every column of the right except the key. The left side streams; the right is read into a hash index the first time the + * result is pulled, so the right table is the one that has to fit in memory. + * + * A column name on both sides is a compile error - rename or drop it first. Keys are matched with `==`, so an `Option` key column matches `None` to `None`. + */ + inline def join[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("join: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("join: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("join: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "join"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]](itr, that, key.value, key.value, false) + end join + + /** Inner join where the key is named differently on each side. See [[join]] for everything else. + * {{{ + * orders.joinOn(customers)["customer_id", "id"] + * }}} + * The output keeps the *left* key's name; the right key column is dropped, since the left table's columns are carried over whole. + */ + inline def joinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("joinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("joinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("joinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]](itr, that, lk.value, rk.value, false) + end joinOn + + /** Left outer join on a column of the same name in both tables - every left row survives, and the right hand columns become `Option`. + * {{{ + * orders.leftJoin(customers)["custId"] // (custId: Int, qty: Int, name: Option[String]) + * }}} + * A right hand column that is *already* an `Option` stays as it is rather than nesting, which does mean an unmatched row and a matched row holding `None` look the same. + */ + inline def leftJoin[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("leftJoin: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("leftJoin: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("leftJoin: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](itr, that, key.value, key.value, true) + end leftJoin + + /** Left outer join where the key is named differently on each side. See [[leftJoin]] and [[joinOn]]. */ + inline def leftJoinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("leftJoinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("leftJoinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("leftJoinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "leftJoinOn"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]](itr, that, lk.value, rk.value, true) + end leftJoinOn end extension extension [CC[X] <: Iterable[X], K <: Tuple, V <: Tuple](nt: CC[NamedTuple[K, V]]) @@ -502,5 +665,101 @@ object NamedTupleIteratorExtensions: ): CC[NamedTuple[ReplaceOneName[K, From, To], V]] = bf.fromSpecific(nt)(nt.view.map(_.withNames[ReplaceOneName[K, From, To]].asInstanceOf[NamedTuple[ReplaceOneName[K, From, To], V]])) + /** Inner join on a column of the same name in both tables, preserving the collection type. See the `Iterator` overload for the semantics. + * {{{ + * orders.join(customers)["custId"] + * }}} + */ + inline def join[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("join: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("join: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("join: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "join"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]](nt.iterator, that, key.value, key.value, false) + ) + end join + + /** Inner join where the key is named differently on each side, preserving the collection type. See the `Iterator` overload. */ + inline def joinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("joinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("joinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("joinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]](nt.iterator, that, lk.value, rk.value, false) + ) + end joinOn + + /** Left outer join on a column of the same name in both tables, preserving the collection type. See the `Iterator` overload. */ + inline def leftJoin[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("leftJoin: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("leftJoin: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("leftJoin: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](nt.iterator, that, key.value, key.value, true) + ) + end leftJoin + + /** Left outer join where the key is named differently on each side, preserving the collection type. See the `Iterator` overload. */ + inline def leftJoinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("leftJoinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("leftJoinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("leftJoinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "leftJoinOn"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]( + nt.iterator, + that, + lk.value, + rk.value, + true + ) + ) + end leftJoinOn + end extension end NamedTupleIteratorExtensions diff --git a/scautable/src/columnType.scala b/scautable/src/columnType.scala index b5404c0ee..53250ca69 100644 --- a/scautable/src/columnType.scala +++ b/scautable/src/columnType.scala @@ -113,6 +113,20 @@ object ColumnTyped: case Option[a] => Option[Tag] case _ => Tag + /** `Option[T]`, idempotently - a column that is already optional does not become `Option[Option[_]]`. + * + * Treats `Option` the same way [[Unwrapped]] and [[Tagged]] do. `optionalise` in `NamedTupleIteratorExtensions` is the runtime counterpart and must agree with it: a left join + * would otherwise hand back values whose shape does not match their static type. + */ + type Optional[T] = T match + case Option[a] => Option[a] + case _ => Option[T] + + /** [[Optional]] applied to every element - the value types of the right hand table of a left join. */ + type Optionalize[T <: Tuple] <: Tuple = T match + case EmptyTuple => EmptyTuple + case head *: tail => Optional[head] *: Optionalize[tail] + /** Marker returned by [[ColTypeAtName]] when a name is not a column at all. */ sealed trait NoSuchColumn @@ -127,6 +141,16 @@ object ColumnTyped: case (EmptyTuple, ?) => NoSuchColumn case (?, EmptyTuple) => NoSuchColumn + /** Whether `Name` is one of `Names`. + * + * Same job as [[IsColumn]], but `Name` is unbounded and appears as the *pattern*, for the reason given on [[ColTypeAtName]] - which is what lets it be driven by a name captured + * by an inline match, where the `<: String` bound is lost. + */ + type NameIn[Names <: Tuple, Name] <: Boolean = Names match + case Name *: ? => true + case ? *: rest => NameIn[rest, Name] + case EmptyTuple => false + /** [[ReplaceOneTypeAtName]] with an unbounded `Name`, for the same reason as [[ColTypeAtName]]. A name that is not a column leaves `V` alone. */ type ReplaceTypeAtName[K <: Tuple, V <: Tuple, Name, A] <: Tuple = (K, V) match case (Name *: ?, ? *: vs) => A *: vs diff --git a/scautable/test/src/JoinSuite.scala b/scautable/test/src/JoinSuite.scala new file mode 100644 index 000000000..56595d5cd --- /dev/null +++ b/scautable/test/src/JoinSuite.scala @@ -0,0 +1,121 @@ +package io.github.quafadas.scautable + +import scala.NamedTuple.NamedTuple + +import io.github.quafadas.table.* + +class JoinSuite extends munit.FunSuite: + + val orders = Seq((custId = 1, qty = 5), (custId = 2, qty = 7), (custId = 1, qty = 9)) + val customers = Seq((custId = 1, name = "ada"), (custId = 2, name = "bob")) + + test("inner join, one to one") { + val out = Seq((custId = 1, qty = 5)).join(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out.toList, List((custId = 1, qty = 5, name = "ada"))) + } + + test("inner join fans a left row out over every matching right row") { + val many = Seq((custId = 1, item = "a"), (custId = 1, item = "b"), (custId = 2, item = "c")) + val out = Seq((custId = 1, qty = 5)).join(many)["custId"].toList + // right hand rows keep their encounter order + assertEquals(out.map(_.item), List("a", "b")) + } + + test("inner join drops rows that match nothing, on either side") { + val out = Seq((custId = 1, qty = 5), (custId = 99, qty = 1)).join(customers)["custId"].toList + assertEquals(out.map(_.custId), List(1)) + } + + test("join preserves left order and multiplies matches") { + val out = orders.join(customers)["custId"].toList + assertEquals(out.map(r => (r.custId, r.qty, r.name)), List((1, 5, "ada"), (2, 7, "bob"), (1, 9, "ada"))) + } + + test("joinOn with differing key names agrees with renameColumn + join") { + val rightNamed = Seq((id = 1, name = "ada"), (id = 2, name = "bob")) + val viaJoinOn = orders.joinOn(rightNamed)["custId", "id"].toList + val viaChain = orders.join(rightNamed.renameColumn["id", "custId"])["custId"].toList + assertEquals(viaJoinOn, viaChain) + // the output keeps the LEFT key's name + summon[viaJoinOn.type <:< List[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + } + + test("leftJoin keeps unmatched left rows and optionalises the right") { + val out = Seq((custId = 1, qty = 5), (custId = 99, qty = 1)).leftJoin(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, Option[String])]]] + assertEquals(out.toList.map(_.name), List(Some("ada"), None)) + } + + test("leftJoin does not nest an already optional right column") { + val right = Seq((custId = 1, nick = Option("a")), (custId = 2, nick = Option.empty[String])) + val out = Seq((custId = 1, qty = 5), (custId = 2, qty = 1), (custId = 9, qty = 0)).leftJoin(right)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "nick"), (Int, Int, Option[String])]]] + // and the runtime agrees with that type - no Some(None) anywhere + assertEquals(out.toList.map(_.nick), List(Some("a"), None, None)) + } + + test("leftJoinOn with differing key names") { + val rightNamed = Seq((id = 1, name = "ada")) + val out = orders.leftJoinOn(rightNamed)["custId", "id"].toList + assertEquals(out.map(_.name), List(Some("ada"), None, Some("ada"))) + } + + test("a right table that is nothing but the key still joins") { + val keyOnly = Seq((custId = 1)) + val out = Seq((custId = 1, qty = 5), (custId = 2, qty = 7)).join(keyOnly)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty"), (Int, Int)]]] + assertEquals(out.toList, List((custId = 1, qty = 5))) + } + + test("the Iterator overload is lazy - building a join drains neither side") { + var forcedLeft = 0 + val left = LazyList((custId = 1, qty = 5), (custId = 2, qty = 7)).map { r => forcedLeft += 1; r } + var forcedRight = 0 + val right = LazyList((custId = 1, name = "ada")).map { r => forcedRight += 1; r } + + val joined = left.iterator.join(right)["custId"] + assertEquals(forcedLeft, 0) + assertEquals(forcedRight, 0) + + assertEquals(joined.next().name, "ada") + assertEquals(forcedRight, 1) + } + + test("the Iterable overload preserves the collection type") { + val out = Vector((custId = 1, qty = 5)).join(customers)["custId"] + summon[out.type <:< Vector[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out, Vector((custId = 1, qty = 5, name = "ada"))) + } + + test("keys are matched with ==, so None matches None") { + val left = Seq((k = Option.empty[Int], a = 1)) + val right = Seq((k = Option.empty[Int], b = 2)) + assertEquals(left.join(right)["k"].toList.map(_.b), List(2)) + } + + test("a column on both sides does not compile, and says which one") { + val errs = compileErrors(""" + val l = Seq((custId = 1, name = "x")) + val r = Seq((custId = 1, name = "ada")) + l.join(r)["custId"] + """) + assert(errs.contains("join: column name exists on both sides"), errs) + } + + test("an unknown key does not compile, and says which side") { + val l = """val l = Seq((custId = 1, qty = 2)); val r = Seq((id = 1, name = "a")); """ + assert(compileErrors(l + """l.joinOn(r)["nope", "id"]""").contains("joinOn: no column named"), "left key") + assert(compileErrors(l + """l.joinOn(r)["custId", "nope"]""").contains("joinOn: no column named"), "right key") + } + + test("keys whose types differ do not compile") { + val errs = compileErrors(""" + val l = Seq((custId = 1, qty = 2)) + val r = Seq((id = "1", name = "a")) + l.joinOn(r)["custId", "id"] + """) + assert(errs.contains("have different types"), errs) + } + +end JoinSuite diff --git a/scautable/test/src/typedy.scala b/scautable/test/src/typedy.scala index 358055927..8dc7fba34 100644 --- a/scautable/test/src/typedy.scala +++ b/scautable/test/src/typedy.scala @@ -260,4 +260,45 @@ class NamedTupleTypeTest extends munit.FunSuite: } + test("NameIn") { + + type Cols = "a" *: "b" *: "c" *: EmptyTuple + + summon[ColumnTyped.NameIn[Cols, "a"] =:= true] + summon[ColumnTyped.NameIn[Cols, "c"] =:= true] + summon[ColumnTyped.NameIn[Cols, "f"] =:= false] + summon[ColumnTyped.NameIn[EmptyTuple, "a"] =:= false] + + } + + test("Optionalize is idempotent over already optional columns") { + + summon[ColumnTyped.Optional[String] =:= Option[String]] + summon[ColumnTyped.Optional[Option[String]] =:= Option[String]] + summon[ColumnTyped.Optionalize[(Int, Option[String])] =:= (Option[Int], Option[String])] + summon[ColumnTyped.Optionalize[EmptyTuple] =:= EmptyTuple] + + } + + test("the shape of a join result") { + + type LeftK = ("custId", "qty") + type LeftV = (Int, Int) + type RightK = ("id", "name") + type RightV = (Int, String) + + // inner: every left column, then every right column but the key + summon[Tuple.Concat[LeftK, ColumnTyped.DropOneName[RightK, "id"]] =:= ("custId", "qty", "name")] + summon[Tuple.Concat[LeftV, ColumnTyped.DropOneTypeAtName[RightK, "id", RightV]] =:= (Int, Int, String)] + + // left outer: the right hand types become optional + summon[ + Tuple.Concat[LeftV, ColumnTyped.Optionalize[ColumnTyped.DropOneTypeAtName[RightK, "id", RightV]]] =:= (Int, Int, Option[String]) + ] + + // a right table that is nothing but the key contributes no columns at all + summon[Tuple.Concat[LeftK, ColumnTyped.DropOneName["id" *: EmptyTuple, "id"]] =:= LeftK] + + } + end NamedTupleTypeTest diff --git a/site/docs/cheatsheet.md b/site/docs/cheatsheet.md index 29dfd1a9f..f1e82ec65 100644 --- a/site/docs/cheatsheet.md +++ b/site/docs/cheatsheet.md @@ -33,6 +33,21 @@ e.g. `val data : Seq[(col1 : String, col2 : Int, col3 : Double)] = ???` | format one column | `data.formatColumn["col3", Decimals[2]].ptbln` | +## Joining + +Assuming two `Iterator` / `Iterable` of named tuples. The key goes *after* the right hand table. + +| Want | Hints | +|-|-| +| inner join, same key name | `orders.join(customers)["custId"]` | +| inner join, different key names | `orders.joinOn(people)["custId", "id"]` | +| left join, right columns become `Option` | `orders.leftJoin(customers)["custId"]` | +| left join, different key names | `orders.leftJoinOn(people)["custId", "id"]` | +| right join | swap the tables and use `leftJoin` | +| a name is on both sides | compile error - `renameColumn` or `dropColumn` first | + +The right hand table is read into memory; the left streams. Put the smaller table on the right. + ## Excel Operations (JVM only) TBD diff --git a/site/docs/csv.md b/site/docs/csv.md index e921dd3ee..d51435cae 100644 --- a/site/docs/csv.md +++ b/site/docs/csv.md @@ -162,3 +162,73 @@ We can delegate all such concerns, to the standard library in the usual way - as colmanipuluation.filter(_.col4_renamed > 20).groupMapReduce(_.col1)(_.col4_renamed)(_ + _) ``` + +### Joining + +Joining is the one relational operation the standard library cannot do for us. It can group and +fold rows perfectly well, but it has no way to work out the *type* of two named tuples stitched +together on a key - so `join` is a first class operation here. + +```scala mdoc +val orders = Seq( + (custId = 1, qty = 5), + (custId = 2, qty = 7), + (custId = 1, qty = 9) +) + +val customers = Seq( + (custId = 1, name = "ada"), + (custId = 2, name = "bob") +) + +orders.join(customers)["custId"].consoleFormatNt(fansi = false) +``` + +The key is written *after* the right hand table, because the compiler works that table's shape out +from the argument and only the key needs spelling out. The result is every column of the left +table, then every column of the right one except the key. + +Where the key is named differently on each side, use `joinOn`. The output keeps the *left* name. + +```scala mdoc +val people = Seq((id = 1, name = "ada"), (id = 2, name = "bob")) + +orders.joinOn(people)["custId", "id"].consoleFormatNt(fansi = false) +``` + +`leftJoin` and `leftJoinOn` keep every left row, and the right hand columns become `Option`: + +```scala mdoc +val sparse = Seq((custId = 1, name = "ada")) + +orders.leftJoin(sparse)["custId"].consoleFormatNt(fansi = false) +``` + +A right hand column that is *already* an `Option` is left as it is rather than nesting into +`Option[Option[_]]` - which does mean an unmatched row and a matched row holding `None` look +alike. + +A column name that appears on both sides is a compile error naming the offender, rather than a +silently duplicated or overwritten column. Rename or drop it first. + +```scala mdoc:fail sc:nocompile +Seq((custId = 1, name = "x")).join(customers)["custId"] +``` + +Unknown keys, and keys whose types disagree, are caught in the same way: + +```scala mdoc:fail sc:nocompile +orders.joinOn(people)["custId", "nope"] +``` + +Worth knowing: + +- The left side streams; the right is read into a hash index the first time the result is pulled. + The right table is therefore the one that has to fit in memory - put the smaller table there. +- Keys are compared with `==`, so an `Option` key column matches `None` to `None`. Pandas would + drop those rows; scautable does not. +- Left order is preserved, and a left row matching several right rows emits them in the right + table's own order. +- Key types must agree exactly. A column carrying a display tag from `formatColumn` will not match + an untagged one - join first, format afterwards. +- Only inner and left joins are built in. A right join is a left join with the tables swapped. From 4fdd9f2ea0a41e56c8519237cbe2796ed703b59f Mon Sep 17 00:00:00 2001 From: Simon Parten Date: Wed, 30 Sep 2026 11:57:36 +0200 Subject: [PATCH 2/3] A None join key never matches Treating None as an ordinary value meant missing data squared itself: a join with three None keys on each side emitted nine rows carrying no information, and a 10k row join 30% missing in its key produced millions of junk rows. The key column is exactly where a duplicated value is least likely to mean "these rows belong together". So None is now "unknown", and two unknowns are not a match. This is what SQL does, where NULL = NULL is never true, and what pandas does, dropping missing keys from a merge. leftJoin still keeps such a row, with its right hand columns all None. To match missing to missing deliberately, map the key to a sentinel first: mapColumn["k", Int](_.getOrElse(-1)). Co-Authored-By: Claude Opus 5 (1M context) --- scautable/src/columnExtensions.scala | 13 ++++++++++++- scautable/test/src/JoinSuite.scala | 26 +++++++++++++++++++++++--- site/docs/cheatsheet.md | 2 ++ site/docs/csv.md | 8 ++++++-- 4 files changed, 43 insertions(+), 6 deletions(-) diff --git a/scautable/src/columnExtensions.scala b/scautable/src/columnExtensions.scala index e6d8f336f..58012fead 100644 --- a/scautable/src/columnExtensions.scala +++ b/scautable/src/columnExtensions.scala @@ -93,6 +93,15 @@ object NamedTupleIteratorExtensions: case x => Some(x) }) + /** A `None` key means "unknown", and two unknowns are not a match. + * + * This is what SQL does - `NULL = NULL` is never true - and what pandas does, dropping missing keys from a merge. Treating `None` as an ordinary value would also quietly square + * the missing rows: three `None` keys on each side is nine output rows carrying no information. A left join still keeps such a row, with its right hand columns all `None`. + * + * To match missing to missing deliberately, map the key to a sentinel first: `mapColumn["k", Int](_.getOrElse(-1))`. + */ + private def isMissingKey(key: Any): Boolean = key == None + /** Hash join. The left side streams; the right is materialised, but not until the result is first pulled - so building a join does not drain its argument. * * `rightArity` is the number of right hand columns *after* the key has been dropped, and is only used to shape the all-`None` row a left join emits for a left row that matched @@ -110,6 +119,7 @@ object NamedTupleIteratorExtensions: // `Map` alone would resolve to `NamedTuple.Map`, courtesy of the wildcard import at the top of this file. lazy val index: scala.collection.immutable.Map[Any, Seq[Tuple]] = right.iterator + .filterNot(t => isMissingKey(t.productElement(rIdx))) .map { t => val rest = removeAt(t, rIdx) t.productElement(rIdx) -> (if leftOuter then optionalise(rest) else rest) @@ -120,7 +130,8 @@ object NamedTupleIteratorExtensions: lazy val nones: Tuple = Tuple.fromArray(Array.fill[Object](rightArity)(None)) left.flatMap { lt => - val matches = index.getOrElse(lt.productElement(lIdx), Nil) + val key = lt.productElement(lIdx) + val matches = if isMissingKey(key) then Nil else index.getOrElse(key, Nil) if matches.nonEmpty then matches.iterator.map(rt => lt ++ rt) else if leftOuter then Iterator.single(lt ++ nones) else Iterator.empty diff --git a/scautable/test/src/JoinSuite.scala b/scautable/test/src/JoinSuite.scala index 56595d5cd..bc59f091e 100644 --- a/scautable/test/src/JoinSuite.scala +++ b/scautable/test/src/JoinSuite.scala @@ -88,9 +88,29 @@ class JoinSuite extends munit.FunSuite: assertEquals(out, Vector((custId = 1, qty = 5, name = "ada"))) } - test("keys are matched with ==, so None matches None") { - val left = Seq((k = Option.empty[Int], a = 1)) - val right = Seq((k = Option.empty[Int], b = 2)) + test("a None key never matches, as in SQL and pandas") { + val left = Seq((k = Option(1), a = 1), (k = Option.empty[Int], a = 2)) + val right = Seq((k = Option(1), b = 10), (k = Option.empty[Int], b = 20)) + assertEquals(left.join(right)["k"].toList.map(r => (r.a, r.b)), List((1, 10))) + } + + test("None keys do not multiply out into a cartesian block") { + val left = Seq.tabulate(4)(i => (k = Option.empty[Int], a = i)) + val right = Seq.tabulate(3)(i => (k = Option.empty[Int], b = i)) + // were None to match None this would be 4 * 3 + assertEquals(left.join(right)["k"].size, 0) + } + + test("leftJoin keeps a None keyed row, with the right hand side all None") { + val left = Seq((k = Option(1), a = 1), (k = Option.empty[Int], a = 2)) + val right = Seq((k = Option(1), b = 10), (k = Option.empty[Int], b = 20)) + val out = left.leftJoin(right)["k"].toList + assertEquals(out.map(r => (r.a, r.b)), List((1, Some(10)), (2, None))) + } + + test("a None key can be matched deliberately by mapping it to a sentinel") { + val left = Seq((k = Option.empty[Int], a = 1)).mapColumn["k", Int](_.getOrElse(-1)) + val right = Seq((k = Option.empty[Int], b = 2)).mapColumn["k", Int](_.getOrElse(-1)) assertEquals(left.join(right)["k"].toList.map(_.b), List(2)) } diff --git a/site/docs/cheatsheet.md b/site/docs/cheatsheet.md index f1e82ec65..a71f38778 100644 --- a/site/docs/cheatsheet.md +++ b/site/docs/cheatsheet.md @@ -45,6 +45,8 @@ Assuming two `Iterator` / `Iterable` of named tuples. The key goes *after* the r | left join, different key names | `orders.leftJoinOn(people)["custId", "id"]` | | right join | swap the tables and use `leftJoin` | | a name is on both sides | compile error - `renameColumn` or `dropColumn` first | +| key column is `Option` | `None` never matches, as in SQL / pandas | +| really want `None` to match | `mapColumn["k", Int](_.getOrElse(-1))` first | The right hand table is read into memory; the left streams. Put the smaller table on the right. diff --git a/site/docs/csv.md b/site/docs/csv.md index d51435cae..dde5f7a53 100644 --- a/site/docs/csv.md +++ b/site/docs/csv.md @@ -225,8 +225,12 @@ Worth knowing: - The left side streams; the right is read into a hash index the first time the result is pulled. The right table is therefore the one that has to fit in memory - put the smaller table there. -- Keys are compared with `==`, so an `Option` key column matches `None` to `None`. Pandas would - drop those rows; scautable does not. +- A `None` key never matches anything - as in SQL, where `NULL = NULL` is never true, and pandas, + which drops missing keys from a merge. `leftJoin` still keeps such a row, with its right hand + columns all `None`. Were `None` to match `None` instead, missing data would square itself: three + missing keys on each side would be nine output rows carrying no information. To match missing to + missing deliberately, map the key to a sentinel first - + `mapColumn["custId", Int](_.getOrElse(-1))`. - Left order is preserved, and a left row matching several right rows emits them in the right table's own order. - Key types must agree exactly. A column carrying a display tag from `formatColumn` will not match From ffdb0faaa0266e517d4d3d0b691cbafd18481168 Mon Sep 17 00:00:00 2001 From: Simon Parten Date: Wed, 30 Sep 2026 12:17:13 +0200 Subject: [PATCH 3/3] An inner join narrows an optional key After an inner join a None key cannot have matched - the row would have been dropped - so an Option key is provably present and the Option is noise. join on custId: Option[Int] now hands back custId: Int, sparing callers a .get that could never have thrown. leftJoin deliberately does not narrow: an unmatched left row survives still holding its None, so Option is the honest type there. Which leaves the two joins as mirror images. leftJoin ADDS optionality to the right hand columns, because an unmatched row appears and those columns really are absent. join REMOVES it from the key, because an unmatched row does not appear at all - and so the right hand columns keep their own types, never Option. Unwrapped is the identity on a non-optional key, so this is a no-op for the ordinary case, and isOptionKey keeps the runtime unwrap exactly in step with NarrowKey. Co-Authored-By: Claude Opus 5 (1M context) --- scautable/src/columnExtensions.scala | 104 +++++++++++++++++++++------ scautable/src/columnType.scala | 11 +++ scautable/test/src/JoinSuite.scala | 33 +++++++++ scautable/test/src/typedy.scala | 7 ++ site/docs/cheatsheet.md | 2 + site/docs/csv.md | 3 + 6 files changed, 140 insertions(+), 20 deletions(-) diff --git a/scautable/src/columnExtensions.scala b/scautable/src/columnExtensions.scala index 58012fead..51540e1be 100644 --- a/scautable/src/columnExtensions.scala +++ b/scautable/src/columnExtensions.scala @@ -93,6 +93,23 @@ object NamedTupleIteratorExtensions: case x => Some(x) }) + /** Whether the join key is optional, so that the runtime unwrap happens exactly where [[ColumnTyped.NarrowKey]] drops the `Option` and nowhere else. */ + private inline def isOptionKey[T]: Boolean = + inline erasedValue[T] match + case _: Option[?] => true + case _ => false + + /** Strip the `Option` off the cell at `idx`, the runtime half of [[ColumnTyped.NarrowKey]]. Only ever called on an inner join's key, which the `None` filter has already made a + * `Some`. + */ + private def unwrapAt(t: Tuple, idx: Int): Tuple = + val cells = t.toArray + cells(idx) = cells(idx) match + case Some(v) => v.asInstanceOf[Object] + case other => other + Tuple.fromArray(cells) + end unwrapAt + /** A `None` key means "unknown", and two unknowns are not a match. * * This is what SQL does - `NULL = NULL` is never true - and what pandas does, dropping missing keys from a merge. Treating `None` as an ordinary value would also quietly square @@ -113,7 +130,8 @@ object NamedTupleIteratorExtensions: right: => IterableOnce[Tuple], rIdx: Int, leftOuter: Boolean, - rightArity: Int + rightArity: Int, + unwrapKey: Boolean ): Iterator[Tuple] = // groupMap keeps right hand rows in encounter order within a key, so the output order is deterministic. // `Map` alone would resolve to `NamedTuple.Map`, courtesy of the wildcard import at the top of this file. @@ -132,7 +150,8 @@ object NamedTupleIteratorExtensions: left.flatMap { lt => val key = lt.productElement(lIdx) val matches = if isMissingKey(key) then Nil else index.getOrElse(key, Nil) - if matches.nonEmpty then matches.iterator.map(rt => lt ++ rt) + lazy val outLeft = if unwrapKey then unwrapAt(lt, lIdx) else lt + if matches.nonEmpty then matches.iterator.map(rt => outLeft ++ rt) else if leftOuter then Iterator.single(lt ++ nones) else Iterator.empty end if @@ -145,7 +164,8 @@ object NamedTupleIteratorExtensions: that: IterableOnce[NamedTuple[K2, V2]], leftKey: String, rightKey: String, - leftOuter: Boolean + leftOuter: Boolean, + unwrapKey: Boolean ): Iterator[NamedTuple[OutK, OutV]] = val rightHeaders = constValueTuple[K2].toList.map(_.toString()) hashJoin( @@ -154,7 +174,8 @@ object NamedTupleIteratorExtensions: that.iterator.map(_.toTuple), rightHeaders.indexOf(rightKey), leftOuter, - rightHeaders.size - 1 + rightHeaders.size - 1, + unwrapKey ).map(_.withNames[OutK].asInstanceOf[NamedTuple[OutK, OutV]]) end joinCore @@ -368,9 +389,16 @@ object NamedTupleIteratorExtensions: @implicitNotFound("join: column ${Key} has a different type on each side") evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], key: ValueOf[Key] - ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] = + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] = checkNoSharedNames[K, DropOneName[K2, Key], "join"] - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]](itr, that, key.value, key.value, false) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]( + itr, + that, + key.value, + key.value, + false, + isOptionKey[GetTypeAtName[K, Key, V]] + ) end join /** Inner join where the key is named differently on each side. See [[join]] for everything else. @@ -388,9 +416,16 @@ object NamedTupleIteratorExtensions: evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], lk: ValueOf[LeftKey], rk: ValueOf[RightKey] - ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] = + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] = checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]](itr, that, lk.value, rk.value, false) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]( + itr, + that, + lk.value, + rk.value, + false, + isOptionKey[GetTypeAtName[K, LeftKey, V]] + ) end joinOn /** Left outer join on a column of the same name in both tables - every left row survives, and the right hand columns become `Option`. @@ -409,7 +444,7 @@ object NamedTupleIteratorExtensions: key: ValueOf[Key] ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](itr, that, key.value, key.value, true) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](itr, that, key.value, key.value, true, false) end leftJoin /** Left outer join where the key is named differently on each side. See [[leftJoin]] and [[joinOn]]. */ @@ -424,7 +459,14 @@ object NamedTupleIteratorExtensions: rk: ValueOf[RightKey] ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] = checkNoSharedNames[K, DropOneName[K2, RightKey], "leftJoinOn"] - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]](itr, that, lk.value, rk.value, true) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]( + itr, + that, + lk.value, + rk.value, + true, + false + ) end leftJoinOn end extension @@ -691,13 +733,20 @@ object NamedTupleIteratorExtensions: key: ValueOf[Key], bf: BuildFrom[ CC[NamedTuple[K, V]], - NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]], - CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] + NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] ] - ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]]] = + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] = checkNoSharedNames[K, DropOneName[K2, Key], "join"] bf.fromSpecific(nt)( - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, DropOneTypeAtName[K2, Key, V2]]](nt.iterator, that, key.value, key.value, false) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]( + nt.iterator, + that, + key.value, + key.value, + false, + isOptionKey[GetTypeAtName[K, Key, V]] + ) ) end join @@ -713,13 +762,20 @@ object NamedTupleIteratorExtensions: rk: ValueOf[RightKey], bf: BuildFrom[ CC[NamedTuple[K, V]], - NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]], - CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] + NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] ] - ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]]] = + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] = checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] bf.fromSpecific(nt)( - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, DropOneTypeAtName[K2, RightKey, V2]]](nt.iterator, that, lk.value, rk.value, false) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]( + nt.iterator, + that, + lk.value, + rk.value, + false, + isOptionKey[GetTypeAtName[K, LeftKey, V]] + ) ) end joinOn @@ -740,7 +796,14 @@ object NamedTupleIteratorExtensions: ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] bf.fromSpecific(nt)( - joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](nt.iterator, that, key.value, key.value, true) + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]( + nt.iterator, + that, + key.value, + key.value, + true, + false + ) ) end leftJoin @@ -767,7 +830,8 @@ object NamedTupleIteratorExtensions: that, lk.value, rk.value, - true + true, + false ) ) end leftJoinOn diff --git a/scautable/src/columnType.scala b/scautable/src/columnType.scala index 53250ca69..00f501578 100644 --- a/scautable/src/columnType.scala +++ b/scautable/src/columnType.scala @@ -122,6 +122,17 @@ object ColumnTyped: case Option[a] => Option[a] case _ => Option[T] + /** The left table's value types with its join key narrowed. + * + * An inner join emits a row only where the keys matched, and a `None` key never matches - so after one, an `Option` key is provably present and the `Option` is noise. Dropping + * it spares callers a `.get` that could never have thrown. + * + * [[Unwrapped]] is the identity on a key that is not optional, so this is a no-op for the ordinary case. A *left* join must not use this: an unmatched left row survives, still + * holding its `None`. + */ + type NarrowKey[K <: Tuple, V <: Tuple, Key <: String] = + ReplaceOneTypeAtName[K, Key, V, Unwrapped[GetTypeAtName[K, Key, V]]] + /** [[Optional]] applied to every element - the value types of the right hand table of a left join. */ type Optionalize[T <: Tuple] <: Tuple = T match case EmptyTuple => EmptyTuple diff --git a/scautable/test/src/JoinSuite.scala b/scautable/test/src/JoinSuite.scala index bc59f091e..3a88036a3 100644 --- a/scautable/test/src/JoinSuite.scala +++ b/scautable/test/src/JoinSuite.scala @@ -88,6 +88,39 @@ class JoinSuite extends munit.FunSuite: assertEquals(out, Vector((custId = 1, qty = 5, name = "ada"))) } + test("an inner join narrows an Option key, which cannot be None once it has matched") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5), (orderId = 2, custId = Option.empty[Int], qty = 3)) + val customers = Seq((custId = Option(10), name = "ada")) + + val out = orders.join(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Int, Int, String)]]] + // no .get needed - the key is an Int now + assertEquals(out.toList.map(r => r.custId + 1), List(11)) + } + + test("joinOn narrows the left key, not the dropped right one") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5)) + val people = Seq((id = Option(10), name = "ada")) + val out = orders.joinOn(people)["custId", "id"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Int, Int, String)]]] + assertEquals(out.toList.map(_.custId), List(10)) + } + + test("narrowing is a no-op for a key that was never optional") { + val out = Seq((custId = 1, qty = 5)).join(Seq((custId = 1, name = "ada")))["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out.toList.map(_.custId), List(1)) + } + + test("a left join does NOT narrow the key - an unmatched row keeps its None") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5), (orderId = 2, custId = Option.empty[Int], qty = 3)) + val customers = Seq((custId = Option(10), name = "ada")) + + val out = orders.leftJoin(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Option[Int], Int, Option[String])]]] + assertEquals(out.toList.map(r => (r.custId, r.name)), List((Some(10), Some("ada")), (None, None))) + } + test("a None key never matches, as in SQL and pandas") { val left = Seq((k = Option(1), a = 1), (k = Option.empty[Int], a = 2)) val right = Seq((k = Option(1), b = 10), (k = Option.empty[Int], b = 20)) diff --git a/scautable/test/src/typedy.scala b/scautable/test/src/typedy.scala index 8dc7fba34..6594357f7 100644 --- a/scautable/test/src/typedy.scala +++ b/scautable/test/src/typedy.scala @@ -299,6 +299,13 @@ class NamedTupleTypeTest extends munit.FunSuite: // a right table that is nothing but the key contributes no columns at all summon[Tuple.Concat[LeftK, ColumnTyped.DropOneName["id" *: EmptyTuple, "id"]] =:= LeftK] + // an inner join narrows an optional key; a left join leaves it alone + type OptK = ("custId", "qty") + type OptV = (Option[Int], Int) + summon[ColumnTyped.NarrowKey[OptK, OptV, "custId"] =:= (Int, Int)] + // identity where the key was never optional + summon[ColumnTyped.NarrowKey[LeftK, LeftV, "custId"] =:= LeftV] + } end NamedTupleTypeTest diff --git a/site/docs/cheatsheet.md b/site/docs/cheatsheet.md index a71f38778..c92a78355 100644 --- a/site/docs/cheatsheet.md +++ b/site/docs/cheatsheet.md @@ -46,6 +46,8 @@ Assuming two `Iterator` / `Iterable` of named tuples. The key goes *after* the r | right join | swap the tables and use `leftJoin` | | a name is on both sides | compile error - `renameColumn` or `dropColumn` first | | key column is `Option` | `None` never matches, as in SQL / pandas | +| key column is `Option`, after `join` | narrowed - `Option[Int]` becomes `Int` | +| key column is `Option`, after `leftJoin` | left alone - the `None` rows are still there | | really want `None` to match | `mapColumn["k", Int](_.getOrElse(-1))` first | The right hand table is read into memory; the left streams. Put the smaller table on the right. diff --git a/site/docs/csv.md b/site/docs/csv.md index dde5f7a53..aa23d9bd4 100644 --- a/site/docs/csv.md +++ b/site/docs/csv.md @@ -225,6 +225,9 @@ Worth knowing: - The left side streams; the right is read into a hash index the first time the result is pulled. The right table is therefore the one that has to fit in memory - put the smaller table there. +- An inner join *narrows* an optional key. A `None` key cannot have matched, so `join` on a + `custId: Option[Int]` hands back `custId: Int` - no `.get` that could never have thrown. A + `leftJoin` leaves the key alone, because an unmatched row survives still holding its `None`. - A `None` key never matches anything - as in SQL, where `NULL = NULL` is never true, and pandas, which drops missing keys from a merge. `leftJoin` still keeps such a row, with its right hand columns all `None`. Were `None` to match `None` instead, missing data would square itself: three