diff --git a/scautable/src/columnExtensions.scala b/scautable/src/columnExtensions.scala index 79fca7196..51540e1be 100644 --- a/scautable/src/columnExtensions.scala +++ b/scautable/src/columnExtensions.scala @@ -62,6 +62,123 @@ object NamedTupleIteratorExtensions: Tuple.fromArray(cells) end applyPlan + /** Fail compilation if any name appears on both sides of a join. `Who` is the calling method, as in [[checkSpecNames]]. + * + * `Tuple.Disjoint[Left, Right] =:= true` would also reject these, but it cannot say *which* column is at fault, and that is the whole of the message's value. + */ + private inline def checkNoSharedNames[Left <: Tuple, Right <: Tuple, Who <: String]: Unit = + inline erasedValue[Right] match + case _: EmptyTuple => () + case _: (n *: rest) => + inline erasedValue[NameIn[Left, n]] match + case _: true => error(constValue[Who] + ": column " + constValue[n & String] + " exists on both sides - rename or drop it first") + case _ => checkNoSharedNames[Left, rest, Who] + + /** Drop the cell at `idx`. The `EmptyTuple` case is not redundant - it is what a right hand table consisting of nothing but the join key hits. */ + private def removeAt(t: Tuple, idx: Int): Tuple = + val (head, tail) = t.splitAt(idx) + head match + case _: EmptyTuple => tail.tail + case _ => head ++ tail.tail + end match + end removeAt + + /** Runtime counterpart of [[ColumnTyped.Optional]]: wrap each cell in `Some` unless it already is an `Option`. + * + * The two must agree. Wrapping unconditionally would hand back `Some(None)` where the static type says `None`. + */ + private def optionalise(t: Tuple): Tuple = + Tuple.fromArray(t.toArray.map { + case o: Option[?] => o + case x => Some(x) + }) + + /** Whether the join key is optional, so that the runtime unwrap happens exactly where [[ColumnTyped.NarrowKey]] drops the `Option` and nowhere else. */ + private inline def isOptionKey[T]: Boolean = + inline erasedValue[T] match + case _: Option[?] => true + case _ => false + + /** Strip the `Option` off the cell at `idx`, the runtime half of [[ColumnTyped.NarrowKey]]. Only ever called on an inner join's key, which the `None` filter has already made a + * `Some`. + */ + private def unwrapAt(t: Tuple, idx: Int): Tuple = + val cells = t.toArray + cells(idx) = cells(idx) match + case Some(v) => v.asInstanceOf[Object] + case other => other + Tuple.fromArray(cells) + end unwrapAt + + /** A `None` key means "unknown", and two unknowns are not a match. + * + * This is what SQL does - `NULL = NULL` is never true - and what pandas does, dropping missing keys from a merge. Treating `None` as an ordinary value would also quietly square + * the missing rows: three `None` keys on each side is nine output rows carrying no information. A left join still keeps such a row, with its right hand columns all `None`. + * + * To match missing to missing deliberately, map the key to a sentinel first: `mapColumn["k", Int](_.getOrElse(-1))`. + */ + private def isMissingKey(key: Any): Boolean = key == None + + /** Hash join. The left side streams; the right is materialised, but not until the result is first pulled - so building a join does not drain its argument. + * + * `rightArity` is the number of right hand columns *after* the key has been dropped, and is only used to shape the all-`None` row a left join emits for a left row that matched + * nothing. + */ + private def hashJoin( + left: Iterator[Tuple], + lIdx: Int, + right: => IterableOnce[Tuple], + rIdx: Int, + leftOuter: Boolean, + rightArity: Int, + unwrapKey: Boolean + ): Iterator[Tuple] = + // groupMap keeps right hand rows in encounter order within a key, so the output order is deterministic. + // `Map` alone would resolve to `NamedTuple.Map`, courtesy of the wildcard import at the top of this file. + lazy val index: scala.collection.immutable.Map[Any, Seq[Tuple]] = + right.iterator + .filterNot(t => isMissingKey(t.productElement(rIdx))) + .map { t => + val rest = removeAt(t, rIdx) + t.productElement(rIdx) -> (if leftOuter then optionalise(rest) else rest) + } + .toSeq + .groupMap(_._1)(_._2) + + lazy val nones: Tuple = Tuple.fromArray(Array.fill[Object](rightArity)(None)) + + left.flatMap { lt => + val key = lt.productElement(lIdx) + val matches = if isMissingKey(key) then Nil else index.getOrElse(key, Nil) + lazy val outLeft = if unwrapKey then unwrapAt(lt, lIdx) else lt + if matches.nonEmpty then matches.iterator.map(rt => outLeft ++ rt) + else if leftOuter then Iterator.single(lt ++ nones) + else Iterator.empty + end if + } + end hashJoin + + /** The shared body of every join: resolve both key positions, join, and re-attach the output names. Callers own the compile time checks, so that errors name *their* method. */ + private inline def joinCore[K <: Tuple, V <: Tuple, K2 <: Tuple, V2 <: Tuple, OutK <: Tuple, OutV <: Tuple]( + itr: Iterator[NamedTuple[K, V]], + that: IterableOnce[NamedTuple[K2, V2]], + leftKey: String, + rightKey: String, + leftOuter: Boolean, + unwrapKey: Boolean + ): Iterator[NamedTuple[OutK, OutV]] = + val rightHeaders = constValueTuple[K2].toList.map(_.toString()) + hashJoin( + itr.map(_.toTuple), + constValueTuple[K].toList.map(_.toString()).indexOf(leftKey), + that.iterator.map(_.toTuple), + rightHeaders.indexOf(rightKey), + leftOuter, + rightHeaders.size - 1, + unwrapKey + ).map(_.withNames[OutK].asInstanceOf[NamedTuple[OutK, OutV]]) + end joinCore + extension [K <: Tuple, V <: Tuple](itr: Iterator[NamedTuple[K, V]]) def sample(frac: Double, deterministic: Boolean = false): Iterator[NamedTuple[K, V]] = @@ -252,6 +369,105 @@ object NamedTupleIteratorExtensions: end match } end dropColumn + + /** Inner join on a column of the same name in both tables. + * + * The key is given *after* the right hand table, because the compiler infers that table's shape from the argument and only the key needs writing out: + * {{{ + * orders.join(customers)["custId"] + * }}} + * The output is every column of the left table, then every column of the right except the key. The left side streams; the right is read into a hash index the first time the + * result is pulled, so the right table is the one that has to fit in memory. + * + * A column name on both sides is a compile error - rename or drop it first. Keys are matched with `==`, so an `Option` key column matches `None` to `None`. + */ + inline def join[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("join: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("join: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("join: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "join"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]( + itr, + that, + key.value, + key.value, + false, + isOptionKey[GetTypeAtName[K, Key, V]] + ) + end join + + /** Inner join where the key is named differently on each side. See [[join]] for everything else. + * {{{ + * orders.joinOn(customers)["customer_id", "id"] + * }}} + * The output keeps the *left* key's name; the right key column is dropped, since the left table's columns are carried over whole. + */ + inline def joinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("joinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("joinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("joinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]( + itr, + that, + lk.value, + rk.value, + false, + isOptionKey[GetTypeAtName[K, LeftKey, V]] + ) + end joinOn + + /** Left outer join on a column of the same name in both tables - every left row survives, and the right hand columns become `Option`. + * {{{ + * orders.leftJoin(customers)["custId"] // (custId: Int, qty: Int, name: Option[String]) + * }}} + * A right hand column that is *already* an `Option` stays as it is rather than nesting, which does mean an unmatched row and a matched row holding `None` look the same. + */ + inline def leftJoin[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("leftJoin: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("leftJoin: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("leftJoin: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]](itr, that, key.value, key.value, true, false) + end leftJoin + + /** Left outer join where the key is named differently on each side. See [[leftJoin]] and [[joinOn]]. */ + inline def leftJoinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("leftJoinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("leftJoinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("leftJoinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey] + ): Iterator[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "leftJoinOn"] + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]( + itr, + that, + lk.value, + rk.value, + true, + false + ) + end leftJoinOn end extension extension [CC[X] <: Iterable[X], K <: Tuple, V <: Tuple](nt: CC[NamedTuple[K, V]]) @@ -502,5 +718,123 @@ object NamedTupleIteratorExtensions: ): CC[NamedTuple[ReplaceOneName[K, From, To], V]] = bf.fromSpecific(nt)(nt.view.map(_.withNames[ReplaceOneName[K, From, To]].asInstanceOf[NamedTuple[ReplaceOneName[K, From, To], V]])) + /** Inner join on a column of the same name in both tables, preserving the collection type. See the `Iterator` overload for the semantics. + * {{{ + * orders.join(customers)["custId"] + * }}} + */ + inline def join[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("join: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("join: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("join: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "join"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[NarrowKey[K, V, Key], DropOneTypeAtName[K2, Key, V2]]]( + nt.iterator, + that, + key.value, + key.value, + false, + isOptionKey[GetTypeAtName[K, Key, V]] + ) + ) + end join + + /** Inner join where the key is named differently on each side, preserving the collection type. See the `Iterator` overload. */ + inline def joinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("joinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("joinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("joinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "joinOn"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[NarrowKey[K, V, LeftKey], DropOneTypeAtName[K2, RightKey, V2]]]( + nt.iterator, + that, + lk.value, + rk.value, + false, + isOptionKey[GetTypeAtName[K, LeftKey, V]] + ) + ) + end joinOn + + /** Left outer join on a column of the same name in both tables, preserving the collection type. See the `Iterator` overload. */ + inline def leftJoin[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[Key <: String](using + @implicitNotFound("leftJoin: no column named ${Key} on the left") + evL: IsColumn[Key, K] =:= true, + @implicitNotFound("leftJoin: no column named ${Key} on the right") + evR: IsColumn[Key, K2] =:= true, + @implicitNotFound("leftJoin: column ${Key} has a different type on each side") + evT: GetTypeAtName[K, Key, V] =:= GetTypeAtName[K2, Key, V2], + key: ValueOf[Key], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, Key], "leftJoin"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, Key]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, Key, V2]]]]( + nt.iterator, + that, + key.value, + key.value, + true, + false + ) + ) + end leftJoin + + /** Left outer join where the key is named differently on each side, preserving the collection type. See the `Iterator` overload. */ + inline def leftJoinOn[K2 <: Tuple, V2 <: Tuple](that: IterableOnce[NamedTuple[K2, V2]])[LeftKey <: String, RightKey <: String](using + @implicitNotFound("leftJoinOn: no column named ${LeftKey} on the left") + evL: IsColumn[LeftKey, K] =:= true, + @implicitNotFound("leftJoinOn: no column named ${RightKey} on the right") + evR: IsColumn[RightKey, K2] =:= true, + @implicitNotFound("leftJoinOn: key ${LeftKey} and key ${RightKey} have different types") + evT: GetTypeAtName[K, LeftKey, V] =:= GetTypeAtName[K2, RightKey, V2], + lk: ValueOf[LeftKey], + rk: ValueOf[RightKey], + bf: BuildFrom[ + CC[NamedTuple[K, V]], + NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]], + CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] + ] + ): CC[NamedTuple[Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]] = + checkNoSharedNames[K, DropOneName[K2, RightKey], "leftJoinOn"] + bf.fromSpecific(nt)( + joinCore[K, V, K2, V2, Tuple.Concat[K, DropOneName[K2, RightKey]], Tuple.Concat[V, Optionalize[DropOneTypeAtName[K2, RightKey, V2]]]]( + nt.iterator, + that, + lk.value, + rk.value, + true, + false + ) + ) + end leftJoinOn + end extension end NamedTupleIteratorExtensions diff --git a/scautable/src/columnType.scala b/scautable/src/columnType.scala index b5404c0ee..00f501578 100644 --- a/scautable/src/columnType.scala +++ b/scautable/src/columnType.scala @@ -113,6 +113,31 @@ object ColumnTyped: case Option[a] => Option[Tag] case _ => Tag + /** `Option[T]`, idempotently - a column that is already optional does not become `Option[Option[_]]`. + * + * Treats `Option` the same way [[Unwrapped]] and [[Tagged]] do. `optionalise` in `NamedTupleIteratorExtensions` is the runtime counterpart and must agree with it: a left join + * would otherwise hand back values whose shape does not match their static type. + */ + type Optional[T] = T match + case Option[a] => Option[a] + case _ => Option[T] + + /** The left table's value types with its join key narrowed. + * + * An inner join emits a row only where the keys matched, and a `None` key never matches - so after one, an `Option` key is provably present and the `Option` is noise. Dropping + * it spares callers a `.get` that could never have thrown. + * + * [[Unwrapped]] is the identity on a key that is not optional, so this is a no-op for the ordinary case. A *left* join must not use this: an unmatched left row survives, still + * holding its `None`. + */ + type NarrowKey[K <: Tuple, V <: Tuple, Key <: String] = + ReplaceOneTypeAtName[K, Key, V, Unwrapped[GetTypeAtName[K, Key, V]]] + + /** [[Optional]] applied to every element - the value types of the right hand table of a left join. */ + type Optionalize[T <: Tuple] <: Tuple = T match + case EmptyTuple => EmptyTuple + case head *: tail => Optional[head] *: Optionalize[tail] + /** Marker returned by [[ColTypeAtName]] when a name is not a column at all. */ sealed trait NoSuchColumn @@ -127,6 +152,16 @@ object ColumnTyped: case (EmptyTuple, ?) => NoSuchColumn case (?, EmptyTuple) => NoSuchColumn + /** Whether `Name` is one of `Names`. + * + * Same job as [[IsColumn]], but `Name` is unbounded and appears as the *pattern*, for the reason given on [[ColTypeAtName]] - which is what lets it be driven by a name captured + * by an inline match, where the `<: String` bound is lost. + */ + type NameIn[Names <: Tuple, Name] <: Boolean = Names match + case Name *: ? => true + case ? *: rest => NameIn[rest, Name] + case EmptyTuple => false + /** [[ReplaceOneTypeAtName]] with an unbounded `Name`, for the same reason as [[ColTypeAtName]]. A name that is not a column leaves `V` alone. */ type ReplaceTypeAtName[K <: Tuple, V <: Tuple, Name, A] <: Tuple = (K, V) match case (Name *: ?, ? *: vs) => A *: vs diff --git a/scautable/test/src/JoinSuite.scala b/scautable/test/src/JoinSuite.scala new file mode 100644 index 000000000..3a88036a3 --- /dev/null +++ b/scautable/test/src/JoinSuite.scala @@ -0,0 +1,174 @@ +package io.github.quafadas.scautable + +import scala.NamedTuple.NamedTuple + +import io.github.quafadas.table.* + +class JoinSuite extends munit.FunSuite: + + val orders = Seq((custId = 1, qty = 5), (custId = 2, qty = 7), (custId = 1, qty = 9)) + val customers = Seq((custId = 1, name = "ada"), (custId = 2, name = "bob")) + + test("inner join, one to one") { + val out = Seq((custId = 1, qty = 5)).join(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out.toList, List((custId = 1, qty = 5, name = "ada"))) + } + + test("inner join fans a left row out over every matching right row") { + val many = Seq((custId = 1, item = "a"), (custId = 1, item = "b"), (custId = 2, item = "c")) + val out = Seq((custId = 1, qty = 5)).join(many)["custId"].toList + // right hand rows keep their encounter order + assertEquals(out.map(_.item), List("a", "b")) + } + + test("inner join drops rows that match nothing, on either side") { + val out = Seq((custId = 1, qty = 5), (custId = 99, qty = 1)).join(customers)["custId"].toList + assertEquals(out.map(_.custId), List(1)) + } + + test("join preserves left order and multiplies matches") { + val out = orders.join(customers)["custId"].toList + assertEquals(out.map(r => (r.custId, r.qty, r.name)), List((1, 5, "ada"), (2, 7, "bob"), (1, 9, "ada"))) + } + + test("joinOn with differing key names agrees with renameColumn + join") { + val rightNamed = Seq((id = 1, name = "ada"), (id = 2, name = "bob")) + val viaJoinOn = orders.joinOn(rightNamed)["custId", "id"].toList + val viaChain = orders.join(rightNamed.renameColumn["id", "custId"])["custId"].toList + assertEquals(viaJoinOn, viaChain) + // the output keeps the LEFT key's name + summon[viaJoinOn.type <:< List[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + } + + test("leftJoin keeps unmatched left rows and optionalises the right") { + val out = Seq((custId = 1, qty = 5), (custId = 99, qty = 1)).leftJoin(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, Option[String])]]] + assertEquals(out.toList.map(_.name), List(Some("ada"), None)) + } + + test("leftJoin does not nest an already optional right column") { + val right = Seq((custId = 1, nick = Option("a")), (custId = 2, nick = Option.empty[String])) + val out = Seq((custId = 1, qty = 5), (custId = 2, qty = 1), (custId = 9, qty = 0)).leftJoin(right)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "nick"), (Int, Int, Option[String])]]] + // and the runtime agrees with that type - no Some(None) anywhere + assertEquals(out.toList.map(_.nick), List(Some("a"), None, None)) + } + + test("leftJoinOn with differing key names") { + val rightNamed = Seq((id = 1, name = "ada")) + val out = orders.leftJoinOn(rightNamed)["custId", "id"].toList + assertEquals(out.map(_.name), List(Some("ada"), None, Some("ada"))) + } + + test("a right table that is nothing but the key still joins") { + val keyOnly = Seq((custId = 1)) + val out = Seq((custId = 1, qty = 5), (custId = 2, qty = 7)).join(keyOnly)["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty"), (Int, Int)]]] + assertEquals(out.toList, List((custId = 1, qty = 5))) + } + + test("the Iterator overload is lazy - building a join drains neither side") { + var forcedLeft = 0 + val left = LazyList((custId = 1, qty = 5), (custId = 2, qty = 7)).map { r => forcedLeft += 1; r } + var forcedRight = 0 + val right = LazyList((custId = 1, name = "ada")).map { r => forcedRight += 1; r } + + val joined = left.iterator.join(right)["custId"] + assertEquals(forcedLeft, 0) + assertEquals(forcedRight, 0) + + assertEquals(joined.next().name, "ada") + assertEquals(forcedRight, 1) + } + + test("the Iterable overload preserves the collection type") { + val out = Vector((custId = 1, qty = 5)).join(customers)["custId"] + summon[out.type <:< Vector[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out, Vector((custId = 1, qty = 5, name = "ada"))) + } + + test("an inner join narrows an Option key, which cannot be None once it has matched") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5), (orderId = 2, custId = Option.empty[Int], qty = 3)) + val customers = Seq((custId = Option(10), name = "ada")) + + val out = orders.join(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Int, Int, String)]]] + // no .get needed - the key is an Int now + assertEquals(out.toList.map(r => r.custId + 1), List(11)) + } + + test("joinOn narrows the left key, not the dropped right one") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5)) + val people = Seq((id = Option(10), name = "ada")) + val out = orders.joinOn(people)["custId", "id"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Int, Int, String)]]] + assertEquals(out.toList.map(_.custId), List(10)) + } + + test("narrowing is a no-op for a key that was never optional") { + val out = Seq((custId = 1, qty = 5)).join(Seq((custId = 1, name = "ada")))["custId"] + summon[out.type <:< Seq[NamedTuple[("custId", "qty", "name"), (Int, Int, String)]]] + assertEquals(out.toList.map(_.custId), List(1)) + } + + test("a left join does NOT narrow the key - an unmatched row keeps its None") { + val orders = Seq((orderId = 1, custId = Option(10), qty = 5), (orderId = 2, custId = Option.empty[Int], qty = 3)) + val customers = Seq((custId = Option(10), name = "ada")) + + val out = orders.leftJoin(customers)["custId"] + summon[out.type <:< Seq[NamedTuple[("orderId", "custId", "qty", "name"), (Int, Option[Int], Int, Option[String])]]] + assertEquals(out.toList.map(r => (r.custId, r.name)), List((Some(10), Some("ada")), (None, None))) + } + + test("a None key never matches, as in SQL and pandas") { + val left = Seq((k = Option(1), a = 1), (k = Option.empty[Int], a = 2)) + val right = Seq((k = Option(1), b = 10), (k = Option.empty[Int], b = 20)) + assertEquals(left.join(right)["k"].toList.map(r => (r.a, r.b)), List((1, 10))) + } + + test("None keys do not multiply out into a cartesian block") { + val left = Seq.tabulate(4)(i => (k = Option.empty[Int], a = i)) + val right = Seq.tabulate(3)(i => (k = Option.empty[Int], b = i)) + // were None to match None this would be 4 * 3 + assertEquals(left.join(right)["k"].size, 0) + } + + test("leftJoin keeps a None keyed row, with the right hand side all None") { + val left = Seq((k = Option(1), a = 1), (k = Option.empty[Int], a = 2)) + val right = Seq((k = Option(1), b = 10), (k = Option.empty[Int], b = 20)) + val out = left.leftJoin(right)["k"].toList + assertEquals(out.map(r => (r.a, r.b)), List((1, Some(10)), (2, None))) + } + + test("a None key can be matched deliberately by mapping it to a sentinel") { + val left = Seq((k = Option.empty[Int], a = 1)).mapColumn["k", Int](_.getOrElse(-1)) + val right = Seq((k = Option.empty[Int], b = 2)).mapColumn["k", Int](_.getOrElse(-1)) + assertEquals(left.join(right)["k"].toList.map(_.b), List(2)) + } + + test("a column on both sides does not compile, and says which one") { + val errs = compileErrors(""" + val l = Seq((custId = 1, name = "x")) + val r = Seq((custId = 1, name = "ada")) + l.join(r)["custId"] + """) + assert(errs.contains("join: column name exists on both sides"), errs) + } + + test("an unknown key does not compile, and says which side") { + val l = """val l = Seq((custId = 1, qty = 2)); val r = Seq((id = 1, name = "a")); """ + assert(compileErrors(l + """l.joinOn(r)["nope", "id"]""").contains("joinOn: no column named"), "left key") + assert(compileErrors(l + """l.joinOn(r)["custId", "nope"]""").contains("joinOn: no column named"), "right key") + } + + test("keys whose types differ do not compile") { + val errs = compileErrors(""" + val l = Seq((custId = 1, qty = 2)) + val r = Seq((id = "1", name = "a")) + l.joinOn(r)["custId", "id"] + """) + assert(errs.contains("have different types"), errs) + } + +end JoinSuite diff --git a/scautable/test/src/typedy.scala b/scautable/test/src/typedy.scala index 358055927..6594357f7 100644 --- a/scautable/test/src/typedy.scala +++ b/scautable/test/src/typedy.scala @@ -260,4 +260,52 @@ class NamedTupleTypeTest extends munit.FunSuite: } + test("NameIn") { + + type Cols = "a" *: "b" *: "c" *: EmptyTuple + + summon[ColumnTyped.NameIn[Cols, "a"] =:= true] + summon[ColumnTyped.NameIn[Cols, "c"] =:= true] + summon[ColumnTyped.NameIn[Cols, "f"] =:= false] + summon[ColumnTyped.NameIn[EmptyTuple, "a"] =:= false] + + } + + test("Optionalize is idempotent over already optional columns") { + + summon[ColumnTyped.Optional[String] =:= Option[String]] + summon[ColumnTyped.Optional[Option[String]] =:= Option[String]] + summon[ColumnTyped.Optionalize[(Int, Option[String])] =:= (Option[Int], Option[String])] + summon[ColumnTyped.Optionalize[EmptyTuple] =:= EmptyTuple] + + } + + test("the shape of a join result") { + + type LeftK = ("custId", "qty") + type LeftV = (Int, Int) + type RightK = ("id", "name") + type RightV = (Int, String) + + // inner: every left column, then every right column but the key + summon[Tuple.Concat[LeftK, ColumnTyped.DropOneName[RightK, "id"]] =:= ("custId", "qty", "name")] + summon[Tuple.Concat[LeftV, ColumnTyped.DropOneTypeAtName[RightK, "id", RightV]] =:= (Int, Int, String)] + + // left outer: the right hand types become optional + summon[ + Tuple.Concat[LeftV, ColumnTyped.Optionalize[ColumnTyped.DropOneTypeAtName[RightK, "id", RightV]]] =:= (Int, Int, Option[String]) + ] + + // a right table that is nothing but the key contributes no columns at all + summon[Tuple.Concat[LeftK, ColumnTyped.DropOneName["id" *: EmptyTuple, "id"]] =:= LeftK] + + // an inner join narrows an optional key; a left join leaves it alone + type OptK = ("custId", "qty") + type OptV = (Option[Int], Int) + summon[ColumnTyped.NarrowKey[OptK, OptV, "custId"] =:= (Int, Int)] + // identity where the key was never optional + summon[ColumnTyped.NarrowKey[LeftK, LeftV, "custId"] =:= LeftV] + + } + end NamedTupleTypeTest diff --git a/site/docs/cheatsheet.md b/site/docs/cheatsheet.md index 29dfd1a9f..c92a78355 100644 --- a/site/docs/cheatsheet.md +++ b/site/docs/cheatsheet.md @@ -33,6 +33,25 @@ e.g. `val data : Seq[(col1 : String, col2 : Int, col3 : Double)] = ???` | format one column | `data.formatColumn["col3", Decimals[2]].ptbln` | +## Joining + +Assuming two `Iterator` / `Iterable` of named tuples. The key goes *after* the right hand table. + +| Want | Hints | +|-|-| +| inner join, same key name | `orders.join(customers)["custId"]` | +| inner join, different key names | `orders.joinOn(people)["custId", "id"]` | +| left join, right columns become `Option` | `orders.leftJoin(customers)["custId"]` | +| left join, different key names | `orders.leftJoinOn(people)["custId", "id"]` | +| right join | swap the tables and use `leftJoin` | +| a name is on both sides | compile error - `renameColumn` or `dropColumn` first | +| key column is `Option` | `None` never matches, as in SQL / pandas | +| key column is `Option`, after `join` | narrowed - `Option[Int]` becomes `Int` | +| key column is `Option`, after `leftJoin` | left alone - the `None` rows are still there | +| really want `None` to match | `mapColumn["k", Int](_.getOrElse(-1))` first | + +The right hand table is read into memory; the left streams. Put the smaller table on the right. + ## Excel Operations (JVM only) TBD diff --git a/site/docs/csv.md b/site/docs/csv.md index e921dd3ee..aa23d9bd4 100644 --- a/site/docs/csv.md +++ b/site/docs/csv.md @@ -162,3 +162,80 @@ We can delegate all such concerns, to the standard library in the usual way - as colmanipuluation.filter(_.col4_renamed > 20).groupMapReduce(_.col1)(_.col4_renamed)(_ + _) ``` + +### Joining + +Joining is the one relational operation the standard library cannot do for us. It can group and +fold rows perfectly well, but it has no way to work out the *type* of two named tuples stitched +together on a key - so `join` is a first class operation here. + +```scala mdoc +val orders = Seq( + (custId = 1, qty = 5), + (custId = 2, qty = 7), + (custId = 1, qty = 9) +) + +val customers = Seq( + (custId = 1, name = "ada"), + (custId = 2, name = "bob") +) + +orders.join(customers)["custId"].consoleFormatNt(fansi = false) +``` + +The key is written *after* the right hand table, because the compiler works that table's shape out +from the argument and only the key needs spelling out. The result is every column of the left +table, then every column of the right one except the key. + +Where the key is named differently on each side, use `joinOn`. The output keeps the *left* name. + +```scala mdoc +val people = Seq((id = 1, name = "ada"), (id = 2, name = "bob")) + +orders.joinOn(people)["custId", "id"].consoleFormatNt(fansi = false) +``` + +`leftJoin` and `leftJoinOn` keep every left row, and the right hand columns become `Option`: + +```scala mdoc +val sparse = Seq((custId = 1, name = "ada")) + +orders.leftJoin(sparse)["custId"].consoleFormatNt(fansi = false) +``` + +A right hand column that is *already* an `Option` is left as it is rather than nesting into +`Option[Option[_]]` - which does mean an unmatched row and a matched row holding `None` look +alike. + +A column name that appears on both sides is a compile error naming the offender, rather than a +silently duplicated or overwritten column. Rename or drop it first. + +```scala mdoc:fail sc:nocompile +Seq((custId = 1, name = "x")).join(customers)["custId"] +``` + +Unknown keys, and keys whose types disagree, are caught in the same way: + +```scala mdoc:fail sc:nocompile +orders.joinOn(people)["custId", "nope"] +``` + +Worth knowing: + +- The left side streams; the right is read into a hash index the first time the result is pulled. + The right table is therefore the one that has to fit in memory - put the smaller table there. +- An inner join *narrows* an optional key. A `None` key cannot have matched, so `join` on a + `custId: Option[Int]` hands back `custId: Int` - no `.get` that could never have thrown. A + `leftJoin` leaves the key alone, because an unmatched row survives still holding its `None`. +- A `None` key never matches anything - as in SQL, where `NULL = NULL` is never true, and pandas, + which drops missing keys from a merge. `leftJoin` still keeps such a row, with its right hand + columns all `None`. Were `None` to match `None` instead, missing data would square itself: three + missing keys on each side would be nine output rows carrying no information. To match missing to + missing deliberately, map the key to a sentinel first - + `mapColumn["custId", Int](_.getOrElse(-1))`. +- Left order is preserved, and a left row matching several right rows emits them in the right + table's own order. +- Key types must agree exactly. A column carrying a display tag from `formatColumn` will not match + an untagged one - join first, format afterwards. +- Only inner and left joins are built in. A right join is a left join with the tables swapped.