diff --git a/Cargo.lock b/Cargo.lock index 3855a109..5270e5c2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,11 +2,17 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "Inflector" +version = "0.11.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe438c63458706e03479442743baae6c88256498e6431708f6dfc520a26515d3" + [[package]] name = "actix-codec" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f7b0a21988c1bf877cf4759ef5ddaac04c1c9fe808c9142ecb78ba97d97a28a" +checksum = "31404e1443b7b7bcaa311c1af456775cc3f668e34573f1503d7ce4844327629e" dependencies = [ "bitflags", "bytes", @@ -21,12 +27,11 @@ dependencies = [ [[package]] name = "actix-http" -version = "3.13.1" +version = "3.13.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48e2faa3e7418ed780cca54829d32782a4008a077230f67457caa063415e99c2" +checksum = "86d62d1a48894ec9450bcde7ef1e3681205771ff6ea9c61cc5ae031145192787" dependencies = [ "actix-codec", - "actix-rt", "actix-service", "actix-utils", "base64", @@ -49,7 +54,7 @@ dependencies = [ "mime", "percent-encoding", "pin-project-lite", - "rand 0.10.1", + "rand 0.10.2", "sha1 0.11.0", "smallvec", "tokio", @@ -65,7 +70,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e01ed3140b2f8d422c68afa1ed2e85d996ea619c988ac834d255db32138655cb" dependencies = [ "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -85,9 +90,9 @@ dependencies = [ [[package]] name = "actix-rt" -version = "2.11.0" +version = "2.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92589714878ca59a7626ea19734f0e07a6a875197eec751bb5d3f99e64998c63" +checksum = "6a16bf2f19c2ad84842bdfe6f3665620e93197d5c607889bfa3aaac45e762fd1" dependencies = [ "futures-core", "tokio", @@ -95,9 +100,9 @@ dependencies = [ [[package]] name = "actix-server" -version = "2.6.0" +version = "2.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a65064ea4a457eaf07f2fba30b4c695bf43b721790e9530d26cb6f9019ff7502" +checksum = "5d44ae8a6516f4ac7bfc7b61aabcd286104e96b4b24c747ce220832a016056d9" dependencies = [ "actix-rt", "actix-service", @@ -105,7 +110,7 @@ dependencies = [ "futures-core", "futures-util", "mio", - "socket2 0.5.10", + "socket2", "tokio", "tracing", ] @@ -132,9 +137,9 @@ dependencies = [ [[package]] name = "actix-web" -version = "4.14.0" +version = "4.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df09e2d9239703dd64056359c920c7f3fba6535ec61a0059e0f44e095ffe02b4" +checksum = "bbacab3593b6b4f7be815076fc52d60a83c873426824675417e2abdd229e2e36" dependencies = [ "actix-codec", "actix-http", @@ -167,7 +172,7 @@ dependencies = [ "serde_json", "serde_urlencoded", "smallvec", - "socket2 0.6.4", + "socket2", "time", "tracing", "url", @@ -182,7 +187,7 @@ dependencies = [ "actix-router", "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -229,9 +234,9 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] @@ -274,9 +279,9 @@ checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" [[package]] name = "android_system_properties" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" dependencies = [ "libc", ] @@ -339,15 +344,15 @@ dependencies = [ [[package]] name = "anyhow" -version = "1.0.103" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2a4385e2e34eb35d6b3efe798b9eb88096925d87726c0798709bf56d9ed84af3" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "ar_archive_writer" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4087686b4b0a3427190bae57a1d9a478dbb2d40c5dc1bd6e2b6d797913bdd348" +checksum = "73cd58deff2140a0a8eae87e417bd01db68a33e148aa93d1e8cd837e55e312b6" dependencies = [ "object", ] @@ -382,17 +387,11 @@ version = "2.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3d62b7694a562cdf5a74227903507c56ab2cc8bdd1f781ed5cb4cf9c9f810bfc" -[[package]] -name = "arrayref" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76a2e8124351fda1ef8aaaa3bbd7ebbcb486bbcd4225aca0aa0d84bb2db8fecb" - [[package]] name = "arrayvec" -version = "0.7.7" +version = "0.7.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f02882884d3e1bc524fb12c79f107f6ad0e1cfd498c536ffb494301740995dfe" +checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" [[package]] name = "arrow" @@ -534,9 +533,9 @@ checksum = "5a0d5eb3fe25337ff83e8333a08379bdd1540b0961b1c888f6e505d971c198e1" [[package]] name = "arrow-schema" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f633dbfdf39c039ada1bf9e34c694816eb71fbb7dc78f613993b7245e078a1ed" +checksum = "21ca356ad6425cecb6eb7b28e4f659f1ee7880fbb1a16127de7dd62901efee9e" dependencies = [ "serde", "serde_core", @@ -615,18 +614,18 @@ checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.92" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -691,12 +690,12 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06" [[package]] -name = "bijou64" -version = "0.2.1" +name = "bijoux" +version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d50212f5f294273afafebfb0e5b5feca77a03beaca06a9209396db7df69e6766" +checksum = "e0df3f02c7ce18790c3c2f547305fa61c8c9a73425d681bae7f978ccbf5abbc4" dependencies = [ - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -725,9 +724,9 @@ dependencies = [ [[package]] name = "bitflags" -version = "2.13.0" +version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4388bee8683e3d04af747c73422af53102d2bd24d9eadb6cbc100baef4b43f8" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" [[package]] name = "bitvec" @@ -743,16 +742,15 @@ dependencies = [ [[package]] name = "blake3" -version = "1.8.5" +version = "1.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0aa83c34e62843d924f905e0f5c866eb1dd6545fc4d719e803d9ba6030371fce" +checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" dependencies = [ - "arrayref", "arrayvec", "cc", "cfg-if", "constant_time_eq 0.4.2", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", ] [[package]] @@ -825,22 +823,22 @@ dependencies = [ [[package]] name = "bytemuck" -version = "1.25.0" +version = "1.25.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" dependencies = [ "bytemuck_derive", ] [[package]] name = "bytemuck_derive" -version = "1.10.2" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9abbd1bc6865053c427f7198e6af43bfdedc55ab791faed4fbd361d789575ff" +checksum = "fc0e56a716f1e132ff6bf4bdac1c944a3fcdc1cae65f70a4a2a1ac3b401d2d1f" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -851,9 +849,9 @@ checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" [[package]] name = "bytes" -version = "1.12.0" +version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ae3f5d315924270530207e2a68396c3cc547f6dca3fbdca317cfb1a51edb593" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" [[package]] name = "bytestring" @@ -881,9 +879,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" [[package]] name = "cc" -version = "1.2.65" +version = "1.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e228eec9be7c17ccb640b59b36a5cd805ea2a564a4c5e162c2f659fea30d3b96" +checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" dependencies = [ "find-msvc-tools", "jobserver", @@ -899,18 +897,18 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfg_aliases" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" -version = "0.10.1" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" +checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "rand_core 0.10.1", ] @@ -966,9 +964,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.1" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -976,9 +974,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.0" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -989,14 +987,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck 0.5.0", "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -1043,7 +1041,7 @@ version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0fa961b519f0b462e3a3b4a34b64d119eeaca1d59af726fe450bbba07a9fc0a1" dependencies = [ - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -1052,15 +1050,16 @@ version = "0.1.0" dependencies = [ "anyhow", "arrow", - "coln-integrator", ] [[package]] name = "coln-flir-rs" version = "0.1.0" dependencies = [ + "coln-flir-rs", "serde", "serde_json", + "specta", ] [[package]] @@ -1070,7 +1069,7 @@ dependencies = [ "anyhow", "coln-store", "criterion", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -1095,12 +1094,27 @@ version = "0.1.0" dependencies = [ "anyhow", "cli-table", + "coln-batch", "coln-flir-rs", + "coln-query", "criterion", "dbsp", "feldera-size-of", "rkyv", - "thiserror 2.0.18", + "thiserror 2.0.20", +] + +[[package]] +name = "coln-rpc" +version = "0.0.1" +dependencies = [ + "coln-flir-rs", + "coln-store", + "serde", + "serde_json", + "specta", + "specta-serde", + "specta-typescript", ] [[package]] @@ -1126,11 +1140,12 @@ dependencies = [ "serde", "serde_json", "shlex", + "specta", "sqlparser", "subduction_core", "subduction_crypto", "subduction_websocket", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "tracing-subscriber", @@ -1263,9 +1278,9 @@ dependencies = [ [[package]] name = "cpufeatures" -version = "0.3.0" +version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" dependencies = [ "libc", ] @@ -1296,9 +1311,9 @@ dependencies = [ [[package]] name = "crc32fast" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9481c1c90cbf2ac953f07c8d4a58aa3945c425b7185c9154d67a65e4230da511" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" dependencies = [ "cfg-if", ] @@ -1359,18 +1374,18 @@ dependencies = [ [[package]] name = "crossbeam-channel" -version = "0.5.15" +version = "0.5.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2" +checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-deque" -version = "0.8.6" +version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9dd111b7b7f7d55b72c0a6ae361660ee5853c9af73f70c3c2ef6858b950e2e51" +checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb" dependencies = [ "crossbeam-epoch", "crossbeam-utils", @@ -1378,27 +1393,27 @@ dependencies = [ [[package]] name = "crossbeam-epoch" -version = "0.9.18" +version = "0.9.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" +checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-queue" -version = "0.3.12" +version = "0.3.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f58bbc28f91df819d0aa2a2c00cd19754769c2fad90579b3592b1c9ba7a3115" +checksum = "803d13fb3b09d88be9f4dbc29062c66b19bf7170867ceb746d2a8689bf6c7a26" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-utils" -version = "0.8.21" +version = "0.8.22" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" [[package]] name = "crunchy" @@ -1476,7 +1491,7 @@ checksum = "f46882e17999c6cc590af592290432be3bce0428cb0d5f8b6715e4dc7b383eb3" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -1494,9 +1509,9 @@ dependencies = [ [[package]] name = "data-encoding" -version = "2.11.0" +version = "2.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" +checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06" [[package]] name = "dbsp" @@ -1550,7 +1565,7 @@ dependencies = [ "petgraph 0.6.5", "pin-project-lite", "ptr_meta 0.2.0", - "rand 0.8.6", + "rand 0.8.8", "rand_chacha 0.3.1", "rkyv", "rmp-serde", @@ -1564,7 +1579,7 @@ dependencies = [ "static_assertions", "tempfile", "textwrap", - "thiserror 2.0.18", + "thiserror 2.0.20", "thread-id", "time", "tokio", @@ -1598,7 +1613,6 @@ version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" dependencies = [ - "powerfmt", "serde_core", ] @@ -1610,7 +1624,7 @@ checksum = "1e567bd82dcff979e4b03460c307b3cdc9e96fde3d73bed1496d2bc75d9dd62a" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -1639,7 +1653,7 @@ checksum = "cb7330aeadfbe296029522e6c40f315320aba36fc43a5b3632f3795348f3bd22" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "unicode-xid", ] @@ -1653,7 +1667,7 @@ dependencies = [ "proc-macro2", "quote", "rustc_version", - "syn 2.0.118", + "syn 2.0.119", "unicode-xid", ] @@ -1681,13 +1695,13 @@ dependencies = [ [[package]] name = "displaydoc" -version = "0.2.6" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -1723,9 +1737,9 @@ dependencies = [ [[package]] name = "either" -version = "1.16.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91622ff5e7162018101f2fea40d6ebf4a78bbe5a49736a2020649edf9693679e" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "embedded-io" @@ -1780,7 +1794,7 @@ checksum = "f282cfdfe92516eb26c2af8589c274c7c17681f5ecc03c18255fe741c6aa64eb" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -1810,17 +1824,16 @@ dependencies = [ [[package]] name = "error-code" -version = "3.3.2" +version = "3.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dea2df4cf52843e0452895c455a1a2cfbb842a1e7329671acf418fdc53ed4c59" +checksum = "0b5343afd4a8365a643ac588dab4cf234a190c7f6c88c9f6dd6ffe00837661b7" [[package]] name = "event-listener" -version = "5.4.1" +version = "5.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" dependencies = [ - "concurrent-queue", "parking", "pin-project-lite", ] @@ -1843,15 +1856,15 @@ checksum = "4e7f34442dbe69c60fe8eaf58a8cafff81a1f278816d8ab4db255b3bef4ac3c4" dependencies = [ "getrandom 0.3.4", "libm", - "rand 0.9.4", + "rand 0.9.5", "siphasher", ] [[package]] name = "fastrand" -version = "2.4.1" +version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "fdlimit" @@ -1898,7 +1911,7 @@ dependencies = [ "prettyplease", "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -1918,7 +1931,7 @@ dependencies = [ "serde", "serde_json", "serde_json_path_to_error", - "thiserror 2.0.18", + "thiserror 2.0.20", "thread-id", "tokio", "tracing", @@ -1958,7 +1971,7 @@ dependencies = [ "rkyv", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "uuid", @@ -1972,7 +1985,7 @@ checksum = "bbe442bb15bc050c0bcdb391ebef272adb3baae3393ca2df38fb1f190adb259c" dependencies = [ "actix-web", "anyhow", - "arrow-schema 58.3.0", + "arrow-schema 58.4.0", "bytemuck", "chrono", "clap", @@ -1984,7 +1997,7 @@ dependencies = [ "regex", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", "time", "utoipa", "uuid", @@ -1998,9 +2011,9 @@ checksum = "28dea519a9695b9977216879a3ebfddf92f1c08c05d984f8996aecd6ecdc811d" [[package]] name = "find-msvc-tools" -version = "0.1.9" +version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" [[package]] name = "fixedbitset" @@ -2026,9 +2039,9 @@ dependencies = [ [[package]] name = "flate2" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" dependencies = [ "crc32fast", "miniz_oxide", @@ -2101,14 +2114,14 @@ checksum = "6658fc211f4fa8521b72c00efe6626be82dd9d5bde5042137bcb90b02caee84d" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "futures" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" dependencies = [ "futures-channel", "futures-core", @@ -2121,9 +2134,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" dependencies = [ "futures-core", "futures-sink", @@ -2131,15 +2144,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" dependencies = [ "futures-core", "futures-task", @@ -2148,32 +2161,32 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-timer" @@ -2183,9 +2196,9 @@ checksum = "af43fadb8a98512d547e37b4e92e0ced13e205c061b87b4623eff01d918d6968" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-channel", "futures-core", @@ -2228,11 +2241,9 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" dependencies = [ "cfg-if", - "js-sys", "libc", "r-efi 5.3.0", "wasip2", - "wasm-bindgen", ] [[package]] @@ -2242,9 +2253,11 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", + "js-sys", "libc", "r-efi 6.0.0", "rand_core 0.10.1", + "wasm-bindgen", ] [[package]] @@ -2281,16 +2294,16 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.15" +version = "0.4.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" +checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" dependencies = [ "atomic-waker", "bytes", "fnv", "futures-core", "futures-sink", - "http 1.4.2", + "http 1.5.0", "indexmap", "slab", "tokio", @@ -2396,15 +2409,18 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] name = "hermit-abi" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" +checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284" [[package]] name = "hex" version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" +dependencies = [ + "serde", +] [[package]] name = "hexane" @@ -2413,7 +2429,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "81df830f11babb0eee9a082e09135ef0a7d8791f3fe9b279bb9485bf7eaf23e0" dependencies = [ "leb128", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -2447,9 +2463,9 @@ dependencies = [ [[package]] name = "http" -version = "1.4.2" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6970f50e31d6fc17d3fa27329444bfa74e196cf62e95052a3f6fee181dba6425" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -2457,23 +2473,23 @@ dependencies = [ [[package]] name = "http-body" -version = "1.0.1" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", - "http 1.4.2", + "http 1.5.0", ] [[package]] name = "http-body-util" -version = "0.1.3" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b021d93e26becf5dc7e1b75b1bed1fd93124b374ceb73f43d4d4eafec896a64a" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", - "http 1.4.2", + "http 1.5.0", "http-body", "pin-project-lite", ] @@ -2492,31 +2508,31 @@ checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" [[package]] name = "humantime" -version = "2.3.0" +version = "2.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" +checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" [[package]] name = "hybrid-array" -version = "0.4.13" +version = "0.4.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "818356c5132c1fede50f837ca96afbe78ff42413047f4abb886217845e1b6c8c" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" dependencies = [ "typenum", ] [[package]] name = "hyper" -version = "1.10.1" +version = "1.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55281c53a1894c864990125767da440a4e630446785086f52523b20033b74498" +checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" dependencies = [ "atomic-waker", "bytes", "futures-channel", "futures-core", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body", "httparse", "itoa", @@ -2532,7 +2548,7 @@ version = "0.27.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "33ca68d021ef39cf6463ab54c1d0f5daf03377b70561305bb89a8f83aab66e0f" dependencies = [ - "http 1.4.2", + "http 1.5.0", "hyper", "hyper-util", "rustls", @@ -2552,14 +2568,14 @@ dependencies = [ "bytes", "futures-channel", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body", "hyper", "ipnet", "libc", "percent-encoding", "pin-project-lite", - "socket2 0.6.4", + "socket2", "tokio", "tower-service", "tracing", @@ -2591,9 +2607,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" dependencies = [ "displaydoc", "potential_utf", @@ -2605,9 +2621,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" dependencies = [ "displaydoc", "litemap", @@ -2618,9 +2634,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -2632,16 +2648,17 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" [[package]] name = "icu_properties" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" dependencies = [ + "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -2652,15 +2669,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" [[package]] name = "icu_provider" -version = "2.2.0" +version = "2.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" dependencies = [ "displaydoc", "icu_locale_core", @@ -2694,9 +2711,9 @@ dependencies = [ [[package]] name = "impl-more" -version = "0.3.1" +version = "0.3.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35a84fd5aa25fae5c0f4a33d9cac2ca017fc622cbd089be2229993514990f870" +checksum = "277ff51754a3f68f12f58446c5d006aa8baa4914ea273cce24a599cfaff33d4f" [[package]] name = "impl-trait-for-tuples" @@ -2706,14 +2723,14 @@ checksum = "a0eb5a3343abf848c0984fe4604b2b105da9539376e24fc0a3b0007411ae4fd9" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "indexmap" -version = "2.14.0" +version = "2.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb" dependencies = [ "equivalent", "hashbrown 0.17.1", @@ -2741,9 +2758,9 @@ dependencies = [ [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" [[package]] name = "is_terminal_polyfill" @@ -2777,19 +2794,19 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jobserver" -version = "0.1.34" +version = "0.1.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9afb3de4395d6b3e67a780b6de64b51c978ecf11cb9a462c66be7d4ca9039d33" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" dependencies = [ - "getrandom 0.3.4", + "getrandom 0.4.3", "libc", ] [[package]] name = "js-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" dependencies = [ "cfg-if", "futures-util", @@ -2879,9 +2896,9 @@ checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c" [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "libm" @@ -2897,9 +2914,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" [[package]] name = "litemap" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" [[package]] name = "local-channel" @@ -2929,9 +2946,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.33" +version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" [[package]] name = "lru-slab" @@ -2976,9 +2993,9 @@ dependencies = [ [[package]] name = "memchr" -version = "2.8.2" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88904434abc2901f197fe8cc55f0445e7ded921dba5911dad2e2b39b48e663c4" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] name = "memory-stats" @@ -3018,9 +3035,9 @@ checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" [[package]] name = "miniz_oxide" -version = "0.8.9" +version = "0.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" dependencies = [ "adler2", "simd-adler32", @@ -3028,9 +3045,9 @@ dependencies = [ [[package]] name = "mio" -version = "1.2.1" +version = "1.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "02bd0af71c67b473010cbbc60715ee815645a4dc942899111f494b4b737d6fda" +checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" dependencies = [ "libc", "log", @@ -3130,9 +3147,9 @@ dependencies = [ [[package]] name = "num-bigint" -version = "0.4.6" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a5e44f723f1133c9deac646763579fdb3ac745e418f2a7af9cd0c431da1f20b9" +checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" dependencies = [ "num-integer", "num-traits", @@ -3161,7 +3178,7 @@ checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -3176,20 +3193,19 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.46" +version = "0.1.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" dependencies = [ "num-traits", ] [[package]] name = "num-iter" -version = "0.1.45" +version = "0.1.46" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1429034a0490724d0075ebb2bc9e875d6503c3cf69e235a8941aa757d83ef5bf" +checksum = "c92800bd69a1eac91786bcfe9da64a897eb72911b8dc3095decbd07429e8048b" dependencies = [ - "autocfg", "num-integer", "num-traits", ] @@ -3227,9 +3243,9 @@ dependencies = [ [[package]] name = "object" -version = "0.37.3" +version = "0.39.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff76201f031d8863c38aa7f905eca4f53abbfa15f609db4277d44cd8938f33fe" +checksum = "2e5a6c098c7a3b6547378093f5cc30bc54fd361ce711e05293a5cc589562739b" dependencies = [ "memchr", ] @@ -3246,7 +3262,7 @@ dependencies = [ "chrono", "form_urlencoded", "futures", - "http 1.4.2", + "http 1.5.0", "http-body-util", "httparse", "humantime", @@ -3256,14 +3272,14 @@ dependencies = [ "parking_lot", "percent-encoding", "quick-xml", - "rand 0.9.4", + "rand 0.9.5", "reqwest", "ring", "rustls-pemfile", "serde", "serde_json", "serde_urlencoded", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -3312,7 +3328,7 @@ checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -3349,7 +3365,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7bb71e1b3fa6ca1c61f383464aaf2bb0e2f8e772a1f01d486832464de363b951" dependencies = [ "num-traits", - "rand 0.8.6", + "rand 0.8.8", "rkyv", "serde", ] @@ -3375,7 +3391,7 @@ dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -3485,9 +3501,9 @@ dependencies = [ [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "plotters" @@ -3519,9 +3535,9 @@ dependencies = [ [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "postcard" @@ -3538,9 +3554,9 @@ dependencies = [ [[package]] name = "potential_utf" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" dependencies = [ "zerovec", ] @@ -3553,9 +3569,9 @@ checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" [[package]] name = "ppmd-rust" -version = "1.4.0" +version = "1.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "efca4c95a19a79d1c98f791f10aebd5c1363b473244630bb7dbde1dc98455a24" +checksum = "9e9219bcb9d7aca6b2f63c83cf100cf78bcd619ac46e6ecbd0dd90869a39345d" [[package]] name = "ppv-lite86" @@ -3573,7 +3589,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" dependencies = [ "proc-macro2", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -3602,9 +3618,9 @@ dependencies = [ [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -3617,16 +3633,16 @@ checksum = "af066a9c399a26e020ada66a034357a868728e72cd426f3adcd35f80d88d88c8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "version_check", "yansi", ] [[package]] name = "psm" -version = "0.1.31" +version = "0.1.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "645dbe486e346d9b5de3ef16ede18c26e6c70ad97418f4874b8b1889d6e761ea" +checksum = "4dcd034599e63b970727f70d79e02d62390a4a84f7c6b827c27c46d5ac3fa622" dependencies = [ "ar_archive_writer", "cc", @@ -3707,8 +3723,8 @@ dependencies = [ "quinn-udp", "rustc-hash", "rustls", - "socket2 0.6.4", - "thiserror 2.0.18", + "socket2", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -3716,20 +3732,21 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.15" +version = "0.11.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4fcb935c5bec503c2f0e306bdd3e58bb9029dcb14fa8d9ac76e3a5256ac0763e" +checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83" dependencies = [ "bytes", - "getrandom 0.3.4", + "getrandom 0.4.3", "lru-slab", - "rand 0.9.4", + "rand 0.10.2", + "rand_pcg", "ring", "rustc-hash", "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -3737,23 +3754,23 @@ dependencies = [ [[package]] name = "quinn-udp" -version = "0.5.14" +version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "addec6a0dcad8a8d96a771f815f0eaf55f9d1805756410b39f5fa81332574cbd" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" dependencies = [ "cfg_aliases", "libc", "once_cell", - "socket2 0.6.4", + "socket2", "tracing", - "windows-sys 0.60.2", + "windows-sys 0.61.2", ] [[package]] name = "quote" -version = "1.0.46" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfbc457d0c7a0759a614551b11a6409e5951f6c7537be1f1b7682b9ae9230368" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -3788,9 +3805,9 @@ dependencies = [ [[package]] name = "rand" -version = "0.8.6" +version = "0.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca0ecfa931c29007047d1bc58e623ab12e5590e8c7cc53200d5202b69266d8a" +checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" dependencies = [ "libc", "rand_chacha 0.3.1", @@ -3800,9 +3817,9 @@ dependencies = [ [[package]] name = "rand" -version = "0.9.4" +version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" dependencies = [ "rand_chacha 0.9.0", "rand_core 0.9.5", @@ -3810,9 +3827,9 @@ dependencies = [ [[package]] name = "rand" -version = "0.10.1" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2e8e8bcc7961af1fdac401278c6a831614941f6164ee3bf4ce61b7edb162207" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" dependencies = [ "chacha20", "getrandom 0.4.3", @@ -3864,6 +3881,15 @@ version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + [[package]] name = "rayon" version = "1.12.0" @@ -3901,7 +3927,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "76009fbe0614077fc1a2ce255e3a1881a2e3a3527097d5dc6d8212c585e7e38b" dependencies = [ "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -3915,9 +3941,9 @@ dependencies = [ [[package]] name = "regex" -version = "1.12.4" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -3927,9 +3953,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -3967,8 +3993,8 @@ dependencies = [ "bytes", "futures-core", "futures-util", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body", "http-body-util", "hyper", @@ -4073,9 +4099,9 @@ dependencies = [ [[package]] name = "rustc-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" [[package]] name = "rustc_version" @@ -4101,9 +4127,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.41" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b92b125634d9b795e7beca796cc790df15a7fb38323bf3196fda83292d06b1f" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "once_cell", "ring", @@ -4136,9 +4162,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.15.0" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "764899a24af3980067ee14bc143654f297b22eaebfe3c7b6b211920a5a59b046" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -4146,9 +4172,9 @@ dependencies = [ [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "ring", "rustls-pki-types", @@ -4157,9 +4183,9 @@ dependencies = [ [[package]] name = "rustversion" -version = "1.0.22" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" [[package]] name = "rustyline" @@ -4243,12 +4269,12 @@ dependencies = [ [[package]] name = "sedimentree_core" -version = "0.13.0" -source = "git+https://github.com/inkandswitch/subduction#03d322c0313578e232d989697d0f47f772083f25" +version = "0.14.0" +source = "git+https://github.com/inkandswitch/subduction#61d1030f805b469ee9df39bfbc944c33f5605058" dependencies = [ "arbitrary", "async-lock", - "bijou64", + "bijoux", "blake3", "future_form", "futures", @@ -4256,10 +4282,10 @@ dependencies = [ "getrandom 0.2.17", "nonempty", "num-bigint", - "rand 0.8.6", + "rand 0.8.8", "serde", "siphasher", - "thiserror 2.0.18", + "thiserror 2.0.20", "tracing", ] @@ -4277,9 +4303,9 @@ checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -4298,22 +4324,22 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -4324,14 +4350,14 @@ checksum = "18d26a20a969b9e3fdf2fc2d9f21eda6c40e2de84c9408bb5d3b05d499aae711" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "serde_json" -version = "1.0.150" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "itoa", "memchr", @@ -4376,9 +4402,9 @@ dependencies = [ [[package]] name = "sha1" -version = "0.10.6" +version = "0.10.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3bf829a2d51ab4a5ddf1352d8470c140cadc8301b2ae1789db023f01cedd6ba" +checksum = "a978451301f4db1d02937a4ab3ccce137717b81826e79b7d49ffe3244a13c3b8" dependencies = [ "cfg-if", "cpufeatures 0.2.17", @@ -4392,7 +4418,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "aacc4cc499359472b4abe1bf11d0b12e688af9a805fa5e3016f9a386dc2d0214" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "digest 0.11.3", ] @@ -4443,9 +4469,9 @@ dependencies = [ [[package]] name = "simd-adler32" -version = "0.3.9" +version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "703d5c7ef118737c72f1af64ad2f6f8c5e1921f818cdcb97b8fe6fc69bf66214" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" [[package]] name = "simdutf8" @@ -4478,9 +4504,9 @@ checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" [[package]] name = "smallvec" -version = "1.15.2" +version = "1.16.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" +checksum = "b9be42f50aa861c555654aa3a37f52f4b1074bacf4e48fe0ef7fa584e80f1f0f" dependencies = [ "serde", ] @@ -4493,35 +4519,67 @@ checksum = "e8e2fb0f499abb4d162f2bedad68f5ef91a1682b5a03596ddb67efd37768d100" [[package]] name = "snap" -version = "1.1.1" +version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" [[package]] name = "socket2" -version = "0.5.10" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] -name = "socket2" -version = "0.6.4" +name = "specta" +version = "2.0.0-rc.25" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52d1cfed4120b4d927bf7c0f86d2087a4a7d6027c906d9f9d525a80573b9be51" +checksum = "38f9a30cbcbb7011f1da7d73483983bf838af123883e45f2b36ed76328df9c50" dependencies = [ - "libc", - "windows-sys 0.61.2", + "rustc_version", + "specta-macros", +] + +[[package]] +name = "specta-macros" +version = "2.0.0-rc.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ce14957ecc2897f1f848b8255b6531d13ddf49cbcf506b7c2c9fb1d005593bb" +dependencies = [ + "Inflector", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "specta-serde" +version = "0.0.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee8a72b755ddb8949fd8f17c5db43f0e8a806ea587d9bc602ee3f73240c00029" +dependencies = [ + "specta", + "specta-macros", +] + +[[package]] +name = "specta-typescript" +version = "0.0.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "639404ee95557f2f8b7e4cb773ffefd45304c7ab8ba21ac83b69051595e083c0" +dependencies = [ + "serde", + "specta", ] [[package]] name = "spin" -version = "0.9.8" +version = "0.9.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" +checksum = "3763264f6b73151db08c50ff20d7d8a0b8796e021cdea7ceedad07b80155fa0e" dependencies = [ "lock_api", ] @@ -4554,9 +4612,9 @@ checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" [[package]] name = "stacker" -version = "0.1.24" +version = "0.1.25" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "640c8cdd92b6b12f5bcb1803ca3bbf5ab96e5e6b6b96b9ab77dabe9e880b3190" +checksum = "707f49d46706bacf8a2b00d51dace3f9de527c13eec3778f570c411f89e69967" dependencies = [ "cc", "cfg-if", @@ -4579,12 +4637,12 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" [[package]] name = "subduction_core" -version = "0.16.0" -source = "git+https://github.com/inkandswitch/subduction#03d322c0313578e232d989697d0f47f772083f25" +version = "0.18.0" +source = "git+https://github.com/inkandswitch/subduction#61d1030f805b469ee9df39bfbc944c33f5605058" dependencies = [ "async-channel", "async-lock", - "bijou64", + "bijoux", "blake3", "ed25519-dalek", "future_form", @@ -4594,15 +4652,15 @@ dependencies = [ "sedimentree_core", "siphasher", "subduction_crypto", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", ] [[package]] name = "subduction_crypto" -version = "0.8.0" -source = "git+https://github.com/inkandswitch/subduction#03d322c0313578e232d989697d0f47f772083f25" +version = "0.9.0" +source = "git+https://github.com/inkandswitch/subduction#61d1030f805b469ee9df39bfbc944c33f5605058" dependencies = [ "ed25519-dalek", "future_form", @@ -4610,13 +4668,13 @@ dependencies = [ "getrandom 0.2.17", "sedimentree_core", "serde", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] name = "subduction_websocket" -version = "0.11.0" -source = "git+https://github.com/inkandswitch/subduction#03d322c0313578e232d989697d0f47f772083f25" +version = "0.12.0" +source = "git+https://github.com/inkandswitch/subduction#61d1030f805b469ee9df39bfbc944c33f5605058" dependencies = [ "async-channel", "async-lock", @@ -4626,11 +4684,11 @@ dependencies = [ "futures", "futures-timer", "futures-util", - "rand 0.8.6", + "rand 0.8.8", "sedimentree_core", "subduction_core", "subduction_crypto", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tokio-util", "tracing", @@ -4656,9 +4714,20 @@ dependencies = [ [[package]] name = "syn" -version = "2.0.118" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b9ae57f904213ebb649ce6895b8a66c66f0203b9319718f69a5612a065b1422" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -4682,7 +4751,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -4745,11 +4814,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.20", ] [[package]] @@ -4760,18 +4829,18 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -4786,21 +4855,20 @@ dependencies = [ [[package]] name = "thread_local" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f60246a4944f24f6e018aa17cdeffb7818b76356965d03b07d6a9886e8962185" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" dependencies = [ "cfg-if", ] [[package]] name = "time" -version = "0.3.47" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "743bd48c283afc0388f9b8827b976905fb217ad9e647fae3a379a9283c4def2c" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", - "itoa", "num-conv", "powerfmt", "serde_core", @@ -4810,15 +4878,15 @@ dependencies = [ [[package]] name = "time-core" -version = "0.1.8" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7694e1cfe791f8d31026952abf09c69ca6f6fa4e1a1229e18988f06a04a12dca" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" [[package]] name = "time-macros" -version = "0.2.27" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e70e4c5a0e0a8a4823ad65dfe1a6930e4f4d756dcd9dd7939022b5e8c501215" +checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" dependencies = [ "num-conv", "time-core", @@ -4835,9 +4903,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" dependencies = [ "displaydoc", "zerovec", @@ -4855,9 +4923,9 @@ dependencies = [ [[package]] name = "tinyvec" -version = "1.11.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e61e67053d25a4e82c844e8424039d9745781b3fc4f32b8d55ed50f5f667ef3" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" dependencies = [ "tinyvec_macros", ] @@ -4870,9 +4938,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" [[package]] name = "tokio" -version = "1.52.3" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -4880,20 +4948,20 @@ dependencies = [ "parking_lot", "pin-project-lite", "signal-hook-registry", - "socket2 0.6.4", + "socket2", "tokio-macros", "windows-sys 0.61.2", ] [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -4918,14 +4986,15 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-sink", "futures-util", + "libc", "pin-project-lite", "tokio", ] @@ -4954,7 +5023,7 @@ dependencies = [ "bitflags", "bytes", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body", "pin-project-lite", "tower", @@ -4995,7 +5064,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -5045,9 +5114,9 @@ checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" [[package]] name = "tsify" -version = "0.5.6" +version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ec5505497c87f1c050b4392d3f11b49a04537fcb9dc0da57bc0af168a6331f2" +checksum = "ea3af6b48c5c5624f7eb3c2f578d1335fadac433321980cae0063e8eb0acaa44" dependencies = [ "gloo-utils", "serde", @@ -5058,14 +5127,14 @@ dependencies = [ [[package]] name = "tsify-macros" -version = "0.5.6" +version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fc2c44dc9fe4baf55b88e032621b7a11b215a1f0a7de8d0aa04367207d915bc" +checksum = "ef1e2e93369379d0a527c7d6632b58c55d52db7a106d8afbde3b354f3a57a9f1" dependencies = [ "proc-macro2", "quote", "serde_derive_internals", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -5076,13 +5145,13 @@ checksum = "eadc29d668c91fcc564941132e17b28a7ceb2f3ebf0b9dae3e03fd7a6748eb0d" dependencies = [ "bytes", "data-encoding", - "http 1.4.2", + "http 1.5.0", "httparse", "log", "native-tls", - "rand 0.9.4", - "sha1 0.10.6", - "thiserror 2.0.18", + "rand 0.9.5", + "sha1 0.10.7", + "thiserror 2.0.20", "utf-8", ] @@ -5194,15 +5263,15 @@ dependencies = [ "proc-macro-error", "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "uuid", ] [[package]] name = "uuid" -version = "1.23.4" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf80a72845275afea99e7f2b434723d3bc7e38470fcd1c7ed39a599c73319a53" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" dependencies = [ "getrandom 0.4.3", "js-sys", @@ -5264,9 +5333,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" dependencies = [ "cfg-if", "once_cell", @@ -5277,9 +5346,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.76" +version = "0.4.77" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" +checksum = "6b7777d5cc23d0e91404e53ce2d5e8ec7acae3026b16233dba62cd3246457950" dependencies = [ "js-sys", "wasm-bindgen", @@ -5287,9 +5356,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -5297,22 +5366,22 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "wasm-bindgen-shared", ] [[package]] name = "wasm-bindgen-shared" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" dependencies = [ "unicode-ident", ] @@ -5332,9 +5401,9 @@ dependencies = [ [[package]] name = "web-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" +checksum = "c435338968042f4f59a557f690a253676d47ce13ceb55d70100e7facf6620a30" dependencies = [ "js-sys", "wasm-bindgen", @@ -5402,7 +5471,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -5413,7 +5482,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -5446,16 +5515,7 @@ version = "0.52.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" dependencies = [ - "windows-targets 0.52.6", -] - -[[package]] -name = "windows-sys" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2f500e4d28234f72040990ec9d39e3a6b950f9f22d3dba18416c35882612bcb" -dependencies = [ - "windows-targets 0.53.5", + "windows-targets", ] [[package]] @@ -5473,31 +5533,14 @@ version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" dependencies = [ - "windows_aarch64_gnullvm 0.52.6", - "windows_aarch64_msvc 0.52.6", - "windows_i686_gnu 0.52.6", - "windows_i686_gnullvm 0.52.6", - "windows_i686_msvc 0.52.6", - "windows_x86_64_gnu 0.52.6", - "windows_x86_64_gnullvm 0.52.6", - "windows_x86_64_msvc 0.52.6", -] - -[[package]] -name = "windows-targets" -version = "0.53.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4945f9f551b88e0d65f3db0bc25c33b8acea4d9e41163edf90dcd0b19f9069f3" -dependencies = [ - "windows-link", - "windows_aarch64_gnullvm 0.53.1", - "windows_aarch64_msvc 0.53.1", - "windows_i686_gnu 0.53.1", - "windows_i686_gnullvm 0.53.1", - "windows_i686_msvc 0.53.1", - "windows_x86_64_gnu 0.53.1", - "windows_x86_64_gnullvm 0.53.1", - "windows_x86_64_msvc 0.53.1", + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", ] [[package]] @@ -5506,96 +5549,48 @@ version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" - [[package]] name = "windows_aarch64_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" -[[package]] -name = "windows_aarch64_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" - [[package]] name = "windows_i686_gnu" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" -[[package]] -name = "windows_i686_gnu" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "960e6da069d81e09becb0ca57a65220ddff016ff2d6af6a223cf372a506593a3" - [[package]] name = "windows_i686_gnullvm" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" -[[package]] -name = "windows_i686_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" - [[package]] name = "windows_i686_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" -[[package]] -name = "windows_i686_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" - [[package]] name = "windows_x86_64_gnu" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" -[[package]] -name = "windows_x86_64_gnu" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" - [[package]] name = "windows_x86_64_gnullvm" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" - [[package]] name = "windows_x86_64_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" -[[package]] -name = "windows_x86_64_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650" - [[package]] name = "wit-bindgen" version = "0.57.1" @@ -5604,9 +5599,9 @@ checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" [[package]] name = "writeable" -version = "0.6.3" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" [[package]] name = "wyz" @@ -5619,9 +5614,9 @@ dependencies = [ [[package]] name = "xxhash-rust" -version = "0.8.15" +version = "0.8.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fdd20c5420375476fbd4394763288da7eb0cc0b8c11deed431a91562af7335d3" +checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6" [[package]] name = "yansi" @@ -5648,28 +5643,28 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "synstructure", ] [[package]] name = "zerocopy" -version = "0.8.52" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce1022995ff5ff5d841ad7d994facc23098cd40152f2c1d11cd607c6f530653f" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.52" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ae7f38b72ec2a254e2b87ef277cf2cd4fb97cbebf944faa6f33354da0867930" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] @@ -5689,7 +5684,7 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", "synstructure", ] @@ -5710,14 +5705,14 @@ checksum = "3c50655cbb0fe3fc43170059e702f1ce5e19b84cec58dc87b037a09935c2f328" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 2.0.119", ] [[package]] name = "zerotrie" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" dependencies = [ "displaydoc", "yoke", @@ -5726,9 +5721,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.6" +version = "0.11.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" dependencies = [ "yoke", "zerofrom", @@ -5737,13 +5732,13 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.3" +version = "0.11.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn 3.0.4", ] [[package]] @@ -5766,7 +5761,7 @@ dependencies = [ "memchr", "pbkdf2", "ppmd-rust", - "sha1 0.10.6", + "sha1 0.10.7", "time", "zeroize", "zopfli", @@ -5775,15 +5770,15 @@ dependencies = [ [[package]] name = "zlib-rs" -version = "0.6.4" +version = "0.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "977347db8caa080403f6b6b7c1cda9479a8e869316f7e13a59b19076a40f94e3" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" [[package]] name = "zmij" -version = "1.0.21" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" [[package]] name = "zopfli" diff --git a/Cargo.toml b/Cargo.toml index 480deff6..61fb7bda 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -5,7 +5,9 @@ members = [ "packages/coln-store", "packages/coln-query", "packages/coln-integrator", - "packages/coln-batch" + "packages/coln-batch", + "packages/coln-batch", + "packages/coln-rpc" ] resolver = "3" @@ -22,6 +24,7 @@ anyhow = "1.0.102" criterion = "0.8.2" serde = { version = "1.0", features = ["derive"] } serde_json = "1.0.149" +specta = { version = "2.0.0-rc.25", features = ["derive"] } thiserror = "2.0.18" tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter", "fmt"] } diff --git a/packages/coln-batch/Cargo.toml b/packages/coln-batch/Cargo.toml index d9b102de..c1e855cc 100644 --- a/packages/coln-batch/Cargo.toml +++ b/packages/coln-batch/Cargo.toml @@ -10,4 +10,3 @@ readme = "README.md" [dependencies] anyhow = { workspace = true } arrow = { version = "56", default-features = false, features = ["ipc"] } -coln-integrator = { path = "../coln-integrator" } diff --git a/packages/coln-flir-rs/Cargo.toml b/packages/coln-flir-rs/Cargo.toml index 7e6b0bb5..16f344e4 100644 --- a/packages/coln-flir-rs/Cargo.toml +++ b/packages/coln-flir-rs/Cargo.toml @@ -12,8 +12,15 @@ exclude = ["/.gitignore"] [lib] +[features] +test-utils = [] + [dependencies] -serde = { version = "1.0", features = ["derive"] } +serde.workspace = true +serde_json.workspace = true +specta.workspace = true [dev-dependencies] -serde_json = "1.0" +# This is a self-referential dev-dependency to have the feature-gated +# test-utils be available in this crate's integration tests, too. +coln-flir-rs = { path = ".", features = ["test-utils"] } diff --git a/packages/coln-flir-rs/src/ir/mod.rs b/packages/coln-flir-rs/src/ir/mod.rs index 5a146e15..75d39ebd 100644 --- a/packages/coln-flir-rs/src/ir/mod.rs +++ b/packages/coln-flir-rs/src/ir/mod.rs @@ -6,18 +6,28 @@ pub mod path; use serde::de::Error as DeError; use serde::{Deserialize, Deserializer, Serialize, Serializer}; +use specta::Type; // A QName is a vec of string, potentially separated by a forward slash / pub type QName = Vec; // For example a G.V would become [["G"], ["V"]], this is at a higher level than // QName because V would be a query inside a theory G -#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize, Type)] #[serde(transparent)] pub struct Path(pub Vec); -type ColName = Path; -pub type FId = i64; +/// A column name is given by a [`Path`]. +pub type ColName = Path; + +/// An index into the [`varNames`](Rule::var_names) and +/// [`varTypes`](Rule::var_types) arrays of a [`Rule`]. +/// +/// Note: An `FId` in `coln-compiler`. +pub type VarIdx = u64; + +/// An index into a relation's physical [`columns`](Schema::columns). +pub type ColumnIdx = u64; #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum BuiltinTy { @@ -62,9 +72,10 @@ impl<'de> Deserialize<'de> for BuiltinTy { #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "tag", rename_all = "camelCase")] pub enum ColType { - RowId { - path: Path, - }, + /// A foreign key into another table by referencing its _row id_ through + /// the provided path. + RowId { path: Path }, + /// A data column with the scalar type [`BuiltinTy`]. #[serde(rename = "builtin")] BuiltinTy { #[serde(rename = "type")] @@ -89,8 +100,11 @@ pub enum IndexMethod { #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "tag", rename_all = "camelCase")] pub enum EntityVariant { + /// A base table of the extensional database (EDB). Table, + /// A derived view of the intensional database (IDB). View(Materialization), + /// Tell `coln-store` to create an index and possibly hint to `coln-query`. Index { method: IndexMethod, columns: Vec, @@ -105,20 +119,30 @@ pub struct ColumnEntry { pub col_type: ColType, } -// This is really Entity on the Haskell IR side, but I feel schema matches it better +/// Describes a schema of a relation. +/// +/// Note: An `Entity` in `coln-compiler`. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub struct Schema { pub entity_variant: EntityVariant, + /// The columns of the table in their physical order. pub columns: Vec, + /// A `None` indicates that there is no primary key. `Some(vec![])` means + /// that there is at most one row in the table. `Some(vec![ColA, ColB])` + /// encodes a compound primary key consisting of the columns `ColA` and + /// `ColB`. + /// + /// At the moment there is only support for a single (compound) primary key. pub primary_key: Option>, } +/// A literal expression. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "tag", rename_all = "lowercase")] pub enum Lit { #[serde(rename = "int")] - Int { value: i64 }, + Int { value: i32 }, #[serde(rename = "string")] String { value: String }, } @@ -127,50 +151,87 @@ pub enum Lit { #[serde(tag = "tag", rename_all = "lowercase")] pub enum Term { Lit { lit: Lit }, - Var { index: FId }, + Var { index: VarIdx }, } #[derive(Debug, Clone, Serialize, Deserialize)] pub struct ValueEntry { - pub column: i64, + pub column: ColumnIdx, pub term: Term, } +/// An [`Atom`] references an entity (a relation or a table) to bring some of +/// its fields into the scope of a [`Rule`]. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub struct Atom { + /// The "name" of the entity being referenced by this [`Atom`]. pub entity: Path, + /// To bring the `row_id` of the [`Entity`](Self::entity) into scope. + /// + /// Note: A [`Some(Term::Lit)`](Term::Lit) does not make sense in this + /// context, as we do not support a row id literal at the moment, I suppose. pub row_id: Option, + /// To bring some columns of the [`Entity`](Self::entity) into scope. pub values: Vec, } #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(tag = "tag", rename_all = "lowercase")] pub enum Prop { - Atom { atom: Atom }, - Eq { left: Term, right: Term }, + Atom { + atom: Atom, + }, + Eq { + #[serde(flatten)] + equality: Equality, + }, } +/// An equality condition between the left and the right term, that is, +/// we assert `left == right`. #[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Equality { + pub left: Term, + pub right: Term, +} + +#[derive(Debug, Copy, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub enum RuleVariant { + /// _Chased_ rules are not yet fully alive but become relevant once initial + /// models land. Chased, + /// Violations of _enforced_ rules cause a transaction to abort. Enforced, + /// Violations of _monitored_ rules are just reported back to the user but + /// still allow a transaction to commit. Monitored, } +/// A `Rule` is an implication and must be true in all valid states of +/// `coln-store` and `coln-query`. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub struct Rule { pub rule_variant: RuleVariant, + /// Assigns some names to the variables the rule binds. + /// + /// Note: Must be of the same arity as [`Self::var_types`]. pub var_names: Vec, + /// Tells the types of the variables the rule binds. + /// + /// Note: Must be of the same arity as [`Self::var_names`]. pub var_types: Vec, + /// The left-hand side of the implication. pub antecedents: Vec, + /// The right-hand side of the implication. pub consequents: Vec, } #[derive(Debug, Clone, Serialize, Deserialize)] pub struct TableEntry { + /// The "name" of the table. pub path: Path, #[serde(rename = "value")] pub table: Schema, @@ -178,14 +239,18 @@ pub struct TableEntry { #[derive(Debug, Clone, Serialize, Deserialize)] pub struct RuleEntry { + /// The "name" of the rule. pub path: Path, #[serde(rename = "value")] pub rule: Rule, } +/// The top-level type of a flattened realm and the starting point of the FLIR. #[derive(Debug, Clone, Serialize, Deserialize)] pub struct FlatRealm { + /// The tables of the flattened realm. #[serde(rename = "entities")] pub tables: Vec, + /// The rules (laws) of the flattened realm. pub rules: Vec, } diff --git a/packages/coln-flir-rs/src/ir/path.rs b/packages/coln-flir-rs/src/ir/path.rs index 792eb116..9347a6e0 100644 --- a/packages/coln-flir-rs/src/ir/path.rs +++ b/packages/coln-flir-rs/src/ir/path.rs @@ -19,6 +19,25 @@ impl Deref for Path { } } +impl Path { + pub fn append(mut self, name: &str) -> Self { + self.0.push(vec![name.to_string()]); + self + } +} + +impl From for String { + fn from(value: Path) -> Self { + value.to_string() + } +} + +impl From<&Path> for String { + fn from(value: &Path) -> Self { + value.to_string() + } +} + impl Display for Path { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { for (i, qname) in self.0.iter().enumerate() { diff --git a/packages/coln-flir-rs/src/lib.rs b/packages/coln-flir-rs/src/lib.rs index 7eea2a82..171dfcf7 100644 --- a/packages/coln-flir-rs/src/lib.rs +++ b/packages/coln-flir-rs/src/lib.rs @@ -3,3 +3,6 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT pub mod ir; +pub mod schema; +#[cfg(feature = "test-utils")] +pub mod test_utils; diff --git a/packages/coln-flir-rs/src/schema.rs b/packages/coln-flir-rs/src/schema.rs new file mode 100644 index 00000000..ede59de2 --- /dev/null +++ b/packages/coln-flir-rs/src/schema.rs @@ -0,0 +1,379 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! This module expresses the different schema views according to coln-compiler, +//! coln-store, and coln-query in code. + +use crate::ir::{self, Path}; +use std::ops::Range; + +#[derive(Debug)] +pub struct BaseTableSchema { + /// The table's unique identifier/name. + name: ir::Path, + /// Fields of the table in their physical order from the perspective + /// of the _compiler_. The columns _do not_ include the implicit row id. + cols_compiler: CompilerCols, + /// Fields of the table in their physical order from the perspective + /// of the _storage engine_. + cols_store: StoreEngineCols, + /// Fields of the table in their physical order from the perspective + /// of the _query engine_. + cols_query: QueryEngineCols, + /// The list of (possibly compound) primary keys into the table, specified + /// as indices into the [compiler view](Self::cols_compiler). + primary_keys: Vec>, +} + +impl BaseTableSchema { + /// The name of the base table. + pub fn name(&self) -> &ir::Path { + &self.name + } + /// Returns [`None`] if `idx` is an index for an implicit row id. + pub fn get_compiler_col(&self, idx: CompilerColIdx) -> Option<&CompilerCol> { + match idx { + CompilerColIdx::RowId => None, + CompilerColIdx::Column(idx) => Some(&self.cols_compiler.0[idx as usize]), + } + } + pub fn get_storage_col(&self, idx: StoreEngineColIdx) -> &StoreEngineCol { + &self.cols_store.0[idx.0] + } + pub fn query_cols(&self) -> &QueryEngineCols { + &self.cols_query + } + pub fn get_query_col(&self, idx: QueryEngineColIdx) -> &QueryEngineCol { + &self.cols_query.0[idx.0] + } + /// Given a [`CompilerColIdx`] from the FLIR, indexing into the columns of + /// the compiler view, what are the corresponding column(s) according to the + /// query engine's view? This translation is necessary because row ids + /// flatten into two columns from the perspective of the query engine, + /// hence, a (compiler) index resolving to a row id column can result in + /// two columns. A (compiler) index to a non row id column results in + /// exactly one column. + pub fn resolve_query_cols(&self, idx: CompilerColIdx) -> impl Iterator { + self.resolve_query_col_range(idx) + .map(|idx| &self.cols_query.0[idx]) + } + /// The same translation as [`resolve_query_cols`](Self::resolve_query_cols), + /// but as the *indices* into the query engine's column view rather than the + /// columns themselves. What a consumer restating one of this table's + /// [`primary_keys`](Self::primary_keys) in the query engine's view needs, + /// since a key is a list of column positions. + pub fn resolve_query_col_range(&self, idx: CompilerColIdx) -> Range { + match idx { + CompilerColIdx::RowId => 0..StoreEngineCols::ROW_ID_COLS, + CompilerColIdx::Column(target_idx) => { + assert!( + (target_idx as usize) < self.cols_compiler.0.len(), + "Compiler idx out of bounds" + ); + // We account for the implicit row id columns by offsetting. + let mut query_idx = StoreEngineCols::ROW_ID_COLS; + let mut iter = self.cols_compiler.0.iter().enumerate(); + let target_col = loop { + let (idx, col) = iter.next().unwrap(); + if idx >= target_idx as usize { + break col; + } + match &col.ty { + // A column of a native scalar type also takes just one column. + ir::ColType::BuiltinTy { builtin_ty: _ } => query_idx += 1, + // A row id flattens into multiple columns in the query engine's + // view, so we have to advance more columns. + ir::ColType::RowId { path: _ } => query_idx += StoreEngineCols::ROW_ID_COLS, + }; + }; + match &target_col.ty { + ir::ColType::BuiltinTy { builtin_ty: _ } => query_idx..query_idx + 1, + ir::ColType::RowId { path: _ } => query_idx..query_idx + 2, + } + } + } + } + /// The list of (compound) primary key(s), given as indexes into the + /// compiler's column view. + /// + /// Hint: Compiler indexes can be converted into other views using the + /// [`resolve_*`](Self::resolve_query_cols) methods. + pub fn primary_keys(&self) -> &Vec> { + &self.primary_keys + } +} + +impl From<&ir::TableEntry> for Option { + fn from(value: &ir::TableEntry) -> Self { + let path = &value.path; + let schema = &value.table; + if !matches!(schema.entity_variant, ir::EntityVariant::Table) { + return None; // Only base tables allowed. + } + let columns_compiler = CompilerCols::from(schema.columns.as_slice()); + let columns_store = StoreEngineCols::from(columns_compiler.0.as_slice()); + let columns_query = QueryEngineCols::from(columns_store.0.as_slice()); + let primary_key = schema + .primary_key + .as_ref() + // Currently, `null` in JSON becomes the empty vector. + .map_or(Vec::new(), |compound_primary_key| { + compound_primary_key + .iter() + .map(|primary_key_column| { + schema + .columns + .iter() + .position(|column| column.path == *primary_key_column) + .map(|idx| CompilerColIdx::Column(idx as u64)) + .unwrap_or_else(|| panic!("Primary key column {primary_key_column} not found in base table {path}")) + }) + .collect::>() + }); + // Currently, the compiler supports only a single primary key. + let primary_keys = vec![primary_key]; + Some(BaseTableSchema { + name: path.clone(), + cols_compiler: columns_compiler, + cols_store: columns_store, + cols_query: columns_query, + primary_keys, + }) + } +} + +// Scalar types. + +/// Scalar types which are supported natively by both coln-store and coln-query. +#[derive(Clone, Copy, Debug)] +pub enum NativeScalarType { + /// Signed 64-bit integer. + Iint, + /// Unsigned 64-bit integer. + Uint, + /// String. + String, + // Add more :) +} + +impl From for NativeScalarType { + fn from(value: ir::BuiltinTy) -> Self { + match value { + ir::BuiltinTy::BuiltinStr => NativeScalarType::String, + ir::BuiltinTy::BuiltinInt => NativeScalarType::Iint, + // So far, no builtin uint. + } + } +} + +/// Scalar types which are supported by coln-store. +#[derive(Clone, Copy, Debug)] +pub enum StoreEngineScalarType { + /// A row id becomes a pair of `(CommitHash, Counter)`. + CommitHash, + /// A row id becomes a pair of `(CommitHash, Counter)`. + Counter, + Native(NativeScalarType), +} + +/// Scalar types which are supported by coln-query. +#[derive(Clone, Copy, Debug)] +pub enum QueryEngineScalarType { + Native(NativeScalarType), +} + +impl From for QueryEngineScalarType { + fn from(value: StoreEngineScalarType) -> Self { + match value { + StoreEngineScalarType::CommitHash => { + QueryEngineScalarType::Native(NativeScalarType::Uint) + } + StoreEngineScalarType::Counter => QueryEngineScalarType::Native(NativeScalarType::Uint), + StoreEngineScalarType::Native(native) => QueryEngineScalarType::Native(native), + } + } +} + +/// Generic column metadata representation. +#[derive(Debug)] +pub struct Col { + /// The column's name. + name: ir::ColName, + /// The column's (scalar) type. + ty: T, + /// If the column is (part of) a foreign key, this links the referenced table. + references: R, +} + +impl Col { + pub fn name(&self) -> &ir::ColName { + &self.name + } + /// The column's (scalar) type, in whichever engine's view `T` belongs to. + pub fn ty(&self) -> &T { + &self.ty + } +} + +/// Column metadata from the perspective of the compiler. +/// +/// The compiler encodes foreign keys as part of the type of a column (see the +/// [`ir::ColType::RowId`] variant of [`ir::ColType`]). +/// Hence, `R` becomes the unit type and is not required in this case. +pub type CompilerCol = Col; + +#[derive(Copy, Clone, Debug)] +pub enum CompilerColIdx { + /// A reference to the table's row id (the implicit primary key). + RowId, + /// A reference to a column is a (zero-indexed) column index. + Column(ir::ColumnIdx), +} + +impl CompilerColIdx { + pub fn for_row_id() -> Self { + CompilerColIdx::RowId + } +} + +impl From for CompilerColIdx { + fn from(value: ir::ColumnIdx) -> Self { + CompilerColIdx::Column(value) + } +} + +#[derive(Debug)] +pub struct CompilerCols(Vec); + +pub type StoreEngineCol = Col>; + +#[derive(Copy, Clone, Debug)] +pub struct StoreEngineColIdx(usize); + +#[derive(Debug)] +pub struct StoreEngineCols(Vec); + +impl StoreEngineCols { + /// To how many columns a row id expands to. + pub const ROW_ID_COLS: usize = 2; + /// The suffix of the hash column of a row id. + pub const HASH_COL_SUFFIX: &'static str = "RowIdHash"; + /// The suffix of the counter column of a row id. + pub const CTR_COL_SUFFIX: &'static str = "RowIdCtr"; + + /// From the perspective of coln-store, every base table has two implicitly + /// defined columns: The commit hash from the transaction which created the + /// row and a counter value, rendering the hash-counter-pair unique among + /// all insertions of a transaction. Coln-store assigns these counters. + fn implicit_row_id_cols() -> [StoreEngineCol; Self::ROW_ID_COLS] { + [ + StoreEngineCol { + name: Path::from(Self::HASH_COL_SUFFIX), + ty: StoreEngineScalarType::CommitHash, + references: None, + }, + StoreEngineCol { + name: Path::from(Self::CTR_COL_SUFFIX), + ty: StoreEngineScalarType::Counter, + references: None, + }, + ] + } + fn foreign_key_cols( + name: &ir::ColName, + foreign_entity: &Path, + ) -> [StoreEngineCol; Self::ROW_ID_COLS] { + [ + StoreEngineCol { + name: name.clone().append(Self::HASH_COL_SUFFIX), + ty: StoreEngineScalarType::CommitHash, + references: Some(foreign_entity.clone()), + }, + StoreEngineCol { + name: name.clone().append(Self::CTR_COL_SUFFIX), + ty: StoreEngineScalarType::Counter, + references: Some(foreign_entity.clone()), + }, + ] + } +} + +pub type QueryEngineCol = Col>; + +#[derive(Copy, Clone, Debug)] +pub struct QueryEngineColIdx(usize); + +#[derive(Debug)] +pub struct QueryEngineCols(Vec); + +impl QueryEngineCols { + pub fn iter(&self) -> std::slice::Iter<'_, QueryEngineCol> { + self.0.iter() + } +} + +// Conversions from one view into another view. + +impl From<&[ir::ColumnEntry]> for CompilerCols { + fn from(ir_cols: &[ir::ColumnEntry]) -> Self { + CompilerCols( + ir_cols + .iter() + // It's an one-to-one mapping from FLIR's JSON representation + // to this intermediate representation. + .map(|col| CompilerCol { + name: col.path.clone(), + ty: col.col_type.clone(), + // Foreign keys are encoded in the `ty` for a CompilerColumn. + // Hence, references becomes the unit type. + references: (), + }) + .collect(), + ) + } +} + +impl From<&[CompilerCol]> for StoreEngineCols { + fn from(compiler_cols: &[CompilerCol]) -> Self { + let prepended_row_id_cols = StoreEngineCols::implicit_row_id_cols().into_iter(); + let schema_cols = compiler_cols.iter().flat_map(|col| { + let name = col.name.clone(); + let (first, second) = match &col.ty { + ir::ColType::RowId { path } => { + let [hash_col, ctr_col] = StoreEngineCols::foreign_key_cols(&name, path); + (hash_col, Some(ctr_col)) + } + ir::ColType::BuiltinTy { builtin_ty } => ( + StoreEngineCol { + name, + ty: StoreEngineScalarType::Native(NativeScalarType::from(*builtin_ty)), + references: None, + }, + None, + ), + }; + std::iter::once(first).chain(second) + }); + StoreEngineCols(prepended_row_id_cols.chain(schema_cols).collect()) + } +} + +impl From<&[StoreEngineCol]> for QueryEngineCols { + fn from(store_engine_cols: &[StoreEngineCol]) -> Self { + QueryEngineCols( + store_engine_cols + .iter() + // It's a one-to-one mapping from the storage engine's schema + // view to the query engine's schema view; only the scalar types + // are different: The commit hash and counter become plain, + // unsigned ints, each. + .map(|col| QueryEngineCol { + name: col.name.clone(), + ty: QueryEngineScalarType::from(col.ty), + references: col.references.clone(), + }) + .collect(), + ) + } +} diff --git a/packages/coln-flir-rs/tests/common/mod.rs b/packages/coln-flir-rs/src/test_utils.rs similarity index 52% rename from packages/coln-flir-rs/tests/common/mod.rs rename to packages/coln-flir-rs/src/test_utils.rs index bd5ef5dd..51dd766f 100644 --- a/packages/coln-flir-rs/tests/common/mod.rs +++ b/packages/coln-flir-rs/src/test_utils.rs @@ -2,11 +2,16 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use std::path::PathBuf; +//! Shared test utils (feature gated behind `"test-utils"`) to have them +//! accessible for not only this crate's integration tests and unit tests but +//! also available for other crate's integration and unit tests. -use coln_flir_rs::ir::FlatRealm; +use crate::ir::FlatRealm; +use std::path::PathBuf; -pub fn load_theory(name: &str) -> FlatRealm { +/// Reads and parses the coln-compiler's JSON FLIR output with `name` stored in +/// `coln-flir-rs/tests/data` into a [`FlatRealm`]. +pub fn load_theory_from_json(name: &str) -> FlatRealm { let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) .join("tests/data") .join(name); diff --git a/packages/coln-flir-rs/tests/data/Graph.json b/packages/coln-flir-rs/tests/data/Graph.json index 329354f2..df99172b 100644 --- a/packages/coln-flir-rs/tests/data/Graph.json +++ b/packages/coln-flir-rs/tests/data/Graph.json @@ -1 +1,92 @@ -{"entities":[{"path":[["Graph"],["E"]],"value":{"entityVariant":{"tag":"table"},"columns":[{"path":[["a"]],"type":{"tag":"rowId","path":[["Graph"],["V"]]}},{"path":[["b"]],"type":{"tag":"rowId","path":[["Graph"],["V"]]}}],"primaryKey":null}},{"path":[["Graph"],["V"]],"value":{"entityVariant":{"tag":"table"},"columns":[],"primaryKey":null}}],"rules":[{"path":[["Graph"],["E"],["foreignKey"]],"value":{"ruleVariant":"enforced","varNames":[[["a"]],[["b"]]],"varTypes":[{"tag":"rowId","path":[["Graph"],["V"]]},{"tag":"rowId","path":[["Graph"],["V"]]}],"antecedents":[{"tag":"atom","atom":{"entity":[["Graph"],["E"]],"rowId":null,"values":[{"column":0,"term":{"tag":"var","index":0}},{"column":1,"term":{"tag":"var","index":1}}]}}],"consequents":[{"tag":"atom","atom":{"entity":[["Graph"],["V"]],"rowId":{"tag":"var","index":0},"values":[]}},{"tag":"atom","atom":{"entity":[["Graph"],["V"]],"rowId":{"tag":"var","index":1},"values":[]}}]}},{"path":[["Graph"],["V"],["foreignKey"]],"value":{"ruleVariant":"enforced","varNames":[],"varTypes":[],"antecedents":[{"tag":"atom","atom":{"entity":[["Graph"],["V"]],"rowId":null,"values":[]}}],"consequents":[]}}]} \ No newline at end of file +{ + "entities": [ + { + "path": [["Graph"], ["E"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { "tag": "rowId", "path": [["Graph"], ["V"]] } + }, + { + "path": [["b"]], + "type": { "tag": "rowId", "path": [["Graph"], ["V"]] } + } + ], + "primaryKey": null + } + }, + { + "path": [["Graph"], ["V"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [], + "primaryKey": null + } + } + ], + "rules": [ + { + "path": [["Graph"], ["E"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]], [["b"]]], + "varTypes": [ + { "tag": "rowId", "path": [["Graph"], ["V"]] }, + { "tag": "rowId", "path": [["Graph"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["Graph"], ["E"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["Graph"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["Graph"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + } + ] + } + }, + { + "path": [["Graph"], ["V"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [], + "varTypes": [], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["Graph"], ["V"]], + "rowId": null, + "values": [] + } + } + ], + "consequents": [] + } + } + ] +} diff --git a/packages/coln-flir-rs/tests/data/GraphOfGraphs.json b/packages/coln-flir-rs/tests/data/GraphOfGraphs.json new file mode 100644 index 00000000..bfe94a53 --- /dev/null +++ b/packages/coln-flir-rs/tests/data/GraphOfGraphs.json @@ -0,0 +1,741 @@ +{ + "entities": [ + { + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [], + "primaryKey": null + } + }, + { + "path": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["b"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + } + ], + "primaryKey": null + } + }, + { + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + } + ], + "primaryKey": null + } + }, + { + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["b"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + }, + { + "path": [["c"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + } + ], + "primaryKey": null + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["V"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["v0"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["v1"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] + } + }, + { + "path": [["b"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + }, + { + "path": [["c"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + } + ], + "primaryKey": [[["a"]], [["b"]], [["v0"]], [["v1"]]] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["E"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["v0"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["v1"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] + } + }, + { + "path": [["a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] + } + }, + { + "path": [["v0", "a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + }, + { + "path": [["v1", "a"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + } + }, + { + "path": [["b"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]] + } + }, + { + "path": [["c"]], + "type": { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]] + } + } + ], + "primaryKey": [ + [["a"]], + [["b"]], + [["v0"]], + [["v1"]], + [["v0", "a"]], + [["v1", "a"]] + ] + } + } + ], + "rules": [ + { + "path": [["GraphOfGraphsRealm"], ["base"], ["V"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [], + "varTypes": [], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": null, + "values": [] + } + } + ], + "consequents": [] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["base"], ["E"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]], [["b"]]], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]]], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": null, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]], [["b"]], [["c"]]], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["E"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 2 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["V"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["v0"]], [["v1"]], [["a"]], [["b"]], [["c"]]], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["V"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 3 } }, + { "column": 4, "term": { "tag": "var", "index": 4 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "rowId": { "tag": "var", "index": 2 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 3 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 4 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 1 } }] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["V"], ["total"]], + "value": { + "ruleVariant": "monitored", + "varNames": [[["v0"]], [["v1"]], [["a"]], [["b"]]], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "rowId": { "tag": "var", "index": 2 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 3 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["V"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 3 } } + ] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["E"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [ + [["v0"]], + [["v1"]], + [["a"]], + [["v0", "a"]], + [["v1", "a"]], + [["b"]], + [["c"]], + [["c"], ["b"]], + [["c"], ["c"]] + ], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]] + }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]] + }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["E"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 3 } }, + { "column": 4, "term": { "tag": "var", "index": 4 } }, + { "column": 5, "term": { "tag": "var", "index": 5 } }, + { "column": 6, "term": { "tag": "var", "index": 6 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["V"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 3 } }, + { "column": 4, "term": { "tag": "var", "index": 7 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["V"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 4 } }, + { "column": 4, "term": { "tag": "var", "index": 8 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "rowId": { "tag": "var", "index": 2 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 3 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 4 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["E"]], + "rowId": { "tag": "var", "index": 5 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 3 } }, + { "column": 2, "term": { "tag": "var", "index": 4 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["E"]], + "rowId": { "tag": "var", "index": 6 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 1 } }, + { "column": 1, "term": { "tag": "var", "index": 7 } }, + { "column": 2, "term": { "tag": "var", "index": 8 } } + ] + } + } + ] + } + }, + { + "path": [["GraphOfGraphsRealm"], ["action"], ["E"], ["total"]], + "value": { + "ruleVariant": "monitored", + "varNames": [ + [["v0"]], + [["v1"]], + [["a"]], + [["v0", "a"]], + [["v1", "a"]], + [["b"]] + ], + "varTypes": [ + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["V"]] }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["base"], ["E"]] }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { + "tag": "rowId", + "path": [["GraphOfGraphsRealm"], ["fiber"], ["V"]] + }, + { "tag": "rowId", "path": [["GraphOfGraphsRealm"], ["fiber"], ["E"]] } + ], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 0 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["V"]], + "rowId": { "tag": "var", "index": 1 }, + "values": [] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["base"], ["E"]], + "rowId": { "tag": "var", "index": 2 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } } + ] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 3 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["V"]], + "rowId": { "tag": "var", "index": 4 }, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + }, + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["fiber"], ["E"]], + "rowId": { "tag": "var", "index": 5 }, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 3 } }, + { "column": 2, "term": { "tag": "var", "index": 4 } } + ] + } + } + ], + "consequents": [ + { + "tag": "atom", + "atom": { + "entity": [["GraphOfGraphsRealm"], ["action"], ["E"]], + "rowId": null, + "values": [ + { "column": 0, "term": { "tag": "var", "index": 0 } }, + { "column": 1, "term": { "tag": "var", "index": 1 } }, + { "column": 2, "term": { "tag": "var", "index": 2 } }, + { "column": 3, "term": { "tag": "var", "index": 3 } }, + { "column": 4, "term": { "tag": "var", "index": 4 } }, + { "column": 5, "term": { "tag": "var", "index": 5 } } + ] + } + } + ] + } + } + ] +} diff --git a/packages/coln-flir-rs/tests/data/Prim.json b/packages/coln-flir-rs/tests/data/Prim.json index dd64b14a..e4bc7124 100644 --- a/packages/coln-flir-rs/tests/data/Prim.json +++ b/packages/coln-flir-rs/tests/data/Prim.json @@ -1 +1,70 @@ -{"entities":[{"path":[["T"],["N"]],"value":{"entityVariant":{"tag":"table"},"columns":[{"path":[["a"]],"type":{"tag":"builtin","type":"builtinInt"}}],"primaryKey":null}},{"path":[["T"],["S"]],"value":{"entityVariant":{"tag":"table"},"columns":[{"path":[["a"]],"type":{"tag":"builtin","type":"builtinString"}}],"primaryKey":null}}],"rules":[{"path":[["T"],["N"],["foreignKey"]],"value":{"ruleVariant":"enforced","varNames":[[["a"]]],"varTypes":[{"tag":"builtin","type":"builtinInt"}],"antecedents":[{"tag":"atom","atom":{"entity":[["T"],["N"]],"rowId":null,"values":[{"column":0,"term":{"tag":"var","index":0}}]}}],"consequents":[]}},{"path":[["T"],["S"],["foreignKey"]],"value":{"ruleVariant":"enforced","varNames":[[["a"]]],"varTypes":[{"tag":"builtin","type":"builtinString"}],"antecedents":[{"tag":"atom","atom":{"entity":[["T"],["S"]],"rowId":null,"values":[{"column":0,"term":{"tag":"var","index":0}}]}}],"consequents":[]}}]} \ No newline at end of file +{ + "entities": [ + { + "path": [["T"], ["N"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { "tag": "builtin", "type": "builtinInt" } + } + ], + "primaryKey": null + } + }, + { + "path": [["T"], ["S"]], + "value": { + "entityVariant": { "tag": "table" }, + "columns": [ + { + "path": [["a"]], + "type": { "tag": "builtin", "type": "builtinString" } + } + ], + "primaryKey": null + } + } + ], + "rules": [ + { + "path": [["T"], ["N"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]]], + "varTypes": [{ "tag": "builtin", "type": "builtinInt" }], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["T"], ["N"]], + "rowId": null, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + } + ], + "consequents": [] + } + }, + { + "path": [["T"], ["S"], ["foreignKey"]], + "value": { + "ruleVariant": "enforced", + "varNames": [[["a"]]], + "varTypes": [{ "tag": "builtin", "type": "builtinString" }], + "antecedents": [ + { + "tag": "atom", + "atom": { + "entity": [["T"], ["S"]], + "rowId": null, + "values": [{ "column": 0, "term": { "tag": "var", "index": 0 } }] + } + } + ], + "consequents": [] + } + } + ] +} diff --git a/packages/coln-flir-rs/tests/test_theory.rs b/packages/coln-flir-rs/tests/test_theory.rs index 21f75d2a..9fe42086 100644 --- a/packages/coln-flir-rs/tests/test_theory.rs +++ b/packages/coln-flir-rs/tests/test_theory.rs @@ -2,9 +2,8 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -mod common; - use coln_flir_rs::ir::Path; +use coln_flir_rs::test_utils; // TODO add more theory json files const THEORY_FIXTURES: &[&str] = &["Graph.json", "Prim.json"]; @@ -12,13 +11,13 @@ const THEORY_FIXTURES: &[&str] = &["Graph.json", "Prim.json"]; #[test] fn deserialises_all_theory_fixtures() { for name in THEORY_FIXTURES { - common::load_theory(name); + test_utils::load_theory_from_json(name); } } #[test] fn deserialises_graph_theory() { - let theory = common::load_theory("Graph.json"); + let theory = test_utils::load_theory_from_json("Graph.json"); assert_eq!(theory.tables.len(), 2); assert_eq!(theory.rules.len(), 2); diff --git a/packages/coln-js-runtime/Cargo.toml b/packages/coln-js-runtime/Cargo.toml index 238df69e..014653e6 100644 --- a/packages/coln-js-runtime/Cargo.toml +++ b/packages/coln-js-runtime/Cargo.toml @@ -14,7 +14,7 @@ js-sys = "0.3.83" serde = { version = "1", features = ["derive"] } serde-wasm-bindgen = "0.6.5" serde_json = "1.0.150" -tsify = "0.5.6" +tsify = "0.5.8" wasm-bindgen = "0.2.123" [lib] diff --git a/packages/coln-js-runtime/src/rust/dto.rs b/packages/coln-js-runtime/src/rust/dto.rs index 47cb9f53..83b8d908 100644 --- a/packages/coln-js-runtime/src/rust/dto.rs +++ b/packages/coln-js-runtime/src/rust/dto.rs @@ -4,20 +4,20 @@ use serde::{Deserialize, Serialize}; use std::array::TryFromSliceError; -use tsify::Tsify; +use tsify::{Ts, Tsify}; +use wasm_bindgen::JsValue; use wasm_bindgen::prelude::wasm_bindgen; use coln_store::{ commit::hash::CommitHash as StoreCommitHash, store::CommitChunk as StoreCommitChunk, - table::{CellValue as StoreCellValue, RowId as StoreRowId, RowView as StoreRowView}, - txn::{RowHandle, TxnValue as StoreTxnValue}, + table::{RowView as StoreRowView, WireRowId as StoreRowId, WireValue as StoreCellValue}, + txn::{TxnLiveRowId, TxnLiveValue as StoreTxnValue}, }; -use crate::error::BoundaryError; +use crate::error::{BoundaryError, js_error}; #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi, from_wasm_abi)] #[serde(rename_all = "camelCase")] pub struct CommitChunk { pub hash: CommitHash, @@ -36,7 +36,6 @@ impl From for CommitChunk { } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi, from_wasm_abi)] #[serde(transparent)] pub struct CommitHash { value: String, @@ -60,7 +59,6 @@ impl TryFrom for StoreCommitHash { /// TempRowId for JS runtime, different from TempRowId in coln-store #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi)] #[serde(rename_all = "camelCase")] pub struct TempRowId { pub tx_id: u64, @@ -68,7 +66,6 @@ pub struct TempRowId { } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi, from_wasm_abi)] #[serde(rename_all = "camelCase")] pub struct RowId { pub commit: CommitHash, @@ -76,7 +73,6 @@ pub struct RowId { } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi, from_wasm_abi)] #[serde(rename_all = "camelCase")] pub enum RowRef { Pending(TempRowId), @@ -108,12 +104,11 @@ impl TryFrom for StoreRowId { } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(from_wasm_abi, into_wasm_abi)] #[serde(tag = "tag", content = "value", rename_all = "lowercase")] pub enum Value { #[serde(rename = "row_id")] Id(RowRef), - Int(i64), + Int(i32), String(String), } @@ -137,13 +132,19 @@ impl Value { } #[wasm_bindgen(js_name = valueEqual)] -pub fn value_equal(v0: Value, v1: Value) -> bool { - v0 == v1 +pub fn value_equal(v0: Ts, v1: Ts) -> Result { + let v0 = v0.to_rust().map_err(js_error)?; + let v1 = v1.to_rust().map_err(js_error)?; + + Ok(v0 == v1) } #[wasm_bindgen(js_name = getRowRef)] -pub fn value_row_ref(v: Value) -> Option { - v.row_ref() +pub fn value_row_ref(v: Ts) -> Result>, JsValue> { + match v.to_rust().map_err(js_error)?.row_ref() { + Some(row_ref) => row_ref.into_ts().map(Some).map_err(js_error), + None => Ok(None), + } } // For reading @@ -163,9 +164,9 @@ impl From for StoreTxnValue { Value::Id(row_ref) => { let handle = match row_ref { RowRef::Pending(temp_row_id) => { - RowHandle::from_pending(temp_row_id.tx_id.into(), temp_row_id.counter) + TxnLiveRowId::from_pending(temp_row_id.tx_id.into(), temp_row_id.counter) } - RowRef::Existing(row_id) => RowHandle::from_existing( + RowRef::Existing(row_id) => TxnLiveRowId::from_existing( row_id.try_into().expect("commit hash not messed up"), ), }; @@ -180,7 +181,6 @@ impl From for StoreTxnValue { // We use Value for both row_id and values to simplify the interface of the app // developer #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Tsify)] -#[tsify(into_wasm_abi, from_wasm_abi)] #[serde(rename_all = "camelCase")] pub struct RowView { pub row_id: Value, diff --git a/packages/coln-js-runtime/src/rust/handles.rs b/packages/coln-js-runtime/src/rust/handles.rs index 6446a8f6..b4c3bdf2 100644 --- a/packages/coln-js-runtime/src/rust/handles.rs +++ b/packages/coln-js-runtime/src/rust/handles.rs @@ -6,14 +6,15 @@ use coln_flir_rs::ir; use coln_store::{ commit::{chunk::Chunk, hash::CommitHash as StoreCommitHash}, store::Store, - table::RowId as StoreRowId, - txn::{OwnedTransaction, RowHandle as StoreRowHandle}, + table::WireRowId as StoreRowId, + txn::{OwnedTransaction, TxnLiveRowId as StoreRowHandle}, }; use js_sys::Reflect; use crate::dto::{CommitChunk, CommitHash, RowId, RowRef, RowView, Value}; use crate::error::js_error; +use tsify::{Ts, Tsify}; use wasm_bindgen::JsValue; use wasm_bindgen::prelude::wasm_bindgen; @@ -78,9 +79,13 @@ fn resolve_value_id(js_value: &JsValue, row_id: RowId) -> Result<(), JsValue> { #[wasm_bindgen] impl TransactionHandle { - pub fn add(&mut self, path: String, values: Vec) -> Result { + pub fn add(&mut self, path: String, values: Vec>) -> Result { let path = ir::Path::from(path); - let values = values.into_iter().map(|v| v.into()).collect::>(); + let values = values + .iter() + .map(|value| value.to_rust()) + .collect::, _>>() + .map_err(js_error)?; let handle = self.tx()?.add(&path, values).map_err(js_error)?; let (tx_id, counter) = handle.pending_ids().map_err(js_error)?; @@ -176,7 +181,7 @@ impl StoreHandle { } #[wasm_bindgen(js_name = scanTable)] - pub fn scan_table(&self, path: String) -> Result, JsValue> { + pub fn scan_table(&self, path: String) -> Result>, JsValue> { let path = ir::Path::from(path); let rows = self .store()? @@ -184,15 +189,26 @@ impl StoreHandle { .map(|rows| rows.map(RowView::from).collect::>()) .unwrap_or_default(); - Ok(rows) + rows.iter() + .map(|row| row.into_ts()) + .collect::, _>>() + .map_err(js_error) } #[wasm_bindgen(js_name = rowById)] - pub fn row_by_id(&self, path: String, row_id: RowRef) -> Result, JsValue> { + pub fn row_by_id( + &self, + path: String, + row_id: Ts, + ) -> Result>, JsValue> { let path = ir::Path::from(path); + let row_id = row_id.to_rust().map_err(js_error)?; let row_id = StoreRowId::try_from(row_id).map_err(js_error)?; - Ok(self.store()?.row_by_id(&path, row_id).map(RowView::from)) + match self.store()?.row_by_id(&path, row_id) { + Some(row) => RowView::from(row).into_ts().map(Some).map_err(js_error), + None => Ok(None), + } } #[wasm_bindgen(js_name = beginTransaction)] @@ -228,7 +244,7 @@ impl StoreHandle { impl StoreHandle { // For automerge-repo interfacing - pub fn heads(&self) -> Result, JsValue> { + pub fn heads(&self) -> Result>, JsValue> { let heads = match &self.state { StoreHandleState::Uninitialized { .. } => return Ok(Vec::new()), StoreHandleState::Ready { store, .. } => store, @@ -243,31 +259,37 @@ impl StoreHandle { .map(CommitHash::from) .collect::>(); - Ok(heads) + heads + .iter() + .map(|hash| hash.into_ts()) + .collect::, _>>() + .map_err(js_error) } #[wasm_bindgen(js_name = commitChunksAfter)] pub fn commit_chunks_after( &self, - have_heads: Vec, - ) -> Result, JsValue> { + have_heads: Vec>, + ) -> Result>, JsValue> { if matches!(self.state, StoreHandleState::Uninitialized { .. }) { return Ok(Vec::new()); } let have_heads = have_heads + .iter() + .map(|hash| hash.to_rust()) + .collect::, _>>() + .map_err(js_error)? .into_iter() .map(StoreCommitHash::try_from) .collect::, _>>() .map_err(js_error)?; - let chunks = self - .store()? + self.store()? .commit_chunks_after(&have_heads) .into_iter() - .map(CommitChunk::from) - .collect::>(); - - Ok(chunks) + .map(|chunk| CommitChunk::from(chunk).into_ts()) + .collect::, _>>() + .map_err(js_error) } #[wasm_bindgen(js_name = applyChunkBytes)] @@ -403,7 +425,7 @@ mod tests { let mut store = Store::try_from_ir(theory).expect("store"); let mut transaction = store.transaction(); transaction - .add(&Path::from("T"), vec![42_i64.into()]) + .add(&Path::from("T"), vec![42_i32]) .expect("add row"); transaction.commit().expect("commit"); store @@ -437,7 +459,7 @@ mod tests { let mut source = source_store(); let mut transaction = source.transaction(); transaction - .add(&Path::from("T"), vec![84_i64.into()]) + .add(&Path::from("T"), vec![84_i32]) .expect("add second row"); transaction.commit().expect("second commit"); @@ -472,7 +494,7 @@ mod tests { let mut source = source_store(); let mut transaction = source.transaction(); transaction - .add(&Path::from("T"), vec![84_i64.into()]) + .add(&Path::from("T"), vec![84_i32]) .expect("add second row"); transaction.commit().expect("second commit"); @@ -514,7 +536,7 @@ mod tests { transaction .tx() .expect("owned transaction") - .add(&Path::from("T"), vec![84_i64.into()]) + .add(&Path::from("T"), vec![84_i32]) .expect("stage row"); let recovered = transaction.take_store().expect("recover store"); diff --git a/packages/coln-query/Cargo.toml b/packages/coln-query/Cargo.toml index ec7fd45e..89e15731 100644 --- a/packages/coln-query/Cargo.toml +++ b/packages/coln-query/Cargo.toml @@ -2,31 +2,61 @@ name = "coln-query" version = "0.1.0" edition = "2024" -license = "MIT OR Apache-2.0" description = "The query engine of the Coln language." -repository = "https://github.com/coln-project/Coln" readme = "README.md" +repository = "https://github.com/coln-project/Coln" +license = "MIT OR Apache-2.0" exclude = [ - "/.editorconfig", - "/.gitattributes", - "/.gitignore", - "/AGENTS.md", - "/flake.nix", - "/justfile", - "/scripts/", + "/.editorconfig", + "/.gitattributes", + "/.gitignore", + "/AGENTS.md", + "/flake.nix", + "/justfile", + "/scripts/", ] -[features] -test = [] - [dependencies] anyhow = { workspace = true } cli-table = "0.4.9" coln-flir-rs = { path = "../coln-flir-rs" } dbsp = "0.305.0" -rkyv = { version = "0.7.46", default-features = false, features = [ "std", "size_64", "validation", "uuid", ] } -size-of = { version = "0.1.7", package = "feldera-size-of", features = [ "hashbrown", "time-std", "xxhash-xxh3", "arcstr", "chrono", "ordered-float", ] } +rkyv = { + version = "0.7.46", + default-features = false, + features = [ + "size_64", + "std", + "uuid", + "validation", + ] +} +size-of = { + package = "feldera-size-of", + version = "0.1.7", + features = [ + "arcstr", + "chrono", + "hashbrown", + "ordered-float", + "time-std", + "xxhash-xxh3", + ] +} thiserror = { workspace = true } +# The batch backend lowers plans into coln-batch Datalog programs and +# executes them with its engine. Data flows store -> pipeline -> batch, +# so this direction creates no cycle. +coln-batch = { path = "../coln-batch" } [dev-dependencies] +# Import coln-flir-rs again but with the test-utils feature flag to have +# its test utils available in this crate's test context. +coln-flir-rs = { path = "../coln-flir-rs", features = ["test-utils"] } +# This is a self-referential dev-dependency to have the feature-gated +# test-utils be available in this crate's integration tests, too. +coln-query = { path = ".", features = ["test-utils"] } criterion = { workspace = true } + +[features] +test-utils = ["coln-flir-rs/test-utils"] diff --git a/packages/coln-query/justfile b/packages/coln-query/justfile index a6dafa16..37b0f633 100644 --- a/packages/coln-query/justfile +++ b/packages/coln-query/justfile @@ -1,52 +1,68 @@ # Task runner: https://github.com/casey/just -# From packages/coln-query, with Rust on PATH (e.g. `nix develop`): -# just # list available recipes -# just check # fmt-check + clippy + test -# just cargo-check # type-check coln-query targets -# just import -# just fix # apply cargo fixes + format -# just coverage # llvm-cov + Cobertura crate := "coln-query" +# List all just targets. default: @just --list +# Auto fix code formatting. fmt: - cargo fmt -p {{crate}} + cargo fmt -p {{ crate }} +# Check code formatting. fmt-check: - cargo fmt -p {{crate}} --check + cargo fmt -p {{ crate }} --check -cargo-check: - cargo check -p {{crate}} --all-targets +# Auto fix linting issues. +lint: + cargo fix -p {{ crate }} --all-targets --all-features --allow-dirty --allow-staged + cargo clippy -p {{ crate }} --fix --all-targets --all-features --allow-dirty --allow-staged -- -D warnings -clippy: - cargo clippy -p {{crate}} --all-targets --all-features -- -D warnings +# Check linting. +lint-check: + cargo check -p {{ crate }} --all-targets --all-features + cargo clippy -p {{ crate }} --all-targets --all-features -- -D warnings -import: - cargo fix -p {{crate}} --all-targets --all-features --allow-dirty --allow-staged - -fix: import fmt - -test: - cargo nextest run -p {{crate}} --all-targets +# Run tests with stdout and stderr suppressed. +test *TESTS: + cargo nextest run -p {{ crate }} --all-targets {{ TESTS }} # As of now nextest does not support doc tests, so we fallback to # cargo test to run them. - cargo test --doc -p {{crate}} + cargo test --doc -p {{ crate }} {{ TESTS }} + +alias t := test-verbose +# Run tests but with stdout and stderr printed. +test-verbose *TESTS: + cargo nextest run -p {{ crate }} --all-targets --nocapture {{ TESTS }} -doc-build: - cargo doc +# Build the docs. +doc: + cargo doc -p {{ crate }} +# Check the docs for broken links. doc-check: - cargo doc --no-deps --document-private-items --config 'build.rustdocflags=["-D", "warnings"]' + cargo doc -p {{ crate }} --no-deps --document-private-items --config 'build.rustdocflags=["-D", "warnings"]' -check: fmt-check clippy test doc-check +alias c := check +# Run all code checks. Global command referenced by the root justfile. +check: fmt-check lint-check test doc-check +# Run all automated code fixes. Global command referenced by the root justfile. +fix: fmt lint + +# Define what to do in the CI pipeline. ci: check +# Test coverage. coverage: - cargo llvm-cov -p {{crate}} --cobertura --output-path cobertura.xml + cargo llvm-cov -p {{ crate }} --cobertura --output-path cobertura.xml + +alias b := build +# Build the crate with all features and targets. +build: + cargo build -p {{ crate }} --all-targets --all-features +# Run the crate. run: - cargo run -p {{crate}} + cargo run -p {{ crate }} diff --git a/packages/coln-query/src/api/deltas.rs b/packages/coln-query/src/api/deltas.rs index a41bfb05..fbae0cef 100644 --- a/packages/coln-query/src/api/deltas.rs +++ b/packages/coln-query/src/api/deltas.rs @@ -3,44 +3,278 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT //! An interface for passing deltas of row-oriented data. There is -//! [RowDelta], [TableDelta], [StoreDelta], and [DerivedDataDelta]. +//! [ZRow], [TableDelta], [StoreDelta], and [DerivedDataDelta]. -use super::schema::TableRef; +use crate::relational::schema::EntityRef; +// Re-exported, not merely imported: [`ZRow::new`] takes a [`TupleValue`] built +// from [`ScalarTypedValue`]s, so a caller outside this crate cannot construct +// one of the deltas this module is about without both names in reach. +pub use crate::relational::TupleValue; pub use crate::scalarial::ScalarTypedValue; +use std::borrow::Borrow; pub type ZWeight = i64; -/// An update of a row of some base table. -/// It either represents an insertion or a deletion of a row from a table, -/// see [`z_weight`](`Self::z_weight`) documentation. -pub struct RowDelta { +/// An update of a row of some table. It either represents an insertion or a +/// deletion of a row from a table, see [`zweight`](Self::zweight()) documentation. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ZRow { + /// See [Self::zweight] documentation. + zweight: ZWeight, + /// The row-oriented data. + row: TupleValue, +} + +impl ZRow { + /// Create a new [`ZRow`] but filters out deltas with a `zweight` of 0 + /// in which case `None` is returned. + pub fn new(zweight: ZWeight, row: TupleValue) -> Option { + if zweight == 0 { + None + } else { + Some(Self { zweight, row }) + } + } /// A ZWeight value ... /// - `== 0` behaves as if there was no change happening at all. /// - `n if n > 0` represents an insertion. If `n > 1` it is a duplicated /// insertion, that is, the row is inserted n-times. /// - `n if n < 0` represents a deletion. If `n < 1` we remove the row /// n-times. - z_weight: ZWeight, - /// The row-oriented data. - row: Vec, + pub fn zweight(&self) -> ZWeight { + self.zweight + } + /// Returns `true` if the [`zweight`](Self::zweight) is (strictly) positive. + /// + /// Can occur for both deltas (changes) or sets (snapshots). + pub fn is_assertion(&self) -> bool { + self.zweight() > 0 + } + /// Returns `true` if the [`zweight`](Self::zweight) is negative in which + /// case a previous assertion of this row is now retracted. + /// + /// Only occurs if the [`ZRow`] stores a delta (change). + pub fn is_retraction(&self) -> bool { + self.zweight() < 0 + } + pub fn into_row(self) -> TupleValue { + self.row + } + /// Flips the [ZWeight](Self::zweight) to retract a previously fed fact. + /// Useful for rolling back a transaction. + fn retract(&mut self) { + self.zweight = -self.zweight; + } +} + +impl std::fmt::Display for ZRow { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "zweight: {:02}, row: {}", self.zweight(), self.row) + } } /// An update to a base table (part of the EDB). +#[derive(Debug, Clone, PartialEq, Eq)] pub struct TableDelta { - /// A unique identifier to a table. - table: TableRef, + /// A unique identifier of a table. + entity: EntityRef, /// The row-oriented updates of the table. - delta: Vec, + inner: Vec, +} + +impl TableDelta { + pub fn new>(for_entity: T, delta: Vec) -> Self { + Self { + entity: for_entity.into(), + inner: delta, + } + } + pub fn is_empty(&self) -> bool { + self.inner.is_empty() + } + pub fn iter(&self) -> impl Iterator { + self.into_iter() + } + pub fn for_entity(&self) -> &EntityRef { + &self.entity + } + pub fn delta(&self) -> &[ZRow] { + &self.inner + } + pub fn into_delta(self) -> Vec { + self.inner + } + /// Retracts all contained [`ZRow`]s. + fn retract(&mut self) { + self.inner.iter_mut().for_each(|delta| delta.retract()); + } +} + +impl IntoIterator for TableDelta { + type Item = ZRow; + type IntoIter = std::vec::IntoIter; + + fn into_iter(self) -> Self::IntoIter { + self.inner.into_iter() + } +} + +impl<'a> IntoIterator for &'a TableDelta { + type Item = &'a ZRow; + type IntoIter = std::slice::Iter<'a, ZRow>; + + fn into_iter(self) -> Self::IntoIter { + self.inner.iter() + } +} + +pub trait ZRowIterExt: Iterator + Sized { + fn assertions(self) -> impl Iterator; + fn retractions(self) -> impl Iterator; +} + +impl ZRowIterExt for I +where + I: Iterator + Sized, + I::Item: Borrow, +{ + fn assertions(self) -> impl Iterator { + self.filter(|row| row.borrow().is_assertion()) + } + + fn retractions(self) -> impl Iterator { + self.filter(|row| row.borrow().is_retraction()) + } +} + +// TODO: Offer proper cli-table based formatting, for anything that implements +// AsRef>. +impl std::fmt::Display for TableDelta { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "Entity {}{}", + self.for_entity(), + if self.is_empty() { " " } else { "\n" } + )?; + for delta in self.delta() { + writeln!(f, "{}", delta)?; + } + Ok(()) + } } /// An update of the EDB, that is, insertions or deletions of base facts. +/// +/// Ideally, there is at most one entry per table. +#[derive(Default, Debug, Clone, PartialEq, Eq)] pub struct StoreDelta { - pub inner: Vec, + inner: Vec, +} + +impl StoreDelta { + pub fn empty() -> Self { + Self { inner: Vec::new() } + } + pub fn with_deltas(deltas: Vec) -> Self { + Self { inner: deltas } + } + pub fn extend>(&mut self, deltas: I) { + self.inner.extend(deltas); + } + pub fn into_table_deltas(self) -> Vec { + self.inner + } + /// Inverses all facts of each contained [`TableDelta`]. Useful for + /// retractions and rolling back a transaction. Applying this twice yields + /// the original state, that is: + /// + /// ``` + /// # use coln_query::api::deltas::{ + /// # ScalarTypedValue, StoreDelta, TableDelta, TupleValue, ZRow, + /// # }; + /// # + /// # let row: TupleValue = [ScalarTypedValue::from(9_i64)].into_iter().collect(); + /// # let row_delta = ZRow::new(1, row).unwrap(); + /// # let row_deltas = vec![row_delta.clone(), row_delta.clone()]; + /// # let table_delta = TableDelta::new("SomeTable", row_deltas); + /// # let store_delta = StoreDelta::with_deltas(vec![table_delta]); + /// assert_eq!(store_delta, store_delta.clone().retract().retract()); + /// ``` + pub fn retract(mut self) -> Self { + self.inner.iter_mut().for_each(|table| table.retract()); + self + } } /// An update of the IDB, that is, insertions or deletions of derived facts. +#[derive(Default, Clone, Debug)] pub struct DerivedDataDelta { /// Contains the delta in the IDB after applying a delta in the EDB (the /// latter is a [`StoreDelta`]). inner: Vec, } + +impl DerivedDataDelta { + pub fn empty() -> Self { + Self { inner: Vec::new() } + } + pub fn is_empty(&self) -> bool { + self.inner.iter().all(|table_delta| table_delta.is_empty()) + } + pub fn with_deltas(deltas: Vec) -> Self { + Self { inner: deltas } + } + pub fn extend>(&mut self, deltas: I) { + self.inner.extend(deltas); + } + pub fn into_table_deltas(self) -> Vec { + self.inner + } +} + +impl std::fmt::Display for DerivedDataDelta { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "DerivedDataDelta{}", + if self.is_empty() { " " } else { "\n" } + )?; + for delta in &self.inner { + write!(f, "{}", delta)?; + } + Ok(()) + } +} + +// TODO: Snapshot with row and column views. + +#[cfg(test)] +mod test { + use super::*; + + fn row_delta() -> ZRow { + ZRow::new( + 2, + [ + ScalarTypedValue::from("String"), + ScalarTypedValue::from(1_i64), + ] + .into_iter() + .collect(), + ) + .expect("non-zero z-weight") + } + + fn table_delta>(name: T) -> TableDelta { + TableDelta::new(name.into(), vec![row_delta(), row_delta()]) + } + + #[test] + fn retracting_twice_restores_the_original_state() { + let store_delta = + StoreDelta::with_deltas(vec![table_delta("BaseTable1"), table_delta("BaseTable2")]); + + assert_eq!(store_delta, store_delta.clone().retract().retract()); + } +} diff --git a/packages/coln-query/src/api/error.rs b/packages/coln-query/src/api/error.rs new file mode 100644 index 00000000..ba2e001c --- /dev/null +++ b/packages/coln-query/src/api/error.rs @@ -0,0 +1,25 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +use crate::api::violations::ViolationsSet; +pub use crate::error::QueryEngineError; +use thiserror::Error; + +#[derive(Error, Debug, Clone, PartialEq, Eq)] +/// Public error type for the coln frontend. +pub enum ColnQueryError { + /// A error in the underlying query engine(s). + #[error(transparent)] + Engine(#[from] QueryEngineError), + #[error(transparent)] + UnsafeApply(#[from] UnsafeApplyError), +} + +#[derive(Error, Debug, Clone, PartialEq, Eq)] +#[error("Violations occurred during unsafe application of facts: {violations}")] +/// A hard constraint has been violated but this should not have happened. +/// This usually indicates a bug. +pub struct UnsafeApplyError { + pub violations: ViolationsSet, +} diff --git a/packages/coln-query/src/api/mod.rs b/packages/coln-query/src/api/mod.rs index ea9a2973..062e7ac2 100644 --- a/packages/coln-query/src/api/mod.rs +++ b/packages/coln-query/src/api/mod.rs @@ -5,12 +5,466 @@ //! This module defines the public API of the query engine intended to be mainly //! used by `coln-store`. See the other module level documentations. -// Receiving `coln-compiler`'s IR is blocked by its stabiliztion and hence still -// missing in here. +use crate::{ + api::{ + deltas::{DerivedDataDelta, StoreDelta, TableDelta, ZRowIterExt}, + error::ColnQueryError, + store::TxStore, + transaction::{DataDelta, TxOutcome}, + violations::{ViolationsDelta, ViolationsSet}, + }, + error::{QueryEngineError, RuntimeError}, + pipeline::Pipeline, + relational::{Runtime, expr::SourceId, incremental::DbspRuntime}, +}; +use coln_flir_rs::ir::{self, FlatRealm}; +use query::FlirProgram; pub mod deltas; +pub mod error; pub mod query; -pub mod schema; -pub mod store; +mod store; pub mod transaction; pub mod violations; + +/// The main entrypoint for coln-store. +/// +/// Keeps the program next to the runnable artifact built from it, because +/// [`take_code`](crate::program::QueryProgram::take_code) empties a program's +/// *statements* while leaving its per-rule metadata behind which is what says +/// whether a violated rule aborts a transaction or merely warns, +/// and what schema to rebuild a [`TableDelta`] by. +/// It doubles as the [`Catalog`](crate::relational::catalog::Catalog) +/// an ad-hoc query is compiled against. +#[derive(Debug)] +pub struct ColnQuery { + flir_program: FlirProgram, + /// The standing incremental computation over every rule declared at init. + incremental_runtime: DbspRuntime, + /// If a transaction is currently being applied, this is `Some(StoreDelta)`. + ongoing_tx: Option, +} + +impl ColnQuery { + pub fn init(realm: &FlatRealm) -> Result { + Ok(FlirProgram::from_flat_realm(realm) + .map_err(QueryEngineError::from) + .and_then(Self::with_flir_program)?) + } + fn with_flir_program(mut flir_program: FlirProgram) -> Result { + let incremental_runtime = Pipeline::incremental().runtime(&mut flir_program)?; + Ok(Self { + flir_program, + incremental_runtime, + ongoing_tx: None, + }) + } + /// Compile one ad-hoc query and evaluate it on the batch backend. + /// + /// The query is a conjunctive query's worth of [`ir::Prop`]s rather than a + /// whole realm, so it lowers through the same machinery a rule's body does. + /// What comes back is a [`Snapshot`](crate::relational::batch::Snapshot) + /// and not a [`Delta`](crate::relational::incremental::dbsp::DbspOutputDelta): + /// there is no previous state of an ad-hoc query for a delta to be relative to. + fn adhoc_query(&mut self, query: Vec) -> Result<(), QueryEngineError> { + // ```ignore + // pub fn adhoc_query( + // &mut self, + // query: Vec, + // source: &dyn RelationSource, <-- This is TBD. + // ) -> Result + // ``` + // + // Adhoc queries need to have a catalog available with both base tables + // and derived views.. + // + // It answers for **derived views as well as base tables**. A view's rows + // already travel back to coln-store as a `DerivedDataDelta`, so coln-store + // holds them and can serve them like any table — which is what lets an + // ad-hoc query read one, and is why the trait is not named for tables. + // + // That has a consequence for the catalog, and it is the part to settle + // first. An ad-hoc plan names a view with a `SourceExpr` leaf, because + // nothing binds it to a host variable the way a rule's statement does — but + // `impl Catalog for FlirProgram` answers for base tables *only*, deliberately + // (see its docs). The fix is not to widen that impl, which would let a rule + // body reference a view as a source and bypass the circuit, but to add a + // second view of the same program for ad-hoc compilation: + // `fn adhoc_catalog(&self) -> impl Catalog + '_`, answering from + // `base_tables` and then from `derived_views`, whose `RuleMeta` already + // carries the `output_schema` such a leaf needs. + // + // Which also means a rule name and a base table name must not collide. + // `base_table` and `rule_declaration` each reject duplicates within their own + // map, and nothing checks across the two — harmless while the two namespaces + // are separate, ambiguous the moment one catalog answers from both. + todo!("Run adhoc query on batch query engine"); + } + /// This intended for use during restarts. We already know that the data + /// we are feeding in fulfills all constraints, so the bookkeeping to + /// potentially undo a transaction can be skipped. + /// + /// # Correctness + /// + /// **Never** use this for unchecked inputs, otherwise, constraints may be + /// violated without notice. If this returns an `Err`, the engine is in + /// an unrecoverable state and it indicates a genuine bug. [`ColnQuery`] + /// has to be recreated to recover from this. + pub fn unsafe_apply(&mut self, delta: StoreDelta) -> Result { + self.internal_apply(delta)?; + Ok(DataDelta::try_from(self.interpret_outputs()?)?) + } + fn internal_apply(&mut self, delta: StoreDelta) -> Result<(), QueryEngineError> { + for delta in delta.into_table_deltas() { + let source_id = SourceId::from(delta.for_entity()); + let delta = delta.into_delta().into_iter(); + let effective = self.incremental_runtime.feed(&source_id, delta)?; + if !effective { + // TODO: Find a solution for logging. + println!("Delta references unknown source '{}'", source_id); + } + } + self.incremental_runtime.commit()?; + Ok(()) + } + /// Drain every readable output, discarding what they hold. For resetting the + /// handles after a commit whose results must not be observed, e.g., a rollback. + fn discard_outputs(&self) { + for (_sink_id, _delta) in self.incremental_runtime.all_outputs() {} + } + fn interpret_outputs(&mut self) -> Result { + let mut hard_violations = ViolationsSet::empty(); + let mut soft_violations = ViolationsDelta::empty(); + let mut derived_data_delta = DerivedDataDelta::empty(); + // We must drain all outputs first, so upon short-circuiting due to an + // error while processing below, we have absorbed all effects of the + // ongoing commit and they don't leak into the next commit. + let drained: Vec<_> = self.incremental_runtime.all_outputs().collect(); + for (sink_id, delta) in drained { + let sink_meta = self.flir_program.sink_meta(sink_id).ok_or_else(|| { + RuntimeError::new(format!( + "Bug: FLIR program does not know output sink {}", + sink_id + )) + })?; + let delta = TableDelta::new(sink_id, delta.as_zrows().collect()); + if delta.is_empty() { + continue; + } + match sink_meta.kind() { + // How to deal with the schema mismatch between coln-query, + // coln-store, and coln-compiler? Reporting may require the + // latter view, while coln-store may want to store it in its + // view. Looks like we need transformations in all directions... + ir::RuleVariant::Enforced => { + // An enforced rule's violation set is empty after every + // committed transaction, because any transaction that violates + // one is rolled back. So there is never a hard violation + // for a transaction to retract, and a negative zweight here + // means that an invariant broke, that is, some path fed the + // circuit without checking: `unsafe_apply` returning an + // `UnsafeApplyError` is the one that can. + debug_assert!( + delta.iter().retractions().next().is_none(), + "enforced rule {} retracts a violation it never reported", + delta.for_entity() + ); + hard_violations.extend(Some(delta)); + } + ir::RuleVariant::Monitored => { + soft_violations.extend(Some(delta)); + } + ir::RuleVariant::Chased => { + derived_data_delta.extend(Some(delta)); + } + } + } + // Both guards are the same emptiness check, but they answer different + // questions, and [`TxOutcome`] documents why: for an enforced rule the + // circuit reports against an empty set, so a non-empty delta *is* the + // violation set; for a monitored one it reports against what previous + // transactions left behind, so a non-empty delta means the set changed. + // Which is why the retractions are passed on rather than filtered out. + if !hard_violations.is_empty() { + return Ok(TxOutcome::HardViolationsSet(hard_violations)); + } + Ok(TxOutcome::DerivedDataDelta(DataDelta::new( + derived_data_delta, + soft_violations, + ))) + } +} + +impl TxStore for ColnQuery { + type Error = ColnQueryError; + + fn apply(&mut self, delta: StoreDelta) -> Result { + // Prepare to potentially undo the transaction. + self.ongoing_tx = Some(delta.clone()); + self.internal_apply(delta)?; + self.interpret_outputs().map_err(ColnQueryError::from) + } + + fn rollback(&mut self) -> Result<(), Self::Error> { + let retracted_delta = if let Some(tx) = self.ongoing_tx.take() { + tx.retract() + } else { + return Ok(()); + }; + self.internal_apply(retracted_delta)?; + self.discard_outputs(); + Ok(()) + } + + fn commit(&mut self) -> Result<(), Self::Error> { + self.ongoing_tx = None; + Ok(()) + } +} + +#[cfg(test)] +mod test { + use super::*; + use crate::{ + api::{ + deltas::{TableDelta, ZRow}, + transaction::{TryCommitErr, TryCommitOk, Tx}, + }, + test_utils::{ + self, + graph_flir::{self, Entity, JsonFlir}, + }, + }; + use anyhow::{Error, Result}; + + /// @Vincent, does this API work for you? + fn main_usage() -> Result<(), Error> { + let flat_realm = FlatRealm { + tables: vec![], + rules: vec![], + }; + let mut coln_query = ColnQuery::init(&flat_realm)?; + + let mut tx = Tx::new(StoreDelta::empty()); + // You can add deltas for tables until you call try_commit() on the tx. + tx.insert(std::iter::once(TableDelta::new("SomeTable", vec![]))); + + // If you are ready, try committing the transaction and pass in the handle: + match tx.try_commit(&mut coln_query) { + Ok(TryCommitOk::Pending(pending)) => { + // Call either commit or abort on `pending`. If commit() isn't + // called the tx is rolled back upon dropping `pending`! + let mut committed = pending.commit()?; + // This is how you can access the results: + let derived_data_delta = committed.take_derived_data_delta(); + let monitored_constraints = committed.take_soft_violations(); + // Commit the tx in coln-store, too, and call it a day :) + // + // If for some reason you don't want to commit the transaction + // you should better explicitly call abort() (and deal with + // the potential error) instead of relying on the fallback + // mechanism on dropping `pending`. + // This is how you can call abort: + // let aborted = pending.abort()?; + } + Ok(TryCommitOk::Rejected(mut rejected)) => { + // A hard constraint has been violated. + // At this point in time coln-query has already rolled back + // its internal state. + // In this arm, coln-store must roll back to keep the two in + // sync and report back the violations: + let violations = rejected.take_hard_violations(); + } + // The next two can also be combined, as they share the same error + // type and should be considered a bug, I guess. + // Err(TryCommitErr::TxApplyError(err)) | Err(TryCommitErr::RollbackError(err)) => {} + Err(TryCommitErr::TxApplyError(err)) => { + // An error while trying to apply the tx. + } + Err(TryCommitErr::RollbackError(err)) => { + // An error during the rollback of the pending tx. + } + }; + + Ok(()) + } + + /// @Vincent, does this API work for you? + fn restart_usage() -> Result<(), Error> { + let flat_realm = FlatRealm { + tables: vec![], + rules: vec![], + }; + let mut coln_query = ColnQuery::init(&flat_realm)?; + + // In a previous session all data we feed in here must have gone through + // the transaction dance shown above. Make sure to feed in the whole DB + // upon restarting. + let trusted_history = StoreDelta::empty(); + + match coln_query.unsafe_apply(trusted_history) { + Ok(mut data_delta) => { + // Update coln-store with the derived delta and report back the + // delta of soft violations caused by monitored rules. + let derived_data_delta = data_delta.take_derived_data_delta(); + let soft_violations_delta = data_delta.take_soft_violations(); + } + Err(err) => { + // This is bad: If it is an UnsafeApplyError, a hard constraint + // has been violated but the data should have been checked during + // a previous run. A query engine error indicates a bug, too. + } + }; + + Ok(()) + } + + #[test] + fn graph_flir() -> Result<()> { + let mut graph_flir = test_utils::graph_flir::GraphFlir::init(); + let flat_realm = graph_flir.load(); + let flir_program = FlirProgram::from_flat_realm(&flat_realm)?; + let mut coln_query = ColnQuery::with_flir_program(flir_program)?; + + let mut tx0 = Tx::empty(); + let v0 = graph_flir.insert_vertex(); + let v1 = graph_flir.insert_vertex(); + let v2 = graph_flir.insert_vertex(); + tx0.insert(graph_flir.next_epoch().into_table_deltas()); + let mut tx0 = tx0.try_commit(&mut coln_query)?.expect_pending_and_commit(); + assert!(tx0.take_derived_data_delta().is_empty()); + assert!(tx0.take_soft_violations().is_empty()); + + let mut tx1 = Tx::empty(); + let e0 = graph_flir.insert_edge(&v0, &v1); + let e1 = graph_flir.insert_edge(&v1, &v2); + tx1.insert(graph_flir.next_epoch().into_table_deltas()); + let mut tx1 = tx1.try_commit(&mut coln_query)?.expect_pending_and_commit(); + assert!(tx1.take_derived_data_delta().is_empty()); + assert!(tx1.take_soft_violations().is_empty()); + + let mut tx2 = Tx::empty(); + // Just some ints which haven't been used yet for sure. + let dangling_hash = 999; + let dangling_ctr = 999; + // Although the vertex does not violate a contraint, this vertex must be + // rolled back because tx3 is invalid due to the other inserts. + let v_rollback = graph_flir.insert_vertex(); + let invalid_edge_to = graph_flir::Edge::new( + graph_flir.epoch(), + graph_flir.next_ctr(), + v0.row_id().hash(), + v0.row_id().ctr(), + dangling_hash, + dangling_ctr, + ); + let invalid_edge_from = graph_flir::Edge::new( + graph_flir.epoch(), + graph_flir.next_ctr(), + dangling_hash, + dangling_ctr, + v1.row_id().hash(), + v1.row_id().ctr(), + ); + let invalid_edge = graph_flir::Edge::new( + graph_flir.epoch(), + graph_flir.next_ctr(), + dangling_hash, + dangling_ctr, + dangling_hash + 1, + dangling_ctr + 1, + ); + graph_flir.insert_raw_edge(invalid_edge_to); + graph_flir.insert_raw_edge(invalid_edge_from); + graph_flir.insert_raw_edge(invalid_edge); + tx2.insert(graph_flir.next_epoch().into_table_deltas()); + let mut tx2 = tx2.try_commit(&mut coln_query)?.expect_rejected(); + let violations = tx2.take_hard_violations(); + println!("{}", violations); + let violations = violations.into_inner(); + assert_eq!(violations.len(), 1); + let violation = &violations[0]; + assert_eq!(violation.for_entity().id(), "Graph.E.foreignKey"); + assert_eq!(violation.delta().len(), 3); + + let mut tx3 = Tx::empty(); + let e0 = graph_flir.insert_edge(&v0, &v_rollback); + tx3.insert(graph_flir.next_epoch().into_table_deltas()); + let mut tx3 = tx3.try_commit(&mut coln_query)?.expect_rejected(); + let violations = tx3.take_hard_violations(); + println!("{}", violations); + let violations = violations.into_inner(); + assert_eq!(violations.len(), 1); + let violation = &violations[0]; + assert_eq!(violation.for_entity().id(), "Graph.E.foreignKey"); + assert_eq!(violation.delta().len(), 1); + + Ok(()) + } + + /// One transaction to break a monitored rule, one to repair it. The point of + /// the pair is the *second* one: it changes the monitored violations without + /// introducing any, which is the only case where the delta a monitored sink + /// reports and the set of violations that exist come apart. + #[test] + fn a_transaction_repairing_a_monitored_violation_reports_it_as_resolved() -> Result<()> { + use test_utils::monitored_flir::{self as monitored, PERMITTED, RULE, TABLE}; + + let mut coln_query = ColnQuery::init(&monitored::realm())?; + // `PERMITTED` is the only value of `a` the rule tolerates, so this row + // violates it and no other. + let offending = monitored::row(0, 0, PERMITTED + 1); + let tx_with = |zweight| { + let mut tx = Tx::empty(); + tx.insert(Some(TableDelta::new( + TABLE, + vec![ZRow::new(zweight, offending.clone()).expect("non-zero zweight")], + ))); + tx + }; + + // Inserting it makes the violation appear. Monitored, so the transaction + // still commits. + let mut committed = tx_with(1) + .try_commit(&mut coln_query)? + .expect_pending_and_commit(); + let appeared = committed.take_soft_violations(); + assert_eq!( + appeared + .iter() + .map(|delta| delta.for_entity().id()) + .collect::>(), + vec![RULE], + "the monitored rule must report on its own sink" + ); + assert!( + appeared + .iter() + .all(|table| table.iter().all(ZRow::is_assertion)), + "a violation that appeared is asserted, not retracted: {appeared}" + ); + + // Retracting the same row repairs it. + let mut committed = tx_with(-1) + .try_commit(&mut coln_query)? + .expect_pending_and_commit(); + let resolved = committed.take_soft_violations(); + assert_eq!( + appeared + .iter() + .map(|delta| delta.for_entity().id()) + .collect::>(), + vec![RULE], + "the monitored rule must report on its own sink and a retraction of \ + a monitored violation must not be swallowed" + ); + assert!( + resolved + .iter() + .all(|table| table.iter().all(ZRow::is_retraction)), + "repairing a violation must not read as introducing one: {resolved}" + ); + + Ok(()) + } +} diff --git a/packages/coln-query/src/api/query.rs b/packages/coln-query/src/api/query.rs index 952cd071..8e094e07 100644 --- a/packages/coln-query/src/api/query.rs +++ b/packages/coln-query/src/api/query.rs @@ -3,4 +3,1548 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT //! This module converts coln's flattened lowered intermediate representation -//! (FLIR) into an AST of relational operators. +//! (FLIR) into a query program expressed in +//! [`Statements`](crate::host::stmt::Stmt), +//! using [`HostExprs`](crate::host::expr::Expr) and +//! [`RelExprs`](crate::relational::expr::RelExpr). + +use crate::error::SyntaxError; +use crate::host::QueryIr; +use crate::host::expr::{BinaryExpr, Expr, Literal, LiteralExpr, VarExpr}; +use crate::host::operator::Operator; +use crate::host::stmt::{Stmt, VarStmt}; +use crate::program::QueryProgram; +use crate::relational::catalog::Catalog; +use crate::relational::expr::{ + AntiJoinExpr, JoinVariable, MultiWayEquiJoinExpr, OutputExpr, OutputKind, ProjectionExpr, + RelationIdx, SelectionExpr, SinkId, SourceExpr, SourceId, +}; +use crate::relational::schema::{Column, EntityRef, TableSchema}; +use crate::scalarial::ScalarType; +use coln_flir_rs::ir::{ + self, Atom, EntityVariant, Equality, FlatRealm, Path, Prop, RuleEntry, TableEntry, Term, +}; +use coln_flir_rs::schema::{ + BaseTableSchema, CompilerColIdx, NativeScalarType, QueryEngineCol, QueryEngineScalarType, + StoreEngineCols, +}; +use std::borrow::Cow; +use std::collections::hash_map::Entry; +use std::collections::{BTreeMap, HashMap}; + +type BaseTableName = EntityRef; +type DerivedViewName = EntityRef; + +/// coln's FLIR frontend's [`QueryProgram`]: what a [`FlatRealm`] lowers to. +/// +/// The [`Catalog`] half is served straight out of [`base_tables`](Self::base_tables), +/// which stores FLIR's own richer [`BaseTableSchema`] which includes the schema +/// view according to coln-compiler and coln-store next coln-query's. +#[derive(Debug)] +pub struct FlirProgram { + /// The (raw, that is, unresolved, unoptimized) statements themselves. + code: QueryIr, + /// The declared base tables. Doubles as this program's [`Catalog`]: every + /// [`SourceExpr`] the lowering mints names one of these. + base_tables: HashMap, + /// The relations the program itself defines, that is, one per declared rule. + /// + /// This doubles as the set of derived views an [`Atom`] may reference, so + /// that what [`rule_declaration`](Self::rule_declaration) writes is exactly + /// what [`derived_view_var_expr`](Self::derived_view_var_expr) reads. + derived_views: HashMap, +} + +#[derive(Debug)] +pub struct RuleMeta { + kind: ir::RuleVariant, + output_schema: TableSchema, +} + +impl RuleMeta { + fn new(kind: ir::RuleVariant, output_schema: TableSchema) -> Self { + RuleMeta { + kind, + output_schema, + } + } + pub fn kind(&self) -> ir::RuleVariant { + self.kind + } + pub fn schema(&self) -> &TableSchema { + &self.output_schema + } +} + +/// Projects FLIR's per-engine schema down to the one thing the layers below +/// share: the query engine's columns, and the table's key(s) restated over +/// them. +/// +/// Both halves change coordinates on the way. The columns are the *query* +/// engine's view, where a row id has already flattened into a hash and a +/// counter column; FLIR states its keys as indices into the *compiler's* +/// view, so each component of a key travels through +/// [`resolve_query_col_range`](BaseTableSchema::resolve_query_col_range) to +/// become the one or two positions it occupies here. +/// +/// The implicit row id leads the list of keys: it is the only key a base table +/// is guaranteed to have and to be unique on, so a backend that can index by +/// just one key (DBSP) picks it by taking the first. +impl From<&BaseTableSchema> for TableSchema { + fn from(value: &BaseTableSchema) -> Self { + let columns = value + .query_cols() + .iter() + .map(|col| Column::new(col.name(), *col.ty())) + .collect(); + let row_id_key = value + .resolve_query_col_range(CompilerColIdx::for_row_id()) + .collect(); + let declared_keys = value + .primary_keys() + .iter() + // The compiler reports a table without a declared primary key as one + // empty key rather than no key at all. + .filter(|key| !key.is_empty()) + .map(|key| { + key.iter() + .flat_map(|idx| value.resolve_query_col_range(*idx)) + .collect() + }); + TableSchema::new( + EntityRef::from(value.name()), + columns, + std::iter::once(row_id_key).chain(declared_keys).collect(), + ) + } +} + +impl FlirProgram { + fn empty() -> Self { + Self { + code: QueryIr::default(), + base_tables: HashMap::new(), + derived_views: HashMap::new(), + } + } + pub fn from_flat_realm(flat_realm: &FlatRealm) -> Result { + let mut builder = FlirProgram::empty(); + for table in &flat_realm.tables { + builder.table_declaration(table)?; + } + for rule in &flat_realm.rules { + if rule.rule.consequents.is_empty() { + // The compiler does not clean up after the lowering and emits + // useless rules after lowering, so we vacuum-clean here instead. + continue; + } + builder.rule_declaration(rule)?; + } + Ok(builder) + } + + pub fn sink_meta(&self, sink: &SinkId) -> Option<&RuleMeta> { + self.derived_views.get(&EntityRef::from(sink)) + } + + fn table_declaration(&mut self, table_entry: &TableEntry) -> Result<(), SyntaxError> { + match &table_entry.table.entity_variant { + EntityVariant::Table => self.base_table(table_entry), + EntityVariant::View(materialization) => { + unimplemented!("[Initial models] Materialized views defined through a query"); + } + EntityVariant::Index { method, columns } => { + unimplemented!("[Not-yet specified] Indexes") + } + } + } + fn base_table(&mut self, table_entry: &ir::TableEntry) -> Result<(), SyntaxError> { + let name = BaseTableName::from(&table_entry.path); + let table_schema = + Option::::from(table_entry).expect("Broken precondition"); + if self + .base_tables + .insert(name.clone(), table_schema) + .is_some() + { + return Err(SyntaxError::new(format!( + "Base table {name} defined multiple times" + ))); + } + Ok(()) + } + + fn rule_declaration(&mut self, rule_entry: &RuleEntry) -> Result<(), SyntaxError> { + let name = DerivedViewName::from(&rule_entry.path); + let Some(rule) = FriendlyRule::from(&rule_entry.rule) else { + // The rule is filtered out but not an error case. + return Ok(()); + }; + let (stmt, output_bindings) = self.rule(name.id().to_string(), &rule)?; + self.code.push(stmt); + let rule_meta = RuleMeta::new(rule.kind, rule_output_schema(&name, &output_bindings)); + // See `base_table` on the direction of this check. + if self.derived_views.insert(name.clone(), rule_meta).is_some() { + return Err(SyntaxError::new(format!( + "Rule {name} defined multiple times" + ))); + } + Ok(()) + } + /// Lowers one rule into the statement that binds its name, and reports the + /// [`Binding`]s of the relation that statement evaluates to, so the caller + /// can describe the rule's output schema. + fn rule( + &mut self, + name: String, + rule: &FriendlyRule, + ) -> Result<(Stmt, Vec), SyntaxError> { + let (left, left_bindings) = self.conjunctive_query(&rule.lhs, &rule.vars)?; + let (right, right_bindings) = self.conjunctive_query(&rule.rhs, &rule.vars)?; + let rule_as_stmt = Stmt::from(VarStmt { + name: name.clone(), + initializer: Some(Expr::from(OutputExpr { + id: SinkId::from(name), + kind: OutputKind::Channel, + relation: Expr::from(AntiJoinExpr { + left, + right, + on: antijoin_key(&left_bindings, &right_bindings), + }), + })), + }); + // An antijoin carries the left relation's tuple through unchanged, so + // the rule's output is shaped by its antecedents. + Ok((rule_as_stmt, left_bindings)) + } + /// Lowers one side of a rule into a relational expression, and reports which + /// variable parts that expression binds so the enclosing [`AntiJoinExpr`] + /// can work out what to compare on. + fn conjunctive_query( + &mut self, + query: &ConjunctiveQuery, + vars: &[FriendlyVar], + ) -> Result<(Expr, Vec), SyntaxError> { + if query.atoms.is_empty() { + return Err(SyntaxError::new( + "FLIR emits conjunctive query with no atom", + )); + } + + let plans = query + .atoms + .iter() + .map(|atom| self.atom(atom, vars)) + .collect::, _>>()?; + + // A part bound by several atoms is a single binding of the conjunctive + // query as a whole, because the join keeps one active copy of it. We + // keep the first, matching the join's left-to-right shadowing. + let mut bindings: BTreeMap<(ir::VarIdx, VarPart), Binding> = BTreeMap::new(); + for plan in &plans { + for binding in &plan.bindings { + bindings + .entry((binding.var, binding.part)) + .or_insert_with(|| binding.clone()); + } + } + let bindings = bindings.into_values().collect(); + + let on = join_variables(&plans); + let mut relations: Vec = plans.into_iter().map(|plan| plan.relation).collect(); + let joined_atoms = if relations.len() == 1 { + // A single atom has nothing to join against, and the join operators + // require at least two relations. + relations.pop().expect("Length checked") + } else { + Expr::from(MultiWayEquiJoinExpr::new(relations, on, None)?) + }; + + let with_conditions = query + .conditions + .iter() + .map(|condition| self.selection(condition, vars)) + // All conditions get compiled into one condition by ANDing them. + .try_reduce(|left, right| { + Expr::from(BinaryExpr { + operator: Operator::And, + left, + right, + }) + })? + .into_iter() + // We fold the Option: If there are no conditions at all, we return + // `joined_atoms` as is and otherwise, we wrap it in a SelectionExpr + // whose condition embodies all conditions. + .fold(joined_atoms, |joined_atoms, conditions| { + Expr::from(SelectionExpr { + relation: joined_atoms, + condition: conditions, + }) + }); + + Ok((with_conditions, bindings)) + } + /// Generates a condition which possibly expands to two ANDed conditions + /// due to row ids being flattening to two variables. + /// + /// Currently, the compiler only supports equality conditions. + fn selection( + &mut self, + condition: &Equality, + vars: &[FriendlyVar], + ) -> Result { + let left = self.term(&condition.left, vars)?; + let right = self.term(&condition.right, vars)?; + // Things get a bit ugly unfortunately due to the flattening of row ids. + let conditions: Box> = match (left.len(), right.len()) { + (2, 2) => { + // This case compares two row ids which expand to two variables + // each and thus we have to create two conditions. + // The underlying condition has to be true for both the hash + // _and_ the counter. In code that translates to the diagonal of + // the terms. + let diagonal = left.into_iter().zip(right); + Box::new(diagonal) + } + _ => { + // This case deals with comparing: + // 1. An already flat variable with a literal. + // But also covers two nonsense cases at the moment: + // 1. A row id with a literal or an already flat variable. + // 2. Two literals. + // In code this boils down to computing all pairs of the terms. + let cartesian_product = left + .into_iter() + .flat_map(|left| right.iter().map(move |right| (left.clone(), right.clone()))); + Box::new(cartesian_product) + } + }; + + Ok(conditions + .map(|(left, right)| { + Expr::from(BinaryExpr { + operator: Operator::Equal, + left, + right, + }) + }) + .reduce(|acc, condition| { + Expr::from(BinaryExpr { + operator: Operator::And, + left: acc, + right: condition, + }) + }) + .expect("A FLIR condition must produce at least one condition")) + } + // Scoped to this function because of the derived-view `todo!()` below; + // the rest of the module is checked for unreachable code again. + #[allow(unreachable_code)] + fn atom(&mut self, atom: &Atom, vars: &[FriendlyVar]) -> Result { + let (source, schema): (Expr, &BaseTableSchema) = + if let Some((source_expr, schema)) = self.base_table_source_expr(&atom.entity) { + (Expr::from(source_expr), schema) + } else if let Some(var_expr) = self.derived_view_var_expr(&atom.entity) { + ( + Expr::from(var_expr), + todo!("Generic schema representation for derived views"), + ) + } else { + return Err(SyntaxError::new(format!( + "Atom references undeclared entity '{}'", + atom.entity + ))); + }; + + let mut binder = AtomBinder::default(); + + // The row id, if this atom brings it into scope. + if let Some(row_id) = &atom.row_id { + match row_id { + ir::Term::Var { index } => { + let var = friendly_var(vars, *index)?; + if !var.is_row_id() { + return Err(SyntaxError::new( + "FLIR wants to assign a row id to a variable of native scalar type", + )); + } + binder.bind( + *index, + var, + schema.resolve_query_cols(CompilerColIdx::for_row_id()), + )?; + } + ir::Term::Lit { lit: _ } => { + // Matching [`ir::Atom::row_id`]'s own note: a literal row id + // is not something we can express. + return Err(SyntaxError::new( + "FLIR equates a row id with a literal, which is not supported", + )); + } + } + } + + // The value columns this atom constrains or brings into scope. + for value in &atom.values { + let mut columns = schema.resolve_query_cols(CompilerColIdx::from(value.column)); + match &value.term { + ir::Term::Lit { lit } => { + let column = columns.next().ok_or_else(|| { + SyntaxError::new("FLIR compares a literal against a column that does not resolve to any query column") + })?; + binder.conditions.push(Expr::from(BinaryExpr { + operator: Operator::Equal, + left: Expr::from(VarExpr::new(column.name())), + right: Expr::from(LiteralExpr::from(Literal::from(lit))), + })); + } + ir::Term::Var { index } => { + binder.bind(*index, friendly_var(vars, *index)?, columns)?; + } + } + } + + let with_selection = binder + .conditions + .into_iter() + .reduce(|acc, condition| { + Expr::from(BinaryExpr { + operator: Operator::And, + left: acc, + right: condition, + }) + }) + .into_iter() + .fold(source, |source, root_condition| { + Expr::from(SelectionExpr { + relation: source, + condition: root_condition, + }) + }); + + let relation = Expr::from(ProjectionExpr { + relation: with_selection, + attributes: binder.attributes, + }); + + Ok(AtomPlan { + relation, + bindings: binder.bindings, + }) + } + fn term(&mut self, term: &Term, vars: &[FriendlyVar]) -> Result, SyntaxError> { + match term { + Term::Lit { lit } => Ok(vec![Expr::from(LiteralExpr::from(Literal::from(lit)))]), + Term::Var { index } => Ok(friendly_var(vars, *index)? + .parts() + .map(|(_part, name)| Expr::from(VarExpr::new(name))) + .collect()), + } + } + /// If the entity referenced by `Path` is part of the extensional database + /// (EDB) and present in the base tables, the function returns a + /// [`SourceExpr`] referencing that entity. Otherwise, [`None`] is returned. + fn base_table_source_expr(&mut self, name: &Path) -> Option<(SourceExpr, &BaseTableSchema)> { + self.base_tables + .get(&BaseTableName::from(name)) + .map(|base_table_schema| { + // The leaf names the table; `Catalog::source_schema` below is + // what turns that name back into a schema, and both sides go + // through `BaseTableSchema::name` so they cannot disagree. + ( + SourceExpr::new(base_table_schema.name().to_string()), + base_table_schema, + ) + }) + } + /// If the entity referenced by `Path` is part of the intensional database + /// (IDB) and present in the derived views, the function returns a + /// [`VarExpr`] referencing that entity. Due to coln-compiler declaring + /// tables and views prior to the rules, said entity must be known at this + /// point. Otherwise, [`None`] is returned. + fn derived_view_var_expr(&mut self, name: &Path) -> Option { + self.derived_views + .get(&DerivedViewName::from(name)) + .map(|_derived_view_schema| VarExpr::new(name.to_string())) + } +} + +impl Catalog for FlirProgram { + /// Projects FLIR's [`BaseTableSchema`] down to the [`TableSchema`] a plan + /// needs, on demand. [`Cow::Owned`] rather than a borrow precisely so that + /// the richer schema stays the only stored copy. + /// + /// Only base tables answer here: a rule's output is bound to a host variable + /// and referenced by [`VarExpr`], never by a [`SourceExpr`], so + /// [`derived_views`](Self::derived_views) is no part of the catalog. + fn source_schema(&self, id: &SourceId) -> Option> { + self.base_tables + .get(&BaseTableName::from(id)) + .map(|base_table_schema| Cow::Owned(TableSchema::from(base_table_schema))) + } +} + +impl QueryProgram for FlirProgram { + fn code(&self) -> &QueryIr { + &self.code + } + + fn take_code(&mut self) -> QueryIr { + std::mem::take(&mut self.code) + } +} + +/// Just like [`ir::Rule`] but friendlier because: +/// +/// 1. Meaningless rules with an empty [consequent](ir::Rule::consequents) are +/// skipped and chased rules panic at the moment due to open questions. +/// 2. It zips the [`ir::Rule::var_names`] and the [`ir::Rule::var_types`] into one +/// array of [`FriendlyVar`]s. +/// 3. It converts [`ir::Rule::antecedents`] and [`ir::Rule::consequents`] into a +/// [`ConjunctiveQuery`], each. +struct FriendlyRule { + kind: ir::RuleVariant, + vars: Vec, + lhs: ConjunctiveQuery, + rhs: ConjunctiveQuery, +} + +impl FriendlyRule { + fn from(rule: &ir::Rule) -> Option { + if rule.consequents.is_empty() { + return None; + } + if matches!(rule.rule_variant, ir::RuleVariant::Chased) { + unimplemented!( + "[Unclear] Chased rules produce a materialized view; how are they different from a materialized view defined in the table/entities section?" + ); + } + assert!( + rule.var_names.len() == rule.var_types.len(), + "var_names and var_types arrays do not size match" + ); + let vars = rule + .var_names + .iter() + .zip(rule.var_types.iter()) + .map(|(path, col_type)| FriendlyVar { + name: path.clone(), + ty: col_type.clone(), + }) + .collect(); + let lhs = ConjunctiveQuery::from(&rule.antecedents); + let rhs = ConjunctiveQuery::from(&rule.consequents); + Some(FriendlyRule { + kind: rule.rule_variant, + vars, + lhs, + rhs, + }) + } +} + +/// Prepares either a [left-hand side](ir::Rule::antecedents) or a +/// [right-hand side](ir::Rule::consequents) of a [`ir::Rule`] for inclusion +/// in an antijoin by partitioning a `Vec` into atoms and conditions. +/// This is useful because applying all atoms first, guarantees that every +/// variable a condition may refer to is in scope already. +struct ConjunctiveQuery { + atoms: Vec, + // Currently, only equality conditions are part of the FLIR. + conditions: Vec, +} + +impl ConjunctiveQuery { + fn from(props: &[Prop]) -> Self { + let (atoms, conditions) = + props + .iter() + .fold((vec![], vec![]), |(mut atoms, mut conditions), prop| { + match prop { + ir::Prop::Atom { atom } => atoms.push(atom.clone()), + ir::Prop::Eq { equality } => conditions.push(equality.clone()), + } + (atoms, conditions) + }); + Self { atoms, conditions } + } +} + +/// All information from [`ir::Rule::var_names`] and [`ir::Rule::var_types`] but +/// _zipped_. +struct FriendlyVar { + name: ir::Path, + ty: ir::ColType, // either a row id or a builtin type +} + +impl FriendlyVar { + fn is_row_id(&self) -> bool { + matches!(self.ty, ir::ColType::RowId { path: _ }) + } + /// The attribute name(s) this variable expands to in an atom's projected + /// schema: one for a builtin scalar type, two for a row id, which flattens + /// into a commit hash and a counter column. + /// + /// This is the single place that flattening happens, and its order matches + /// the order + /// [`resolve_query_cols`](BaseTableSchema::resolve_query_cols) yields the + /// corresponding columns in, so the two can be zipped. + fn parts(&self) -> impl Iterator { + match &self.ty { + ir::ColType::BuiltinTy { builtin_ty: _ } => { + vec![(VarPart::Scalar, self.name.to_string())] + } + ir::ColType::RowId { path: _ } => vec![ + ( + VarPart::RowIdHash, + self.name + .clone() + .append(StoreEngineCols::HASH_COL_SUFFIX) + .to_string(), + ), + ( + VarPart::RowIdCtr, + self.name + .clone() + .append(StoreEngineCols::CTR_COL_SUFFIX) + .to_string(), + ), + ], + } + .into_iter() + } +} + +/// Which of the attributes a FLIR variable expands to, see +/// [`FriendlyVar::parts`]. The derived ordering keeps a row id's two halves +/// adjacent and in flattening order. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +enum VarPart { + /// The variable in its entirety. It is of a builtin scalar type. + Scalar, + /// The commit-hash half of a row id. + RowIdHash, + /// The counter half of a row id. + RowIdCtr, +} + +/// One attribute an atom's projection exposes, tagged with the FLIR variable it +/// originates from. +/// +/// The join condition of a conjunctive query is derived by grouping these on +/// [`var`](Self::var) and [`part`](Self::part) rather than on +/// [`name`](Self::name): the variable index is exact and free, whereas grouping +/// by name would additionally assume that rendering [`ir::Path`]s into strings +/// is injective, and would have to tell a row id's two halves apart by parsing +/// their suffixes back off. +#[derive(Clone, Debug)] +struct Binding { + var: ir::VarIdx, + part: VarPart, + /// The attribute's name in the atom's projected schema. + name: String, + /// The type of the query column this part is bound to. Taken from the + /// column rather than from the FLIR variable, because that is where a row + /// id's halves have already been resolved to their query-engine types. + scalar_type: ScalarType, +} + +/// The relational plan for one FLIR [`Atom`], together with the [`Binding`]s +/// its projection exposes. +/// +/// Reporting the bindings is what lets the enclosing conjunctive query derive +/// its join variables without re-deriving them from the projection it just +/// built. +struct AtomPlan { + /// Essentially, a `Projection(Selection(atom's source relation))`. + relation: Expr, + /// The [`Binding`]s of the plan. + bindings: Vec, +} + +/// Accumulates what one atom contributes while its row id and value terms are +/// walked. +#[derive(Default)] +struct AtomBinder { + /// Conditions local to this atom: literal comparisons, plus the equalities + /// that a variable repeated within this one atom gives rise to, that is, + /// `atom(x, x)`. + conditions: Vec, + /// The atom's projection, mapping each bound variable part onto the query + /// column carrying it. + attributes: Vec<(String, Expr)>, + bindings: Vec, + /// The query column each variable part was *first* bound to in this atom, so + /// a repeated occurrence can be turned into an equality against it. + bound: HashMap<(ir::VarIdx, VarPart), String>, +} + +impl AtomBinder { + /// Binds `var`'s parts to `columns`, which must resolve to one query column + /// per part. + fn bind<'a>( + &mut self, + index: ir::VarIdx, + var: &FriendlyVar, + columns: impl Iterator, + ) -> Result<(), SyntaxError> { + let parts: Vec<(VarPart, String)> = var.parts().collect(); + let columns: Vec<&QueryEngineCol> = columns.collect(); + if parts.len() != columns.len() { + return Err(SyntaxError::new(format!( + "FLIR binds variable '{}', which flattens into {} column(s), to a \ + column resolving to {} query column(s)", + var.name, + parts.len(), + columns.len() + ))); + } + for ((part, name), column) in parts.into_iter().zip(columns) { + let scalar_type = ScalarType::from(*column.ty()); + let column = column.name().to_string(); + match self.bound.entry((index, part)) { + Entry::Vacant(slot) => { + self.attributes + .push((name.clone(), Expr::from(VarExpr::new(column.clone())))); + self.bindings.push(Binding { + var: index, + part, + name, + scalar_type, + }); + slot.insert(column); + } + Entry::Occupied(first) => { + // The variable is repeated within this single atom, as in + // `R(x, x)`. That is a local equality condition on this one + // relation rather than a join condition, and the projection + // has to expose the attribute exactly once — two attributes + // of the same name would collide in the projected schema. + self.conditions.push(Expr::from(BinaryExpr { + operator: Operator::Equal, + left: Expr::from(VarExpr::new(first.get().clone())), + right: Expr::from(VarExpr::new(column)), + })); + } + } + } + Ok(()) + } +} + +/// The schema of the relation a rule evaluates to. +/// +/// Its columns are the parts the rule's output binds, in the +/// `(VarIdx, VarPart)` order [`FlirProgram::conjunctive_query`] reports +/// them in, and their types come from the query columns those parts resolve to +/// rather than from the FLIR variables — a row id's two halves reach the query +/// engine as plain unsigned integers, which the variable's [`ir::ColType`] does +/// not say. +fn rule_output_schema(name: &EntityRef, bindings: &[Binding]) -> TableSchema { + TableSchema::new( + name.clone(), + bindings + .iter() + .map(|binding| Column::new(binding.name.clone(), binding.scalar_type)) + .collect(), + // A rule declares no key of its own, and nothing consumes the primary + // keys of a derived relation yet. The row id parts it binds would be the + // candidate once something does. + vec![], + ) +} + +fn friendly_var(vars: &[FriendlyVar], index: ir::VarIdx) -> Result<&FriendlyVar, SyntaxError> { + vars.get(index as usize) + .ok_or_else(|| SyntaxError::new(format!("FLIR var idx {index} out of bounds"))) +} + +/// Derives the join condition of a conjunctive query: one [`JoinVariable`] per +/// variable part that more than one atom binds. +/// +/// A part bound by a single atom is dropped. It is not an equality class, so it +/// is not part of a join condition — it reaches the output through its atom's +/// schema, which is also how it stays available to an enclosing antijoin. +/// +/// Grouping runs through a [`BTreeMap`] keyed on `(VarIdx, VarPart)`, so the +/// resulting order follows the FLIR variable indices instead of a hash order. +/// Plans have to be reproducible for a given input. +fn join_variables(plans: &[AtomPlan]) -> Vec { + let mut occurrences: BTreeMap<(ir::VarIdx, VarPart), Vec<(RelationIdx, String)>> = + BTreeMap::new(); + for (relation, plan) in plans.iter().enumerate() { + for binding in &plan.bindings { + occurrences + .entry((binding.var, binding.part)) + .or_default() + .push((relation, binding.name.clone())); + } + } + occurrences + .into_values() + .filter(|occurrences| occurrences.len() > 1) + .map(|occurrences| JoinVariable { + // Every atom projects a given part onto the same name, so the first + // occurrence's name is the shared output name — and it is the copy + // the join keeps active, since shadowing favours the earlier + // relation. + name: occurrences[0].1.clone(), + occurrences: occurrences + .into_iter() + .map(|(relation, name)| (relation, Expr::from(VarExpr::new(name)))) + .collect(), + }) + .collect() +} + +/// The key an [`AntiJoinExpr`] between the two sides of a rule compares on: +/// every variable part that both sides bind. +/// +/// Note that a part occurring only once *within* a side belongs here all the +/// same. It is not a join variable of that side's conjunctive query, but it is +/// bound by that side's schema, and the antijoin does have to compare on it. +fn antijoin_key(left: &[Binding], right: &[Binding]) -> Vec<(Expr, Expr)> { + let right: BTreeMap<(ir::VarIdx, VarPart), &str> = right + .iter() + .map(|binding| ((binding.var, binding.part), binding.name.as_str())) + .collect(); + left.iter() + .filter_map(|binding| { + right + .get(&(binding.var, binding.part)) + .map(|counterpart| (binding.name.as_str(), *counterpart)) + }) + .map(|(left, right)| { + ( + Expr::from(VarExpr::new(left)), + Expr::from(VarExpr::new(right)), + ) + }) + .collect() +} + +pub trait TryReduce: Iterator> { + /// Reduces to a single item, short-circuiting on the first [`Err`]. + /// + /// Unlike collecting into a `Vec` first, nothing is allocated, and unlike + /// [`Iterator::reduce`] the items may fail. Note that `f` itself is + /// infallible: the fallibility belongs to the items, not to the step that + /// combines two of them. + fn try_reduce(mut self, mut f: impl FnMut(T, T) -> T) -> Result, E> + where + Self: Sized, + { + let Some(first) = self.next().transpose()? else { + return Ok(None); + }; + self.try_fold(first, |acc, item| Ok(f(acc, item?))) + .map(Some) + } +} + +impl TryReduce for I where I: Iterator> {} + +impl From<&ir::Path> for EntityRef { + fn from(value: &ir::Path) -> Self { + EntityRef::from(value.to_string()) + } +} + +impl From<&ir::Lit> for Literal { + fn from(value: &ir::Lit) -> Self { + match value { + ir::Lit::Int { value } => Literal::Iint((*value).into()), + ir::Lit::String { value } => Literal::String(value.clone()), + } + } +} + +impl From for ScalarType { + fn from(value: NativeScalarType) -> Self { + match value { + NativeScalarType::Iint => ScalarType::Iint, + NativeScalarType::Uint => ScalarType::Uint, + NativeScalarType::String => ScalarType::String, + } + } +} + +impl From for ScalarType { + fn from(value: QueryEngineScalarType) -> Self { + match value { + // A row id's two halves reach the query engine as plain unsigned + // integers, so every query-engine type is a native one by this + // point. + QueryEngineScalarType::Native(native) => ScalarType::from(native), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::relational::expr::RelExpr; + use crate::test_utils::flir::{ + atom, atom_props, builtin_int, equality, lit_term, rule_entry, table_entry, var_term, + }; + + /// The one base table these tests declare, and the entity every atom here is + /// over. The shared builders are told a name; this is the only one needed. + const T: &str = "t"; + + /// A builder with one base table `t` whose columns are given as + /// `(name, type)` pairs, so [`FlirProgram::atom`] can be driven + /// directly. + fn builder_with_table(columns: Vec<(&str, ir::ColType)>) -> FlirProgram { + let mut builder = FlirProgram::empty(); + builder + .table_declaration(&table_entry(T, columns)) + .expect("A single base table declaration must succeed"); + builder + } + + /// Destructures the `Projection(Selection?(Source))` shape an atom lowers to. + fn assert_projection(expr: &Expr) -> &ProjectionExpr { + match expr { + Expr::Relational(RelExpr::Projection(projection)) => projection, + other => panic!("Expected a relational projection expression, got {other:?}"), + } + } + + /// The selection an atom's local conditions produce, if it has any. + fn maybe_assert_selection(expr: &Expr) -> Option<&SelectionExpr> { + match expr { + Expr::Relational(RelExpr::Selection(selection)) => Some(selection), + _ => None, + } + } + + fn attribute_names(projection: &ProjectionExpr) -> Vec<&str> { + projection + .attributes + .iter() + .map(|(name, _)| name.as_str()) + .collect() + } + + #[test] + fn declaring_the_same_base_table_twice_is_an_error() { + // `HashMap::insert` returns the previous value, so the check's direction + // matters: the first declaration must pass and the second must not. + let mut builder = FlirProgram::empty(); + let entry = table_entry(T, vec![("a", builtin_int())]); + builder + .table_declaration(&entry) + .expect("The first declaration of a base table must succeed"); + assert!( + builder.table_declaration(&entry).is_err(), + "A second declaration of the same base table must be rejected" + ); + } + + #[test] + fn an_atom_projects_each_bound_variable_onto_its_column() { + let mut builder = builder_with_table(vec![("a", builtin_int()), ("b", builtin_int())]); + let vars = vec![scalar_var("x"), scalar_var("y")]; + let plan = builder + .atom( + &atom(T, None, vec![(0, var_term(0)), (1, var_term(1))]), + &vars, + ) + .expect("A well-formed atom lowers"); + + assert_eq!( + attribute_names(assert_projection(&plan.relation)), + vec!["x", "y"] + ); + assert_eq!(plan.bindings.len(), 2); + // No local conditions, so no selection between projection and source. + assert!(maybe_assert_selection(&assert_projection(&plan.relation).relation).is_none()); + } + + #[test] + fn a_literal_becomes_a_local_condition_rather_than_a_binding() { + let mut builder = builder_with_table(vec![("a", builtin_int())]); + let plan = builder + .atom(&atom(T, None, vec![(0, lit_term(42))]), &[]) + .expect("An atom comparing a column to a literal lowers"); + + assert!(plan.bindings.is_empty()); + assert!(attribute_names(assert_projection(&plan.relation)).is_empty()); + assert!( + maybe_assert_selection(&assert_projection(&plan.relation).relation).is_some(), + "The literal must become a selection beneath the projection" + ); + } + + #[test] + fn a_variable_repeated_within_one_atom_is_bound_once_and_equated() { + // `t(x, x)` must not project two attributes called `x` — they would + // collide in the projected schema. The repetition is a local equality + // condition on this one relation instead, which is also what keeps the + // join's relation indices distinct per variable. + let mut builder = builder_with_table(vec![("a", builtin_int()), ("b", builtin_int())]); + let vars = vec![scalar_var("x")]; + let plan = builder + .atom( + &atom(T, None, vec![(0, var_term(0)), (1, var_term(0))]), + &vars, + ) + .expect("A repeated variable lowers"); + + assert_eq!( + attribute_names(assert_projection(&plan.relation)), + vec!["x"] + ); + assert_eq!(plan.bindings.len(), 1); + + let selection = maybe_assert_selection(&assert_projection(&plan.relation).relation) + .expect("The repetition must produce a selection"); + match &selection.condition { + Expr::Binary(binary) => { + assert_eq!(binary.operator, Operator::Equal); + assert_ne!( + binary.left, binary.right, + "The equality must compare the two distinct columns" + ); + } + other => panic!("Expected an equality condition, got {other:?}"), + } + } + + #[test] + fn a_row_id_variable_binds_both_of_its_halves() { + let mut builder = builder_with_table(vec![( + "a", + ir::ColType::RowId { + path: ir::Path::from("other"), + }, + )]); + let vars = vec![row_id_var("x")]; + let plan = builder + .atom(&atom(T, None, vec![(0, var_term(0))]), &vars) + .expect("A row id valued column lowers"); + + assert_eq!(plan.bindings.len(), 2); + assert_eq!(attribute_names(assert_projection(&plan.relation)).len(), 2); + assert_eq!( + plan.bindings + .iter() + .map(|binding| binding.part) + .collect::>(), + vec![VarPart::RowIdHash, VarPart::RowIdCtr] + ); + } + + #[test] + fn binding_a_row_id_to_a_scalar_variable_is_an_error() { + let mut builder = builder_with_table(vec![("a", builtin_int())]); + let vars = vec![scalar_var("x")]; + assert!( + builder + .atom(&atom(T, Some(var_term(0)), vec![]), &vars) + .is_err() + ); + } + + /// An enforced rule `antecedents => consequents` over the given variables, + /// both sides being nothing but atoms. Every rule these tests state has that + /// shape; [`flir::rule_entry`](crate::test_helper::flir::rule_entry) is the + /// general form. + fn enforced_rule<'a>( + name: &str, + vars: impl IntoIterator, + antecedents: Vec, + consequents: Vec, + ) -> ir::RuleEntry { + rule_entry( + name, + ir::RuleVariant::Enforced, + vars, + atom_props(antecedents), + atom_props(consequents), + ) + } + + #[test] + fn a_flat_realm_lowers_into_a_program() { + let realm = FlatRealm { + tables: vec![table_entry( + T, + vec![("a", builtin_int()), ("b", builtin_int())], + )], + rules: vec![enforced_rule( + "r", + [("x", builtin_int()), ("y", builtin_int())], + // t(x, y) and t(x, _) share `x`, so the body is a real join. + vec![ + atom(T, None, vec![(0, var_term(0)), (1, var_term(1))]), + atom(T, None, vec![(0, var_term(0))]), + ], + vec![atom(T, None, vec![(0, var_term(0))])], + )], + }; + + let builder = FlirProgram::from_flat_realm(&realm).expect("The realm lowers"); + + assert_eq!(builder.code().len(), 1, "One rule is one statement"); + let schema = &builder + .derived_views + .get(&EntityRef::from(&ir::Path::from("r"))) + .expect("The rule must be registered under its own name") + .output_schema; + // The antecedents bind `x` and `y`, so both are output columns, with the + // types of the query columns they resolve to. + assert_eq!( + schema + .columns() + .iter() + .map(|column| (column.name(), column.scalar_type())) + .collect::>(), + vec![("x", ScalarType::Iint), ("y", ScalarType::Iint)] + ); + } + + #[test] + fn a_lowered_program_passes_the_resolver() { + // Whatever the lowering emits has to be a well-formed program: every + // variable resolves, and every relational operator's invariants hold. + // This is what actually reaches `MultiWayEquiJoinExpr::validate`. + // + // It stops short of `Pipeline::runtime`, which would go on to build the + // DBSP circuit and hit the backend's `unimplemented!` for multi way + // joins — that needs the fold-into-binary-joins pass. + let realm = FlatRealm { + tables: vec![table_entry( + T, + vec![("a", builtin_int()), ("b", builtin_int())], + )], + rules: vec![enforced_rule( + "r", + [("x", builtin_int()), ("y", builtin_int())], + vec![ + atom(T, None, vec![(0, var_term(0)), (1, var_term(1))]), + atom(T, None, vec![(0, var_term(0))]), + ], + vec![atom(T, None, vec![(0, var_term(0))])], + )], + }; + let builder = FlirProgram::from_flat_realm(&realm).expect("The realm lowers"); + + crate::host::resolver::ResolvedCode::from(builder.code) + .expect("The lowered program must resolve"); + } + + #[test] + fn declaring_the_same_rule_twice_is_an_error() { + let rule = enforced_rule( + "r", + [("x", builtin_int())], + vec![atom(T, None, vec![(0, var_term(0))])], + vec![atom(T, None, vec![(0, var_term(0))])], + ); + let realm = FlatRealm { + tables: vec![table_entry(T, vec![("a", builtin_int())])], + rules: vec![rule.clone(), rule], + }; + assert!(FlirProgram::from_flat_realm(&realm).is_err()); + } + + fn multi_way_join(expr: &Expr) -> &MultiWayEquiJoinExpr { + match expr { + Expr::Relational(rel) => match rel { + RelExpr::MultiWayEquiJoin(join) => join, + other => panic!("Expected a multi way equi join, got {other:?}"), + }, + other => panic!("Expected a relational expression, got {other:?}"), + } + } + + fn conjunctive_query(atoms: Vec, conditions: Vec) -> ConjunctiveQuery { + ConjunctiveQuery { atoms, conditions } + } + + fn assert_selection(expr: &Expr) -> &SelectionExpr { + match maybe_assert_selection(expr) { + Some(selection) => selection, + None => panic!("Expected a relational selection expression, got {expr:?}"), + } + } + + #[test] + fn a_single_atom_conjunctive_query_needs_no_join() { + // There is nothing to equate across atoms, and the join operators reject + // fewer than two relations, so the atom must come through as-is. + let mut builder = builder_with_table(vec![("a", builtin_int())]); + let query = conjunctive_query(vec![atom(T, None, vec![(0, var_term(0))])], vec![]); + let (expr, bindings) = builder + .conjunctive_query(&query, &[scalar_var("x")]) + .expect("A one-atom conjunctive query lowers"); + + assert_eq!(attribute_names(assert_projection(&expr)), vec!["x"]); + assert_eq!(bindings.len(), 1); + } + + #[test] + fn two_atoms_sharing_a_variable_lower_to_a_join_on_that_variable() { + let mut builder = builder_with_table(vec![("a", builtin_int()), ("b", builtin_int())]); + // t(x, y) and t(x, z): `x` is shared, `y` and `z` are not. + let query = conjunctive_query( + vec![ + atom(T, None, vec![(0, var_term(0)), (1, var_term(1))]), + atom(T, None, vec![(0, var_term(0)), (1, var_term(2))]), + ], + vec![], + ); + let vars = vec![scalar_var("x"), scalar_var("y"), scalar_var("z")]; + let (expr, bindings) = builder + .conjunctive_query(&query, &vars) + .expect("A two-atom conjunctive query lowers"); + + let join = multi_way_join(&expr); + assert_eq!(join.relations.len(), 2); + assert_eq!( + summary(&join.on), + vec![("x".to_string(), vec![0, 1])], + "Only the shared variable may appear in the join condition" + ); + // `y` and `z` are still bound by the query as a whole, which is what + // keeps them available to an enclosing antijoin. + assert_eq!(bindings.len(), 3); + assert!(join.validate().is_ok()); + } + + #[test] + fn two_atoms_sharing_nothing_lower_to_a_cartesian_product() { + let mut builder = builder_with_table(vec![("a", builtin_int())]); + let query = conjunctive_query( + vec![ + atom(T, None, vec![(0, var_term(0))]), + atom(T, None, vec![(0, var_term(1))]), + ], + vec![], + ); + let vars = vec![scalar_var("x"), scalar_var("y")]; + let (expr, _bindings) = builder + .conjunctive_query(&query, &vars) + .expect("Atoms sharing no variable still lower"); + + assert!( + multi_way_join(&expr).on.is_empty(), + "An empty join condition is how a cartesian product is expressed" + ); + } + + #[test] + fn several_conditions_lower_to_one_selection_on_top_of_the_join() { + // The conditions are ANDed into a single condition, so exactly one + // selection sits on top of the join rather than one selection per + // condition chained after another. + let mut builder = builder_with_table(vec![("a", builtin_int()), ("b", builtin_int())]); + // t(x, y) and t(x, z), with `y = 1` and `z = 2`. + let query = conjunctive_query( + vec![ + atom(T, None, vec![(0, var_term(0)), (1, var_term(1))]), + atom(T, None, vec![(0, var_term(0)), (1, var_term(2))]), + ], + vec![ + equality(var_term(1), lit_term(1)), + equality(var_term(2), lit_term(2)), + ], + ); + let vars = vec![scalar_var("x"), scalar_var("y"), scalar_var("z")]; + let (expr, _bindings) = builder + .conjunctive_query(&query, &vars) + .expect("A conjunctive query with conditions lowers"); + + let selection = assert_selection(&expr); + // What sits directly beneath the selection is the join itself, and not + // another selection carrying the second condition. + multi_way_join(&selection.relation); + match &selection.condition { + Expr::Binary(and) => { + assert_eq!(and.operator, Operator::And); + for side in [&and.left, &and.right] { + match side { + Expr::Binary(equality) => assert_eq!(equality.operator, Operator::Equal), + other => panic!("Expected an equality condition, got {other:?}"), + } + } + } + other => panic!("Expected the two conditions to be ANDed, got {other:?}"), + } + } + + #[test] + fn a_conjunctive_query_without_atoms_is_an_error() { + let mut builder = builder_with_table(vec![("a", builtin_int())]); + assert!( + builder + .conjunctive_query(&conjunctive_query(vec![], vec![]), &[]) + .is_err() + ); + } + + #[test] + fn an_atom_over_an_undeclared_entity_is_an_error() { + let mut builder = builder_with_table(vec![("a", builtin_int())]); + let atom = ir::Atom { + entity: ir::Path::from("nonexistent"), + row_id: None, + values: vec![], + }; + assert!(builder.atom(&atom, &[]).is_err()); + } + + fn scalar_var(name: &str) -> FriendlyVar { + FriendlyVar { + name: ir::Path::from(name), + ty: ir::ColType::BuiltinTy { + builtin_ty: ir::BuiltinTy::BuiltinInt, + }, + } + } + + fn row_id_var(name: &str) -> FriendlyVar { + FriendlyVar { + name: ir::Path::from(name), + ty: ir::ColType::RowId { + path: ir::Path::from("some_table"), + }, + } + } + + fn binding(var: ir::VarIdx, part: VarPart, name: &str) -> Binding { + Binding { + var, + part, + name: name.to_string(), + // Irrelevant to join-variable and antijoin-key derivation; the + // schema tests below assert on types via `atom` instead. + scalar_type: ScalarType::Null, + } + } + + /// An [`AtomPlan`] whose relation is a stand-in: only the bindings matter to + /// [`join_variables`] and [`antijoin_key`]. + fn plan(bindings: Vec) -> AtomPlan { + AtomPlan { + relation: Expr::from(VarExpr::new("atom")), + bindings, + } + } + + /// The relation indices and output name of each derived join variable. + fn summary(variables: &[JoinVariable]) -> Vec<(String, Vec)> { + variables + .iter() + .map(|variable| { + ( + variable.name.clone(), + variable + .occurrences + .iter() + .map(|(relation, _)| *relation) + .collect(), + ) + }) + .collect() + } + + #[test] + fn a_scalar_variable_flattens_into_one_part() { + let parts: Vec<(VarPart, String)> = scalar_var("x").parts().collect(); + assert_eq!(parts, vec![(VarPart::Scalar, "x".to_string())]); + } + + #[test] + fn a_row_id_variable_flattens_into_a_hash_and_a_counter_part() { + let var = row_id_var("x"); + let parts: Vec = var.parts().map(|(part, _)| part).collect(); + assert_eq!(parts, vec![VarPart::RowIdHash, VarPart::RowIdCtr]); + // The names have to differ, or the projection would collide with itself. + let names: Vec = var.parts().map(|(_, name)| name).collect(); + assert_ne!(names[0], names[1]); + assert!(names.iter().all(|name| name.contains('x'))); + } + + #[test] + fn a_variable_shared_by_two_atoms_becomes_one_join_variable() { + let plans = vec![ + plan(vec![binding(0, VarPart::Scalar, "x")]), + plan(vec![binding(0, VarPart::Scalar, "x")]), + ]; + assert_eq!( + summary(&join_variables(&plans)), + vec![("x".to_string(), vec![0, 1])] + ); + } + + #[test] + fn a_variable_bound_by_a_single_atom_is_not_a_join_variable() { + // It constrains nothing, and it still reaches the output through its + // atom's schema — which is what keeps it available to the antijoin. + let plans = vec![ + plan(vec![ + binding(0, VarPart::Scalar, "x"), + binding(1, VarPart::Scalar, "lonely"), + ]), + plan(vec![binding(0, VarPart::Scalar, "x")]), + ]; + assert_eq!( + summary(&join_variables(&plans)), + vec![("x".to_string(), vec![0, 1])] + ); + } + + #[test] + fn atoms_sharing_no_variable_yield_an_empty_join_condition() { + let plans = vec![ + plan(vec![binding(0, VarPart::Scalar, "x")]), + plan(vec![binding(1, VarPart::Scalar, "y")]), + ]; + assert!(join_variables(&plans).is_empty()); + } + + #[test] + fn a_shared_row_id_variable_yields_one_join_variable_per_half() { + // Equality on a row id is equality on the hash *and* the counter, so the + // two halves are two independent equality classes. + let plans = vec![ + plan(vec![ + binding(0, VarPart::RowIdHash, "xRowIdHash"), + binding(0, VarPart::RowIdCtr, "xRowIdCtr"), + ]), + plan(vec![ + binding(0, VarPart::RowIdHash, "xRowIdHash"), + binding(0, VarPart::RowIdCtr, "xRowIdCtr"), + ]), + ]; + assert_eq!( + summary(&join_variables(&plans)), + vec![ + ("xRowIdHash".to_string(), vec![0, 1]), + ("xRowIdCtr".to_string(), vec![0, 1]), + ] + ); + } + + #[test] + fn a_variable_shared_by_three_atoms_has_three_occurrences() { + let plans = vec![ + plan(vec![binding(0, VarPart::Scalar, "x")]), + plan(vec![binding(0, VarPart::Scalar, "x")]), + plan(vec![binding(0, VarPart::Scalar, "x")]), + ]; + assert_eq!( + summary(&join_variables(&plans)), + vec![("x".to_string(), vec![0, 1, 2])] + ); + } + + #[test] + fn join_variables_are_ordered_by_flir_variable_index() { + // Not by the order the atoms happened to bind them in, so that the same + // input always lowers to the same plan. + let plans = vec![ + plan(vec![ + binding(2, VarPart::Scalar, "c"), + binding(0, VarPart::Scalar, "a"), + ]), + plan(vec![ + binding(0, VarPart::Scalar, "a"), + binding(2, VarPart::Scalar, "c"), + binding(1, VarPart::Scalar, "b"), + ]), + plan(vec![binding(1, VarPart::Scalar, "b")]), + ]; + let names: Vec = join_variables(&plans) + .into_iter() + .map(|variable| variable.name) + .collect(); + assert_eq!(names, vec!["a", "b", "c"]); + } + + #[test] + fn derived_join_variables_satisfy_the_join_operator_invariants() { + // The two halves of this change have to fit: whatever `join_variables` + // derives must be constructible, which is the check that no singleton + // and no repeated relation index slips through. + let plans = vec![ + plan(vec![ + binding(0, VarPart::Scalar, "x"), + binding(1, VarPart::Scalar, "only_here"), + ]), + plan(vec![ + binding(0, VarPart::Scalar, "x"), + binding(2, VarPart::Scalar, "y"), + ]), + plan(vec![binding(2, VarPart::Scalar, "y")]), + ]; + let on = join_variables(&plans); + let relations = plans.into_iter().map(|plan| plan.relation).collect(); + MultiWayEquiJoinExpr::new(relations, on, None) + .expect("Derived join variables must satisfy the operator's invariants"); + } + + #[test] + fn the_antijoin_key_is_the_intersection_of_both_sides() { + let left = vec![ + binding(0, VarPart::Scalar, "shared"), + binding(1, VarPart::Scalar, "left_only"), + ]; + let right = vec![ + binding(0, VarPart::Scalar, "shared"), + binding(2, VarPart::Scalar, "right_only"), + ]; + let key = antijoin_key(&left, &right); + assert_eq!(key.len(), 1); + assert_eq!( + key[0], + ( + Expr::from(VarExpr::new("shared")), + Expr::from(VarExpr::new("shared")) + ) + ); + } + + #[test] + fn the_antijoin_key_matches_parts_rather_than_variables() { + // Both sides bind variable 0, but the hash half only appears on the + // left, so only the counter half may be compared. + let left = vec![ + binding(0, VarPart::RowIdHash, "xRowIdHash"), + binding(0, VarPart::RowIdCtr, "xRowIdCtr"), + ]; + let right = vec![binding(0, VarPart::RowIdCtr, "xRowIdCtr")]; + let key = antijoin_key(&left, &right); + assert_eq!(key.len(), 1); + assert_eq!(key[0].0, Expr::from(VarExpr::new("xRowIdCtr"))); + } + + #[test] + fn disjoint_sides_produce_an_empty_antijoin_key() { + let left = vec![binding(0, VarPart::Scalar, "x")]; + let right = vec![binding(1, VarPart::Scalar, "y")]; + assert!(antijoin_key(&left, &right).is_empty()); + } + + fn translate_json_flir(file_name: &str) -> FlirProgram { + let flat_realm = coln_flir_rs::test_utils::load_theory_from_json(file_name); + FlirProgram::from_flat_realm(&flat_realm) + .unwrap_or_else(|_| panic!("{file_name} is convertible to a query program")) + } + + #[test] + fn graph_flir() { + let program = translate_json_flir("Graph.json"); + println!("{}", program.to_tree()); + } + + #[test] + fn graph_of_graphs_flir() { + let program = translate_json_flir("GraphOfGraphs.json"); + println!("{}", program.to_tree()); + } +} diff --git a/packages/coln-query/src/api/schema.rs b/packages/coln-query/src/api/schema.rs deleted file mode 100644 index cee5c575..00000000 --- a/packages/coln-query/src/api/schema.rs +++ /dev/null @@ -1,31 +0,0 @@ -// SPDX-FileCopyrightText: 2026 Coln contributors -// -// SPDX-License-Identifier: Apache-2.0 OR MIT - -//! This module provides implementations for passing references to tables -//! ([TableRef]) and communicating a schema of a table ([TableSchema]). - -use crate::scalarial::ScalarType; - -/// An identifier that uniquely identifies a table (globally across the store). -#[derive(Eq, PartialEq, Hash, Debug, Clone)] -pub struct TableRef { - inner: String, -} - -pub struct TableSchema { - /// The table's unique identifier/name. - name: TableRef, - /// All fields of the table in their physical order. - columns: Vec, - /// The list of (possibly compound) primary keys into the table, specified - /// as indices into the [`columns`](Self::columns). - primary_keys: Vec>, -} - -pub struct Column { - /// The column's name. - name: String, - /// Ihe column's (scalar) type. - scalar_type: ScalarType, -} diff --git a/packages/coln-query/src/api/store.rs b/packages/coln-query/src/api/store.rs index a65a3eb7..b38df132 100644 --- a/packages/coln-query/src/api/store.rs +++ b/packages/coln-query/src/api/store.rs @@ -2,26 +2,51 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -//! This defines the [API for a transactional store](TxStore). +//! How coln-store tells us about new base facts: the [API for a transactional +//! store](TxStore), and coln-query's side of it. +//! +//! This is the *push* direction, and the only one the incremental backend needs. +//! A DBSP circuit holds the base data it was fed, so evaluating a standing query +//! never asks anybody for anything. coln-store hands over a [`StoreDelta`], we +//! route it into the circuit's inputs, step, and report what changed downstream. +//! +//! Note which way the trait points. [`TxStore`] is implemented by _this_ crate +//! and called by coln-store, because applying a delta is the engine's operation +//! to define. That makes the whole push path one-directional: nothing in here +//! ever calls back into coln-store. +//! +//! # Why this module is private +//! +//! Deliberately so, and *not* an oversight to be corrected by a `pub`: the +//! module's privacy is what keeps [`TxStore`]'s methods out of a caller's reach. +//! +//! [`Tx`](super::transaction::Tx) is the only thing that may drive a +//! [`TxStore`]: calling [`apply`](TxStore::apply) on an implementor (such as +//! [`ColnQuery`](super::ColnQuery)) directly would step the circuit +//! behind the typestate machine's back, leaving a transaction half-applied +//! with no rollback guard to undo it. +//! +//! `pub(crate)` is not an option, because the trait appears in the bounds of +//! public items and must stay reachable to the type system. But *reachable* and +//! *nameable* are two different things: a `pub` trait in a private module still +//! discharges those bounds, while a downstream crate has no path to `use` it — +//! and without the trait in scope, no call to its methods resolves. Sealing it +//! the usual way, with a private supertrait, would not do this; that stops +//! foreign `impl`s but not foreign calls. -use super::{deltas::StoreDelta, transaction::ApplicationOutcome}; -use std::error::Error; +use super::{deltas::StoreDelta, transaction::TxOutcome}; -/// A generic transactional engine/store. +/// A generic transactional engine/store. Implemented by +/// [`ColnQuery`](super::ColnQuery), callable only from within this crate. pub trait TxStore { - type ApplicationOk: Into; - type ApplicationError: Error + Clone; - type RollbackError: Error + Clone; - type CommitError: Error + Clone; + type Error: std::error::Error + Clone; /// Executes and applies the transaction given by the updates in `delta`. - fn apply(&mut self, delta: StoreDelta) -> Result; + fn apply(&mut self, delta: StoreDelta) -> Result; /// Undoes the last transaction by rolling back every state change caused /// by that transaction. Should only fail in exceptional circumstances. - fn rollback(&mut self) -> Result<(), Self::RollbackError>; + fn rollback(&mut self) -> Result<(), Self::Error>; /// Commits the last transaction. Possibly, a no-op or do some cleanup. /// Should only fail in exceptional circumstances. - fn commit(&mut self) -> Result<(), Self::CommitError>; + fn commit(&mut self) -> Result<(), Self::Error>; } - -// TODO: Implement for the engine! diff --git a/packages/coln-query/src/api/transaction.rs b/packages/coln-query/src/api/transaction.rs index c62236a0..536cd6b6 100644 --- a/packages/coln-query/src/api/transaction.rs +++ b/packages/coln-query/src/api/transaction.rs @@ -7,69 +7,215 @@ use super::{ deltas::{DerivedDataDelta, StoreDelta, TableDelta}, + error::UnsafeApplyError, store::TxStore, - violations::Violations, + violations::{ViolationsDelta, ViolationsSet}, }; -/// We use the Typestate-Pattern for compile-time enforced transaction states -/// and their transitions to provide a hard-to-abuse API. +/// The query engine's primary API is a transaction represented as a state +/// machine with the following five states: +/// +/// 1. [`Prepare`]: Data can be fed into the base tables (EDPs) in +/// row-oriented format. +/// 2. [`Pending`]: The query processing finished and all hard (enforced) +/// constraints are met. The **caller still has to explicitly call +/// [`commit`](Tx::commit) to avoid a rollback**, or can call +/// [`abort`](Tx::abort) to rollback the transaction for some +/// reason (maybe an end-user requested an abort). +/// 3. [`Committed`]: The transaction cannot be rolled back anymore and has +/// been committed. Any query results (derived views, monitored constraint +/// violations) can now be obtained. +/// 4. [`Aborted`]: For some reason the caller decided to abort. +/// The query engine's state has been rolled back as if the transaction +/// never happened. +/// 5. [`Rejected`]: The transaction could not commit because of some violations +/// of hard (enforced) constraints. Any state caused by the transaction +/// is already rolled back within the query engine. Any violation can be +/// reported back. +/// +/// ```text +/// +------------+ +/// | Prepare | +/// +------+-----+ +/// | +/// | try_commit() +/// | (runs query engine and checks hard constraints) +/// | +/// +----------------+ +/// | | +/// met violated +/// | | +/// v v +/// +------------+ +------------+ +/// | Pending | | Rejected | +/// +------+-----+ +------------+ +/// | +/// +----------------+ +/// | | +/// commit() abort() +/// | | +/// v v +/// +------------+ +------------+ +/// | Committed | | Aborted | +/// +------------+ +------------+ +/// ``` +#[derive(Debug)] pub struct Tx { state: State, } -/// This is the initial state of a Transaction and it is open to receive table -/// deltas. +/// This is the initial state of a [transaction](Tx) and it is open to receive +/// table deltas. +#[derive(Debug)] pub struct Prepare { delta: StoreDelta, } -/// The transaction is ready to apply in theory, that is, all _mandatory_ +/// The transaction is ready to apply in theory, that is, all _enforced_ /// constraints are met (although some _monitored_ constraints may be violated). /// Yet, the transaction awaits either an approval or an end user abort. Without /// an explicit approval, any state change caused by the transaction will be /// undone. +#[derive(Debug)] pub struct Pending<'a, Store: TxStore> { + store: RollbackGuard<'a, Store>, + delta: DataDelta, +} + +/// Rolls the store back when dropped, unless it has been +/// [`disarm`](Self::disarm)ed first. This is what makes the API foolproof: a +/// caller who neither commits nor aborts gets the conservative outcome. +/// +/// The guard is a field of [`Pending`] rather than [`Pending`] itself carrying +/// the [`Drop`] impl, and that is the whole point of it existing. A type that +/// implements `Drop` cannot be destructured — Rust has to keep it whole to hand +/// it to `drop` — so `commit` and `abort` would have to move each field out from +/// under the destructor by hand, with a `ManuallyDrop` and one `ptr::read` per +/// field. Confining `Drop` to the one field that actually needs it leaves +/// `Pending` an ordinary struct that can be taken apart safely, and adding a +/// field to it costs nothing. +#[derive(Debug)] +struct RollbackGuard<'a, Store: TxStore> { store: &'a mut Store, - derived_data_delta: DerivedDataDelta, - soft_violations: Violations, + armed: bool, +} + +impl<'a, Store: TxStore> RollbackGuard<'a, Store> { + fn armed(store: &'a mut Store) -> Self { + Self { store, armed: true } + } + /// The store, with the guard left armed, so an operation that fails still + /// rolls back on the way out. + fn store(&mut self) -> &mut Store { + self.store + } + /// Stop the implicit rollback: from here on dropping the guard does nothing. + fn disarm(&mut self) { + self.armed = false; + } +} + +impl Drop for RollbackGuard<'_, Store> { + fn drop(&mut self) { + if self.armed { + // Happens in best-effort manner to avoid panicking in Drop impls. + // Possibly log the event, though. + let _ = self.store.rollback(); + } + } } -/// The transaction is finalized and applied to both the storage and query -/// engine. Any state caused by the transaction is already committed. +/// The transaction is done and fully applied to the query engine. No rollback +/// is possible anymore. +#[derive(Debug)] pub struct Committed { - derived_data_delta: DerivedDataDelta, + delta: DataDelta, } /// The transaction is committable in theory, that is, it does _not_ violate any -/// constraint but the end user decided to abort regardless. Any state caused by -/// the transaction is already rolled back. +/// constraint, _but the caller decided to abort regardless_. +/// Any state caused by the transaction is already rolled back. +#[derive(Debug)] pub struct Aborted {} -/// The transaction _must be_ rejected because some _mandatory_ constraints are -/// violated. Any state caused by the transaction is already rolled back. +/// The transaction _must be_ rejected because some hard (enforced) constraints +/// are violated. Any state caused by the transaction is already rolled back. +#[derive(Debug)] pub struct Rejected { - violations: Violations, + violations: ViolationsSet, } /// The outcomes that can happen if updates are applied to the store: /// -/// 1. [`Self::DerivedDataDelta`], if no constraints are violated. -/// 2. [`Self::HardViolations`], if mandatory constraints are violated. -/// 3. [`Self::SoftViolations`], if monitored constraints are violated. +/// 1. [`Self::DerivedDataDelta`], if no hard constraints are violated (but +/// possibly soft constraints). +/// 2. [`Self::HardViolationsSet`], if hard constraints are violated. +/// +/// We treat hard constraint violations as perfectly normal use and report them +/// back as part of the `Ok` case of a `Result` and reserve the `Err` case +/// for hard engine errors. +/// +/// # A set on one arm, a delta on the other +/// +/// [`HardViolationsSet`](Self::HardViolationsSet) reports the violations that +/// *exist*. [`DataDelta`](Self::DerivedDataDelta) reports how the monitored +/// violations *changed*. The rows are the same shape either way, which +/// is why they are [two types](super::violations) rather than one. +/// Which of the two a rule's violations carry follows from what the engine +/// does when one occurs: /// -/// We treat constraint violations as perfectly normal use and report them back -/// as part of the `Ok` case of a `Result` and reserve the `Err` case for hard -/// engine errors. -pub enum ApplicationOutcome { +/// A transaction violating an enforced (hard) constraint is rolled back, +/// so the set of those violations is empty after every committed transaction. +/// Whatever the engine reports is therefore measured against nothing, +/// which makes it the whole set. +/// +/// A monitored (soft) violation is tolerated and committed, so *that* set +/// accumulates across transactions, and the engine reports it relative to what +/// was already there: a positive [`ZWeight`](super::deltas::ZWeight) is a +/// violation that appeared, a negative one a violation this transaction resolved. +/// Which is the right shape, because the engine does not keep the set +/// (the caller does, and a caller can only maintain one if needed). +/// To know whether any monitored violation is currently outstanding, +/// apply these deltas to your own set and ask *it* by integrating over the +/// respective `zweight`s. +pub enum TxOutcome { /// All constraints are met and updates in derived data are communicated - /// back. - DerivedDataDelta(DerivedDataDelta), - /// Mandatory constraints are violated. - HardViolations(Violations), - /// Monitored constraints are violated. Since they only issue a warning but - /// are tolerated in general, we nevertheless apply the transaction, obtain - /// the derived data delta, and report back about the violations. - SoftViolations(DerivedDataDelta, Violations), + /// back next to updates in the soft violations produced by monitored rules. + DerivedDataDelta(DataDelta), + /// Enforced constraints are violated. The violations, in full, as a set. + HardViolationsSet(ViolationsSet), +} + +#[derive(Debug, Clone)] +pub struct DataDelta { + derived: DerivedDataDelta, + soft_violations: ViolationsDelta, +} + +impl DataDelta { + pub fn new(derived: DerivedDataDelta, soft_violations: ViolationsDelta) -> Self { + Self { + derived, + soft_violations, + } + } + pub fn take_derived_data_delta(&mut self) -> DerivedDataDelta { + std::mem::take(&mut self.derived) + } + pub fn take_soft_violations(&mut self) -> ViolationsDelta { + std::mem::take(&mut self.soft_violations) + } +} + +impl TryFrom for DataDelta { + type Error = UnsafeApplyError; + + fn try_from(outcome: TxOutcome) -> Result { + match outcome { + TxOutcome::DerivedDataDelta(delta) => Ok(delta), + TxOutcome::HardViolationsSet(violations) => Err(UnsafeApplyError { violations }), + } + } } pub enum TryCommitOk<'a, Store: TxStore> { @@ -77,12 +223,53 @@ pub enum TryCommitOk<'a, Store: TxStore> { Rejected(Tx), } -pub enum TryCommitErr { - ApplicationError(ApplicationError), - RollbackError(RollbackError), +#[cfg(test)] +impl<'a, Store: TxStore + std::fmt::Debug> TryCommitOk<'a, Store> { + pub fn expect_pending_and_commit(self) -> Tx { + match self { + TryCommitOk::Pending(pending) => pending.commit().expect("valid tx"), + TryCommitOk::Rejected(mut rejected) => { + panic!( + "Expected valid tx but got hard constraint {}", + rejected.take_hard_violations() + ) + } + } + } + pub fn expect_rejected(self) -> Tx { + match self { + TryCommitOk::Rejected(rejected) => rejected, + TryCommitOk::Pending(pending) => { + panic!("Expected invalid tx but got pending {:?}", pending) + } + } + } +} + +#[derive(Debug, Clone)] +pub enum TryCommitErr { + TxApplyError(Error), + RollbackError(Error), +} + +impl From> for anyhow::Error { + fn from(value: TryCommitErr) -> Self { + match value { + TryCommitErr::TxApplyError(err) | TryCommitErr::RollbackError(err) => { + anyhow::Error::new(err) + } + } + } } impl Tx { + pub fn empty() -> Self { + Tx { + state: Prepare { + delta: StoreDelta::empty(), + }, + } + } pub fn new(store_delta: StoreDelta) -> Self { Tx { state: Prepare { delta: store_delta }, @@ -90,81 +277,72 @@ impl Tx { } /// Convenience method to add data beyond initialization. pub fn insert>(&mut self, deltas: I) { - self.state.delta.inner.extend(deltas); + self.state.delta.extend(deltas); } pub fn try_commit<'a, Store: TxStore>( self, store: &'a mut Store, - ) -> Result, TryCommitErr> - { + ) -> Result, TryCommitErr> { match store .apply(self.state.delta) - .map_err(TryCommitErr::ApplicationError) - .map(Into::::into)? + .map_err(TryCommitErr::TxApplyError)? { - ApplicationOutcome::DerivedDataDelta(delta) => Ok(TryCommitOk::Pending(Tx { + TxOutcome::DerivedDataDelta(delta) => Ok(TryCommitOk::Pending(Tx { state: Pending { - store, - derived_data_delta: delta, - soft_violations: Violations::none(), + store: RollbackGuard::armed(store), + delta, }, })), - ApplicationOutcome::HardViolations(violations) => { + TxOutcome::HardViolationsSet(violations) => { store.rollback().map_err(TryCommitErr::RollbackError)?; Ok(TryCommitOk::Rejected(Tx { state: Rejected { violations }, })) } - ApplicationOutcome::SoftViolations(delta, violations) => Ok(TryCommitOk::Pending(Tx { - state: Pending { - store, - derived_data_delta: delta, - soft_violations: violations, - }, - })), } } } -impl<'a, Store: TxStore> Tx> { - pub fn commit(self) -> Result, Store::CommitError> { - // Prevent the custom Drop implementation from running at this point. - let md = std::mem::ManuallyDrop::new(self.state); - // Move the store ref. This is safe because `md` will never be dropped, - // so we avoid a double-free, *and* because the returned `Committed` - // state inherits the same lifetime as `self`. - let store = unsafe { std::ptr::read(&md.store) }; - // Move the vector. This is safe because `md` will never be dropped, so - // we avoid a double-free. - let derived_data_delta = unsafe { std::ptr::read(&md.derived_data_delta) }; - store.commit()?; +impl Tx> { + pub fn commit(self) -> Result, Store::Error> { + // Plain destructuring: `Pending` has no `Drop` of its own, only its + // guard field has. + let Pending { mut store, delta } = self.state; + // The guard stays armed across the commit, so a commit that fails is + // undone when the guard drops on the way out — a half-committed + // transaction is never left behind. Only a commit that succeeded has + // nothing left to undo. + store.store().commit()?; + store.disarm(); Ok(Tx { - state: Committed { derived_data_delta }, + state: Committed { delta }, }) } - pub fn abort(self) -> Result, Store::RollbackError> { - // Prevent the custom Drop implementation from running at this point. - let md = std::mem::ManuallyDrop::new(self.state); - // Move the store ref. This is safe because `md` will never be dropped, - // so we avoid a double-free, *and* because the returned `Aborted` - // state inherits the same lifetime as `self`. - let store = unsafe { std::ptr::read(&md.store) }; - // Move the vector. This is safe because `md` will never be dropped, so - // we avoid a double-free, but also required to free the heap allocation - // behind the vector. - let derived_data_delta = unsafe { std::ptr::read(&md.derived_data_delta) }; - store.rollback()?; + pub fn abort(self) -> Result, Store::Error> { + // The two deltas are simply dropped here, as an aborted transaction has + // no results to report. + let Pending { mut store, .. } = self.state; + // Disarmed *before* the rollback, unlike [`commit`](Self::commit) above: + // the guard's own job is to roll back, so leaving it armed here would + // only retry — inside a destructor, and with the error already reported — + // an operation that just failed. + store.disarm(); + store.store().rollback()?; Ok(Tx { state: Aborted {} }) } } -// This is to make the API foolproof: If the caller does neither commit nor -// abort the transaction, we take the conservative approach and rollback any -// state change caused by it. -impl<'a, Store: TxStore> Drop for Pending<'a, Store> { - fn drop(&mut self) { - // Happens in best-effort manner to avoid panicking in Drop impls. - // Possibly log the event, though. - let _ = self.store.rollback(); +impl Tx { + pub fn take_derived_data_delta(&mut self) -> DerivedDataDelta { + self.state.delta.take_derived_data_delta() + } + pub fn take_soft_violations(&mut self) -> ViolationsDelta { + self.state.delta.take_soft_violations() + } +} + +impl Tx { + pub fn take_hard_violations(&mut self) -> ViolationsSet { + std::mem::take(&mut self.state.violations) } } diff --git a/packages/coln-query/src/api/violations.rs b/packages/coln-query/src/api/violations.rs index ff0a2172..b53b2c8b 100644 --- a/packages/coln-query/src/api/violations.rs +++ b/packages/coln-query/src/api/violations.rs @@ -4,23 +4,119 @@ //! This module is concerned with violations and how to report them. Many things //! are still TBD. +//! +//! Violations come in two [interpretations](Interpretation), and they are two +//! types rather than one: a [`ViolationsSet`] says which violations *are*, a +//! [`ViolationsDelta`] says how they *changed*. The rows are identical either +//! way — the same [`TableDelta`]s of ±weighted counterexamples — so nothing but +//! the type keeps a consumer from reading one as the other, and reading a delta +//! as a set is the mistake worth preventing: it reports a repaired constraint as +//! a broken one. +//! +//! Which interpretation a rule's violations carry follows from what the engine +//! does when one occurs. See +//! [`TxOutcome`](super::transaction::TxOutcome), whose two violation arms are +//! where this distinction comes from. use super::deltas::TableDelta; +use std::marker::PhantomData; + +/// Violations that exist: the whole set of them, as of now. +#[derive(Default, Debug, Clone, PartialEq, Eq)] +pub struct Set; + +/// A change in the violations, relative to whatever previous transactions left +/// behind. A positive [`ZWeight`](super::deltas::ZWeight) is a violation that +/// appeared, a negative one a violation that was resolved. +#[derive(Default, Debug, Clone, PartialEq, Eq)] +pub struct Delta; + +/// What a [`Violations`] is a statement about. Implemented by [`Set`] and +/// [`Delta`]; there is no third case. +pub trait Interpretation { + /// How this interpretation names itself when violations are printed. The + /// rows alone do not say which one they are, so the label is the only thing + /// that tells a reader whether a `-1` in the output is bad news or good. + const LABEL: &'static str; +} + +impl Interpretation for Set { + const LABEL: &'static str = "Violations"; +} + +impl Interpretation for Delta { + const LABEL: &'static str = "Violations Delta"; +} + +pub type ViolationsSet = Violations; +pub type ViolationsDelta = Violations; /// For each query which is checking a constraint, this reports back identified /// counterexamples. -pub struct Violations { +/// +/// Always named through [`ViolationsSet`] or [`ViolationsDelta`]; the parameter +/// carries no data and exists only so the two cannot be mixed up. See the +/// [module docs](self). +#[derive(Default, Debug, Clone, PartialEq, Eq)] +pub struct Violations { /// Contains the counter examples for each unmet constraint. Note that - /// [`TableRef`](super::schema::TableRef) refers to a derived view - /// (defined through a query) rather than a physical base table here. + /// [`EntityRef`](crate::relational::schema::EntityRef) refers to a derived + /// view (defined through a query) rather than a physical base table here. inner: Vec, + interpretation: PhantomData, } -impl Violations { +impl Violations { /// Report no violations. - pub fn none() -> Self { + pub fn empty() -> Self { Self { inner: Vec::new(), // Does not allocate. + interpretation: PhantomData, + } + } + pub fn is_empty(&self) -> bool { + self.inner.iter().all(|table_delta| table_delta.is_empty()) + } + pub fn iter(&self) -> impl Iterator { + self.into_iter() + } + pub fn into_inner(self) -> Vec { + self.inner + } + pub fn extend>(&mut self, deltas: T) { + self.inner.extend(deltas); + } +} + +impl IntoIterator for Violations { + type Item = TableDelta; + type IntoIter = std::vec::IntoIter; + + fn into_iter(self) -> Self::IntoIter { + self.inner.into_iter() + } +} + +impl<'a, I: Interpretation> IntoIterator for &'a Violations { + type Item = &'a TableDelta; + type IntoIter = std::slice::Iter<'a, TableDelta>; + + fn into_iter(self) -> Self::IntoIter { + self.inner.iter() + } +} + +impl std::fmt::Display for Violations { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "{}{}", + I::LABEL, + if self.is_empty() { " " } else { "\n" } + )?; + for violation in &self.inner { + write!(f, "{}", violation)?; } + Ok(()) } } diff --git a/packages/coln-query/src/error.rs b/packages/coln-query/src/error.rs index b4bb2727..19b62876 100644 --- a/packages/coln-query/src/error.rs +++ b/packages/coln-query/src/error.rs @@ -14,6 +14,10 @@ pub enum QueryEngineError { /// An error that occurs during an optimization pass prior to runtime. #[error(transparent)] Optimization(#[from] OptimizationError), + /// An error that occurs while lowering the plan into the operator + /// vocabulary of the chosen backend. + #[error(transparent)] + Lowering(#[from] LoweringError), /// An error which occurs during runtime of the circuit constructing, /// tree-walk interpreter. #[error(transparent)] @@ -55,6 +59,85 @@ impl OptimizationError { } } +#[derive(Error, Debug, Clone, PartialEq, Eq)] +#[error("{message}")] +/// An error that occurs while lowering the plan into the operator vocabulary of +/// the chosen backend, see [`Backend::lower`](crate::relational::Backend::lower). +/// +/// Distinct from an [`OptimizationError`] because the two stages fail for +/// different reasons: an optimization is free to decline (and the pipeline is +/// just as correct without it), whereas a lowering that cannot proceed leaves +/// behind a plan the backend has no way to execute. +pub struct LoweringError { + pub message: String, +} + +impl LoweringError { + pub fn new>(message: T) -> Self { + Self { + message: message.into(), + } + } +} + +/// A lowering re-checks the invariants of the nodes it rewrites, because a plan +/// may have been assembled or rewritten by hand. Such a violation is a +/// [`SyntaxError`] by nature, but it surfaces here. +impl From for LoweringError { + fn from(value: SyntaxError) -> Self { + Self { + message: value.message, + } + } +} + +#[derive(Error, Debug, Clone, PartialEq, Eq)] +#[error("{message}")] +/// What a [transformation rule](crate::optimizer::rewrite::TransformationRule), +/// or the driver running one, failed with. +/// +/// Deliberately *not* tied to a pipeline stage: the same rule machinery serves +/// the optimizer and the backend lowerings, so a rule reports in this shared +/// currency and each stage converts it into the error its own contract is +/// phrased in. +pub struct RewriteError { + pub message: String, +} + +impl RewriteError { + pub fn new>(message: T) -> Self { + Self { + message: message.into(), + } + } +} + +/// A rule re-checks the invariants of the nodes it rewrites, for the same +/// reason a lowering does. +impl From for RewriteError { + fn from(value: SyntaxError) -> Self { + Self { + message: value.message, + } + } +} + +impl From for LoweringError { + fn from(value: RewriteError) -> Self { + Self { + message: value.message, + } + } +} + +impl From for OptimizationError { + fn from(value: RewriteError) -> Self { + Self { + message: value.message, + } + } +} + #[derive(Error, Debug, Clone, PartialEq, Eq)] #[error("{message}")] /// An error which occurs during runtime of the circuit constructing, diff --git a/packages/coln-query/src/host/code.rs b/packages/coln-query/src/host/code.rs new file mode 100644 index 00000000..a5951a25 --- /dev/null +++ b/packages/coln-query/src/host/code.rs @@ -0,0 +1,112 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! The host-language program. + +use super::{print, stmt::Stmt}; +use std::ops::{Deref, DerefMut}; + +/// A host-language program containing multiple queries. +/// For instance, this is part of what a +/// [`coln-flir` lowering emits](crate::api::query::FlirProgram::from_flat_realm). +/// +/// This encodes what a query program is _doing_. _Upon what_ it operates (base +/// tables and their schemas) is defined one layer above in +/// a [`QueryProgram`](crate::program::QueryProgram)'s +/// [`Catalog`](crate::relational::catalog::Catalog) +#[derive(Clone, Debug, Default, Eq, PartialEq)] +pub struct QueryIr(Vec); + +impl QueryIr { + pub fn new(stmts: Vec) -> Self { + Self(stmts) + } + + /// This program rendered as an indented node tree, e.g. + /// `println!("{}", query_ir.to_tree())`. See [`super::print`]. + /// + /// [`print::to_tree`] is the same rendering for a bare `[Stmt]`, which is + /// what a sub-forest (a fixed point's step body or a function's body) + /// actually is. + pub fn to_tree(&self) -> String { + print::to_tree(self) + } + + /// Append a statement. The *only* structural mutation a program exposes: + /// a lowering builds one up statement by statement, and nothing else has a + /// reason to change its length. + pub fn push(&mut self, stmt: Stmt) { + self.0.push(stmt); + } + + pub fn into_stmts(self) -> Vec { + self.0 + } +} + +impl From> for QueryIr { + fn from(stmts: Vec) -> Self { + Self(stmts) + } +} + +impl From for Vec { + fn from(code: QueryIr) -> Self { + code.0 + } +} + +impl FromIterator for QueryIr { + fn from_iter>(stmts: I) -> Self { + Self(stmts.into_iter().collect()) + } +} + +/// Derefs to the *slice*, exactly as [`Vec`] itself does, rather than to the +/// `Vec`: a pass gets `iter`, `iter_mut`, `len` and indexing, while `clear`, +/// `truncate` and `drain` stay off a type whose point is to be a complete +/// program. Appending goes through [`QueryIr::push`]. +impl Deref for QueryIr { + type Target = [Stmt]; + + fn deref(&self) -> &Self::Target { + &self.0 + } +} + +impl DerefMut for QueryIr { + fn deref_mut(&mut self) -> &mut Self::Target { + &mut self.0 + } +} + +/// Deref coercion does not apply at a generic bound, so a `&QueryIr` handed to +/// something taking `impl IntoIterator` — as the interpreter does +/// — would not compile without this. +impl<'a> IntoIterator for &'a QueryIr { + type Item = &'a Stmt; + type IntoIter = std::slice::Iter<'a, Stmt>; + + fn into_iter(self) -> Self::IntoIter { + self.0.iter() + } +} + +impl<'a> IntoIterator for &'a mut QueryIr { + type Item = &'a mut Stmt; + type IntoIter = std::slice::IterMut<'a, Stmt>; + + fn into_iter(self) -> Self::IntoIter { + self.0.iter_mut() + } +} + +impl IntoIterator for QueryIr { + type Item = Stmt; + type IntoIter = std::vec::IntoIter; + + fn into_iter(self) -> Self::IntoIter { + self.0.into_iter() + } +} diff --git a/packages/coln-query/src/host/expr.rs b/packages/coln-query/src/host/expr.rs index 327e5e56..190eb68c 100644 --- a/packages/coln-query/src/host/expr.rs +++ b/packages/coln-query/src/host/expr.rs @@ -3,12 +3,10 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT use super::operator::Operator; +use super::resolver::{Named, Resolvable}; use crate::{ - host::stmt::BlockStmt, - host::variable::VariableSlot, - impl_from_auto_box, + host::stmt::BlockStmt, host::variable::VariableSlot, impl_from_auto_box, relational::expr::RelExpr, - util::{MemAddr, Named, Resolvable}, }; use std::fmt::{self, Debug, Display}; @@ -28,7 +26,13 @@ pub enum Expr { Function(Box), /// The single bridge into the relational layer: a relational operator is /// *also* a host expression (bindable to a var, placeable in a tuple, …). - Relational(Box), + /// + /// The only variant that does not box its payload, because there is nothing + /// to gain by it: [`RelExpr`] boxes each of *its* payloads, so it is already + /// a discriminant plus a pointer, and rustc packs this enum's discriminant + /// into the unused values of that one. `Expr` is the same size either way, + /// while every relational node saves an allocation and an indirection. + Relational(RelExpr), } impl_from_auto_box! { @@ -234,6 +238,8 @@ impl Display for Literal { } } +/// Read-only visitor. See [`ExprVisitorOwn`] for which of the three families a +/// given pass belongs in. pub trait ExprVisitor { fn visit_expr(&mut self, expr: &Expr, ctx: C) -> T { match expr { @@ -266,6 +272,8 @@ pub trait ExprVisitor { fn visit_relational_expr(&mut self, expr: &RelExpr, ctx: C) -> T; } +/// Annotating visitor. See [`ExprVisitorOwn`] for which of the three families a +/// given pass belongs in. pub trait ExprVisitorMut { fn visit_expr(&mut self, expr: &mut Expr, ctx: C) -> T { match expr { @@ -296,43 +304,94 @@ pub trait ExprVisitorMut { fn visit_relational_expr(&mut self, expr: &mut RelExpr, ctx: C) -> T; } +/// Restructuring visitor: it consumes the tree and produces a new one. +/// +/// # Which visitor family a pass belongs in +/// +/// The three families differ in what a pass is allowed to *do*, not merely in +/// how it borrows: +/// +/// - [`ExprVisitor`] (`&`) — **read**. The pass derives something from the tree +/// and leaves it untouched (type resolution, printing, interpretation). +/// - [`ExprVisitorMut`] (`&mut`) — **annotate**. The pass fills fields in place +/// but never changes the tree's *shape*; every node stays the node it was +/// (the [`Resolver`](crate::host::resolver) filling variable slots). +/// - [`ExprVisitorOwn`] (owned) — **restructure**. A node may be replaced by a +/// differently shaped one, or become the child of a node that did not exist +/// before (lowering a multi-way join into a fold of binary ones). +/// +/// The rule follows from what Rust permits. Restructuring means moving children +/// out of their parent and re-parenting them, and one cannot move out of a +/// `&mut`. A `&mut` pass would have to leave a placeholder behind for every +/// child it takes, which needs a dummy node the AST does not have, and which +/// leaves a half-rewritten tree behind when the pass fails part-way through. An +/// owned pass just moves values, and a failure drops the partial result. +/// +/// # Why the payloads arrive boxed +/// +/// Every [`Expr`] variant boxes its payload, so unboxing here would make a pass +/// pay a deallocation plus an allocation for each node it walks over — including +/// the overwhelming majority it does not rewrite at all. Handing out the `Box` +/// instead lets an unchanged node go straight back into its enum +/// (`Ok(expr.into())`, via the `From>` impls) for free, while a +/// pass that *does* consume a node still writes `let XxxExpr { .. } = *expr;` +/// exactly as it would have otherwise. +/// +/// Recursing into a child keeps the box, too, because a field can be moved out +/// of a `Box`'s contents and written back: +/// +/// ```ignore +/// fn visit_grouping_expr(&mut self, mut expr: Box, ctx: C) -> T { +/// expr.expr = self.visit_expr(expr.expr, ctx)?; +/// Ok(expr.into()) +/// } +/// ``` pub trait ExprVisitorOwn { fn visit_expr(&mut self, expr: Expr, ctx: C) -> T { match expr { - Expr::Literal(expr) => self.visit_literal_expr(*expr, ctx), - Expr::Tuple(expr) => self.visit_tuple_expr(*expr, ctx), - Expr::GetIndex(expr) => self.visit_get_index_expr(*expr, ctx), - Expr::Grouping(expr) => self.visit_grouping_expr(*expr, ctx), - Expr::Binary(expr) => self.visit_binary_expr(*expr, ctx), - Expr::Unary(expr) => self.visit_unary_expr(*expr, ctx), - Expr::Var(expr) => self.visit_var_expr(*expr, ctx), - Expr::Assign(expr) => self.visit_assign_expr(*expr, ctx), - Expr::Function(expr) => self.visit_function_expr(*expr, ctx), - Expr::Call(expr) => self.visit_call_expr(*expr, ctx), - Expr::Relational(expr) => self.visit_relational_expr(*expr, ctx), + Expr::Literal(expr) => self.visit_literal_expr(expr, ctx), + Expr::Tuple(expr) => self.visit_tuple_expr(expr, ctx), + Expr::GetIndex(expr) => self.visit_get_index_expr(expr, ctx), + Expr::Grouping(expr) => self.visit_grouping_expr(expr, ctx), + Expr::Binary(expr) => self.visit_binary_expr(expr, ctx), + Expr::Unary(expr) => self.visit_unary_expr(expr, ctx), + Expr::Var(expr) => self.visit_var_expr(expr, ctx), + Expr::Assign(expr) => self.visit_assign_expr(expr, ctx), + Expr::Function(expr) => self.visit_function_expr(expr, ctx), + Expr::Call(expr) => self.visit_call_expr(expr, ctx), + Expr::Relational(expr) => self.visit_relational_expr(expr, ctx), } } - fn visit_literal_expr(&mut self, expr: LiteralExpr, ctx: C) -> T; - fn visit_tuple_expr(&mut self, expr: TupleExpr, ctx: C) -> T; - fn visit_get_index_expr(&mut self, expr: GetIndexExpr, ctx: C) -> T; - fn visit_grouping_expr(&mut self, expr: GroupingExpr, ctx: C) -> T; - fn visit_binary_expr(&mut self, expr: BinaryExpr, ctx: C) -> T; - fn visit_unary_expr(&mut self, expr: UnaryExpr, ctx: C) -> T; - fn visit_var_expr(&mut self, expr: VarExpr, ctx: C) -> T; - fn visit_assign_expr(&mut self, expr: AssignExpr, ctx: C) -> T; - fn visit_function_expr(&mut self, expr: FunctionExpr, ctx: C) -> T; - fn visit_call_expr(&mut self, expr: CallExpr, ctx: C) -> T; + fn visit_literal_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_tuple_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_get_index_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_grouping_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_binary_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_unary_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_var_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_assign_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_function_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_call_expr(&mut self, expr: Box, ctx: C) -> T; /// Bridge into the relational layer. See [`ExprVisitor::visit_relational_expr`]. + /// + /// The one payload that does not arrive boxed, because [`Expr::Relational`] + /// does not box it either. fn visit_relational_expr(&mut self, expr: RelExpr, ctx: C) -> T; } -impl MemAddr for Expr {} -impl MemAddr for LiteralExpr {} -impl MemAddr for TupleExpr {} -impl MemAddr for GroupingExpr {} -impl MemAddr for BinaryExpr {} -impl MemAddr for UnaryExpr {} -impl MemAddr for VarExpr {} -impl MemAddr for AssignExpr {} -impl MemAddr for FunctionExpr {} -impl MemAddr for CallExpr {} +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn wrapping_a_relational_operator_in_an_expr_is_free() { + // Why `Relational` is the one variant that does not box its payload: + // `RelExpr` is a discriminant plus a pointer, and rustc packs `Expr`'s + // discriminant into the values that one does not use. Inlining it costs + // no space and saves an allocation per relational node. If this ever + // stops holding, the trade-off is worth revisiting rather than keeping + // by habit. + assert_eq!(size_of::(), size_of::()); + assert_eq!(size_of::(), size_of::>() * 2); + } +} diff --git a/packages/coln-query/src/host/function.rs b/packages/coln-query/src/host/function.rs index 0c944315..55c8fde2 100644 --- a/packages/coln-query/src/host/function.rs +++ b/packages/coln-query/src/host/function.rs @@ -2,7 +2,9 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::{host::expr::FunctionExpr, host::variable::Environment, typing::ExprType}; +use crate::{ + host::expr::FunctionExpr, host::variable::Environment, typing::type_resolver::ExprType, +}; use std::{ cell::RefCell, fmt::{self, Debug, Display}, diff --git a/packages/coln-query/src/host/interpreter.rs b/packages/coln-query/src/host/interpreter.rs index 69e9cc8b..e8d40ef2 100644 --- a/packages/coln-query/src/host/interpreter.rs +++ b/packages/coln-query/src/host/interpreter.rs @@ -35,12 +35,7 @@ use super::{ tuple::Tuple, variable::{Environment, Value}, }; -use crate::{ - error::BuildError, - relational::expr::RelExpr, - relational::relation::{SchemaTuple, Tuple as TupleTrait, TupleSchema}, - scalarial::ScalarTypedValue, -}; +use crate::{error::BuildError, relational::expr::RelExpr, scalarial::ScalarTypedValue}; use std::collections::HashMap; pub type EvalResult = Result; @@ -418,9 +413,11 @@ impl HostInterpreter for ScalarHost { #[derive(Debug)] pub struct InterpreterContext<'a> { pub environment: &'a mut Environment, - /// If the interpreter runs within a DBSP context, we store the currently - /// processing tuple here for making each of its fields accessible - /// as a variable. + /// The fields of the row a backend is currently processing, each accessible + /// as a variable. The host layer only reads this map; filling it from a row + /// is the backend's job, since only the backend knows how its rows are laid + /// out (for instance, see + /// [`DbspTupleContext`](crate::relational::incremental::schema::DbspTupleContext)). // No need to wrap it in an Option because HashMap::new() does not allocate! pub tuple_vars: HashMap, /// Stores the most recent alias for a relation. @@ -441,15 +438,6 @@ impl InterpreterContext<'_> { pub fn consume_alias(&mut self) -> Option { self.alias.take() } - pub fn extend_tuple_ctx( - &mut self, - alias: &Option, - schema: &TupleSchema, - tuple: &T, - ) { - self.tuple_vars - .extend(SchemaTuple::new(schema, tuple).named_fields(alias)); - } pub fn clear_tuple_ctx(&mut self) { self.tuple_vars.clear(); } diff --git a/packages/coln-query/src/host/mod.rs b/packages/coln-query/src/host/mod.rs index 6bb7dc5e..de13d1ba 100644 --- a/packages/coln-query/src/host/mod.rs +++ b/packages/coln-query/src/host/mod.rs @@ -7,15 +7,17 @@ //! The latter is important because `coln-flir` rarely emits just a single //! query but an entire program of queries. +pub mod code; pub mod expr; pub mod function; pub mod interpreter; pub mod operator; +pub mod print; pub mod resolver; pub mod stmt; pub mod tuple; pub mod variable; +pub mod walk; +pub use code::QueryIr; pub use interpreter::{HostInterpreter, InterpreterContext, ScalarHost}; - -pub type Code = Vec; diff --git a/packages/coln-query/src/host/operator.rs b/packages/coln-query/src/host/operator.rs index e99798e4..836f39e1 100644 --- a/packages/coln-query/src/host/operator.rs +++ b/packages/coln-query/src/host/operator.rs @@ -36,3 +36,29 @@ pub enum Operator { /// Eagerly-evaluated binary operation. Division, } + +impl Operator { + pub fn symbol(self) -> &'static str { + match self { + Operator::Equal => "==", + Operator::NotEqual => "!=", + Operator::Less => "<", + Operator::LessEqual => "<=", + Operator::Greater => ">", + Operator::GreaterEqual => ">=", + Operator::And => "&&", + Operator::Or => "||", + Operator::Not => "!", + Operator::Addition => "+", + Operator::Subtraction => "-", + Operator::Multiplication => "*", + Operator::Division => "/", + } + } +} + +impl std::fmt::Display for Operator { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.symbol()) + } +} diff --git a/packages/coln-query/src/host/print.rs b/packages/coln-query/src/host/print.rs new file mode 100644 index 00000000..2f88a874 --- /dev/null +++ b/packages/coln-query/src/host/print.rs @@ -0,0 +1,746 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! Rendering a [`QueryIr`](super::QueryIr) program as an indented node tree. +//! +//! One line per node: how its parent addresses it, its kind, and the payloads +//! that are *not* children — an operator, a schema, whether a variable has been +//! resolved. A `Vec` is a forest, so the top-level statements are rendered +//! flush at column zero, one after the other. +//! +//! # Why this is a visitor and not a walk +//! +//! [`walk`](mod@super::walk) exists so that a **scan** need not restate the +//! AST's shape. This is not a scan: it has to address each child *exactly* — by +//! the field it came from, the name that field associates with it, or the +//! relation it is evaluated against — and none of that is a property of the +//! child. It lives in the parent, next to it. +//! +//! A generic traversal can only hand a consumer a positional tag and hope it +//! reconstructs the rest; that is how `Vec<(String, Expr)>` and +//! `Vec<(RelationIdx, Expr)>` fields end up rendered as bare repeated labels +//! whose meaning the reader has to recover by counting. Visiting instead puts +//! the parent's payload and its children in the same scope, so +//! [`ProjectionExpr`] labels an attribute with its *name* and +//! [`MultiWayEquiJoinExpr`] labels an occurrence with the *relation* it belongs +//! to. The price is that this module states the AST's shape a second time — the +//! cost of fidelity, paid here rather than pushed into the walk's shared types. +//! +//! # Addressing vocabulary +//! +//! A label is a field name, optionally followed by which occurrence in +//! parentheses: +//! +//! | Label | Means | +//! | --------------- | -------------------------------------------------- | +//! | `relation` | the field holds a single child | +//! | `relation(2)` | element 2 of a sequence | +//! | `select(out)` | the entry the field keys as `out` | +//! | `on(0 in left)` | equality 0, as evaluated against the left relation | +//! | `on(y in 2)` | join variable `y`, as evaluated against relation 2 | +//! +//! The last two are the same statement — *which equality*, in *which relation* — +//! for a binary and an N-ary join respectively. + +use super::{ + expr::{ + AssignExpr, BinaryExpr, CallExpr, Expr, ExprVisitor, FunctionExpr, GetIndexExpr, + GroupingExpr, Literal, LiteralExpr, TupleExpr, UnaryExpr, VarExpr, + }, + stmt::{BlockStmt, ExprStmt, Stmt, StmtVisitor, VarStmt}, + variable::VariableSlot, + walk::Node, +}; +use crate::relational::catalog::Catalog; +use crate::relational::expr::{ + AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, + FixedPointIterExpr, MultiWayEquiJoinExpr, OutputExpr, OutputKind, ProjectionExpr, RelExpr, + RelExprVisitor, SelectionExpr, SourceExpr, UnionExpr, +}; +use std::fmt::Write; + +/// Writing into a [`String`] cannot fail, which is why every method of the +/// printer below returns `()` instead of a [`std::fmt::Result`]. +macro_rules! emit { + ($printer:ident, $($arg:tt)*) => { + write!($printer.out, $($arg)*).expect("writing into a String cannot fail") + }; +} + +/// `code` rendered as an indented node tree. +/// +/// A whole program is better reached through +/// [`QueryIr::to_tree`](super::QueryIr::to_tree), which needs no import. +/// This takes a bare `[Stmt]` because a *sub*-forest (a fixed point's step body, +/// a function's body) is a plain `Vec` rather than a +/// [`QueryIr`](super::QueryIr), and inspecting one of those in isolation is +/// exactly when a tree rendering earns its keep. +/// +/// Named after [`ToString`]: it says what it *returns*, not what it does. +/// There is deliberately no [`Display`](std::fmt::Display) wrapper — the +/// printer assembles a subtree's text before it knows the guide lines that +/// prefix it, so it buffers regardless, and a wrapper could only copy that +/// buffer into a formatter: a second allocation, and a second way to say the +/// same thing. +pub fn to_tree(code: &[Stmt]) -> String { + render(code, None) +} + +/// [`to_tree`], with each [`SourceExpr`] leaf described by the [`Catalog`] the +/// code is compiled against — the leaf itself only names its relation. +/// +/// Reach for this over [`to_tree`] whenever a catalog is at hand, which for a +/// whole program it always is: +/// [`QueryProgram::to_tree`](crate::program::QueryProgram::to_tree) is this +/// function applied to a program and its own catalog. Besides showing the schema +/// at all, it is what makes a leaf the catalog does *not* describe visible, which +/// is the one failure mode a name-only leaf introduces. +pub fn to_tree_with(code: &[Stmt], catalog: &dyn Catalog) -> String { + render(code, Some(catalog)) +} + +fn render(code: &[Stmt], catalog: Option<&dyn Catalog>) -> String { + let mut printer = TreePrinter { + out: String::new(), + prefix: String::new(), + catalog, + }; + // A program is a forest, so every root starts a fresh tree at column zero + // rather than hanging off a shared parent. + if let Some((first, rest)) = code.split_first() { + printer.visit_stmt(first, ()); + rest.iter().for_each(|stmt| { + printer.out.push('\n'); + printer.visit_stmt(stmt, ()); + }); + } + printer.out +} + +/// The labelled children of one node, in source order. Built by the node's own +/// visit method, which is the only place that knows how to address them. +type Branches<'a> = Vec<(String, Node<'a>)>; + +struct TreePrinter<'a> { + out: String, + /// The guide lines every line of the current subtree is prefixed with. Owned + /// by the printer and pushed/popped around each child, so a node needs to + /// know nothing about where it sits — hence the `()` visitor context. + prefix: String, + /// What the plan's [`SourceExpr`] leaves name, when the caller has it. + /// [`None`] for a rendering of bare code — a sub-forest, or a plan under + /// test that was never paired with a catalog — where a leaf can only be + /// named, not described. + catalog: Option<&'a dyn Catalog>, +} + +impl TreePrinter<'_> { + /// Emit `node` as a child on its own line, then its subtree one level in. + /// + /// `last` decides the elbow and whether the guide line continues past this + /// child, which is the only reason a parent has to know how many children it + /// has before emitting the first. + fn branch(&mut self, label: &str, node: Node<'_>, last: bool) { + self.out.push('\n'); + self.out.push_str(&self.prefix); + self.out.push_str(if last { "└─ " } else { "├─ " }); + self.out.push_str(label); + self.out.push_str(": "); + let enclosing = self.prefix.len(); + self.prefix.push_str(if last { " " } else { "│ " }); + match node { + Node::Stmt(stmt) => self.visit_stmt(stmt, ()), + Node::Expr(expr) => self.visit_expr(expr, ()), + Node::Rel(rel) => self.visit_rel(rel, ()), + } + self.prefix.truncate(enclosing); + } + + /// Emit every child of the node just described. + fn branches(&mut self, branches: Branches<'_>) { + let last = branches.len().saturating_sub(1); + for (index, (label, node)) in branches.into_iter().enumerate() { + self.branch(&label, node, index == last); + } + } +} + +/// The children of an [`EquiJoinExpr`], shared with the +/// [`CartesianProductExpr`] that delegates to one. +/// +/// `on(0 in left)` reads as "equality 0, as evaluated against the left +/// relation": the two sides of a pair are evaluated against *different* +/// relations, and those relations are the `left` and `right` children. +fn equi_join_branches(expr: &EquiJoinExpr) -> Branches<'_> { + let mut branches = vec![ + ("left".to_string(), Node::from(&expr.left)), + ("right".to_string(), Node::from(&expr.right)), + ]; + for (index, (left, right)) in expr.on.iter().enumerate() { + branches.push((format!("on({index} in left)"), Node::from(left))); + branches.push((format!("on({index} in right)"), Node::from(right))); + } + branches.extend(select_branches(expr.attributes.as_deref())); + branches +} + +/// The children of an optional projection, keyed by the name each attribute +/// produces — the payload that would otherwise be reachable only by counting. +fn select_branches(attributes: Option<&[(String, Expr)]>) -> Branches<'_> { + attributes + .unwrap_or_default() + .iter() + .map(|(name, expr)| (format!("select({name})"), Node::from(expr))) + .collect() +} + +/// One `on` pair of a binary join, for the operators that carry no projection. +fn on_branches(on: &[(Expr, Expr)]) -> Branches<'_> { + on.iter() + .enumerate() + .flat_map(|(index, (left, right))| { + [ + (format!("on({index} in left)"), Node::from(left)), + (format!("on({index} in right)"), Node::from(right)), + ] + }) + .collect() +} + +/// A string payload, with the escapes it needs to stay one token. +fn escaped(value: &str) -> String { + value.replace('\\', "\\\\").replace('"', "\\\"") +} + +/// The resolved slot of a variable reference, as `@scope:index`, or a marker +/// that the resolver has not reached it. Invisible in the plan itself, and +/// exactly what one debugs a resolution with. +fn slot(resolved: Option) -> String { + match resolved { + Some((scope, index)) => format!(" @{scope}:{index}"), + None => " (unresolved)".to_string(), + } +} + +impl StmtVisitor<(), ()> for TreePrinter<'_> { + fn visit_var_stmt(&mut self, stmt: &VarStmt, ctx: ()) { + emit!(self, "VarStmt {}", stmt.name); + // At most one initializer, so there is no occurrence to name. + self.branches( + stmt.initializer + .iter() + .map(|expr| ("init".to_string(), Node::from(expr))) + .collect(), + ); + } + + fn visit_expr_stmt(&mut self, stmt: &ExprStmt, ctx: ()) { + self.out.push_str("ExprStmt"); + self.branches(vec![("expr".to_string(), Node::from(&stmt.expr))]); + } + + fn visit_block_stmt(&mut self, stmt: &BlockStmt, ctx: ()) { + self.out.push_str("Block"); + self.branches(sequence("stmt", &stmt.stmts)); + } +} + +impl ExprVisitor<(), ()> for TreePrinter<'_> { + fn visit_literal_expr(&mut self, expr: &LiteralExpr, ctx: ()) { + match &expr.value { + // `Literal`'s own `Display` prints a string bare, which would make + // it indistinguishable from a variable. + Literal::String(value) => emit!(self, "Literal \"{}\"", escaped(value)), + value => emit!(self, "Literal {value}"), + } + } + + fn visit_tuple_expr(&mut self, expr: &TupleExpr, ctx: ()) { + self.out.push_str("Tuple"); + self.branches(sequence("element", &expr.elements)); + } + + fn visit_get_index_expr(&mut self, expr: &GetIndexExpr, ctx: ()) { + self.out.push_str("GetIndex"); + self.branches(vec![ + ("target".to_string(), Node::from(&expr.target)), + ("index".to_string(), Node::from(&expr.index)), + ]); + } + + fn visit_grouping_expr(&mut self, expr: &GroupingExpr, ctx: ()) { + self.out.push_str("Grouping"); + self.branches(vec![("expr".to_string(), Node::from(&expr.expr))]); + } + + fn visit_binary_expr(&mut self, expr: &BinaryExpr, ctx: ()) { + emit!(self, "Binary {}", expr.operator); + self.branches(vec![ + ("left".to_string(), Node::from(&expr.left)), + ("right".to_string(), Node::from(&expr.right)), + ]); + } + + fn visit_unary_expr(&mut self, expr: &UnaryExpr, ctx: ()) { + emit!(self, "Unary {}", expr.operator); + self.branches(vec![("operand".to_string(), Node::from(&expr.operand))]); + } + + fn visit_var_expr(&mut self, expr: &VarExpr, ctx: ()) { + emit!(self, "Var {}{}", expr.name, slot(expr.resolved)); + } + + fn visit_assign_expr(&mut self, expr: &AssignExpr, ctx: ()) { + emit!(self, "Assign {}{}", expr.name, slot(expr.resolved)); + self.branches(vec![("value".to_string(), Node::from(&expr.value))]); + } + + fn visit_function_expr(&mut self, expr: &FunctionExpr, ctx: ()) { + // The parameters are plain names, so they belong on this line rather + // than among the children. + emit!(self, "Function ({})", expr.parameters.join(", ")); + self.branches(sequence("body", &expr.body.stmts)); + } + + fn visit_call_expr(&mut self, expr: &CallExpr, ctx: ()) { + self.out.push_str("Call"); + let mut branches = vec![("callee".to_string(), Node::from(&expr.callee))]; + branches.extend(sequence("argument", &expr.arguments)); + self.branches(branches); + } + + fn visit_relational_expr(&mut self, expr: &RelExpr, ctx: ()) { + self.visit_rel(expr, ctx); + } +} + +impl RelExprVisitor<(), ()> for TreePrinter<'_> { + fn visit_source_expr(&mut self, expr: &SourceExpr, ctx: ()) { + // A plan leaf only *names* an extensional relation, so the schema comes + // from the catalog. Copy the reference out of `self` first: the `Cow` it + // hands back borrows from the catalog, not from `self`, which is what + // lets `emit!` take `&mut self.out` while the schema is still in hand. + let catalog = self.catalog; + let name = escaped(expr.as_id().as_str()); + match catalog.map(|catalog| catalog.source_schema(&expr.id)) { + Some(Some(schema)) => { + emit!(self, "Source \"{name}\" {}", schema.shape()); + } + // A catalog that does not describe this leaf is worth saying out + // loud: the plan names a relation nothing will bind, and this is the + // rendering that shows it. + Some(None) => emit!(self, "Source \"{name}\" (not in catalog)"), + // No catalog to consult — see [`TreePrinter::catalog`]. + None => emit!(self, "Source \"{name}\""), + } + } + + fn visit_output_expr(&mut self, expr: &OutputExpr, ctx: ()) { + let kind = match expr.kind { + OutputKind::Cli => "cli", + OutputKind::Channel => "channel", + }; + emit!(self, "Output \"{}\" {kind}", escaped(expr.id.as_str())); + self.branches(vec![("relation".to_string(), Node::from(&expr.relation))]); + } + + fn visit_alias_expr(&mut self, expr: &AliasExpr, ctx: ()) { + emit!(self, "Alias {}", expr.alias); + self.branches(vec![("relation".to_string(), Node::from(&expr.relation))]); + } + + fn visit_distinct_expr(&mut self, expr: &DistinctExpr, ctx: ()) { + self.out.push_str("Distinct"); + self.branches(vec![("relation".to_string(), Node::from(&expr.relation))]); + } + + fn visit_union_expr(&mut self, expr: &UnionExpr, ctx: ()) { + self.out.push_str("Union"); + self.branches(sequence("relation", &expr.relations)); + } + + fn visit_difference_expr(&mut self, expr: &DifferenceExpr, ctx: ()) { + self.out.push_str("Difference"); + self.branches(vec![ + ("left".to_string(), Node::from(&expr.left)), + ("right".to_string(), Node::from(&expr.right)), + ]); + } + + fn visit_selection_expr(&mut self, expr: &SelectionExpr, ctx: ()) { + self.out.push_str("Selection"); + self.branches(vec![ + ("relation".to_string(), Node::from(&expr.relation)), + ("condition".to_string(), Node::from(&expr.condition)), + ]); + } + + fn visit_projection_expr(&mut self, expr: &ProjectionExpr, ctx: ()) { + self.out.push_str("Projection"); + let mut branches = vec![("relation".to_string(), Node::from(&expr.relation))]; + branches.extend(select_branches(Some(&expr.attributes))); + self.branches(branches); + } + + fn visit_cartesian_product_expr(&mut self, expr: &CartesianProductExpr, ctx: ()) { + // The projection lives on the equi join this delegates to, so reading the + // children off that delegate is what keeps their names attached. + self.out.push_str("CartesianProduct"); + self.branches(equi_join_branches(&expr.inner)); + } + + fn visit_equi_join_expr(&mut self, expr: &EquiJoinExpr, ctx: ()) { + self.out.push_str("EquiJoin"); + self.branches(equi_join_branches(expr)); + } + + fn visit_multi_way_equi_join_expr(&mut self, expr: &MultiWayEquiJoinExpr, ctx: ()) { + self.out.push_str("MultiWayEquiJoin"); + let mut branches = sequence("relation", &expr.relations); + // `on(y in 2)`: the equality class by name, and the relation the + // occurrence is evaluated against. The relation index is a payload + // rather than a node, so labelling the child with it is the only way it + // appears at all — and it is the entire content of the join condition. + branches.extend(expr.on.iter().flat_map(|variable| { + variable.occurrences.iter().map(|(relation, expr)| { + ( + format!("on({} in {relation})", variable.name), + Node::from(expr), + ) + }) + })); + branches.extend(select_branches(expr.attributes.as_deref())); + self.branches(branches); + } + + fn visit_anti_join_expr(&mut self, expr: &AntiJoinExpr, ctx: ()) { + self.out.push_str("AntiJoin"); + let mut branches = vec![ + ("left".to_string(), Node::from(&expr.left)), + ("right".to_string(), Node::from(&expr.right)), + ]; + branches.extend(on_branches(&expr.on)); + self.branches(branches); + } + + fn visit_fixed_point_iter_expr(&mut self, expr: &FixedPointIterExpr, ctx: ()) { + // The accumulator's name binds the step body, so it belongs on this line. + emit!(self, "FixedPointIter {}", expr.accumulator.0); + let mut branches = vec![("init".to_string(), Node::from(&expr.accumulator.1))]; + branches.extend(sequence("step", &expr.step.stmts)); + self.branches(branches); + } +} + +/// The children of a sequence field, labelled `role(index)`. +fn sequence<'a, T>(role: &str, items: &'a [T]) -> Branches<'a> +where + &'a T: Into>, +{ + items + .iter() + .enumerate() + .map(|(index, item)| (format!("{role}({index})"), item.into())) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::host::QueryIr; + use crate::program::QueryProgram; + use crate::relational::{ + expr::{JoinVariable, SinkId}, + schema::TableSchema, + }; + use crate::scalarial::ScalarType; + use crate::test_utils::{TestProgram, table_schema}; + + fn schema(name: &str) -> TableSchema { + table_schema( + name, + [("x", ScalarType::Uint), ("y", ScalarType::Uint)], + ["x"], + ) + } + + fn var(name: &str) -> Expr { + Expr::from(VarExpr::new(name)) + } + + fn expr_stmt(expr: Expr) -> Stmt { + Stmt::from(ExprStmt { expr }) + } + + fn attributes(attributes: &[(&str, &str)]) -> Vec<(String, Expr)> { + attributes + .iter() + .map(|(name, source)| (name.to_string(), var(source))) + .collect() + } + + /// The plan a transitive-closure lowering produces: a fixed point whose step + /// joins the accumulator against a source and unions the result back in. + fn transitive_closure() -> Vec { + vec![ + Stmt::from(VarStmt { + name: "edge".to_string(), + initializer: Some(Expr::from(SourceExpr::new("edge"))), + }), + Stmt::from(VarStmt { + name: "reach".to_string(), + initializer: Some(Expr::from(FixedPointIterExpr { + accumulator: ("acc".to_string(), var("edge")), + step: BlockStmt { + stmts: vec![expr_stmt(Expr::from(DistinctExpr { + relation: Expr::from(UnionExpr { + relations: vec![ + var("acc"), + Expr::from(EquiJoinExpr { + left: var("acc"), + right: var("edge"), + on: vec![(var("y"), var("x"))], + attributes: Some(attributes(&[("x", "x"), ("y", "y")])), + }), + ], + }), + }))], + }, + })), + }), + expr_stmt(Expr::from(OutputExpr { + relation: var("reach"), + id: SinkId::from("reach"), + kind: OutputKind::Channel, + })), + ] + } + + #[test] + fn a_source_is_described_by_the_catalog_the_program_carries() { + // The leaf names `edge` and nothing more, so the schema on this line can + // only have come from the program's catalog. + let program = TestProgram::new(transitive_closure(), [schema("edge")]); + assert!( + program + .to_tree() + .contains("init: Source \"edge\" (x: uint, y: uint) key(x)"), + "{}", + program.to_tree() + ); + } + + #[test] + fn a_source_no_catalog_describes_is_rendered_as_such() { + // The failure mode a name-only leaf introduces: the plan names a + // relation nothing will bind. Rendering is where it becomes visible. + let program = TestProgram::new(transitive_closure(), []); + assert!( + program + .to_tree() + .contains("init: Source \"edge\" (not in catalog)"), + "{}", + program.to_tree() + ); + } + + #[test] + fn renders_a_forest_of_statements_flush_at_column_zero() { + assert_eq!( + // Through the inherent method, which is how bare code is rendered: + // no import, and no catalog to describe the leaves with — see + // `a_source_is_described_by_the_catalog_the_program_carries` for the + // same plan rendered against one. + QueryIr::from(transitive_closure()).to_tree(), + "\ +VarStmt edge +└─ init: Source \"edge\" +VarStmt reach +└─ init: FixedPointIter acc + ├─ init: Var edge (unresolved) + └─ step(0): ExprStmt + └─ expr: Distinct + └─ relation: Union + ├─ relation(0): Var acc (unresolved) + └─ relation(1): EquiJoin + ├─ left: Var acc (unresolved) + ├─ right: Var edge (unresolved) + ├─ on(0 in left): Var y (unresolved) + ├─ on(0 in right): Var x (unresolved) + ├─ select(x): Var x (unresolved) + └─ select(y): Var y (unresolved) +ExprStmt +└─ expr: Output \"reach\" channel + └─ relation: Var reach (unresolved)" + ); + } + + #[test] + fn prints_a_bare_slice_such_as_a_fixed_points_step_body() { + // The trait is on `[Stmt]`, not on `Vec`, so a sub-forest reached + // through a field prints without being collected first — which is how + // one inspects a single fixed-point step in isolation. + let code = transitive_closure(); + let Stmt::Var(stmt) = &code[1] else { + unreachable!("the second statement binds `reach`") + }; + let Some(Expr::Relational(rel)) = &stmt.initializer else { + unreachable!("`reach` is bound to a relational expression") + }; + let RelExpr::FixedPointIter(fixed_point) = rel else { + unreachable!("`reach` is bound to a fixed point") + }; + assert_eq!( + to_tree(&fixed_point.step.stmts), + "\ +ExprStmt +└─ expr: Distinct + └─ relation: Union + ├─ relation(0): Var acc (unresolved) + └─ relation(1): EquiJoin + ├─ left: Var acc (unresolved) + ├─ right: Var edge (unresolved) + ├─ on(0 in left): Var y (unresolved) + ├─ on(0 in right): Var x (unresolved) + ├─ select(x): Var x (unresolved) + └─ select(y): Var y (unresolved)" + ); + } + + #[test] + fn addresses_a_binary_joins_keys_by_equality_and_relation() { + // Relations, keys and projected attributes are *all* plain `Expr`s. Each + // label says which is which: `on(1 in right)` is the right-hand side of + // the second equality, and `select(keep)` produces the `keep` column. + let code = [expr_stmt(Expr::from(EquiJoinExpr { + left: var("l"), + right: var("r"), + on: vec![(var("a"), var("b")), (var("c"), var("d"))], + attributes: Some(attributes(&[("out", "a"), ("keep", "c")])), + }))]; + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: EquiJoin + ├─ left: Var l (unresolved) + ├─ right: Var r (unresolved) + ├─ on(0 in left): Var a (unresolved) + ├─ on(0 in right): Var b (unresolved) + ├─ on(1 in left): Var c (unresolved) + ├─ on(1 in right): Var d (unresolved) + ├─ select(out): Var a (unresolved) + └─ select(keep): Var c (unresolved)" + ); + } + + #[test] + fn addresses_a_multi_way_joins_occurrences_by_variable_and_relation() { + // The same statement as `on(0 in left)`, for an N-ary join: which + // equality, in which relation. A `RelationIdx` is a payload rather than a + // node, so the label is the only place it can appear — note that + // relation 1 is reachable only through `z`. + let code = [expr_stmt(Expr::from( + MultiWayEquiJoinExpr::new( + vec![var("r0"), var("r1"), var("r2")], + vec![ + JoinVariable { + name: "y".to_string(), + occurrences: vec![(0, var("y")), (2, var("y"))], + }, + JoinVariable { + name: "z".to_string(), + occurrences: vec![(1, var("z")), (2, var("z"))], + }, + ], + None, + ) + .expect("every variable is bound by two relations"), + ))]; + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: MultiWayEquiJoin + ├─ relation(0): Var r0 (unresolved) + ├─ relation(1): Var r1 (unresolved) + ├─ relation(2): Var r2 (unresolved) + ├─ on(y in 0): Var y (unresolved) + ├─ on(y in 2): Var y (unresolved) + ├─ on(z in 1): Var z (unresolved) + └─ on(z in 2): Var z (unresolved)" + ); + } + + #[test] + fn keeps_the_names_of_a_projection_a_cartesian_product_delegates() { + // A `CartesianProductExpr` holds its projection on the `EquiJoinExpr` it + // delegates to; reading the children off that delegate is what keeps + // each attribute expression next to the name it produces. + let code = [expr_stmt(Expr::from(CartesianProductExpr::new( + var("l"), + var("r"), + Some(attributes(&[("out", "a"), ("keep", "b")])), + )))]; + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: CartesianProduct + ├─ left: Var l (unresolved) + ├─ right: Var r (unresolved) + ├─ select(out): Var a (unresolved) + └─ select(keep): Var b (unresolved)" + ); + } + + #[test] + fn guides_lines_through_deeper_siblings() { + // The `│` of the outer union has to continue past the whole first + // branch, otherwise a deep tree cannot be read. + let code = [expr_stmt(Expr::from(UnionExpr { + relations: vec![Expr::from(DistinctExpr { relation: var("a") }), var("b")], + }))]; + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: Union + ├─ relation(0): Distinct + │ └─ relation: Var a (unresolved) + └─ relation(1): Var b (unresolved)" + ); + } + + #[test] + fn shows_a_resolved_variable_slot() { + // The resolver writes into the tree in place, so this is how one sees + // whether it has been here — and what it decided. + let mut code = [expr_stmt(var("x"))]; + let Stmt::Expr(stmt) = &mut code[0] else { + unreachable!("the statement is an expression statement") + }; + let Expr::Var(expr) = &mut stmt.expr else { + unreachable!("the expression is a variable") + }; + expr.resolved = Some((1, 4)); + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: Var x @1:4" + ); + } + + #[test] + fn quotes_and_escapes_string_payloads() { + let code = [expr_stmt(Expr::from(LiteralExpr::from("a\"b")))]; + assert_eq!( + to_tree(&code), + "\ +ExprStmt +└─ expr: Literal \"a\\\"b\"" + ); + } +} diff --git a/packages/coln-query/src/host/resolver.rs b/packages/coln-query/src/host/resolver.rs index e6b59348..ff71aaff 100644 --- a/packages/coln-query/src/host/resolver.rs +++ b/packages/coln-query/src/host/resolver.rs @@ -5,7 +5,7 @@ use crate::{ error::SyntaxError, host::{ - Code, + QueryIr, expr::{ AssignExpr, BinaryExpr, CallExpr, Expr, ExprVisitorMut, FunctionExpr, GetIndexExpr, GroupingExpr, LiteralExpr, TupleExpr, UnaryExpr, VarExpr, @@ -15,13 +15,20 @@ use crate::{ }, relational::expr::{ AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, - FixedPointIterExpr, OutputExpr, ProjectionExpr, RelExpr, RelExprVisitorMut, SelectionExpr, - SourceExpr, UnionExpr, + FixedPointIterExpr, MultiWayEquiJoinExpr, OutputExpr, ProjectionExpr, RelExpr, + RelExprVisitorMut, SelectionExpr, SourceExpr, UnionExpr, }, - util::{Named, Resolvable}, }; use std::collections::HashMap; +pub trait Resolvable { + fn set_resolved(&mut self, resolved: super::variable::VariableSlot); +} + +pub trait Named { + fn name(&self) -> &str; +} + #[derive(Clone, Copy, Debug)] struct VariableMeta { initialized: bool, @@ -86,20 +93,21 @@ impl ScopeStack { /// [`ResolvedCode::from`] mints one, so a backend cannot be handed an /// unprocessed plan. #[derive(Clone)] -pub struct ResolvedCode(Code); +pub struct ResolvedCode(QueryIr); impl ResolvedCode { /// Run the static pipeline over a raw plan and resolve variable slots. - pub fn from(mut code: Code) -> Result { + pub fn from(code: impl Into) -> Result { + let mut code = code.into(); let mut scopes = ScopeStack::new(); let mut ctx = ResolverContext::new(&mut scopes); Resolver::new().resolve(code.iter_mut(), &mut ctx)?; Ok(Self(code)) } - pub fn as_code(&self) -> &Code { + pub fn as_code(&self) -> &QueryIr { &self.0 } - pub fn into_code(self) -> Code { + pub fn into_code(self) -> QueryIr { self.0 } } @@ -388,6 +396,31 @@ impl RelExprVisitorMut> for Resolver { .and_then(|()| self.visit_projection_attributes(expr.attributes.as_mut(), ctx)) } + fn visit_multi_way_equi_join_expr( + &mut self, + expr: &mut MultiWayEquiJoinExpr, + ctx: VisitorCtx<'_, '_>, + ) -> VisitorResult { + // The structural invariants (arity, in-bounds and distinct relation + // indices, at least two occurrences per join variable) are checked here + // rather than re-derived by every consumer, because the fields are + // public and a plan may be assembled or rewritten by hand. + expr.validate()?; + + expr.relations + .iter_mut() + .try_for_each(|relation| self.visit_expr(relation, ctx)) + .and_then(|()| { + expr.on_exprs_mut().try_for_each(|expr| { + ctx.begin_tuple_context(); + let ret = self.visit_expr(expr, ctx); + ctx.end_tuple_context(); + ret + }) + }) + .and_then(|()| self.visit_projection_attributes(expr.attributes.as_mut(), ctx)) + } + fn visit_anti_join_expr(&mut self, expr: &mut AntiJoinExpr, ctx: VisitorCtx) -> VisitorResult { self.visit_expr(&mut expr.left, ctx) .and_then(|()| self.visit_expr(&mut expr.right, ctx)) diff --git a/packages/coln-query/src/host/stmt.rs b/packages/coln-query/src/host/stmt.rs index 137c9f48..fb80b3b9 100644 --- a/packages/coln-query/src/host/stmt.rs +++ b/packages/coln-query/src/host/stmt.rs @@ -3,7 +3,7 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT use super::expr::Expr; -use crate::{impl_from_auto_box, util::MemAddr}; +use crate::impl_from_auto_box; #[derive(Clone, Debug, Eq, PartialEq)] pub enum Stmt { @@ -36,6 +36,7 @@ pub struct BlockStmt { pub stmts: Vec, } +/// Read-only visitor. See [`StmtVisitorOwn`]. pub trait StmtVisitor { fn visit_stmt(&mut self, stmt: &Stmt, ctx: C) -> T { match stmt { @@ -49,6 +50,7 @@ pub trait StmtVisitor { fn visit_block_stmt(&mut self, stmt: &BlockStmt, ctx: C) -> T; } +/// Annotating visitor. See [`StmtVisitorOwn`]. pub trait StmtVisitorMut { fn visit_stmt(&mut self, stmt: &mut Stmt, ctx: C) -> T { match stmt { @@ -62,20 +64,18 @@ pub trait StmtVisitorMut { fn visit_block_stmt(&mut self, stmt: &mut BlockStmt, ctx: C) -> T; } +/// Restructuring visitor. See +/// [`ExprVisitorOwn`](crate::host::expr::ExprVisitorOwn) for which of the three +/// families a given pass belongs in, and why the payloads arrive boxed. pub trait StmtVisitorOwn { fn visit_stmt(&mut self, stmt: Stmt, ctx: C) -> T { match stmt { - Stmt::Var(stmt) => self.visit_var_stmt(*stmt, ctx), - Stmt::Expr(stmt) => self.visit_expr_stmt(*stmt, ctx), - Stmt::Block(stmt) => self.visit_block_stmt(*stmt, ctx), + Stmt::Var(stmt) => self.visit_var_stmt(stmt, ctx), + Stmt::Expr(stmt) => self.visit_expr_stmt(stmt, ctx), + Stmt::Block(stmt) => self.visit_block_stmt(stmt, ctx), } } - fn visit_var_stmt(&mut self, stmt: VarStmt, ctx: C) -> T; - fn visit_expr_stmt(&mut self, stmt: ExprStmt, ctx: C) -> T; - fn visit_block_stmt(&mut self, stmt: BlockStmt, ctx: C) -> T; + fn visit_var_stmt(&mut self, stmt: Box, ctx: C) -> T; + fn visit_expr_stmt(&mut self, stmt: Box, ctx: C) -> T; + fn visit_block_stmt(&mut self, stmt: Box, ctx: C) -> T; } - -impl MemAddr for Stmt {} -impl MemAddr for VarStmt {} -impl MemAddr for ExprStmt {} -impl MemAddr for BlockStmt {} diff --git a/packages/coln-query/src/host/tuple.rs b/packages/coln-query/src/host/tuple.rs index ed145fee..c6fbccb4 100644 --- a/packages/coln-query/src/host/tuple.rs +++ b/packages/coln-query/src/host/tuple.rs @@ -3,7 +3,7 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT use crate::host::variable::Value; -use crate::typing::ExprType; +use crate::typing::type_resolver::ExprType; use std::fmt; use std::rc::Rc; diff --git a/packages/coln-query/src/host/walk.rs b/packages/coln-query/src/host/walk.rs new file mode 100644 index 00000000..366d7424 --- /dev/null +++ b/packages/coln-query/src/host/walk.rs @@ -0,0 +1,497 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! Generic depth-first traversal of the AST, for consumers that only need to +//! *scan* it. +//! +//! # Why this exists next to the visitor traits +//! +//! The AST has two kinds of consumer, and they need opposite things: +//! +//! - A **fold** computes a value per node out of its children's values, and the +//! parent decides whether and in which order children are visited at all +//! (`TypeResolver::visit_unary_expr` answers `Not` without ever looking at the +//! operand; `DbspInterpreter::visit_equi_join_expr` must consume the left +//! alias *between* its two operands). No traversal can be factored out of +//! those — they stay on the [`ExprVisitor`](super::expr::ExprVisitor) family. +//! So does anything that has to *address* a child rather than just reach it, +//! which is why [`print`](super::print) is a visitor too. +//! - A **scan** reads nodes, produces no per-node result, and does not care +//! about order. Everything a scan needs from the AST's shape is "what are this +//! node's children", which is exactly what this module states *once*. +//! +//! So this is not a replacement for the visitors, it is the other half: scans +//! stop restating the tree's shape, and a potential future flatter +//! representation only has to re-implement [`Node::push_children`] to keep +//! every scan working. +//! +//! # Traversal orders +//! +//! [`Walk`] yields [`Enter`](Event::Enter)/[`Leave`](Event::Leave) events, from +//! which every order follows: [`pre_order`] keeps the enters, [`post_order`] +//! keeps the leaves. A static order is not the real execution order either way: +//! A [`CallExpr`](super::expr::CallExpr) jumps into a function body and a +//! [`FixedPointIterExpr`](crate::relational::expr::FixedPointIterExpr) repeats +//! its step. +//! +//! There is deliberately no `&mut` counterpart. A `&mut` to a parent and to its +//! children cannot be held at once, so a mutable walk could only ever be +//! pre-order and could never emit [`Leave`](Event::Leave) — which is precisely +//! the half `Resolver` needs, to pop the scopes and tuple contexts it pushes on +//! the way down. Rewriting passes stay visitors. + +use crate::{ + host::{expr::Expr, stmt::Stmt}, + relational::expr::{EquiJoinExpr, OutputExpr, RelExpr, SourceExpr}, +}; + +/// A borrowed pointer to a node of any of the three mutually recursive node +/// kinds. The kinds interleave in both directions — a +/// [`FunctionExpr`](super::expr::FunctionExpr) body and a +/// [`FixedPointIterExpr`](crate::relational::expr::FixedPointIterExpr) step are +/// statements — so a traversal has to be able to hold any of them. +/// +/// [`Expr::Relational`] never appears as a [`Node::Expr`]: it is a pure wrapper +/// carrying no data of its own, so the walk yields the [`RelExpr`] inside it +/// directly rather than both. +#[derive(Clone, Copy, Debug)] +pub enum Node<'a> { + Stmt(&'a Stmt), + Expr(&'a Expr), + Rel(&'a RelExpr), +} + +impl<'a> From<&'a Stmt> for Node<'a> { + fn from(stmt: &'a Stmt) -> Self { + Node::Stmt(stmt) + } +} + +impl<'a> From<&'a Expr> for Node<'a> { + fn from(expr: &'a Expr) -> Self { + match expr { + // Unwrap the bridge into the relational layer, see [`Node`]. + Expr::Relational(rel) => Node::Rel(rel), + expr => Node::Expr(expr), + } + } +} + +impl<'a> From<&'a RelExpr> for Node<'a> { + fn from(rel: &'a RelExpr) -> Self { + Node::Rel(rel) + } +} + +impl<'a> Node<'a> { + /// Append this node's node-typed children to `out`, in source order. + /// + /// **This is the only place the AST's shape is spelled out for traversal + /// purposes.** Non-node payloads (an operator, an attribute's name, a + /// [`RelationIdx`](crate::relational::expr::RelationIdx)) are not children, + /// and neither is the *position* a child occupies here: a scan reaches every + /// child either way, and a consumer that needs to tell one `on` key from + /// another is addressing rather than scanning, so it visits instead. + pub fn push_children(self, out: &mut Vec>) { + match self { + Node::Stmt(stmt) => Self::push_stmt_children(stmt, out), + Node::Expr(expr) => Self::push_expr_children(expr, out), + Node::Rel(rel) => Self::push_rel_children(rel, out), + } + } + + fn push_stmt_children(stmt: &'a Stmt, out: &mut Vec>) { + match stmt { + Stmt::Var(stmt) => out.extend(stmt.initializer.iter().map(Node::from)), + Stmt::Expr(stmt) => out.push(Node::from(&stmt.expr)), + Stmt::Block(stmt) => out.extend(stmt.stmts.iter().map(Node::from)), + } + } + + fn push_expr_children(expr: &'a Expr, out: &mut Vec>) { + match expr { + Expr::Literal(_) | Expr::Var(_) => {} + Expr::Tuple(expr) => out.extend(expr.elements.iter().map(Node::from)), + Expr::GetIndex(expr) => { + out.extend([Node::from(&expr.target), Node::from(&expr.index)]); + } + Expr::Grouping(expr) => out.push(Node::from(&expr.expr)), + Expr::Binary(expr) => out.extend([Node::from(&expr.left), Node::from(&expr.right)]), + Expr::Unary(expr) => out.push(Node::from(&expr.operand)), + Expr::Assign(expr) => out.push(Node::from(&expr.value)), + Expr::Call(expr) => { + out.push(Node::from(&expr.callee)); + out.extend(expr.arguments.iter().map(Node::from)); + } + // The parameters are plain names, so only the body holds nodes. + Expr::Function(expr) => out.extend(expr.body.stmts.iter().map(Node::from)), + // Normalized away by `Node::from`; reachable only through a + // hand-built `Node::Expr`, which is handled rather than pruned. + Expr::Relational(rel) => out.push(Node::Rel(rel)), + } + } + + fn push_rel_children(rel: &'a RelExpr, out: &mut Vec>) { + match rel { + // A plan leaf: it only *names* an extensional relation. + RelExpr::Source(_) => {} + RelExpr::Output(expr) => out.push(Node::from(&expr.relation)), + RelExpr::Alias(expr) => out.push(Node::from(&expr.relation)), + RelExpr::Distinct(expr) => out.push(Node::from(&expr.relation)), + RelExpr::Union(expr) => out.extend(expr.relations.iter().map(Node::from)), + RelExpr::Difference(expr) => { + out.extend([Node::from(&expr.left), Node::from(&expr.right)]); + } + RelExpr::Selection(expr) => { + out.extend([Node::from(&expr.relation), Node::from(&expr.condition)]); + } + RelExpr::Projection(expr) => { + out.push(Node::from(&expr.relation)); + out.extend(expr.attributes.iter().map(|(_, expr)| Node::from(expr))); + } + // A cartesian product is an equi join with an empty `on`, so it has + // no children beyond that join's. + RelExpr::CartesianProduct(expr) => Self::push_equi_join_children(&expr.inner, out), + RelExpr::EquiJoin(expr) => Self::push_equi_join_children(expr, out), + RelExpr::MultiWayEquiJoin(expr) => { + out.extend(expr.relations.iter().map(Node::from)); + out.extend(expr.on_exprs().map(Node::from)); + out.extend( + expr.attributes + .iter() + .flatten() + .map(|(_, expr)| Node::from(expr)), + ); + } + RelExpr::AntiJoin(expr) => { + out.extend([Node::from(&expr.left), Node::from(&expr.right)]); + out.extend( + expr.on + .iter() + .flat_map(|(left, right)| [Node::from(left), Node::from(right)]), + ); + } + RelExpr::FixedPointIter(expr) => { + out.push(Node::from(&expr.accumulator.1)); + out.extend(expr.step.stmts.iter().map(Node::from)); + } + } + } + + fn push_equi_join_children(expr: &'a EquiJoinExpr, out: &mut Vec>) { + out.extend([Node::from(&expr.left), Node::from(&expr.right)]); + out.extend( + expr.on + .iter() + .flat_map(|(left, right)| [Node::from(left), Node::from(right)]), + ); + out.extend( + expr.attributes + .iter() + .flatten() + .map(|(_, expr)| Node::from(expr)), + ); + } + + /// Every event of the subtree rooted at this node. + pub fn walk(self) -> Walk<'a> { + Walk::new([self]) + } + + /// This node and its descendants, parents before children. + pub fn pre_order(self) -> impl Iterator> { + self.walk().filter_map(|event| event.entered()) + } + + /// This node and its descendants, children before parents. + pub fn post_order(self) -> impl Iterator> { + self.walk().filter_map(|event| event.left()) + } + + pub fn as_stmt(self) -> Option<&'a Stmt> { + match self { + Node::Stmt(stmt) => Some(stmt), + _ => None, + } + } + + pub fn as_expr(self) -> Option<&'a Expr> { + match self { + Node::Expr(expr) => Some(expr), + _ => None, + } + } + + pub fn as_rel(self) -> Option<&'a RelExpr> { + match self { + Node::Rel(rel) => Some(rel), + _ => None, + } + } + + /// The [`SourceExpr`] leaf this node is, if any. What a plan-wide source + /// discovery filters a [walk](Walk) on. + pub fn as_source(self) -> Option<&'a SourceExpr> { + self.as_rel().and_then(|expr| match expr { + RelExpr::Source(source) => Some(source.as_ref()), + _ => None, + }) + } + + /// The [`SourceExpr`] leaf this node is, if any. What a plan-wide output + /// discovery filters a [walk](Walk) on. + pub fn as_output(self) -> Option<&'a OutputExpr> { + self.as_rel().and_then(|expr| match expr { + RelExpr::Output(output) => Some(output.as_ref()), + _ => None, + }) + } +} + +/// One step of a [`Walk`]. Every node is reported twice, so a consumer can pick +/// its order ([`pre_order`], [`post_order`]) or track depth, without the walk +/// having to offer one iterator per traversal. +#[derive(Clone, Copy, Debug)] +pub enum Event<'a> { + Enter(Node<'a>), + Leave(Node<'a>), +} + +impl<'a> Event<'a> { + /// The node, whichever half of its visit this is. + pub fn node(self) -> Node<'a> { + match self { + Event::Enter(node) | Event::Leave(node) => node, + } + } + + pub fn entered(self) -> Option> { + match self { + Event::Enter(node) => Some(node), + Event::Leave(_) => None, + } + } + + pub fn left(self) -> Option> { + match self { + Event::Leave(node) => Some(node), + Event::Enter(_) => None, + } + } +} + +/// A depth-first walk over one or more subtrees, as a stream of [`Event`]s. +/// Iterative rather than recursive, so a deeply nested plan cannot exhaust the +/// stack. +pub struct Walk<'a> { + /// Pending work, innermost last. A node's `Leave` is pushed underneath its + /// children when the node is entered. + pending: Vec>, + /// Scratch space for [`Node::push_children`], reused across nodes so the + /// walk allocates amortized nothing per node. + children: Vec>, +} + +impl<'a> Walk<'a> { + fn new(roots: impl IntoIterator>) -> Self { + let mut pending: Vec> = roots.into_iter().map(Event::Enter).collect(); + // The stack is popped from the back, so the first root has to end up + // last. + pending.reverse(); + Self { + pending, + children: Vec::new(), + } + } +} + +impl<'a> Iterator for Walk<'a> { + type Item = Event<'a>; + + fn next(&mut self) -> Option { + let event = self.pending.pop()?; + if let Event::Enter(node) = event { + // The node's own leave sits below its children, so it is reported + // once the whole subtree is done. + self.pending.push(Event::Leave(node)); + node.push_children(&mut self.children); + self.pending + .extend(self.children.drain(..).rev().map(Event::Enter)); + } + Some(event) + } +} + +/// Every event of `code`, depth-first, statements in order. +pub fn walk(code: &[Stmt]) -> Walk<'_> { + Walk::new(code.iter().map(Node::from)) +} + +/// Every node of `code`, parents before children. +pub fn pre_order(code: &[Stmt]) -> impl Iterator> { + walk(code).filter_map(|event| event.entered()) +} + +/// Every node of `code`, children before parents — the order in which an +/// interpreter reduces them. +pub fn post_order(code: &[Stmt]) -> impl Iterator> { + walk(code).filter_map(|event| event.left()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::host::{ + expr::{BinaryExpr, LiteralExpr, VarExpr}, + operator::Operator, + stmt::{ExprStmt, VarStmt}, + }; + use crate::relational::expr::{DistinctExpr, EquiJoinExpr, SelectionExpr, UnionExpr}; + + /// `1 + 2` as a statement. + fn arithmetic() -> Vec { + vec![Stmt::from(ExprStmt { + expr: Expr::from(BinaryExpr { + operator: Operator::Addition, + left: Expr::from(LiteralExpr::from(1_u64)), + right: Expr::from(LiteralExpr::from(2_u64)), + }), + })] + } + + /// A label per node kind, enough to pin an order down in a test. + fn label(node: Node<'_>) -> String { + match node { + Node::Stmt(stmt) => match stmt { + Stmt::Var(stmt) => format!("var {}", stmt.name), + Stmt::Expr(_) => "stmt".to_string(), + Stmt::Block(_) => "block".to_string(), + }, + Node::Expr(expr) => match expr { + Expr::Literal(expr) => expr.value.to_string(), + Expr::Binary(expr) => expr.operator.to_string(), + Expr::Var(expr) => expr.name.clone(), + other => format!("{other:?}"), + }, + Node::Rel(rel) => match rel { + RelExpr::Source(source) => format!("source {}", source.as_id()), + RelExpr::Distinct(_) => "distinct".to_string(), + RelExpr::Union(_) => "union".to_string(), + RelExpr::Selection(_) => "selection".to_string(), + RelExpr::EquiJoin(_) => "join".to_string(), + other => format!("{other:?}"), + }, + } + } + + fn labels<'a>(nodes: impl Iterator>) -> Vec { + nodes.map(label).collect() + } + + #[test] + fn pre_order_reports_parents_before_children_left_to_right() { + let code = arithmetic(); + assert_eq!(labels(pre_order(&code)), ["stmt", "+", "1", "2"]); + } + + #[test] + fn post_order_is_evaluation_order() { + // The operands are reduced before the operator combining them, which is + // why post-order — not pre-order — is what an interpreter follows. + let code = arithmetic(); + assert_eq!(labels(post_order(&code)), ["1", "2", "+", "stmt"]); + } + + #[test] + fn every_node_is_entered_and_left_exactly_once() { + let code = arithmetic(); + let events: Vec<_> = walk(&code).collect(); + assert_eq!(events.len(), 2 * pre_order(&code).count()); + // A well-formed nesting: the depth returns to zero and never goes below. + let mut depth = 0_isize; + for event in events { + depth += match event { + Event::Enter(_) => 1, + Event::Leave(_) => -1, + }; + assert!(depth >= 0, "left a node that was never entered"); + } + assert_eq!(depth, 0, "entered a node that was never left"); + } + + #[test] + fn the_relational_wrapper_is_not_reported_as_a_node() { + // `Expr::Relational` carries nothing of its own, so `distinct(source)` + // must be two nodes below the statement, not four. + let code = vec![Stmt::from(ExprStmt { + expr: Expr::from(DistinctExpr { + relation: Expr::from(SourceExpr::new("edge")), + }), + })]; + assert_eq!( + labels(pre_order(&code)), + ["stmt", "distinct", "source edge"] + ); + } + + #[test] + fn walking_crosses_between_statements_and_expressions() { + // A source nested in a var initializer inside a union: the walk has to + // change node kind twice to reach it. + let code = vec![Stmt::from(VarStmt { + name: "both".to_string(), + initializer: Some(Expr::from(UnionExpr { + relations: vec![ + Expr::from(SourceExpr::new("left")), + Expr::from(VarExpr::new("right")), + ], + })), + })]; + assert_eq!( + labels(pre_order(&code)), + ["var both", "union", "source left", "right"] + ); + } + + #[test] + fn a_scan_filters_the_walk_instead_of_restating_the_tree() { + // The pattern that replaced the hand-written source collection in the + // DBSP backend: no knowledge of the tree's shape at the call site. + let code = vec![Stmt::from(ExprStmt { + expr: Expr::from(SelectionExpr { + relation: Expr::from(UnionExpr { + relations: vec![ + Expr::from(SourceExpr::new("left")), + Expr::from(SourceExpr::new("right")), + ], + }), + condition: Expr::from(VarExpr::new("a")), + }), + })]; + let sources: Vec<&str> = pre_order(&code) + .filter_map(Node::as_source) + .map(|source| source.as_id().as_str()) + .collect(); + assert_eq!(sources, ["left", "right"]); + } + + #[test] + fn every_operand_of_a_join_is_reached_even_though_none_is_addressable() { + // A scan has to reach the relations, the keys *and* the projected + // attributes; telling them apart is not its job (see `push_children`), + // so this pins reachability and order, nothing more. + let code = vec![Stmt::from(ExprStmt { + expr: Expr::from(EquiJoinExpr { + left: Expr::from(VarExpr::new("l")), + right: Expr::from(VarExpr::new("r")), + on: vec![(Expr::from(VarExpr::new("a")), Expr::from(VarExpr::new("b")))], + attributes: Some(vec![("out".to_string(), Expr::from(VarExpr::new("c")))]), + }), + })]; + assert_eq!( + labels(pre_order(&code)), + ["stmt", "join", "l", "r", "a", "b", "c"] + ); + } +} diff --git a/packages/coln-query/src/lib.rs b/packages/coln-query/src/lib.rs index f53bafde..d27eff77 100644 --- a/packages/coln-query/src/lib.rs +++ b/packages/coln-query/src/lib.rs @@ -5,15 +5,17 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT pub mod api; -pub mod error; -pub mod host; -pub mod optimizer; -pub mod pipeline; -pub mod relational; -pub mod scalarial; -pub mod test_helper; +mod error; +mod host; +mod optimizer; +mod pipeline; +mod program; +mod relational; +mod scalarial; +#[cfg(feature = "test-utils")] +mod test_utils; mod typing; -mod util; +mod utils; #[cfg(test)] mod test { @@ -34,16 +36,16 @@ mod test { expr::{ AliasExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, FixedPointIterExpr, OutputExpr, OutputKind, ProjectionExpr, SelectionExpr, SinkId, - SourceExpr, SourceId, UnionExpr, + SourceExpr, UnionExpr, }, incremental::dbsp::{ZWeight, zset}, relation::TupleValue, }, scalarial::ScalarTypedValue, - test_helper::{person_profession_data, rows, rows_with_weight}, + test_utils::{TestProgram, person_profession_data, rows, rows_with_weight}, }; use ::dbsp::OrdZSet; - use test_helper::{Edge, InputEntity, Person, PlainRelation, PredRel, Profession, SetOp}; + use test_utils::{EdgeRel, InputRel, PersonRel, PlainRel, PredRel, ProfessionRel, SetRel}; /// Tap the relation held by the variable `name` as a named runtime output, /// reusing the variable name as the output's [`SinkId`]. The resulting @@ -196,7 +198,7 @@ mod test { right: Expr::from(LiteralExpr::from(2_u64)), }), }), - relation: Expr::from(SourceExpr::new(Edge::schema())), + relation: Expr::from(SourceExpr::new(EdgeRel::id())), })), }), output_stmt("selected"), @@ -221,24 +223,37 @@ mod test { }), output_stmt("projected"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = + Pipeline::incremental().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()]))?; const STEPS: usize = 3; let mut edges_data = ([ - [Edge::new(0, 1, 1), Edge::new(1, 2, 2), Edge::new(2, 3, 3)] - .map(|e| (e, 2)) - .into_iter() - .collect(), - [Edge::new(3, 4, 1), Edge::new(4, 5, 2), Edge::new(5, 6, 3)] - .map(|e| (e, 1)) - .into_iter() - .collect(), - [Edge::new(0, 1, 1), Edge::new(1, 2, 2), Edge::new(2, 3, 3)] - .map(|e| (e, -1)) - .into_iter() - .collect(), - ] as [Vec<(Edge, ZWeight)>; STEPS]) + [ + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 2), + EdgeRel::new(2, 3, 3), + ] + .map(|e| (e, 2)) + .into_iter() + .collect(), + [ + EdgeRel::new(3, 4, 1), + EdgeRel::new(4, 5, 2), + EdgeRel::new(5, 6, 3), + ] + .map(|e| (e, 1)) + .into_iter() + .collect(), + [ + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 2), + EdgeRel::new(2, 3, 3), + ] + .map(|e| (e, -1)) + .into_iter() + .collect(), + ] as [Vec<(EdgeRel, ZWeight)>; STEPS]) .into_iter(); let mut selected_output = ([ @@ -274,14 +289,14 @@ mod test { .into_iter(); for _ in 1..=STEPS { - rt.feed(&SourceId::from("edges"), rows(edges_data.next().unwrap()))?; + assert!(rt.feed(&EdgeRel::id(), rows(edges_data.next().unwrap()))?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("selected"))?.0, + rt.output(&SinkId::from("selected"))?.to_debug_zset(), selected_output.next().unwrap() ); assert_eq!( - rt.output(&SinkId::from("projected"))?.0, + rt.output(&SinkId::from("projected"))?.to_debug_zset(), projected_output.next().unwrap() ); } @@ -304,7 +319,7 @@ mod test { let plan = vec![ Stmt::from(VarStmt { name: "edges".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Edge::schema()))), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), }), Stmt::from(VarStmt { name: "deduped".to_string(), @@ -329,16 +344,14 @@ mod test { }), output_stmt("downstream"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; - rt.feed( - &SourceId::from("edges"), - rows_with_weight([Edge::new(0, 1, 5)], 1), - )?; + let mut rt = + Pipeline::incremental().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()]))?; + assert!(rt.feed(&EdgeRel::id(), rows_with_weight([EdgeRel::new(0, 1, 5)], 1),)?); rt.commit()?; // The Cli tap did not disturb the flow: the downstream channel is correct. // `Edge` carries an implicit `active` column (defaults to `true`). assert_eq!( - rt.output(&SinkId::from("downstream"))?.0, + rt.output(&SinkId::from("downstream"))?.to_debug_zset(), zset! { tuple!(0_u64, 1_u64, 5_u64, true) => 1 } ); // Reading the Cli tap by name fails loudly instead of returning drained, @@ -360,7 +373,7 @@ mod test { let plan = vec![ Stmt::from(VarStmt { name: "edges".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Edge::schema()))), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), }), Stmt::from(VarStmt { name: "deduped".to_string(), @@ -383,7 +396,9 @@ mod test { }), }), ]; - let Err(err) = Pipeline::incremental().runtime(plan) else { + let Err(err) = + Pipeline::incremental().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()])) + else { panic!("duplicate output names must be rejected at build time"); }; println!("{err}"); @@ -393,16 +408,38 @@ mod test { ); } + #[test] + fn source_no_catalog_describes_is_rejected_at_build_time() { + // A source leaf only *names* its relation, so a plan can name one the + // catalog says nothing about. That is caught up front, before the + // backend builds anything, and the error names the offending source + // rather than surfacing later as an input that was never wired. + let plan = vec![ + Stmt::from(VarStmt { + name: "edges".to_string(), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), + }), + output_stmt("edges"), + ]; + let Err(err) = Pipeline::incremental().runtime(&mut TestProgram::new(plan, [])) else { + panic!("a source the catalog does not describe must be rejected"); + }; + assert!( + err.to_string().contains("edge"), + "expected the error to name the unknown source, got: {err}" + ); + } + #[test] fn test_standard_join() -> Result<(), anyhow::Error> { let plan = vec![ Stmt::from(VarStmt { name: "person".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Person::schema()))), + initializer: Some(Expr::from(SourceExpr::new(PersonRel::id()))), }), Stmt::from(VarStmt { name: "profession".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Profession::schema()))), + initializer: Some(Expr::from(SourceExpr::new(ProfessionRel::id()))), }), Stmt::from(VarStmt { name: "joined".to_string(), @@ -444,19 +481,19 @@ mod test { }), output_stmt("joined"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = Pipeline::incremental().runtime(&mut TestProgram::new( + plan, + [PersonRel::schema(), ProfessionRel::schema()], + ))?; for (person_step, profession_step) in person_profession_data() { - rt.feed(&SourceId::from("person"), rows_with_weight(person_step, 1))?; - rt.feed( - &SourceId::from("profession"), - rows_with_weight(profession_step, 1), - )?; + assert!(rt.feed(&PersonRel::id(), rows_with_weight(person_step, 1))?); + assert!(rt.feed(&ProfessionRel::id(), rows_with_weight(profession_step, 1),)?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("joined"))?.0, + rt.output(&SinkId::from("joined"))?.to_debug_zset(), zset! { tuple!(0_u64, "Alice", 20_u64, 0_u64, "Engineer") => 1, tuple!(2_u64, "Charlie", 40_u64, 0_u64, "Engineer") => 1, @@ -473,11 +510,11 @@ mod test { let plan = vec![ Stmt::from(VarStmt { name: "person".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Person::schema()))), + initializer: Some(Expr::from(SourceExpr::new(PersonRel::id()))), }), Stmt::from(VarStmt { name: "profession".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Profession::schema()))), + initializer: Some(Expr::from(SourceExpr::new(ProfessionRel::id()))), }), Stmt::from(VarStmt { name: "joined".to_string(), @@ -495,19 +532,19 @@ mod test { }), output_stmt("joined"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = Pipeline::incremental().runtime(&mut TestProgram::new( + plan, + [PersonRel::schema(), ProfessionRel::schema()], + ))?; for (person_step, profession_step) in person_profession_data() { - rt.feed(&SourceId::from("person"), rows_with_weight(person_step, 1))?; - rt.feed( - &SourceId::from("profession"), - rows_with_weight(profession_step, 1), - )?; + assert!(rt.feed(&PersonRel::id(), rows_with_weight(person_step, 1))?); + assert!(rt.feed(&ProfessionRel::id(), rows_with_weight(profession_step, 1),)?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("joined"))?.0, + rt.output(&SinkId::from("joined"))?.to_debug_zset(), zset! { tuple!(0_u64, "Alice", 20_u64, 0_u64, 0_u64, "Engineer") => 1, tuple!(0_u64, "Alice", 20_u64, 0_u64, 1_u64, "Doctor") => 1, @@ -527,7 +564,7 @@ mod test { let plan = vec![ Stmt::from(VarStmt { name: "edges".to_string(), - initializer: Some(Expr::from(SourceExpr::new(Edge::schema()))), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), }), Stmt::from(VarStmt { name: "len_1".to_string(), @@ -694,21 +731,22 @@ mod test { }), output_stmt("full_closure"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = + Pipeline::incremental().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()]))?; let init_data = [ - Edge::new(0, 1, 1), + EdgeRel::new(0, 1, 1), // This edge is omitted: Edge::new(1, 2, 1), - Edge::new(2, 3, 2), - Edge::new(3, 4, 2), + EdgeRel::new(2, 3, 2), + EdgeRel::new(3, 4, 2), ]; - rt.feed(&SourceId::from("edges"), rows_with_weight(init_data, 1))?; + assert!(rt.feed(&EdgeRel::id(), rows_with_weight(init_data, 1))?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("full_closure"))?.0, + rt.output(&SinkId::from("full_closure"))?.to_debug_zset(), zset! { tuple!(0_u64, 1_u64, 1_u64, 1_u64) => 1, tuple!(2_u64, 3_u64, 2_u64, 1_u64) => 1, @@ -717,14 +755,14 @@ mod test { } ); - let extra_data = [Edge::new(1, 2, 1)]; + let extra_data = [EdgeRel::new(1, 2, 1)]; - rt.feed(&SourceId::from("edges"), rows_with_weight(extra_data, 1))?; + assert!(rt.feed(&EdgeRel::id(), rows_with_weight(extra_data, 1))?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("full_closure"))?.0, + rt.output(&SinkId::from("full_closure"))?.to_debug_zset(), zset! { tuple!(0_u64, 2_u64, 2_u64, 2_u64) => 1, tuple!(1_u64, 2_u64, 1_u64, 1_u64) => 1, @@ -744,7 +782,7 @@ mod test { Stmt::from(VarStmt { name: "edges".to_string(), initializer: Some(Expr::from(ProjectionExpr { - relation: Expr::from(SourceExpr::new(Edge::schema())), + relation: Expr::from(SourceExpr::new(EdgeRel::id())), attributes: ["from", "to", "weight"] .into_iter() .map(|name| (name.to_string(), Expr::from(VarExpr::new(name)))) @@ -830,21 +868,22 @@ mod test { }), output_stmt("closure"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = + Pipeline::incremental().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()]))?; let init_data = [ - Edge::new(0, 1, 1), - Edge::new(1, 2, 1), - Edge::new(2, 3, 2), - Edge::new(3, 4, 2), + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 1), + EdgeRel::new(2, 3, 2), + EdgeRel::new(3, 4, 2), ]; - rt.feed(&SourceId::from("edges"), rows_with_weight(init_data, 1))?; + assert!(rt.feed(&EdgeRel::id(), rows_with_weight(init_data, 1))?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("closure"))?.0, + rt.output(&SinkId::from("closure"))?.to_debug_zset(), zset! { tuple!(0_u64, 1_u64, 1_u64, 1_u64) => 1, tuple!(0_u64, 2_u64, 2_u64, 2_u64) => 1, @@ -862,20 +901,16 @@ mod test { Ok(()) } - #[test] - fn source_leaf_inside_fixed_point_step_is_bridged() -> Result<(), anyhow::Error> { - // A `SourceExpr` referenced *only* inside a step body is legal: the - // backend wires its root input and `delta0`s it into the nested circuit, - // exactly as it would an outer variable — no explicit imports needed. - // This computes reachability: starting from the seed nodes in `plain`, - // follow `edges` transitively. `edges` appears nowhere but the step, so - // this exercises wiring a source's root input for a step-only source. - let plan = vec![ + /// Reachability from the seed nodes in `plain` over `edges`, + /// arithmetic-free. Shared between the incremental run and its + /// batch twin below. + fn reachability_plan() -> Vec { + vec![ // base = the seed node ids from `plain`, as a single `node` column. Stmt::from(VarStmt { name: "base".to_string(), initializer: Some(Expr::from(ProjectionExpr { - relation: Expr::from(SourceExpr::new(PlainRelation::schema())), + relation: Expr::from(SourceExpr::new(PlainRel::id())), attributes: vec![("node".to_string(), Expr::from(VarExpr::new("a")))], })), }), @@ -892,7 +927,7 @@ mod test { }), // `edges` used inline in the step — its only use. right: Expr::from(AliasExpr { - relation: Expr::from(SourceExpr::new(Edge::schema())), + relation: Expr::from(SourceExpr::new(EdgeRel::id())), alias: "edge".to_string(), }), on: vec![( @@ -909,26 +944,41 @@ mod test { })), }), output_stmt("reachable"), - ]; + ] + } - let mut rt = Pipeline::incremental().runtime(plan)?; + #[test] + fn source_leaf_inside_fixed_point_step_is_bridged() -> Result<(), anyhow::Error> { + // A `SourceExpr` referenced *only* inside a step body is legal: the + // backend wires its root input and `delta0`s it into the nested circuit, + // exactly as it would an outer variable — no explicit imports needed. + // This computes reachability: starting from the seed nodes in `plain`, + // follow `edges` transitively. `edges` appears nowhere but the step, so + // this exercises wiring a source's root input for a step-only source. + let plan = reachability_plan(); + + let mut rt = Pipeline::incremental().runtime(&mut TestProgram::new( + plan, + [PlainRel::schema(), EdgeRel::schema()], + ))?; // Seed node 0; edges 0->1->2->3. - rt.feed( - &SourceId::from("plain"), - rows([(PlainRelation::new(0, 0, 0), 1)]), - )?; - rt.feed( - &SourceId::from("edges"), + assert!(rt.feed(&PlainRel::id(), rows([(PlainRel::new(0, 0, 0), 1)]),)?); + assert!(rt.feed( + &EdgeRel::id(), rows_with_weight( - [Edge::new(0, 1, 1), Edge::new(1, 2, 1), Edge::new(2, 3, 1)], + [ + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 1), + EdgeRel::new(2, 3, 1) + ], 1, ), - )?; + )?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("reachable"))?.0, + rt.output(&SinkId::from("reachable"))?.to_debug_zset(), zset! { tuple!(0_u64) => 1, tuple!(1_u64) => 1, @@ -940,17 +990,188 @@ mod test { Ok(()) } + /// The reachability plan on the batch backend, cross-checked against + /// the incremental run. The first commit from empty state makes the + /// incremental delta equal the full state, so both engines must + /// produce exactly the same rows. + #[test] + fn batch_reachability_matches_incremental() -> Result<(), anyhow::Error> { + let seed = || [(PlainRel::new(0, 0, 0), 1)]; + let edges = || { + [ + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 1), + EdgeRel::new(2, 3, 1), + ] + }; + let schemas = || [PlainRel::schema(), EdgeRel::schema()]; + + let mut batch = + Pipeline::batch().runtime(&mut TestProgram::new(reachability_plan(), schemas()))?; + assert!(batch.feed(&PlainRel::id(), rows(seed()))?); + assert!(batch.feed(&EdgeRel::id(), rows_with_weight(edges(), 1))?); + batch.commit()?; + let snapshot = batch.output(&SinkId::from("reachable"))?; + assert_eq!(snapshot.columns(), ["node"]); + assert_eq!(snapshot.len(), 4); + assert!(!snapshot.is_empty()); + + let mut incremental = Pipeline::incremental() + .runtime(&mut TestProgram::new(reachability_plan(), schemas()))?; + assert!(incremental.feed(&PlainRel::id(), rows(seed()))?); + assert!(incremental.feed(&EdgeRel::id(), rows_with_weight(edges(), 1))?); + incremental.commit()?; + + let expected = zset! { + tuple!(0_u64) => 1, + tuple!(1_u64) => 1, + tuple!(2_u64) => 1, + tuple!(3_u64) => 1, + }; + assert_eq!(snapshot.to_debug_zset(), expected); + assert_eq!( + incremental + .output(&SinkId::from("reachable"))? + .to_debug_zset(), + expected + ); + Ok(()) + } + + /// A plain u64 join through the whole batch pipeline, cross-checked + /// against the incremental backend. + #[test] + fn batch_join_matches_incremental() -> Result<(), anyhow::Error> { + let plan = || { + vec![ + Stmt::from(VarStmt { + name: "edges".to_string(), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), + }), + Stmt::from(VarStmt { + name: "two_hops".to_string(), + initializer: Some(Expr::from(EquiJoinExpr { + left: Expr::from(AliasExpr { + relation: Expr::from(VarExpr::new("edges")), + alias: "h1".to_string(), + }), + right: Expr::from(AliasExpr { + relation: Expr::from(VarExpr::new("edges")), + alias: "h2".to_string(), + }), + on: vec![( + Expr::from(VarExpr::new("to")), + Expr::from(VarExpr::new("from")), + )], + attributes: Some(vec![ + ("start".to_string(), Expr::from(VarExpr::new("h1.from"))), + ("mid".to_string(), Expr::from(VarExpr::new("h1.to"))), + ("end".to_string(), Expr::from(VarExpr::new("h2.to"))), + ]), + })), + }), + output_stmt("two_hops"), + ] + }; + let data = || { + [ + EdgeRel::new(0, 1, 1), + EdgeRel::new(1, 2, 1), + EdgeRel::new(5, 6, 1), + ] + }; + + let mut batch = + Pipeline::batch().runtime(&mut TestProgram::new(plan(), [EdgeRel::schema()]))?; + assert!(batch.feed(&EdgeRel::id(), rows_with_weight(data(), 1))?); + batch.commit()?; + + let mut incremental = + Pipeline::incremental().runtime(&mut TestProgram::new(plan(), [EdgeRel::schema()]))?; + assert!(incremental.feed(&EdgeRel::id(), rows_with_weight(data(), 1))?); + incremental.commit()?; + + let expected = zset! { tuple!(0_u64, 1_u64, 2_u64) => 1 }; + assert_eq!( + batch.output(&SinkId::from("two_hops"))?.to_debug_zset(), + expected + ); + assert_eq!( + incremental + .output(&SinkId::from("two_hops"))? + .to_debug_zset(), + expected + ); + Ok(()) + } + + /// The batch backend's value slice is unsigned integers and booleans. + /// Rows carrying strings fail loudly instead of computing something + /// wrong. + // TODO(Jan): support the remaining scalar types (strings first, via + // dictionary encoding) after the end-to-end slice is complete; then + // `test_standard_join` gets its batch twin, too. + #[test] + fn batch_feed_rejects_strings_for_now() -> Result<(), anyhow::Error> { + use crate::api::deltas::ZRow; + use crate::relational::relation::TupleValue; + use crate::scalarial::ScalarTypedValue; + + let plan = vec![ + Stmt::from(VarStmt { + name: "people".to_string(), + initializer: Some(Expr::from(SourceExpr::new(PersonRel::id()))), + }), + output_stmt("people"), + ]; + let mut rt = + Pipeline::batch().runtime(&mut TestProgram::new(plan, [PersonRel::schema()]))?; + let alice = ZRow::new( + 1, + TupleValue { + data: vec![ + ScalarTypedValue::Uint(0), + ScalarTypedValue::String("Alice".to_string()), + ScalarTypedValue::Uint(20), + ScalarTypedValue::Uint(0), + ], + }, + ) + .expect("non-zero zweight"); + let err = rt.feed(&PersonRel::id(), [alice]).unwrap_err(); + assert!(err.to_string().contains("unsigned integer"), "got: {err}"); + Ok(()) + } + + /// Reading an output before any commit is an error, and feeding a + /// source the plan does not use reports `false` instead of failing. + #[test] + fn batch_output_requires_commit() -> Result<(), anyhow::Error> { + let plan = vec![ + Stmt::from(VarStmt { + name: "edges".to_string(), + initializer: Some(Expr::from(SourceExpr::new(EdgeRel::id()))), + }), + output_stmt("edges"), + ]; + let mut rt = Pipeline::batch().runtime(&mut TestProgram::new(plan, [EdgeRel::schema()]))?; + assert!(!rt.feed(&PlainRel::id(), rows([(PlainRel::new(0, 0, 0), 1)]))?); + let err = rt.output(&SinkId::from("edges")).unwrap_err(); + assert!(err.to_string().contains("commit"), "got: {err}"); + Ok(()) + } + #[test] fn test_mvr_store_crdt() -> Result<(), anyhow::Error> { let plan = vec![ // Inputs start. Stmt::from(VarStmt { name: "pred".to_string(), - initializer: Some(Expr::from(SourceExpr::new(PredRel::schema()))), + initializer: Some(Expr::from(SourceExpr::new(PredRel::id()))), }), Stmt::from(VarStmt { name: "set".to_string(), - initializer: Some(Expr::from(SourceExpr::new(SetOp::schema()))), + initializer: Some(Expr::from(SourceExpr::new(SetRel::id()))), }), // Inputs end. Stmt::from(VarStmt { @@ -1110,7 +1331,10 @@ mod test { }), output_stmt("mvrStore"), ]; - let mut rt = Pipeline::incremental().runtime(plan)?; + let mut rt = Pipeline::incremental().runtime(&mut TestProgram::new( + plan, + [PredRel::schema(), SetRel::schema()], + ))?; // The operation history is as follows: // In first step (just one root operation setting register with key 1 to @@ -1139,9 +1363,9 @@ mod test { ]; let set_op_data = [ - vec![SetOp::new(0, 0, 1, 1)], - vec![SetOp::new(0, 1, 1, 2), SetOp::new(1, 0, 1, 3)], - vec![SetOp::new(1, 2, 1, 4)], + vec![SetRel::new(0, 0, 1, 1)], + vec![SetRel::new(0, 1, 1, 2), SetRel::new(1, 0, 1, 3)], + vec![SetRel::new(1, 2, 1, 4)], ]; let mut expected = [ @@ -1162,13 +1386,13 @@ mod test { .into_iter(); for (pred_rel_step, set_op_step) in pred_rel_data.into_iter().zip(set_op_data) { - rt.feed(&SourceId::from("pred"), rows_with_weight(pred_rel_step, 1))?; - rt.feed(&SourceId::from("set"), rows_with_weight(set_op_step, 1))?; + assert!(rt.feed(&PredRel::id(), rows_with_weight(pred_rel_step, 1))?); + assert!(rt.feed(&SetRel::id(), rows_with_weight(set_op_step, 1))?); rt.commit()?; assert_eq!( - rt.output(&SinkId::from("mvrStore"))?.0, + rt.output(&SinkId::from("mvrStore"))?.to_debug_zset(), expected.next().unwrap() ); } diff --git a/packages/coln-query/src/optimizer/mod.rs b/packages/coln-query/src/optimizer/mod.rs index ba57c3c3..b92c42c8 100644 --- a/packages/coln-query/src/optimizer/mod.rs +++ b/packages/coln-query/src/optimizer/mod.rs @@ -2,15 +2,23 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::{error::OptimizationError, host::Code}; +use crate::{error::OptimizationError, host::QueryIr}; + +pub mod rewrite; /// An optimizer does a series of transformations on the relational algebra IR /// which retain the semantics of the program but intend to improve performance. /// For instance, it could do: +/// - Collapse a PROJECT(JOIN(R, S, ON, []), ATTRS) into a JOIN(R, S, ON, ATTRS). +/// - Some Projections can be turned into a simple schema operations, e.g.: +/// - column rename +/// - column omission/reordering (beware that operations which require schemas +/// to be equal may fail in a "dirty" state, e.g., set difference and union) +/// - Decide a binary join ordering of an NWayJoin /// - predicate pushdown /// - expression simplification pub trait Optimizer: Clone { - fn optimize(self, code: Code) -> Result { + fn optimize(self, code: QueryIr) -> Result { // The default impl does nothing and simply returns the IR as is. Ok(code) } diff --git a/packages/coln-query/src/optimizer/rewrite.rs b/packages/coln-query/src/optimizer/rewrite.rs new file mode 100644 index 00000000..66e8b508 --- /dev/null +++ b/packages/coln-query/src/optimizer/rewrite.rs @@ -0,0 +1,661 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! Shared machinery for the static passes that *rewrite* a plan: the +//! [optimizers](crate::optimizer::Optimizer) and the backend +//! [lowerings](crate::relational::Backend::lower). +//! +//! # What this exists for +//! +//! A rewriting pass is one interesting function surrounded by a traversal. The +//! traversal is the same for every pass: Recurse into the children and rebuild +//! the node. Implementing a folding walk is 27 methods long, one per node kind +//! across the three [owned visitor](crate::host::expr::ExprVisitorOwn) +//! families. Writing those per pass means writing the same 26 uninteresting +//! ones again each time. So they are written *here*, once, and a pass +//! contributes only the interesting part: a [`TransformationRule`]. +//! +//! Note what this is **not** an argument for. Walking the plan once per rule +//! instead of once per pass saves nothing worth having: a plan is tens to +//! hundreds of nodes, walked at compile time, not per row. Bundle rules into +//! one [`RewriteDriver`] because you want them to see each other's output, +//! and split them across several because you do not but never for saving the +//! performance cost of the walk. +//! +//! # Rules see relational nodes +//! +//! [`TransformationRule::apply`] is offered [`RelExpr`] nodes, since that is +//! the vocabulary plan rewrites are phrased in. Scalar rewrites (constant +//! folding in a join condition, say) would want a sibling trait offered +//! [`Expr`] nodes; the traversal below already reaches them, so adding one is a +//! matter of a second hook rather than a second walk. +//! +//! # Scheduling +//! +//! [`RewriteDriver::run`] walks the plan repeatedly until a full walk fires +//! nothing. That is all the scheduling there is, and it is enough for one kind +//! of dependency: if rule `A` *enables* rule `B`, `B` fires on a later round, +//! and neither rule has to know the other exists. +//! +//! The other kind — `B` undoes what `A` did — is not a scheduling problem but a +//! modelling error, and no ordering fixes it. So instead of pretending to solve +//! it, the driver bounds the rounds and, on exhausting them, **names the rules +//! that were still firing**. A cycle then shows up as a legible error during +//! development rather than as a plan that quietly stopped halfway. Rules that +//! genuinely pull in opposite directions belong in separate [`RewriteDriver`]s, +//! run in an order you choose. +//! +//! One consequence worth stating: a *mandatory* rule set (a backend lowering, +//! whose whole point is a post-condition) and an *optional* one (the optimizer, +//! which may always decline) should not share a [`RewriteDriver`] pass. +//! Otherwise a correctness guarantee comes to depend on the scheduler, +//! which is the same mistake as folding +//! [`Backend::lower`](crate::relational::Backend::lower) into the +//! [`Optimizer`](crate::optimizer::Optimizer). Share the machinery, +//! not the rule set, and have the mandatory side *verify* its post-condition +//! afterwards (a [`walk`](mod@crate::host::walk) scan) rather than trust that +//! the rules were scheduled right. + +use crate::{ + error::RewriteError, + host::{ + QueryIr, + expr::{ + AssignExpr, BinaryExpr, CallExpr, Expr, ExprVisitorOwn, FunctionExpr, GetIndexExpr, + GroupingExpr, LiteralExpr, TupleExpr, UnaryExpr, VarExpr, + }, + stmt::{BlockStmt, ExprStmt, Stmt, StmtVisitorOwn, VarStmt}, + }, + relational::expr::{ + AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, + FixedPointIterExpr, JoinVariable, MultiWayEquiJoinExpr, OutputExpr, ProjectionExpr, + RelExpr, RelExprVisitorOwn, RelKind, SelectionExpr, SourceExpr, UnionExpr, + }, +}; + +/// When in the traversal a rule is offered a node. +#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)] +pub enum Direction { + /// Before the node's children are rewritten. What a rule that pushes + /// something *into* a subtree needs (predicate pushdown has to see the + /// selection before descending past it). + TopDown, + /// After the node's children are rewritten, so the rule sees operands that + /// are already in their final form. What canonicalization and lowering + /// need, and the default. + BottomUp, +} + +/// The outcome of offering a node to a rule. +/// +/// A rule that declines hands the node back rather than answering a separate +/// "would you fire?" question first. That keeps the structural precondition +/// ([`TransformationRule::interest`]) apart from the semantic one without +/// making the rule take the node apart twice, and it makes "declined" a state +/// the type system tracks. +pub enum Rewritten { + /// The rule fired; this replaces the node. + Changed(Expr), + /// The rule declined; this is the node it was handed, untouched. + Unchanged(RelExpr), +} + +/// One semantics-preserving rewrite of the plan, independent of the traversal +/// that finds the nodes to apply it to. +pub trait TransformationRule { + /// Identifies the rule in errors and in the driver's cycle report. + fn name(&self) -> &'static str; + + /// The node kinds this rule could fire on. A dispatch filter, not a full + /// precondition. Conditions that need to inspect the node's contents + /// belong in [`apply`](Self::apply), expressed by declining. + fn interest(&self) -> &'static [RelKind]; + + /// When in the traversal this rule wants its nodes. See [`Direction`]. + fn direction(&self) -> Direction { + Direction::BottomUp + } + + /// Rewrite `node`, or hand it back unchanged. + fn apply(&mut self, node: RelExpr) -> Result; +} + +/// A set of rules and the driver that runs them to a fixed point. +pub struct RewriteDriver { + rules: Vec>, + max_rounds: usize, +} + +impl RewriteDriver { + /// Enough rounds for any cascade a sane rule set produces, few enough that + /// a cycle is reported promptly. + pub const DEFAULT_MAX_ROUNDS: usize = 16; + + pub fn new(rules: Vec>) -> Self { + Self { + rules, + max_rounds: Self::DEFAULT_MAX_ROUNDS, + } + } + + pub fn with_max_rounds(mut self, max_rounds: usize) -> Self { + self.max_rounds = max_rounds; + self + } + + /// Walk `plan` until a full walk fires no rule, and return the result. + /// + /// Errors if the rules are still firing after + /// [`max_rounds`](Self::with_max_rounds) walks, naming the ones that were, + /// since that is what a pair of rules undoing each other looks like from + /// here. + pub fn run(&mut self, plan: QueryIr) -> Result { + let mut plan = plan; + let mut fired = Vec::new(); + + for _ in 0..self.max_rounds { + let mut rewriter = Rewriter { + rules: &mut self.rules, + fired: Vec::new(), + }; + plan = plan + .into_iter() + .map(|stmt| rewriter.visit_stmt(stmt, ())) + .collect::>()?; + if rewriter.fired.is_empty() { + return Ok(plan); + } + fired = rewriter.fired; + } + + fired.sort_unstable(); + fired.dedup(); + Err(RewriteError::new(format!( + "Rewriting did not reach a fixed point after {} rounds; these rules were still \ + firing: {}. Rules that undo one another cannot be ordered apart. Run them in \ + separate rewrites, or narrow the condition under which they fire", + self.max_rounds, + fired.join(", ") + ))) + } +} + +/// One walk of the plan. Implements the traversal once for every rewriting +/// pass there will ever be; the passes themselves live in the +/// [rules](TransformationRule). +struct Rewriter<'r> { + /// The rules to apply in this static pass. + rules: &'r mut [Box], + /// The rules that fired during this walk, for the fixed-point test and for + /// the cycle report. + fired: Vec<&'static str>, +} + +type VisitorResult = Result; + +impl Rewriter<'_> { + /// The single recursion point: every child expression passes through here. + /// + /// Having exactly one means a top-down rule is offered each node exactly + /// once per round, *before* that node's children are rewritten and never + /// again on its own output within the round. Which is what keeps a rule + /// that fires on what it just produced from recursing until the stack gives + /// out — it gets its next turn on the next round instead, under the + /// driver's round budget. + fn child(&mut self, expr: Expr) -> VisitorResult { + let expr = self.offer(expr, Direction::TopDown)?; + self.visit_expr(expr, ()) + } + + /// Offers a node to every rule facing `direction` whose interest covers it, + /// stopping at the first that fires. + fn offer(&mut self, expr: Expr, direction: Direction) -> VisitorResult { + let mut node = match expr { + Expr::Relational(node) => node, + // Rules are phrased over relational operators, so a host expression + // is only ever passed through. See the module docs. + other => return Ok(other), + }; + let kind = node.kind(); + + for rule in self.rules.iter_mut() { + if rule.direction() != direction || !rule.interest().contains(&kind) { + continue; + } + match rule.apply(node)? { + Rewritten::Changed(expr) => { + self.fired.push(rule.name()); + return Ok(expr); + } + Rewritten::Unchanged(declined) => node = declined, + } + } + + Ok(Expr::Relational(node)) + } + + fn children(&mut self, exprs: Vec) -> VisitorResult> { + exprs.into_iter().map(|expr| self.child(expr)).collect() + } + + fn stmts(&mut self, stmts: Vec) -> VisitorResult> { + stmts + .into_iter() + .map(|stmt| self.visit_stmt(stmt, ())) + .collect() + } + + fn pairs(&mut self, pairs: Vec<(Expr, Expr)>) -> VisitorResult> { + pairs + .into_iter() + .map(|(left, right)| Ok((self.child(left)?, self.child(right)?))) + .collect() + } + + fn attributes( + &mut self, + attributes: Vec<(String, Expr)>, + ) -> VisitorResult> { + attributes + .into_iter() + .map(|(name, expr)| Ok((name, self.child(expr)?))) + .collect() + } + + fn optional_attributes( + &mut self, + attributes: Option>, + ) -> VisitorResult>> { + attributes + .map(|attributes| self.attributes(attributes)) + .transpose() + } + + fn join_variables(&mut self, on: Vec) -> VisitorResult> { + on.into_iter() + .map(|variable| { + Ok(JoinVariable { + name: variable.name, + occurrences: variable + .occurrences + .into_iter() + .map(|(relation, expr)| Ok((relation, self.child(expr)?))) + .collect::>()?, + }) + }) + .collect() + } +} + +impl StmtVisitorOwn, ()> for Rewriter<'_> { + fn visit_var_stmt(&mut self, mut stmt: Box, _ctx: ()) -> VisitorResult { + stmt.initializer = stmt + .initializer + .map(|initializer| self.child(initializer)) + .transpose()?; + Ok(stmt.into()) + } + + fn visit_expr_stmt(&mut self, mut stmt: Box, _ctx: ()) -> VisitorResult { + stmt.expr = self.child(stmt.expr)?; + Ok(stmt.into()) + } + + fn visit_block_stmt(&mut self, mut stmt: Box, _ctx: ()) -> VisitorResult { + stmt.stmts = self.stmts(stmt.stmts)?; + Ok(stmt.into()) + } +} + +impl ExprVisitorOwn, ()> for Rewriter<'_> { + fn visit_literal_expr(&mut self, expr: Box, _ctx: ()) -> VisitorResult { + Ok(expr.into()) + } + + fn visit_tuple_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.elements = self.children(expr.elements)?; + Ok(expr.into()) + } + + fn visit_get_index_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.target = self.child(expr.target)?; + expr.index = self.child(expr.index)?; + Ok(expr.into()) + } + + fn visit_grouping_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.expr = self.child(expr.expr)?; + Ok(expr.into()) + } + + fn visit_binary_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.left = self.child(expr.left)?; + expr.right = self.child(expr.right)?; + Ok(expr.into()) + } + + fn visit_unary_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.operand = self.child(expr.operand)?; + Ok(expr.into()) + } + + fn visit_var_expr(&mut self, expr: Box, _ctx: ()) -> VisitorResult { + Ok(expr.into()) + } + + fn visit_assign_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.value = self.child(expr.value)?; + Ok(expr.into()) + } + + fn visit_function_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.body.stmts = self.stmts(expr.body.stmts)?; + Ok(expr.into()) + } + + fn visit_call_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.callee = self.child(expr.callee)?; + expr.arguments = self.children(expr.arguments)?; + Ok(expr.into()) + } + + fn visit_relational_expr(&mut self, expr: RelExpr, _ctx: ()) -> VisitorResult { + self.visit_rel(expr, ()) + } +} + +/// Each of these rewrites the node's children and then offers the node itself +/// to the bottom-up rules. The top-down offer already happened, in +/// [`Rewriter::child`], on the way in. +impl RelExprVisitorOwn, ()> for Rewriter<'_> { + fn visit_source_expr(&mut self, expr: Box, _ctx: ()) -> VisitorResult { + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_output_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.relation = self.child(expr.relation)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_alias_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.relation = self.child(expr.relation)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_distinct_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.relation = self.child(expr.relation)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_union_expr(&mut self, mut expr: Box, _ctx: ()) -> VisitorResult { + expr.relations = self.children(expr.relations)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_difference_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.left = self.child(expr.left)?; + expr.right = self.child(expr.right)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_selection_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.relation = self.child(expr.relation)?; + expr.condition = self.child(expr.condition)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_projection_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.relation = self.child(expr.relation)?; + expr.attributes = self.attributes(expr.attributes)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_cartesian_product_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.inner.left = self.child(expr.inner.left)?; + expr.inner.right = self.child(expr.inner.right)?; + expr.inner.attributes = self.optional_attributes(expr.inner.attributes)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_equi_join_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.left = self.child(expr.left)?; + expr.right = self.child(expr.right)?; + expr.on = self.pairs(expr.on)?; + expr.attributes = self.optional_attributes(expr.attributes)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_multi_way_equi_join_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.relations = self.children(expr.relations)?; + expr.on = self.join_variables(expr.on)?; + expr.attributes = self.optional_attributes(expr.attributes)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_anti_join_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.left = self.child(expr.left)?; + expr.right = self.child(expr.right)?; + expr.on = self.pairs(expr.on)?; + self.offer(expr.into(), Direction::BottomUp) + } + + fn visit_fixed_point_iter_expr( + &mut self, + mut expr: Box, + _ctx: (), + ) -> VisitorResult { + expr.accumulator.1 = self.child(expr.accumulator.1)?; + expr.step.stmts = self.stmts(expr.step.stmts)?; + self.offer(expr.into(), Direction::BottomUp) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + host::{ + expr::VarExpr, + walk::{self, Node}, + }, + relational::expr::{DistinctExpr, OutputKind, SinkId}, + }; + + fn relation(name: &str) -> Expr { + Expr::from(VarExpr::new(name)) + } + + fn plan(expr: Expr) -> QueryIr { + QueryIr::new(vec![Stmt::from(ExprStmt { expr })]) + } + + fn kinds(plan: &QueryIr) -> Vec { + walk::pre_order(plan) + .filter_map(Node::as_rel) + .map(RelExpr::kind) + .collect() + } + + /// Wraps every [`RelKind::Distinct`] it is offered in another one, up to + /// `budget` times in total. A rule that fires on its own output, which is + /// what the round budget and the once-per-node offer exist for. + struct Nest { + budget: usize, + direction: Direction, + } + + impl TransformationRule for Nest { + fn name(&self) -> &'static str { + "nest" + } + fn interest(&self) -> &'static [RelKind] { + &[RelKind::Distinct] + } + fn direction(&self) -> Direction { + self.direction + } + fn apply(&mut self, node: RelExpr) -> Result { + if self.budget == 0 { + return Ok(Rewritten::Unchanged(node)); + } + self.budget -= 1; + Ok(Rewritten::Changed(Expr::from(DistinctExpr { + relation: Expr::from(node), + }))) + } + } + + /// Declines everything, so a walk over it must be a no-op. + struct Inert; + + impl TransformationRule for Inert { + fn name(&self) -> &'static str { + "inert" + } + fn interest(&self) -> &'static [RelKind] { + RelKind::ALL + } + fn apply(&mut self, node: RelExpr) -> Result { + Ok(Rewritten::Unchanged(node)) + } + } + + #[test] + fn a_rule_that_declines_everything_leaves_the_plan_alone() { + let original = plan(Expr::from(OutputExpr { + relation: Expr::from(DistinctExpr { + relation: relation("r"), + }), + id: SinkId::from("out"), + kind: OutputKind::Channel, + })); + + let rewritten = RewriteDriver::new(vec![Box::new(Inert)]) + .run(original.clone()) + .expect("A rule that never fires cannot fail to converge"); + + assert_eq!(rewritten, original); + } + + #[test] + fn reaches_a_fixed_point_when_the_rules_stop_firing() { + // Two nestings, so the rule has to be offered its own output — which + // only happens on a later round. + let rewritten = RewriteDriver::new(vec![Box::new(Nest { + budget: 2, + direction: Direction::BottomUp, + })]) + .run(plan(Expr::from(DistinctExpr { + relation: relation("r"), + }))) + .expect("The rule runs out of budget and the walk settles"); + + assert_eq!( + kinds(&rewritten), + vec![RelKind::Distinct, RelKind::Distinct, RelKind::Distinct] + ); + } + + #[test] + fn a_top_down_rule_does_not_recurse_on_its_own_output() { + // The same rule from the other direction. Were the replacement + // re-offered at the node it came from, this would recurse until the + // stack gave out instead of settling after three rounds. + let rewritten = RewriteDriver::new(vec![Box::new(Nest { + budget: 2, + direction: Direction::TopDown, + })]) + .run(plan(Expr::from(DistinctExpr { + relation: relation("r"), + }))) + .expect("The rule runs out of budget and the walk settles"); + + assert_eq!( + kinds(&rewritten), + vec![RelKind::Distinct, RelKind::Distinct, RelKind::Distinct] + ); + } + + #[test] + fn reports_the_rules_that_kept_firing_when_it_cannot_converge() { + // `usize::MAX` budget: the rule never stops, which is what a pair of + // rules undoing each other looks like from the driver's side. + let error = RewriteDriver::new(vec![Box::new(Nest { + budget: usize::MAX, + direction: Direction::BottomUp, + })]) + .with_max_rounds(4) + .run(plan(Expr::from(DistinctExpr { + relation: relation("r"), + }))) + .expect_err("A rule that always fires must not spin forever"); + + assert!(error.message.contains("nest"), "{}", error.message); + assert!(error.message.contains('4'), "{}", error.message); + } + + #[test] + fn skips_rules_whose_interest_does_not_cover_the_node() { + // `Nest` asks for `Distinct` only, so a plan without one must come back + // untouched however many rounds it is given. + let original = plan(Expr::from(OutputExpr { + relation: relation("r"), + id: SinkId::from("out"), + kind: OutputKind::Channel, + })); + + let rewritten = RewriteDriver::new(vec![Box::new(Nest { + budget: usize::MAX, + direction: Direction::BottomUp, + })]) + .run(original.clone()) + .expect("An uninterested rule is never offered the node"); + + assert_eq!(rewritten, original); + } +} diff --git a/packages/coln-query/src/pipeline.rs b/packages/coln-query/src/pipeline.rs index 29866691..46a40f59 100644 --- a/packages/coln-query/src/pipeline.rs +++ b/packages/coln-query/src/pipeline.rs @@ -3,7 +3,7 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT //! The main entrypoint to the query engine. This wires up all stages into one -//! piece. The input to the [`Pipeline`] is a [(logical) query program](`Code`) +//! piece. The input to the [`Pipeline`] is a [(logical) query program](`QueryIr`) //! which is technically abstract syntax _forest_ (ASF). A _single_ query would //! be a tree but a query _program_ can contain multiple queries, hence the name. //! Importantly, said query program must be in a valid execution order already, @@ -14,38 +14,57 @@ //! //! Here's an overview of the stages: //! -//! 1. Typechecker: Takes an [ASF](Code) and type checks it. +//! 1. Typechecker: Takes an [ASF](QueryIr) and type checks it. //! Currently skipped, as `coln-compiler` already emits type checked FLIR. -//! 2. Optimizer: Takes the [type-checked ASF](Code) and optimizes it +//! 2. Optimizer: Takes the [type-checked ASF](QueryIr) and optimizes it //! _logically_. It can thereby rewrite parts of the queries. As of now, //! there is no logical optimization implemented. Returns a type-checked //! and optimized ASF. -//! 3. Resolver: Takes a [type-checked and optimized ASF](`Code`). +//! 3. Lowering: Takes the [type-checked and optimized ASF](QueryIr) and lets the +//! [`Backend`] rewrite it into the operator vocabulary it can actually +//! execute — see [`Backend::lower`]. Unlike the optimizer this is not +//! optional: the [`DbspBackend`] folds every +//! [`MultiWayEquiJoinExpr`](crate::relational::expr::MultiWayEquiJoinExpr) +//! into a sequence of binary joins here, because it has no other way to +//! execute one. It runs *before* the resolver so that the nodes it mints get +//! resolved along with everything else. +//! 4. Resolver: Takes a [type-checked, optimized and lowered ASF](`QueryIr`). //! It resolves all variables (of the host language) to slots in an //! interpretation [`Environment`]. //! in a static pass over the ASF, speeding up variable lookup and checking //! for invalid variable access. Returns a [resolved ASF](ResolvedCode). -//! 4. Build: Takes a [resolved ASF](ResolvedCode) (and maybe type-checked and -//! optimized) and hands it off to the supplied [`Backend`] to prepare -//! for execution. A backend can work incrementally or batchwise. -//! Returns a [`Runtime`](crate::relational::Runtime). -//! 5. Run: [`Runtime`](crate::relational::Runtime) is the runnable artifact: +//! 5. Source resolution: Looks up every +//! [`SourceExpr`](crate::relational::expr::SourceExpr) leaf of the plan in the +//! program's [`Catalog`](crate::relational::catalog::Catalog) (see +//! [`resolve_sources`]). This is the only stage that touches the catalog so far; +//! the [`SourceSchemas`](crate::relational::catalog::SourceSchemas) it +//! produces are what the backend gets. A leaf the catalog does not describe +//! is rejected here, before anything has been built. +//! 6. Build: Takes a [resolved ASF](ResolvedCode) (and maybe type-checked and +//! optimized) plus those source schemas, and hands them off to the supplied +//! [`Backend`] to prepare for execution. A backend can work incrementally or +//! batchwise. Returns a [`Runtime`](crate::relational::Runtime). +//! 7. Run: [`Runtime`](crate::relational::Runtime) is the runnable artifact: //! Feed input changes, advance, and output results. This is where //! incremental vs batch actually differ: DBSP's `commit` runs one //! incremental transaction and yields per-commit -//! [`Delta`](crate::relational::Delta)s; the batch engine recomputes from -//! the accumulated inputs and yields [`Snapshot`](crate::relational::Snapshot)s. +//! [`Delta`](crate::relational::incremental::dbsp::DbspOutputDelta)s; +//! the batch engine recomputes from the accumulated inputs and yields +//! [`Snapshot`](crate::relational::batch::Snapshot)s. use std::num::NonZeroUsize; use crate::{ error::QueryEngineError, host::{ - Code, HostInterpreter, InterpreterContext, ScalarHost, + HostInterpreter, InterpreterContext, QueryIr, ScalarHost, resolver::ResolvedCode, variable::{Environment, Value}, }, optimizer::{NoOptimizer, Optimizer}, - relational::{Backend, batch::BatchBackend, incremental::DbspBackend}, + program::QueryProgram, + relational::{ + Backend, batch::BatchBackend, catalog::resolve_sources, incremental::DbspBackend, + }, }; pub struct Pipeline { @@ -94,15 +113,25 @@ impl Pipeline { self.threads = threads; self } - /// Optimize, resolve and evaluate a self-container **query** program + /// Optimize, lower, resolve and evaluate a self-contained **query** program /// (with relational operators) on the [`Backend`](`Self::backend`) and /// with the [`Optimizer`](`Self::optimizer`). - pub fn runtime(self, plan: Code) -> Result { - let type_checked = plan; // Not for now. + /// + /// The program is taken by value because its code is *moved* through the + /// stages: each one consumes a [`QueryIr`] and returns the rewritten one. + pub fn runtime(self, program: &mut impl QueryProgram) -> Result { + let type_checked = program.take_code(); // Not type checked, for now. let optimized = self.optimizer.optimize(type_checked)?; - let resolved = ResolvedCode::from(optimized)?; + let lowered = self.backend.lower(optimized)?; + let resolved = ResolvedCode::from(lowered)?; + // Resolve what the plan's source leaves name against the program's + // catalog. This is the *only* place a `Catalog` is consulted: everything + // downstream works from the resolved schemas, so no backend has to reach + // back into a frontend's data structure. It runs after lowering, so a + // source a lowering pass minted is resolved along with the rest. + let sources = resolve_sources(resolved.as_code(), program)?; self.backend - .build(self.threads, resolved) + .build(self.threads, resolved, sources) .map_err(|e| e.into().into()) } } @@ -111,8 +140,8 @@ impl Pipeline<(), ()> { /// Resolve and evaluate a self-contained **host-language** program once (no /// relational operators), returning the value of its last statement. For pure /// scalar/host tests. Relational programs must go through [`Self::runtime`]. - pub fn run(host_code: Code) -> Result, QueryEngineError> { - let type_checked = host_code; // Not for now. + pub fn run(host_code: impl Into) -> Result, QueryEngineError> { + let type_checked = host_code.into(); // Not for now. let resolved = ResolvedCode::from(type_checked)?; let mut environment = Environment::default(); diff --git a/packages/coln-query/src/program.rs b/packages/coln-query/src/program.rs new file mode 100644 index 00000000..dca5b787 --- /dev/null +++ b/packages/coln-query/src/program.rs @@ -0,0 +1,49 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! What a frontend hands the [`Pipeline`](crate::pipeline::Pipeline). +//! +//! A query program is two things: the [`QueryIr`] to compile, and the +//! [`Catalog`] describing the extensional relations its +//! [`SourceExpr`](crate::relational::expr::SourceExpr) leaves name. This module +//! is deliberately *not* part of [`api`](crate::api) because that module is +//! coln's FLIR frontend, and a second frontend (a Datalog one, say) must be +//! able to implement [`QueryProgram`] independently. + +use crate::{ + host::{QueryIr, print}, + relational::catalog::Catalog, +}; + +/// A compilable query program: the [`QueryIr`] the pipeline rewrites, plus (via +/// [`Catalog`]) what its source leaves refer to. +/// +/// Each frontend implements this over its own program type, keeping whatever +/// else it needs — coln's FLIR frontend also tracks the derived view per rule — +/// beside the two things the pipeline asks for. +pub trait QueryProgram: Catalog { + /// The program's statements, for inspection. + fn code(&self) -> &QueryIr; + + /// Hand the code over to be rewritten, leaving the catalog behind. + /// + /// Every stage of the pipeline *consumes* the code: The optimizer, the + /// backend's lowering and the resolver each take a [`QueryIr`] and return + /// the rewritten one, while all of them may still need to ask the catalog + /// what a source means. Moving the code out rather than borrowing it is + /// what lets both hold: afterwards this program describes its sources as + /// before, and [`code`](Self::code) reports the empty program that is left. + fn take_code(&mut self) -> QueryIr; + + /// This program rendered as an indented node tree, with each source leaf + /// described by *this* program's catalog. The counterpart of + /// [`QueryIr::to_tree`], which has no catalog to consult and so can only + /// name the leaves. + fn to_tree(&self) -> String + where + Self: Sized, + { + print::to_tree_with(self.code(), self) + } +} diff --git a/packages/coln-query/src/relational/batch/lowering.rs b/packages/coln-query/src/relational/batch/lowering.rs new file mode 100644 index 00000000..422ee4c8 --- /dev/null +++ b/packages/coln-query/src/relational/batch/lowering.rs @@ -0,0 +1,1358 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! Lowering from the logical plan to coln-batch's physical form. +//! +//! The whole statement list becomes one Datalog program: every bound +//! variable turns into a derived relation defined by one or more rules, +//! sources stay stored relations, and a `FixedPointIter` contributes +//! recursive rules for its accumulator. Evaluating the program with +//! coln-batch's semi-naive fixpoint then computes the entire plan, +//! including the dependencies between statements, in one run. +//! +//! Scope of this first slice: purely relational plans. Sources, equi +//! joins on columns (chains are flattened into one n-ary rule body), +//! multi-way equi joins (FLIR's native n-ary node, consumed directly), +//! cartesian products, projections onto columns and literals, equality +//! selections against literals, unions, distinct, and fixed points whose +//! step is a single relational expression. Everything that needs the +//! scalar engine (computed columns, general conditions) and the +//! remaining operators (anti join, difference) fail with a clear error +//! instead of a wrong answer. +//! +//! The tests at the bottom build up in difficulty and double as a guided +//! tour of the translation. + +use std::collections::HashMap; + +use anyhow::{Context, Result, bail}; +use coln_batch::query::{Atom, Term}; +use coln_batch::rule::{Program, Rule}; + +use crate::host::QueryIr; +use crate::host::expr::{Expr, Literal}; +use crate::host::operator::Operator; +use crate::host::stmt::{Stmt, VarStmt}; +use crate::relational::catalog::SourceSchemas; +use crate::relational::expr::{ + EquiJoinExpr, FixedPointIterExpr, MultiWayEquiJoinExpr, OutputKind, RelExpr, SourceExpr, +}; + +/// The result of lowering a logical plan. +#[derive(Debug)] +pub struct LoweredPlan { + /// One Datalog program covering every statement of the plan. + pub program: Program, + /// Source id (schema name) to column names, in tuple order. + pub sources: HashMap>, + /// Sink id to the derived relation `output` reads. + pub outputs: HashMap, + /// Column names per relation, sources and derived alike. + pub schemas: HashMap>, +} + +/// Lower a plan (its statement list) into a [`LoweredPlan`]. `sources` +/// describes the base tables the plan's [`SourceExpr`] leaves name: the +/// batch projection of a [`Catalog`](crate::relational::catalog::Catalog). +pub fn lower(ir: &QueryIr, sources: &SourceSchemas) -> Result { + let available_sources = sources + .iter() + .map(|(id, schema)| { + let columns = schema + .columns() + .iter() + .map(|column| column.name().to_string()) + .collect(); + (id.as_str().to_string(), columns) + }) + .collect(); + let mut lowerer = Lowerer { + available_sources, + ..Lowerer::default() + }; + lowerer.lower_stmts(ir)?; + let mut schemas: HashMap> = lowerer.sources.clone(); + for (name, info) in &lowerer.env { + schemas.insert(name.clone(), info.columns.clone()); + } + Ok(LoweredPlan { + program: Program { + rules: lowerer.rules, + }, + sources: lowerer.sources, + outputs: lowerer.outputs, + schemas, + }) +} + +/// What a plan variable is bound to: a relation plus its column names. +#[derive(Clone)] +struct RelInfo { + relation: String, + columns: Vec, +} + +/// A column binding inside one rule body under construction. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +enum Bind { + Var(usize), + Lit(u64), +} + +impl Bind { + fn term(self, remap: &HashMap) -> Result { + match self { + Bind::Var(v) => remap + .get(&v) + .map(|nv| Term::Var(*nv)) + .context("head column refers to a variable that does not occur in the body"), + Bind::Lit(x) => Ok(Term::Lit(x)), + } + } +} + +/// The visible columns of a lowered sub-expression: the output arity in +/// order, plus a lookup table that also knows alias-qualified names. +/// `None` marks a plain name that became ambiguous after a join. +#[derive(Clone, Default)] +struct Scope { + columns: Vec<(String, Bind)>, + names: HashMap>, +} + +impl Scope { + fn resolve(&self, name: &str) -> Result { + match self.names.get(name) { + Some(Some(bind)) => Ok(*bind), + Some(None) => bail!("column {name} is ambiguous here, qualify it with an alias"), + None => bail!("unknown column {name}"), + } + } + + fn substitute(&mut self, from: usize, to: Bind) { + for (_, bind) in &mut self.columns { + if *bind == Bind::Var(from) { + *bind = to; + } + } + for bind in self.names.values_mut() { + if *bind == Some(Bind::Var(from)) { + *bind = Some(to); + } + } + } +} + +/// One rule body under construction: a variable counter and the atoms. +#[derive(Default)] +struct Frame { + next_var: usize, + atoms: Vec, +} + +impl Frame { + fn fresh(&mut self) -> usize { + self.next_var += 1; + self.next_var - 1 + } + + fn substitute(&mut self, from: usize, to: Bind) { + let to = match to { + Bind::Var(v) => Term::Var(v), + Bind::Lit(x) => Term::Lit(x), + }; + for atom in &mut self.atoms { + for term in &mut atom.terms { + if *term == Term::Var(from) { + *term = to; + } + } + } + } +} + +#[derive(Default)] +struct Lowerer { + rules: Vec, + /// All base tables the catalog offers (id, column names). + available_sources: HashMap>, + /// The subset of `available_sources` the plan actually uses. + sources: HashMap>, + env: HashMap, + outputs: HashMap, +} + +impl Lowerer { + fn lower_stmts(&mut self, stmts: &[Stmt]) -> Result<()> { + for stmt in stmts { + match stmt { + Stmt::Var(var) => self.lower_var_stmt(var)?, + Stmt::Expr(expr_stmt) => self.lower_output_stmt(&expr_stmt.expr)?, + Stmt::Block(_) => bail!("batch lowering does not support nested blocks yet"), + } + } + Ok(()) + } + + /// A variable binding defines a derived relation named after the + /// variable, filled by one rule per union branch (usually one). + fn lower_var_stmt(&mut self, var: &VarStmt) -> Result<()> { + let initializer = var + .initializer + .as_ref() + .with_context(|| format!("variable {} has no initializer", var.name))?; + + let columns = match initializer { + // Aliasing one relation variable to another. + Expr::Var(inner) => { + let info = self.lookup(&inner.name)?.clone(); + self.env.insert(var.name.clone(), info.clone()); + return Ok(()); + } + Expr::Relational(rel) => match rel { + RelExpr::FixedPointIter(fixed_point) => { + self.lower_fixed_point(&var.name, fixed_point)? + } + RelExpr::Union(union) => { + let mut columns: Option> = None; + for branch in &union.relations { + let branch_columns = self.push_rule_for(&var.name, branch)?; + match &columns { + None => columns = Some(branch_columns), + Some(first) => { + if first.len() != branch_columns.len() { + bail!( + "union branches for {} have arities {} and {}", + var.name, + first.len(), + branch_columns.len() + ); + } + } + } + } + columns.with_context(|| format!("union for {} is empty", var.name))? + } + _ => self.push_rule_for(&var.name, initializer)?, + }, + _ => bail!( + "batch lowering only supports relational initializers, {} is a host expression", + var.name + ), + }; + + self.env.insert( + var.name.clone(), + RelInfo { + relation: var.name.clone(), + columns, + }, + ); + Ok(()) + } + + /// Lower one relational expression into a single rule with the given + /// head relation; returns the head's column names. + fn push_rule_for(&mut self, head: &str, expr: &Expr) -> Result> { + let mut frame = Frame::default(); + let scope = self.lower_rel(expr, &mut frame)?; + self.push_rule(head, frame, &scope)?; + Ok(scope.columns.iter().map(|(name, _)| name.clone()).collect()) + } + + /// A fixed point defines its accumulator as a recursive relation: + /// one base rule from the initializer, one rule per step statement, + /// all with the accumulator as their head. Semi-naive evaluation in + /// coln-batch then iterates them to the fixpoint. + fn lower_fixed_point( + &mut self, + head: &str, + fixed_point: &FixedPointIterExpr, + ) -> Result> { + let (accumulator, init) = (&fixed_point.accumulator.0, &fixed_point.accumulator.1); + let columns = self.push_rule_for(head, init)?; + + // Inside the step, the accumulator name refers to the relation + // being defined; that reference is what makes the rule recursive. + let previous = self.env.insert( + accumulator.clone(), + RelInfo { + relation: head.to_string(), + columns: columns.clone(), + }, + ); + + let result = (|| { + let [step] = fixed_point.step.stmts.as_slice() else { + bail!("batch lowering expects exactly one statement in a fixed point step"); + }; + let Stmt::Expr(step_expr) = step else { + bail!("batch lowering expects the fixed point step to be an expression"); + }; + let step_columns = self.push_rule_for(head, &step_expr.expr)?; + if step_columns.len() != columns.len() { + bail!( + "fixed point step for {head} has arity {}, its base has {}", + step_columns.len(), + columns.len() + ); + } + Ok(()) + })(); + + match previous { + Some(info) => { + self.env.insert(accumulator.clone(), info); + } + None => { + self.env.remove(accumulator); + } + } + result?; + Ok(columns) + } + + /// An output statement names a derived relation as a sink. + fn lower_output_stmt(&mut self, expr: &Expr) -> Result<()> { + let Expr::Relational(rel) = expr else { + bail!("batch lowering only supports output statements at the top level"); + }; + let RelExpr::Output(output) = rel else { + bail!("batch lowering only supports output statements at the top level"); + }; + let Expr::Var(var) = &output.relation else { + bail!("batch lowering expects outputs to tap a bound variable"); + }; + let info = self.lookup(&var.name)?; + match output.kind { + OutputKind::Channel => { + self.outputs + .insert(output.id.0.clone(), info.relation.clone()); + } + // Print-only taps have no readable result; nothing to record. + OutputKind::Cli => {} + } + Ok(()) + } + + /// Lower a relational expression into atoms of `frame`, returning the + /// visible columns. + fn lower_rel(&mut self, expr: &Expr, frame: &mut Frame) -> Result { + match expr { + Expr::Var(var) => { + let info = self.lookup(&var.name)?.clone(); + Ok(self.scope_from_atom(&info, frame)) + } + Expr::Relational(rel) => match rel { + RelExpr::Source(source) => self.lower_source(source, frame), + RelExpr::Alias(alias) => { + let inner = self.lower_rel(&alias.relation, frame)?; + let mut names = inner.names.clone(); + for (name, bind) in &inner.columns { + names.insert(format!("{}.{}", alias.alias, name), Some(*bind)); + } + Ok(Scope { + columns: inner.columns, + names, + }) + } + // coln-batch results are sets already, distinct is free. + RelExpr::Distinct(distinct) => self.lower_rel(&distinct.relation, frame), + RelExpr::Projection(projection) => { + let inner = self.lower_rel(&projection.relation, frame)?; + let mut columns = Vec::with_capacity(projection.attributes.len()); + let mut names = HashMap::new(); + for (name, attribute) in &projection.attributes { + let bind = self + .lower_attribute(attribute, &inner) + .with_context(|| format!("in the projection attribute {name}"))?; + columns.push((name.clone(), bind)); + names.insert(name.clone(), Some(bind)); + } + Ok(Scope { columns, names }) + } + RelExpr::Selection(selection) => { + let mut inner = self.lower_rel(&selection.relation, frame)?; + self.apply_selection(&selection.condition, frame, &mut inner)?; + Ok(inner) + } + RelExpr::EquiJoin(join) => self.lower_equi_join(join, frame), + RelExpr::CartesianProduct(product) => self.lower_equi_join(&product.inner, frame), + RelExpr::Union(_) => { + bail!("batch lowering supports union only as a direct variable initializer") + } + RelExpr::FixedPointIter(_) => { + bail!( + "batch lowering supports a fixed point only as a direct variable initializer" + ) + } + RelExpr::Output(_) => bail!("outputs cannot be nested inside other operators"), + RelExpr::MultiWayEquiJoin(join) => self.lower_multi_way(join, frame), + RelExpr::AntiJoin(_) => bail!("batch lowering does not support AntiJoin yet"), + RelExpr::Difference(_) => bail!("batch lowering does not support Difference yet"), + }, + _ => bail!("batch lowering expects a relational expression here"), + } + } + + fn lower_source(&mut self, source: &SourceExpr, frame: &mut Frame) -> Result { + let id = source.as_id().as_str().to_string(); + let columns = self + .available_sources + .get(&id) + .with_context(|| format!("source {id} is not in the catalog"))? + .clone(); + self.sources.insert(id.clone(), columns.clone()); + let info = RelInfo { + relation: id, + columns, + }; + Ok(self.scope_from_atom(&info, frame)) + } + + /// Place one atom over `info`'s relation with fresh variables. + fn scope_from_atom(&mut self, info: &RelInfo, frame: &mut Frame) -> Scope { + let mut columns = Vec::with_capacity(info.columns.len()); + let mut names = HashMap::new(); + let mut terms = Vec::with_capacity(info.columns.len()); + for name in &info.columns { + let var = frame.fresh(); + terms.push(Term::Var(var)); + columns.push((name.clone(), Bind::Var(var))); + names.insert(name.clone(), Some(Bind::Var(var))); + } + frame.atoms.push(Atom { + relation: info.relation.clone(), + terms, + }); + Scope { columns, names } + } + + /// Both join operands are lowered into the same frame; every `on` + /// pair then unifies one variable of each side, which is exactly how + /// a conjunctive query expresses an equi join. Chained joins land in + /// the same frame too, so a chain flattens into one n-ary body and + /// the executor picks its own order. Kept for frontends that emit + /// binary chains; FLIR's native n-ary join takes + /// [`Self::lower_multi_way`]. + fn lower_equi_join(&mut self, join: &EquiJoinExpr, frame: &mut Frame) -> Result { + let left = self.lower_rel(&join.left, frame)?; + let mut right = self.lower_rel(&join.right, frame)?; + let mut left = left; + + for (left_expr, right_expr) in &join.on { + let left_bind = left.resolve(Self::column_name(left_expr)?)?; + let right_bind = right.resolve(Self::column_name(right_expr)?)?; + match (left_bind, right_bind) { + (Bind::Var(l), _) => { + frame.substitute(l, right_bind); + left.substitute(l, right_bind); + } + (Bind::Lit(_), Bind::Var(r)) => { + frame.substitute(r, left_bind); + right.substitute(r, left_bind); + } + (Bind::Lit(a), Bind::Lit(b)) if a == b => {} + (Bind::Lit(a), Bind::Lit(b)) => { + bail!("join condition compares two different literals, {a} and {b}") + } + } + } + + let mut names = left.names.clone(); + for (name, bind) in &right.names { + match names.get(name) { + // A plain name on both sides becomes ambiguous. + Some(_) => { + names.insert(name.clone(), None); + } + None => { + names.insert(name.clone(), *bind); + } + } + } + let merged = Scope { + columns: Vec::new(), + names, + }; + + let columns = match &join.attributes { + Some(attributes) => { + let mut columns = Vec::with_capacity(attributes.len()); + for (name, attribute) in attributes { + let bind = self + .lower_attribute(attribute, &merged) + .with_context(|| format!("in the join attribute {name}"))?; + columns.push((name.clone(), bind)); + } + columns + } + None => left + .columns + .iter() + .chain(right.columns.iter()) + .cloned() + .collect(), + }; + + let mut names = merged.names; + for (name, bind) in &columns { + names.insert(name.clone(), Some(*bind)); + } + Ok(Scope { columns, names }) + } + + /// The n-ary join FLIR emits, consumed natively: every participant + /// is lowered into the same frame, and each join variable unifies + /// one column per occurrence. This is the direct construction of an + /// n-ary rule body, no flattening involved. The output schema + /// follows the documented left-to-right fold: a join variable + /// appears once, named after the variable and carried by the first + /// relation binding it, and a later column is dropped when an + /// earlier relation already contributes an active column of the + /// same name. + fn lower_multi_way(&mut self, join: &MultiWayEquiJoinExpr, frame: &mut Frame) -> Result { + let mut scopes = Vec::with_capacity(join.relations.len()); + for relation in &join.relations { + scopes.push(self.lower_rel(relation, frame)?); + } + + // Unify each join variable's occurrences (the constructor + // guarantees at least two, in relation order) and remember which + // column carries the variable (first occurrence) and which ones + // it deactivates (the rest). + let mut carried: HashMap<(usize, String), Option> = HashMap::new(); + for variable in &join.on { + let mut kept: Option = None; + for (relation, occurrence) in &variable.occurrences { + let column = Self::column_name(occurrence) + .with_context(|| format!("in join variable {}", variable.name))? + .to_string(); + let bind = scopes[*relation] + .resolve(&column) + .with_context(|| format!("in join variable {}", variable.name))?; + kept = Some(match kept { + None => { + carried.insert((*relation, column), Some(variable.name.clone())); + bind + } + Some(kept) => { + carried.insert((*relation, column), None); + match (kept, bind) { + (Bind::Var(l), _) => { + frame.substitute(l, bind); + for scope in &mut scopes { + scope.substitute(l, bind); + } + bind + } + (Bind::Lit(_), Bind::Var(r)) => { + frame.substitute(r, kept); + for scope in &mut scopes { + scope.substitute(r, kept); + } + kept + } + (Bind::Lit(a), Bind::Lit(b)) if a == b => kept, + (Bind::Lit(a), Bind::Lit(b)) => bail!( + "join variable {} equates two different literals, {a} and {b}", + variable.name + ), + } + } + }); + } + } + + // The schema fold, left to right over the participants. + let mut columns: Vec<(String, Bind)> = Vec::new(); + for (relation, scope) in scopes.iter().enumerate() { + for (name, bind) in &scope.columns { + match carried.get(&(relation, name.clone())) { + Some(Some(variable_name)) => columns.push((variable_name.clone(), *bind)), + Some(None) => {} + None => { + if columns.iter().all(|(active, _)| active != name) { + columns.push((name.clone(), *bind)); + } + } + } + } + } + + // Name lookup for a projection step: alias-qualified names from + // the participants (first one wins), plain names from the fold. + let mut names: HashMap> = HashMap::new(); + for scope in &scopes { + for (name, bind) in &scope.names { + if name.contains('.') { + names.entry(name.clone()).or_insert(*bind); + } + } + } + for (name, bind) in &columns { + names.insert(name.clone(), Some(*bind)); + } + let merged = Scope { + columns: columns.clone(), + names, + }; + + let columns = match &join.attributes { + Some(attributes) => { + let mut projected = Vec::with_capacity(attributes.len()); + for (name, attribute) in attributes { + let bind = self + .lower_attribute(attribute, &merged) + .with_context(|| format!("in the join attribute {name}"))?; + projected.push((name.clone(), bind)); + } + projected + } + None => columns, + }; + + let mut names = merged.names; + for (name, bind) in &columns { + names.insert(name.clone(), Some(*bind)); + } + Ok(Scope { columns, names }) + } + + /// The attributes this slice supports: a column reference or a + /// literal. Computed attributes arrive with the scalar engine. + fn lower_attribute(&self, attribute: &Expr, scope: &Scope) -> Result { + match attribute { + Expr::Var(var) => scope.resolve(&var.name), + Expr::Literal(literal) => Ok(Bind::Lit(Self::literal_u64(&literal.value)?)), + _ => bail!( + "computed attributes need the scalar engine and are not supported in this slice" + ), + } + } + + /// The conditions this slice supports: column equals literal (either + /// side). Everything else needs the scalar engine. + fn apply_selection( + &self, + condition: &Expr, + frame: &mut Frame, + scope: &mut Scope, + ) -> Result<()> { + let Expr::Binary(binary) = condition else { + bail!("selection conditions beyond column = literal need the scalar engine"); + }; + if binary.operator != Operator::Equal { + bail!("selection conditions beyond column = literal need the scalar engine"); + } + let (column, literal) = match (&binary.left, &binary.right) { + (Expr::Var(var), Expr::Literal(lit)) => (&var.name, Self::literal_u64(&lit.value)?), + (Expr::Literal(lit), Expr::Var(var)) => (&var.name, Self::literal_u64(&lit.value)?), + _ => bail!("selection conditions beyond column = literal need the scalar engine"), + }; + match scope.resolve(column)? { + Bind::Var(v) => { + frame.substitute(v, Bind::Lit(literal)); + scope.substitute(v, Bind::Lit(literal)); + } + Bind::Lit(existing) if existing == literal => {} + Bind::Lit(existing) => { + bail!("column {column} is already pinned to {existing}, cannot equal {literal}") + } + } + Ok(()) + } + + /// Compact the frame's variables to a dense range and emit the rule. + fn push_rule(&mut self, head_relation: &str, frame: Frame, scope: &Scope) -> Result<()> { + let mut remap: HashMap = HashMap::new(); + for atom in &frame.atoms { + for term in &atom.terms { + if let Term::Var(v) = term { + let next = remap.len(); + remap.entry(*v).or_insert(next); + } + } + } + let body = frame + .atoms + .into_iter() + .map(|atom| Atom { + relation: atom.relation, + terms: atom + .terms + .into_iter() + .map(|term| match term { + Term::Var(v) => Term::Var(remap[&v]), + lit => lit, + }) + .collect(), + }) + .collect(); + let head_terms = scope + .columns + .iter() + .map(|(name, bind)| bind.term(&remap).with_context(|| format!("column {name}"))) + .collect::>>()?; + self.rules.push(Rule { + var_names: (0..remap.len()).map(|i| format!("v{i}")).collect(), + head: Atom { + relation: head_relation.to_string(), + terms: head_terms, + }, + body, + }); + Ok(()) + } + + fn lookup(&self, name: &str) -> Result<&RelInfo> { + self.env + .get(name) + .with_context(|| format!("unknown plan variable {name}")) + } + + fn column_name(expr: &Expr) -> Result<&str> { + match expr { + Expr::Var(var) => Ok(&var.name), + _ => bail!("join conditions must reference columns by name"), + } + } + + fn literal_u64(literal: &Literal) -> Result { + match literal { + Literal::Uint(value) => Ok(*value), + Literal::Bool(value) => Ok(*value as u64), + _ => bail!("only unsigned integer and boolean literals are supported in this slice"), + } + } +} + +#[cfg(test)] +mod tests { + //! A guided tour of the translation, from a single source scan to a + //! recursive fixed point. Read the tests in order; each one adds a + //! single concept on top of the previous. + + use super::*; + use crate::host::expr::{BinaryExpr, Literal, LiteralExpr, VarExpr}; + use crate::host::stmt::{BlockStmt, ExprStmt, VarStmt}; + use crate::relational::expr::{ + AliasExpr, AntiJoinExpr, CartesianProductExpr, DistinctExpr, JoinVariable, OutputExpr, + ProjectionExpr, SelectionExpr, SinkId, SourceId, UnionExpr, + }; + use crate::relational::schema::{Column, EntityRef, TableSchema}; + use crate::scalarial::ScalarType; + use coln_batch::fixpoint::{self, Exec}; + use coln_batch::generic_join; + use coln_batch::query::Catalog; + use coln_batch::relation::Relation; + + // Small plan-building helpers, mirroring how the logical tests in + // lib.rs write their plans, just terser. + + fn src(name: &str) -> SourceId { + SourceId::from(name) + } + + /// One catalog for every rung: the base tables a plan may name. + fn test_sources() -> SourceSchemas { + [ + ("edge", ["from", "to"]), + ("r", ["a", "b"]), + ("s", ["c", "d"]), + ] + .into_iter() + .map(|(name, columns)| { + let columns = columns + .into_iter() + .map(|column| Column::new(column, ScalarType::Uint)) + .collect(); + ( + SourceId::from(name), + TableSchema::new(EntityRef::from(name), columns, vec![]), + ) + }) + .collect() + } + + fn lower_plan(stmts: Vec) -> Result { + lower(&QueryIr::new(stmts), &test_sources()) + } + + fn let_rel(name: &str, initializer: impl Into) -> Stmt { + Stmt::from(VarStmt { + name: name.to_string(), + initializer: Some(initializer.into()), + }) + } + + fn var(name: &str) -> Expr { + Expr::from(VarExpr::new(name)) + } + + fn lit(value: u64) -> Expr { + Expr::from(LiteralExpr { + value: Literal::Uint(value), + }) + } + + fn out(name: &str) -> Stmt { + Stmt::from(ExprStmt { + expr: Expr::from(OutputExpr { + relation: var(name), + id: SinkId::from(name), + kind: OutputKind::Channel, + }), + }) + } + + fn join_variable(name: &str, occurrences: &[(usize, &str)]) -> JoinVariable { + JoinVariable { + name: name.to_string(), + occurrences: occurrences + .iter() + .map(|(relation, column)| (*relation, var(column))) + .collect(), + } + } + + fn named_columns(pairs: &[(&str, &str)]) -> Vec<(String, Expr)> { + pairs + .iter() + .map(|(name, column)| (name.to_string(), var(column))) + .collect() + } + + /// Run a lowered plan over hand-built relations with coln-batch's + /// semi-naive fixpoint and the worst-case-optimal executor. + fn run(plan: &LoweredPlan, edb: Catalog) -> Catalog { + fixpoint::semi_naive(&plan.program, &edb, generic_join::execute as Exec) + .unwrap() + .catalog + } + + fn rel(name: &str, columns: [&str; 2], rows_: &[(u64, u64)]) -> Relation { + Relation::new( + name, + columns, + vec![ + rows_.iter().map(|r| r.0).collect(), + rows_.iter().map(|r| r.1).collect(), + ], + ) + } + + fn rows(relation: &Relation) -> Vec> { + (0..relation.len()).map(|i| relation.row(i)).collect() + } + + /// Step 1: a source becomes a stored atom, a variable becomes a + /// derived relation, and an output names what to read. The smallest + /// possible translation: one rule that copies the source. + #[test] + fn s01_source_becomes_a_stored_atom() { + let plan = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + out("edges"), + ]) + .unwrap(); + + assert_eq!(plan.sources["edge"], vec!["from", "to"]); + assert_eq!(plan.outputs["edges"], "edges"); + assert_eq!(plan.program.rules.len(), 1); + let rule = &plan.program.rules[0]; + assert_eq!(rule.head.relation, "edges"); + assert_eq!(rule.body.len(), 1); + assert_eq!(rule.body[0].relation, "edge"); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(0, 1), (1, 2)])); + edb + }); + assert_eq!( + rows(result.get("edges").unwrap()), + vec![vec![0, 1], vec![1, 2]] + ); + } + + /// Step 2: a projection is just the rule head. Selecting, reordering + /// and renaming columns costs nothing at runtime. + #[test] + fn s02_projection_is_the_rule_head() { + let plan = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + let_rel( + "swapped", + ProjectionExpr { + relation: var("edges"), + attributes: named_columns(&[("target", "to"), ("origin", "from")]), + }, + ), + out("swapped"), + ]) + .unwrap(); + + assert_eq!(plan.schemas["swapped"], vec!["target", "origin"]); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(0, 1)])); + edb + }); + assert_eq!(rows(result.get("swapped").unwrap()), vec![vec![1, 0]]); + } + + /// Step 3: a projection can also pin a literal column. It lands as a + /// literal directly in the rule head. + #[test] + fn s03_projection_can_pin_a_literal() { + let plan = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + let_rel( + "tagged", + ProjectionExpr { + relation: var("edges"), + attributes: vec![ + ("from".to_string(), var("from")), + ("tag".to_string(), lit(7)), + ], + }, + ), + out("tagged"), + ]) + .unwrap(); + + let head = &plan.program.rules[1].head; + assert_eq!(head.terms[1], Term::Lit(7)); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(3, 4)])); + edb + }); + assert_eq!(rows(result.get("tagged").unwrap()), vec![vec![3, 7]]); + } + + /// Step 4: distinct is free. coln-batch results are sets already, so + /// the operator lowers to nothing at all. + #[test] + fn s04_distinct_is_free() { + let plain = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + out("edges"), + ]) + .unwrap(); + let deduped = lower_plan(vec![ + let_rel( + "edges", + DistinctExpr { + relation: Expr::from(SourceExpr::new(src("edge"))), + }, + ), + out("edges"), + ]) + .unwrap(); + + assert_eq!(plain.program.rules.len(), deduped.program.rules.len()); + assert_eq!( + plain.program.rules[0].body.len(), + deduped.program.rules[0].body.len() + ); + } + + /// Step 5: an equality selection pins a column to a literal inside + /// the atom. The executor then only ever sees matching rows; there is + /// no separate filter step. + #[test] + fn s05_selection_pins_a_column() { + let plan = lower_plan(vec![ + let_rel( + "from_three", + SelectionExpr { + relation: Expr::from(SourceExpr::new(src("edge"))), + condition: Expr::from(BinaryExpr { + operator: Operator::Equal, + left: var("from"), + right: lit(3), + }), + }, + ), + out("from_three"), + ]) + .unwrap(); + + let body_atom = &plan.program.rules[0].body[0]; + assert_eq!(body_atom.terms[0], Term::Lit(3)); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(3, 4), (5, 6)])); + edb + }); + assert_eq!(rows(result.get("from_three").unwrap()), vec![vec![3, 4]]); + } + + /// Step 6: an equi join is a shared variable. Both operands become + /// atoms in the same body, and every `on` pair merges one variable of + /// each side into one. That is the whole translation of a join. + #[test] + fn s06_equi_join_is_a_shared_variable() { + let plan = lower_plan(vec![ + let_rel( + "joined", + EquiJoinExpr { + left: Expr::from(SourceExpr::new(src("r"))), + right: Expr::from(SourceExpr::new(src("s"))), + on: vec![(var("b"), var("c"))], + attributes: None, + }, + ), + out("joined"), + ]) + .unwrap(); + + let rule = &plan.program.rules[0]; + assert_eq!(rule.body.len(), 2); + // The join column is the same variable in both atoms. + assert_eq!(rule.body[0].terms[1], rule.body[1].terms[0]); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("r", ["a", "b"], &[(1, 10), (2, 20)])); + edb.insert(rel("s", ["c", "d"], &[(10, 100), (30, 300)])); + edb + }); + assert_eq!( + rows(result.get("joined").unwrap()), + vec![vec![1, 10, 10, 100]] + ); + } + + /// Step 7: chained joins flatten into one n-ary body. The nested + /// logical tree becomes a single rule with three atoms, and the + /// executor picks its own join order. This is the agreed stopgap for + /// the missing n-ary join node in the shared plan. + #[test] + fn s07_join_chains_flatten_into_one_body() { + let plan = lower_plan(vec![ + let_rel("e", SourceExpr::new(src("edge"))), + let_rel( + "three_hops", + EquiJoinExpr { + left: Expr::from(EquiJoinExpr { + left: Expr::from(AliasExpr { + relation: var("e"), + alias: "h1".to_string(), + }), + right: Expr::from(AliasExpr { + relation: var("e"), + alias: "h2".to_string(), + }), + on: vec![(var("to"), var("from"))], + attributes: Some(named_columns(&[("start", "h1.from"), ("mid", "h2.to")])), + }), + right: Expr::from(AliasExpr { + relation: var("e"), + alias: "h3".to_string(), + }), + on: vec![(var("mid"), var("from"))], + attributes: Some(named_columns(&[("start", "start"), ("end", "h3.to")])), + }, + ), + out("three_hops"), + ]) + .unwrap(); + + // One rule, three atoms: the tree flattened. + let rule = &plan.program.rules[1]; + assert_eq!(rule.body.len(), 3); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(0, 1), (1, 2), (2, 3)])); + edb + }); + assert_eq!(rows(result.get("three_hops").unwrap()), vec![vec![0, 3]]); + } + + /// Step 8: a cartesian product is the same translation with an empty + /// `on` list: two atoms that simply share no variable. + #[test] + fn s08_cartesian_product_shares_no_variable() { + let plan = lower_plan(vec![ + let_rel( + "pairs", + CartesianProductExpr::new( + Expr::from(SourceExpr::new(src("r"))), + Expr::from(SourceExpr::new(src("s"))), + None, + ), + ), + out("pairs"), + ]) + .unwrap(); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("r", ["a", "b"], &[(1, 1), (2, 2)])); + edb.insert(rel("s", ["c", "d"], &[(9, 9)])); + edb + }); + assert_eq!(result.get("pairs").unwrap().len(), 2); + } + + /// Step 9: the n-ary join FLIR emits, consumed natively. One + /// MultiWayEquiJoinExpr lowers to a single rule body with one + /// variable per equality class, no flattening involved. The cyclic + /// triangle is the shape that needs this: every pair of atoms + /// shares a variable, and the executor must see all three atoms at + /// once. + #[test] + fn s09_multi_way_join_is_native() { + let triangle = MultiWayEquiJoinExpr::new( + vec![var("e"), var("e"), var("e")], + vec![ + join_variable("x", &[(0, "from"), (2, "to")]), + join_variable("y", &[(0, "to"), (1, "from")]), + join_variable("z", &[(1, "to"), (2, "from")]), + ], + None, + ) + .unwrap(); + let plan = lower_plan(vec![ + let_rel("e", SourceExpr::new(src("edge"))), + let_rel("triangles", triangle), + out("triangles"), + ]) + .unwrap(); + + // One rule, three atoms over the derived edge relation, and the + // documented schema fold: every join variable appears exactly + // once, under its variable name. + let rule = &plan.program.rules[1]; + assert_eq!(rule.body.len(), 3); + assert_eq!(plan.schemas["triangles"], vec!["x", "y", "z"]); + + // Edges 1→2→3→1 close a triangle; 1→4 is a dead end. The three + // result rows are the rotations of the one triangle. + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel( + "edge", + ["from", "to"], + &[(1, 2), (2, 3), (3, 1), (1, 4)], + )); + edb + }); + assert_eq!( + rows(result.get("triangles").unwrap()), + vec![vec![1, 2, 3], vec![2, 3, 1], vec![3, 1, 2]] + ); + } + + /// Step 10: aliases only affect name resolution. `h1.from` and a plain + /// `from` resolve to the same variable; nothing changes in the rule. + #[test] + fn s10_aliases_are_pure_name_resolution() { + let plan = lower_plan(vec![ + let_rel( + "joined", + EquiJoinExpr { + left: Expr::from(AliasExpr { + relation: Expr::from(SourceExpr::new(src("r"))), + alias: "left".to_string(), + }), + right: Expr::from(AliasExpr { + relation: Expr::from(SourceExpr::new(src("s"))), + alias: "right".to_string(), + }), + on: vec![(var("b"), var("c"))], + attributes: Some(named_columns(&[("a", "left.a"), ("d", "right.d")])), + }, + ), + out("joined"), + ]) + .unwrap(); + + assert_eq!(plan.schemas["joined"], vec!["a", "d"]); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("r", ["a", "b"], &[(1, 10)])); + edb.insert(rel("s", ["c", "d"], &[(10, 100)])); + edb + }); + assert_eq!(rows(result.get("joined").unwrap()), vec![vec![1, 100]]); + } + + /// Step 11: referencing a bound variable places an atom over the + /// derived relation, exactly like referencing a source. The chain of + /// statements becomes a chain of rules, and the fixpoint evaluation + /// orders them by data dependency on its own. + #[test] + fn s11_bound_variables_become_derived_atoms() { + let plan = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + let_rel( + "targets", + ProjectionExpr { + relation: var("edges"), + attributes: named_columns(&[("node", "to")]), + }, + ), + out("targets"), + ]) + .unwrap(); + + // The second rule's body reads the *derived* relation "edges". + assert_eq!(plan.program.rules[1].body[0].relation, "edges"); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(0, 1), (1, 2)])); + edb + }); + assert_eq!(rows(result.get("targets").unwrap()), vec![vec![1], vec![2]]); + } + + /// Step 12: a union is one rule per branch with the same head. Set + /// semantics deduplicate overlaps for free. + #[test] + fn s12_union_is_one_rule_per_branch() { + let plan = lower_plan(vec![ + let_rel("left", SourceExpr::new(src("r"))), + let_rel("right", SourceExpr::new(src("s"))), + let_rel( + "both", + UnionExpr { + relations: vec![var("left"), var("right")], + }, + ), + out("both"), + ]) + .unwrap(); + + let heads: Vec<&str> = plan + .program + .rules + .iter() + .filter(|rule| rule.head.relation == "both") + .map(|rule| rule.body[0].relation.as_str()) + .collect(); + assert_eq!(heads, vec!["left", "right"]); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("r", ["a", "b"], &[(1, 1), (2, 2)])); + edb.insert(rel("s", ["c", "d"], &[(2, 2), (3, 3)])); + edb + }); + assert_eq!(result.get("both").unwrap().len(), 3); + } + + /// Step 13: the payoff. A fixed point lowers to a base rule plus a + /// recursive step rule, and coln-batch's semi-naive evaluation + /// computes the closure. This is the arithmetic-free shape of Leo's + /// `test_iteration`, translated end to end. + #[test] + fn s13_fixed_point_reaches_the_closure() { + let plan = lower_plan(vec![ + let_rel("edges", SourceExpr::new(src("edge"))), + let_rel( + "closure", + FixedPointIterExpr { + accumulator: ("cur".to_string(), var("edges")), + step: BlockStmt { + stmts: vec![Stmt::from(ExprStmt { + expr: Expr::from(EquiJoinExpr { + left: Expr::from(AliasExpr { + relation: var("cur"), + alias: "walk".to_string(), + }), + right: Expr::from(AliasExpr { + relation: var("edges"), + alias: "step".to_string(), + }), + on: vec![(var("to"), var("from"))], + attributes: Some(named_columns(&[ + ("from", "walk.from"), + ("to", "step.to"), + ])), + }), + })], + }, + }, + ), + out("closure"), + ]) + .unwrap(); + + // A base rule and a recursive rule, both defining "closure"; the + // step rule reads "closure" in its own body. + let closure_rules: Vec<_> = plan + .program + .rules + .iter() + .filter(|rule| rule.head.relation == "closure") + .collect(); + assert_eq!(closure_rules.len(), 2); + assert!( + closure_rules[1] + .body + .iter() + .any(|atom| atom.relation == "closure") + ); + + let result = run(&plan, { + let mut edb = Catalog::new(); + edb.insert(rel("edge", ["from", "to"], &[(0, 1), (1, 2), (2, 3)])); + edb + }); + assert_eq!( + rows(result.get("closure").unwrap()), + vec![ + vec![0, 1], + vec![0, 2], + vec![0, 3], + vec![1, 2], + vec![1, 3], + vec![2, 3], + ] + ); + } + + /// Step 14: everything outside this slice fails loudly with a clear + /// message instead of producing a wrong answer. + #[test] + fn s14_unsupported_features_fail_loudly() { + let err = lower_plan(vec![let_rel( + "anti", + Expr::from(AntiJoinExpr { + left: Expr::from(SourceExpr::new(src("r"))), + right: Expr::from(SourceExpr::new(src("s"))), + on: vec![(var("a"), var("c"))], + }), + )]) + .unwrap_err(); + assert!(err.to_string().contains("AntiJoin")); + + let err = lower_plan(vec![let_rel( + "computed", + ProjectionExpr { + relation: Expr::from(SourceExpr::new(src("r"))), + attributes: vec![( + "sum".to_string(), + Expr::from(BinaryExpr { + operator: Operator::Addition, + left: var("a"), + right: var("b"), + }), + )], + }, + )]) + .unwrap_err(); + assert!(format!("{err:#}").contains("scalar engine")); + } +} diff --git a/packages/coln-query/src/relational/batch/mod.rs b/packages/coln-query/src/relational/batch/mod.rs index 6cfb8e03..b2e4d3de 100644 --- a/packages/coln-query/src/relational/batch/mod.rs +++ b/packages/coln-query/src/relational/batch/mod.rs @@ -3,25 +3,74 @@ // SPDX-License-Identifier: Apache-2.0 OR MIT //! A batch backend optimized for efficient evaluation of non-binary joins. +//! +//! The non-incremental half of the pipeline. [`Backend::build`] lowers the +//! resolved plan into one coln-batch Datalog program, [`Runtime::feed`] +//! stages rows per source, [`Runtime::commit`] recomputes the whole result +//! eagerly (a semi-naive fixpoint over worst-case-optimal joins), and +//! [`Runtime::output`] hands back the full current state of a sink as a +//! [`Snapshot`]. Where the incremental backend reports deltas, this backend +//! reports states. +//! +//! Value scope of this slice: unsigned integers and booleans, mapped onto +//! the engine's u64 universe. +// TODO(Jan): the remaining scalar types (strings first, via dictionary +// encoding) land after the end-to-end slice is complete. +//! +//! # Interim: base tables arrive by push +//! +//! The pipeline has one input door, [`Runtime::feed`], and it speaks +//! deltas: rows with z-weights, pushed by the store transaction by +//! transaction. That is what the incremental backend needs. A batch +//! backend wants the opposite: pull the full snapshot of every base table +//! at query time. That pull API does not exist in the pipeline yet; the +//! store side of it does (`SortedTableSnapshot` in coln-store). +//! +//! Until it lands, this backend integrates the pushed deltas itself: +//! [`Runtime::feed`] keeps a net z-weight per row and +//! `materialize_sources` turns that into the base tables right before +//! every recomputation. This is correct as long as the runtime sees every +//! delta from the start, which holds for what a batch query is today: one +//! feed-commit-output cycle over a snapshot pushed through `feed`. It is a +//! stopgap, not the design. The store already holds these tables, copying +//! them through `feed` is wasted work, and a runtime created later cannot +//! catch up on deltas it never saw. +// TODO(Jan): replace the push-side integration with the pull API once the +// pipeline offers one. `materialize_sources` is the single seam to swap; +// lowering, fixpoint, and output stay as they are. +mod lowering; + +use std::collections::HashMap; use std::num::NonZeroUsize; +use coln_batch::fixpoint::{self, Exec}; +use coln_batch::generic_join; +use coln_batch::query::Catalog as BatchCatalog; +use coln_batch::relation::Relation; +use coln_batch::rule::Program; +use dbsp::{OrdZSet, utils::Tup2}; + +use self::lowering::{LoweredPlan, lower}; use super::{Backend, Runtime}; use crate::{ - api::deltas::ZWeight, + api::deltas::{ZRow, ZWeight}, error::{BuildError, RuntimeError}, host::resolver::ResolvedCode, relational::{ - Snapshot, + catalog::SourceSchemas, expr::{SinkId, SourceId}, relation::TupleValue, }, - scalarial::{ColumnScalarEngine, column::VectorizedScalarEngine}, + scalarial::{ColumnScalarEngine, ScalarTypedValue, column::VectorizedScalarEngine}, }; -/// The non-incremental backend: evaluates the plan eagerly over materialized -/// Z-sets. Bodies are the next slice of work. +/// The non-incremental backend: lowers the plan to a coln-batch Datalog +/// program at build time and recomputes it eagerly on every commit. pub struct BatchBackend { + // Reserved for the scalar slice: computed columns and general + // conditions will run on this engine. + #[allow(dead_code)] scalar_engine: E, } @@ -40,14 +89,142 @@ impl Backend for BatchBackend { fn build( self, _threads: NonZeroUsize, - _plan: ResolvedCode, + plan: ResolvedCode, + sources: SourceSchemas, ) -> Result { - todo!("eager batch backend: build a RelExprVisitor over Z-sets") + let LoweredPlan { + program, + sources: used_sources, + outputs, + schemas, + } = lower(plan.as_code(), &sources) + .map_err(|error| BuildError::new(format!("{error:#}")))?; + let inputs = used_sources + .into_keys() + .map(|source| (source, HashMap::new())) + .collect(); + let sinks = outputs + .keys() + .map(|sink| SinkId::from(sink.as_str())) + .collect(); + Ok(BatchRuntime { + program, + outputs, + schemas, + inputs, + sinks, + results: None, + }) + } +} + +/// Staged source rows plus the compiled program; [`Runtime::commit`] +/// recomputes the full result from the accumulated inputs. +pub struct BatchRuntime { + program: Program, + /// Sink id to the derived relation `output` reads. + outputs: HashMap, + /// Column names per relation, sources and derived alike. + schemas: HashMap>, + /// Per used source: the net z-weight of every row fed so far. This is + /// the interim snapshot store described in the module docs; the pull + /// API replaces it. + inputs: HashMap, ZWeight>>, + sinks: Vec, + /// The relations of the last commit. + results: Option, +} + +/// The full current state of a result relation, the natural output of a +/// batch backend (the incremental backend reports deltas instead). +/// +/// Rows are sorted and deduplicated: results are sets. Values come back as +/// unsigned integers, matching the value slice the backend accepts. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct Snapshot { + columns: Vec, + rows: Vec, +} + +impl Snapshot { + fn from_relation(columns: Vec, relation: &Relation) -> Self { + let rows = (0..relation.len()) + .map(|row| TupleValue { + data: relation + .row(row) + .into_iter() + .map(ScalarTypedValue::Uint) + .collect(), + }) + .collect(); + Self { columns, rows } + } + + /// The result's column names, in tuple order. + pub fn columns(&self) -> &[String] { + &self.columns + } + + /// The rows, sorted and deduplicated. + pub fn rows(&self) -> &[TupleValue] { + &self.rows + } + + pub fn len(&self) -> usize { + self.rows.len() + } + + pub fn is_empty(&self) -> bool { + self.rows.is_empty() + } + + /// The snapshot as a z-set with every weight `+1`, for set-level + /// comparison against the incremental backend's consolidated output. + pub fn to_debug_zset(&self) -> OrdZSet { + let keys = self + .rows + .iter() + .map(|row| Tup2(row.clone(), 1)) + .collect::>(); + OrdZSet::from_keys((), keys) } } -/// Accumulated source Z-sets + the compiled plan; `commit` recomputes the result. -pub struct BatchRuntime; +impl BatchRuntime { + /// Build the base tables for one recomputation from the deltas fed so + /// far: a row is present when its net z-weight is positive. + /// + /// Interim, see the module docs. This is the one seam where the pull + /// API hooks in later: read the store's sorted snapshots instead of + /// integrating pushed deltas. Nothing downstream needs to change. + // TODO(Jan): swap for the pull API once the pipeline offers one. + fn materialize_sources(&self) -> Result { + let mut edb = BatchCatalog::new(); + for (source, staged) in &self.inputs { + let columns = self.schemas.get(source).cloned().unwrap_or_default(); + let mut data: Vec> = vec![Vec::new(); columns.len()]; + for (row, weight) in staged { + match weight { + weight if *weight < 0 => { + return Err(RuntimeError::new(format!( + "source '{source}': a row was deleted more often than inserted \ + (net weight {weight})" + ))); + } + 0 => {} + // Sets: duplicated insertions collapse into one row. + _ => { + for (column, value) in data.iter_mut().zip(row) { + column.push(*value); + } + } + } + } + edb.insert(Relation::new(source.clone(), columns, data)); + } + Ok(edb) + } +} impl Runtime for BatchRuntime { type Output = Snapshot; @@ -55,17 +232,82 @@ impl Runtime for BatchRuntime { fn feed( &mut self, - _source: &SourceId, - _rows: impl IntoIterator, - ) -> Result<(), Self::Error> { - todo!("eager batch feed: accumulate into source tables") + source: &SourceId, + rows: impl IntoIterator, + ) -> Result { + // Mirrors the incremental backend: a source the plan does not use + // is `Ok(false)`, not an error. The caller decides what that means. + let arity = self.schemas.get(source.as_str()).map_or(0, Vec::len); + let Some(staged) = self.inputs.get_mut(source.as_str()) else { + return Ok(false); + }; + for zrow in rows { + let weight = zrow.zweight(); + let row = convert_row(source, &zrow.into_row(), arity)?; + *staged.entry(row).or_insert(0) += weight; + } + Ok(true) } fn commit(&mut self) -> Result<(), Self::Error> { - todo!("eager batch commit: recompute the plan from accumulated inputs") + let edb = self.materialize_sources()?; + let result = fixpoint::semi_naive(&self.program, &edb, generic_join::execute as Exec) + .map_err(|error| RuntimeError::new(format!("{error:#}")))?; + self.results = Some(result.catalog); + Ok(()) + } + + fn output(&self, out: &SinkId) -> Result { + let Some(results) = &self.results else { + return Err(RuntimeError::new("commit before reading an output")); + }; + let Some(relation) = self.outputs.get(out.as_str()) else { + return Err(RuntimeError::new(format!( + "unknown output '{}' (print-only CLI taps are not readable)", + out.as_str() + ))); + }; + // The engine names rule variables generically (v0, v1, …); the + // plan's speaking column names live in the schema table. + let columns = self.schemas.get(relation).cloned().unwrap_or_default(); + let relation = results + .get(relation) + .map_err(|error| RuntimeError::new(format!("{error:#}")))?; + Ok(Snapshot::from_relation(columns, relation)) } - fn output(&self, _out: &SinkId) -> Result { - todo!("eager batch output: read the materialized result") + fn list_outputs(&self) -> impl Iterator { + self.sinks.iter() + } +} + +/// Convert one staged row into the engine's u64 universe. +/// +/// This slice accepts unsigned integers and booleans; everything else +/// fails loudly instead of computing something wrong. +// TODO(Jan): remaining scalar types after the end-to-end slice. +fn convert_row( + source: &SourceId, + row: &TupleValue, + arity: usize, +) -> Result, RuntimeError> { + if row.data.len() != arity { + return Err(RuntimeError::new(format!( + "row for source '{}' has {} values, its schema has {arity} columns", + source.as_str(), + row.data.len() + ))); } + row.data + .iter() + .map(|value| match value { + ScalarTypedValue::Uint(value) => Ok(*value), + ScalarTypedValue::Bool(value) => Ok(u64::from(*value)), + other => Err(RuntimeError::new(format!( + "source '{}': the batch backend supports unsigned integer and boolean \ + values for now, got {other:?}", + source.as_str() + ))), + }) + .collect() } diff --git a/packages/coln-query/src/relational/catalog.rs b/packages/coln-query/src/relational/catalog.rs new file mode 100644 index 00000000..f274551f --- /dev/null +++ b/packages/coln-query/src/relational/catalog.rs @@ -0,0 +1,100 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! What a plan's source leaves *mean*. +//! +//! A [`SourceExpr`](crate::relational::expr::SourceExpr) leaf only names an +//! extensional relation. Everything else about that relation — today its +//! [`TableSchema`], tomorrow perhaps cardinality estimates a cost-based +//! optimizer would want — is answered by the [`Catalog`] the plan is compiled +//! against, so that a relation referenced `N` times is still described once. +//! +//! What a catalog answers with is the *neutral* schema: columns, types, and the +//! table's key(s). Turning that into the keyed, positional layout a particular +//! runtime needs is the backend's job (see +//! [`StreamSchema`](crate::relational::incremental::schema::StreamSchema) +//! for the DBSP one). + +use crate::{ + error::BuildError, + host::{ + stmt::Stmt, + walk::{Node, pre_order}, + }, + relational::{expr::SourceId, schema::TableSchema}, +}; +use std::borrow::Cow; +use std::collections::HashMap; + +/// The static description of a plan's extensional inputs, keyed by the +/// [`SourceId`] its leaves carry. +/// +/// Read-only and longer-lived than the plan itself: the optimizer, the lowering +/// and the resolver all rewrite the code, while the catalog they consult stays +/// as it is — which is why this is a trait of its own rather than a method on +/// [`QueryProgram`](crate::program::QueryProgram), whose code gets moved out +/// from under it. A pass is handed the `&dyn Catalog` half and so cannot touch +/// the code half. +pub trait Catalog { + /// The schema of the relation `id` names, or [`None`] if this catalog does + /// not describe it. + /// + /// Returns a [`Cow`] so that an implementation is free to keep a *richer* + /// per-relation description of its own, e.g., coln's FLIR frontend stores a + /// `BaseTableSchema`, with one column view per engine and the index + /// translations between them, and project it down on demand + /// ([`Cow::Owned`]), instead of being forced to store a parallel + /// [`TableSchema`] just to have one to lend out. An implementation that + /// does hold one lends it ([`Cow::Borrowed`]) and allocates nothing. + fn source_schema(&self, id: &SourceId) -> Option>; +} + +/// Every source *one plan* names, with the schema its catalog describes it by: +/// the resolved projection of a [`Catalog`] onto that plan. One entry per +/// *distinct* [`SourceId`], however many leaves reference it. +/// +/// This is what a [`Backend`](crate::relational::Backend) is handed, rather than +/// the catalog itself — see [`resolve_sources`] for why resolving up front is +/// what makes an incremental circuit buildable at all. +pub type SourceSchemas = HashMap; + +/// A resolved projection answers the same questions the [`Catalog`] it came from +/// does, so a consumer that only wants to look a source up — the type resolver, +/// the tree printer — takes a `&dyn Catalog` and can be handed either. One +/// vocabulary, whether the schemas are still to be computed or already resolved. +impl Catalog for SourceSchemas { + fn source_schema(&self, id: &SourceId) -> Option> { + self.get(id).map(Cow::Borrowed) + } +} + +/// Resolve every [`SourceExpr`](crate::relational::expr::SourceExpr) leaf in +/// `code` against `catalog`, up front. The single point at which a [`Catalog`] +/// is consulted: [`Pipeline::runtime`](crate::pipeline::Pipeline::runtime) calls +/// this once, and every stage downstream works from the [`SourceSchemas`] it +/// returns. +/// +/// This is where a plan naming a relation the catalog knows nothing about is +/// caught and covers the one failure mode that name-only leaves introduce. +/// Hence, it fails before a backend has built anything, naming the offending +/// source rather than lazily discovering as a missing input later on. +/// +/// Resolving eagerly is also what lets an incremental backend exist: DBSP's +/// `init_circuit` constructor must be `Send + 'static` (it runs once per worker +/// thread), so a borrowed catalog cannot cross into it, while owned schemas can. +pub fn resolve_sources(code: &[Stmt], catalog: &dyn Catalog) -> Result { + pre_order(code) + .filter_map(Node::as_source) + .map(|source| { + let schema = catalog.source_schema(source.as_id()).ok_or_else(|| { + BuildError::new(format!( + "Source '{}' is not described by the catalog this plan is compiled \ + against, so there is nothing to bind it to", + source.as_id() + )) + })?; + Ok((source.as_id().clone(), schema.into_owned())) + }) + .collect() +} diff --git a/packages/coln-query/src/relational/expr.rs b/packages/coln-query/src/relational/expr.rs index a36de295..fc6f7cdd 100644 --- a/packages/coln-query/src/relational/expr.rs +++ b/packages/coln-query/src/relational/expr.rs @@ -11,10 +11,10 @@ //! preserves relation-valued variables, nested operators, and tuple-of-relations. use crate::{ + error::SyntaxError, host::{expr::Expr, stmt::BlockStmt}, - relational::RelationSchema, - util::MemAddr, }; +use std::collections::HashSet; /// Relational-algebra operator = backend-neutral query-plan vocabulary. /// @@ -29,12 +29,16 @@ pub enum RelExpr { Output(Box), Alias(Box), Distinct(Box), + // A union can also be expressed with a full outer join and a projection. Union(Box), + // As the antijoin is a generalization of the set difference, this may be + // removed in the future. Difference(Box), Selection(Box), Projection(Box), CartesianProduct(Box), EquiJoin(Box), + MultiWayEquiJoin(Box), AntiJoin(Box), FixedPointIter(Box), } @@ -43,6 +47,12 @@ pub enum RelExpr { /// composed `From for Expr` (via [`Expr::Relational`]) so that /// constructing a host expression from a relational operator stays a single /// `Expr::from(..)`/`.into()` call, exactly as before the host/relational split. +/// +/// Each also comes in a `Box` flavour, which reuses the allocation the +/// caller already holds. That is what an owned rewriting pass rebuilds an +/// untouched node with — see [`RelExprVisitorOwn`] — and since +/// [`Expr::Relational`] does not box what it wraps, that route allocates +/// nothing at all. macro_rules! impl_rel_and_expr_from { ($(($variant:path, $expr:ty)),* $(,)?) => { $( @@ -51,9 +61,19 @@ macro_rules! impl_rel_and_expr_from { $variant(Box::new(value)) } } + impl From> for RelExpr { + fn from(value: Box<$expr>) -> Self { + $variant(value) + } + } impl From<$expr> for Expr { fn from(value: $expr) -> Self { - Expr::Relational(Box::new(RelExpr::from(value))) + Expr::Relational(RelExpr::from(value)) + } + } + impl From> for Expr { + fn from(value: Box<$expr>) -> Self { + Expr::Relational(RelExpr::from(value)) } } )* @@ -71,15 +91,82 @@ impl_rel_and_expr_from! { (RelExpr::Projection, ProjectionExpr), (RelExpr::CartesianProduct, CartesianProductExpr), (RelExpr::EquiJoin, EquiJoinExpr), + (RelExpr::MultiWayEquiJoin, MultiWayEquiJoinExpr), (RelExpr::AntiJoin, AntiJoinExpr), (RelExpr::FixedPointIter, FixedPointIterExpr), } /// The single bridge from the relational layer back into the host layer: a -/// relational operator is *also* a host expression. +/// relational operator is *also* a host expression. Free of charge, since +/// [`Expr::Relational`] does not box what it wraps. impl From for Expr { fn from(value: RelExpr) -> Self { - Expr::Relational(Box::new(value)) + Expr::Relational(value) + } +} + +/// [`RelExpr`] without the payloads: which *kind* of operator a node is. +/// +/// This is the vocabulary a rewriting rule declares its interest in, so a +/// driver can skip offering it nodes it could never fire on — see +/// [`TransformationRule::interest`](crate::optimizer::rewrite::TransformationRule::interest). +/// A structural precondition of that shape is worth stating separately from the +/// rewrite itself; a *semantic* one is not, because checking it means taking +/// the node apart, which the rewrite then has to do again. +#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)] +pub enum RelKind { + Source, + Output, + Alias, + Distinct, + Union, + Difference, + Selection, + Projection, + CartesianProduct, + EquiJoin, + MultiWayEquiJoin, + AntiJoin, + FixedPointIter, +} + +impl RelKind { + /// Every kind, for a rule that has to see the whole plan. + pub const ALL: &'static [RelKind] = &[ + RelKind::Source, + RelKind::Output, + RelKind::Alias, + RelKind::Distinct, + RelKind::Union, + RelKind::Difference, + RelKind::Selection, + RelKind::Projection, + RelKind::CartesianProduct, + RelKind::EquiJoin, + RelKind::MultiWayEquiJoin, + RelKind::AntiJoin, + RelKind::FixedPointIter, + ]; +} + +impl RelExpr { + /// Which operator this node is, without looking at its operands. + pub fn kind(&self) -> RelKind { + match self { + RelExpr::Source(_) => RelKind::Source, + RelExpr::Output(_) => RelKind::Output, + RelExpr::Alias(_) => RelKind::Alias, + RelExpr::Distinct(_) => RelKind::Distinct, + RelExpr::Union(_) => RelKind::Union, + RelExpr::Difference(_) => RelKind::Difference, + RelExpr::Selection(_) => RelKind::Selection, + RelExpr::Projection(_) => RelKind::Projection, + RelExpr::CartesianProduct(_) => RelKind::CartesianProduct, + RelExpr::EquiJoin(_) => RelKind::EquiJoin, + RelExpr::MultiWayEquiJoin(_) => RelKind::MultiWayEquiJoin, + RelExpr::AntiJoin(_) => RelKind::AntiJoin, + RelExpr::FixedPointIter(_) => RelKind::FixedPointIter, + } } } @@ -95,42 +182,75 @@ impl SourceId { } } -impl> From for SourceId { - fn from(value: T) -> Self { - Self(value.into()) +impl AsRef for SourceId { + fn as_ref(&self) -> &str { + &self.0 + } +} + +impl From for SourceId { + fn from(value: String) -> Self { + Self(value) } } -/// Backend-neutral relation leaf. Carries a schema but no stream and no table. -/// Its identity is the schema name, so it is derived rather than stored. +impl From<&str> for SourceId { + fn from(value: &str) -> Self { + Self(value.to_string()) + } +} + +/// Sources are named in diagnostics and in rendered plans, so the id renders as +/// the bare name it is. +impl std::fmt::Display for SourceId { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} + +/// So a `HashMap` can be probed with a `&str`, as one keyed by +/// `String` could. Without it every lookup would have to mint an owned +/// [`SourceId`] first. +impl std::borrow::Borrow for SourceId { + fn borrow(&self) -> &str { + &self.0 + } +} + +/// Backend-neutral relation leaf: it *names* an extensional relation and carries +/// nothing else — no schema, no stream, no table. What the name means is answered +/// by the [`Catalog`](crate::relational::catalog::Catalog) the plan is compiled +/// against, so a relation the plan references `N` times is described once instead +/// of `N` times. +/// +/// Naming rather than describing is also what makes the derived [`PartialEq`] +/// mean what it reads as. A backend's physical schema +/// (say a [`StreamSchema`](crate::relational::incremental::schema::StreamSchema)) +/// compares only its key and tuple, deliberately ignoring its `name` (a +/// transformation trace rather than an identity) — so back when this leaf held a +/// schema, two leaves naming *different* relations compared equal whenever their +/// shapes matched. #[derive(Clone, Debug, Eq, PartialEq)] pub struct SourceExpr { - pub schema: RelationSchema, + pub id: SourceId, } impl SourceExpr { - /// Build a source leaf from a schema. This is how a (circuit-free) plan names - /// an extensional input; the backend later binds its [`id`](Self::to_id) to a - /// concrete relation. - pub fn new(schema: RelationSchema) -> Self { - Self { schema } + /// Build a source leaf naming an extensional input. The backend binds the + /// [`SourceId`] to a concrete relation at execution time, and the plan's + /// [`Catalog`](crate::relational::catalog::Catalog) answers what it is. + pub fn new(id: impl Into) -> Self { + Self { id: id.into() } } - pub fn as_id(&self) -> &str { - &self.schema.name - } - - /// The extensional input this leaf names. Derived from the schema name, so it - /// can never disagree with the schema. - pub fn to_id(&self) -> String { - self.schema.name.clone() + pub fn as_id(&self) -> &SourceId { + &self.id } } /// Backend-neutral identity of a query output. Mirrors [`SourceId`] on the input /// side: the plan only ever *names* a sink; the backend maps the name to a live -/// destination (a read handle, a CLI printer, …) at execution time. Replaces the -/// old positional `OutputId` so outputs are addressed by name. +/// destination (a read handle, a CLI printer, …) at execution time. #[derive(Clone, Debug, Eq, PartialEq, Hash)] pub struct SinkId(pub String); @@ -146,6 +266,12 @@ impl> From for SinkId { } } +impl std::fmt::Display for SinkId { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} + /// Where an [`OutputExpr`] sends the rows it taps. Pure data — the backend binds /// each variant to a concrete destination; nothing runtime-stateful lives in the /// plan. @@ -250,14 +376,170 @@ pub struct EquiJoinExpr { pub left: Expr, /// Must evaluate to a relation. pub right: Expr, - /// The attributes to join on. The first element of any pair belongs to the - /// left relation, and the second element of any pair belongs to right relation. - /// Each attribute pair should produce the same type. + /// The attribute(s) to join on. The first element of any pair is evaluated + /// in the context of the left relation, and the second element of any pair + /// is evaluated in the context of the right relation. + /// + /// If `on` is empty, a [`CartesianProduct`](CartesianProductExpr) is computed. pub on: Vec<(Expr, Expr)>, /// An optional projection step. See documentation of [`ProjectionExpr`]. pub attributes: Option>, } +/// The position of a relation within [`MultiWayEquiJoinExpr::relations`]. +pub type RelationIdx = usize; + +/// One equality class of a [`MultiWayEquiJoinExpr`]: every listed occurrence +/// must produce the same value for a tuple to enter the join's output. +/// +/// A variable bound by only *one* relation is deliberately not representable +/// here: It constrains nothing, so it is not part of a join condition. Such a +/// variable still reaches the output, carried by its relation's schema like any +/// other non-join attribute. Keeping them out is what makes +/// [`MultiWayEquiJoinExpr::on`]`.is_empty()` an exact test for "nothing to join +/// on". +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct JoinVariable { + /// The name the joined attribute carries in the output schema. + /// + /// The lowering from coln's FLIR projects every atom onto the names of the + /// variables it binds, so there the occurrences are plain column picks that + /// already agree on this name, and the schema fold described on + /// [`MultiWayEquiJoinExpr::on`] keeps exactly one active copy of it. When + /// the occurrences do *not* agree on a name (`l.a = r.b`), producing this + /// name is the job of whoever lowers the join. + pub name: String, + /// Which relations bind this variable, and how: the [`RelationIdx`] indexes + /// into [`MultiWayEquiJoinExpr::relations`], and the [`Expr`] is evaluated + /// in the context of that relation. + /// + /// Invariants, enforced by [`MultiWayEquiJoinExpr::new`]: at least two + /// occurrences, every index in bounds, indices pairwise distinct, and + /// ordered by index. + pub occurrences: Vec<(RelationIdx, Expr)>, +} + +/// An equijoin involving `N >= 2` relations. A better input than a folded +/// sequence of [binary `EquiJoin`s](EquiJoinExpr) for worst-case optimal join +/// algorithms (such as the leapfrog triejoin), which are variable-oriented: +/// they iterate a variable ordering, which is what [`on`](Self::on) spells out. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct MultiWayEquiJoinExpr { + /// The `N >= 2` relations which participate in the join. Each [`Expr`] must + /// evaluate to a relation. + pub relations: Vec, + /// The join condition, as one [`JoinVariable`] per equality class of + /// attributes that have to agree. + /// + /// If `on` is empty, a [`CartesianProduct`](CartesianProductExpr) over + /// [`relations`](Self::relations) is computed. Since a variable bound by a + /// single relation cannot be a [`JoinVariable`], that test is exact rather + /// than approximate. + /// + /// **Output schema.** Joining folds + /// [`StreamSchema::join`](crate::relational::incremental::schema::StreamSchema::join) + /// left to right, which deactivates an attribute of a later relation when + /// an earlier one already contributes an active attribute of the same name. + /// A join variable whose occurrences agree on their name therefore appears + /// **once** in the output, carried by the first relation that binds it. + /// No de-duplicating projection is required, and no join column is silently + /// duplicated. + pub on: Vec, + /// An optional projection step. See documentation of [`ProjectionExpr`]. + pub attributes: Option>, +} + +impl MultiWayEquiJoinExpr { + /// The only constructor that cannot produce a malformed join: it normalizes + /// each [`JoinVariable`]'s occurrences into relation order and then applies + /// [`validate`](Self::validate). + pub fn new( + relations: Vec, + on: Vec, + attributes: Option>, + ) -> Result { + let mut joined = Self { + relations, + on, + attributes, + }; + for variable in &mut joined.on { + variable.occurrences.sort_by_key(|(relation, _)| *relation); + } + joined.validate()?; + Ok(joined) + } + + /// Checks the invariants documented on [`Self::relations`] and + /// [`JoinVariable::occurrences`]. [`Self::new`] applies this to everything + /// it builds; the resolver re-applies it because the fields are public and + /// a plan may also be assembled or rewritten by hand. + pub fn validate(&self) -> Result<(), SyntaxError> { + if self.relations.len() < 2 { + return Err(SyntaxError::new(format!( + "A multi way equi join requires at least two relations, got {}", + self.relations.len() + ))); + } + let mut names = HashSet::with_capacity(self.on.len()); + for variable in &self.on { + if !names.insert(&variable.name) { + return Err(SyntaxError::new(format!( + "Join variable '{}' is declared twice", + variable.name + ))); + } + if variable.occurrences.len() < 2 { + return Err(SyntaxError::new(format!( + "Join variable '{}' has {} occurrence(s): below two it constrains \ + nothing, and a variable bound by a single relation reaches the \ + output through that relation's schema instead", + variable.name, + variable.occurrences.len() + ))); + } + let mut relations = HashSet::with_capacity(variable.occurrences.len()); + for (relation, _) in &variable.occurrences { + if *relation >= self.relations.len() { + return Err(SyntaxError::new(format!( + "Join variable '{}' refers to relation {relation} but the join \ + has only {} relations", + variable.name, + self.relations.len() + ))); + } + if !relations.insert(relation) { + return Err(SyntaxError::new(format!( + "Join variable '{}' occurs twice in relation {relation}: a \ + variable repeated within one relation is a local equality \ + condition on that relation, not a join condition", + variable.name + ))); + } + } + } + Ok(()) + } + + /// Every [`Expr`] nested in the join condition, in [`on`](Self::on) order. + /// Each one is evaluated in the context of *its own* relation, so a consumer + /// that needs to know which relation must iterate [`on`](Self::on) directly. + pub fn on_exprs(&self) -> impl Iterator { + self.on + .iter() + .flat_map(|variable| variable.occurrences.iter().map(|(_, expr)| expr)) + } + + /// The [`on_exprs`](Self::on_exprs) counterpart for rewriting passes. + /// Handing out `&mut Expr` cannot break any invariant, as those constrain + /// the arity and the relation indices rather than the expressions. + pub fn on_exprs_mut(&mut self) -> impl Iterator { + self.on + .iter_mut() + .flat_map(|variable| variable.occurrences.iter_mut().map(|(_, expr)| expr)) + } +} + /// This is not a commutative operation, that is, swapping the `left` and `right` /// relations may alter the result. This computes `left` setminus `right` while /// only considering the columns specified in `on`. @@ -267,14 +549,22 @@ pub struct AntiJoinExpr { pub left: Expr, /// Must evaluate to a relation. pub right: Expr, - /// The attributes to _not_ join on. The first element of any pair belongs to the - /// left relation, and the second element of any pair belongs to right relation. - /// Each attribute pair should produce the same type. + /// The attributes the two relations are compared on: a `left` row is + /// suppressed exactly when some `right` row agrees with it on all of them. + /// The first element of any pair is evaluated in the context of the left + /// relation, the second in the context of the right one, and each pair + /// should produce the same type. + /// + /// Note that this is the key to match *on*, in the same sense as + /// [`EquiJoinExpr::on`] — the columns that survive into the output are not + /// expressed here at all, since the output carries the left relation's + /// schema unchanged. pub on: Vec<(Expr, Expr)>, } #[derive(Clone, Debug, Eq, PartialEq)] pub struct ThetaJoinExpr { + // Can be subsumed by the EquiJoin/MultiWayJoin, but better /// Must evaluate to a relation. pub left: Expr, /// Must evaluate to a relation. @@ -334,6 +624,7 @@ pub trait RelExprVisitor { RelExpr::Projection(expr) => self.visit_projection_expr(expr, ctx), RelExpr::CartesianProduct(expr) => self.visit_cartesian_product_expr(expr, ctx), RelExpr::EquiJoin(expr) => self.visit_equi_join_expr(expr, ctx), + RelExpr::MultiWayEquiJoin(expr) => self.visit_multi_way_equi_join_expr(expr, ctx), RelExpr::AntiJoin(expr) => self.visit_anti_join_expr(expr, ctx), RelExpr::FixedPointIter(expr) => self.visit_fixed_point_iter_expr(expr, ctx), } @@ -348,10 +639,12 @@ pub trait RelExprVisitor { fn visit_projection_expr(&mut self, expr: &ProjectionExpr, ctx: C) -> T; fn visit_cartesian_product_expr(&mut self, expr: &CartesianProductExpr, ctx: C) -> T; fn visit_equi_join_expr(&mut self, expr: &EquiJoinExpr, ctx: C) -> T; + fn visit_multi_way_equi_join_expr(&mut self, expr: &MultiWayEquiJoinExpr, ctx: C) -> T; fn visit_anti_join_expr(&mut self, expr: &AntiJoinExpr, ctx: C) -> T; fn visit_fixed_point_iter_expr(&mut self, expr: &FixedPointIterExpr, ctx: C) -> T; } +/// Annotating visitor. See [`RelExprVisitorOwn`]. pub trait RelExprVisitorMut { fn visit_rel(&mut self, expr: &mut RelExpr, ctx: C) -> T { match expr { @@ -365,6 +658,7 @@ pub trait RelExprVisitorMut { RelExpr::Projection(expr) => self.visit_projection_expr(expr, ctx), RelExpr::CartesianProduct(expr) => self.visit_cartesian_product_expr(expr, ctx), RelExpr::EquiJoin(expr) => self.visit_equi_join_expr(expr, ctx), + RelExpr::MultiWayEquiJoin(expr) => self.visit_multi_way_equi_join_expr(expr, ctx), RelExpr::AntiJoin(expr) => self.visit_anti_join_expr(expr, ctx), RelExpr::FixedPointIter(expr) => self.visit_fixed_point_iter_expr(expr, ctx), } @@ -379,52 +673,162 @@ pub trait RelExprVisitorMut { fn visit_projection_expr(&mut self, expr: &mut ProjectionExpr, ctx: C) -> T; fn visit_cartesian_product_expr(&mut self, expr: &mut CartesianProductExpr, ctx: C) -> T; fn visit_equi_join_expr(&mut self, expr: &mut EquiJoinExpr, ctx: C) -> T; + fn visit_multi_way_equi_join_expr(&mut self, expr: &mut MultiWayEquiJoinExpr, ctx: C) -> T; fn visit_anti_join_expr(&mut self, expr: &mut AntiJoinExpr, ctx: C) -> T; fn visit_fixed_point_iter_expr(&mut self, expr: &mut FixedPointIterExpr, ctx: C) -> T; } +/// Restructuring visitor for the relational layer, and the family a +/// backend-specific *lowering* pass lives in — see +/// [`ExprVisitorOwn`](crate::host::expr::ExprVisitorOwn) for the rule that +/// decides between the three families, and for why the payloads arrive boxed. pub trait RelExprVisitorOwn { fn visit_rel(&mut self, expr: RelExpr, ctx: C) -> T { match expr { - RelExpr::Source(expr) => self.visit_source_expr(*expr, ctx), - RelExpr::Output(expr) => self.visit_output_expr(*expr, ctx), - RelExpr::Alias(expr) => self.visit_alias_expr(*expr, ctx), - RelExpr::Distinct(expr) => self.visit_distinct_expr(*expr, ctx), - RelExpr::Union(expr) => self.visit_union_expr(*expr, ctx), - RelExpr::Difference(expr) => self.visit_difference_expr(*expr, ctx), - RelExpr::Selection(expr) => self.visit_selection_expr(*expr, ctx), - RelExpr::Projection(expr) => self.visit_projection_expr(*expr, ctx), - RelExpr::CartesianProduct(expr) => self.visit_cartesian_product_expr(*expr, ctx), - RelExpr::EquiJoin(expr) => self.visit_equi_join_expr(*expr, ctx), - RelExpr::AntiJoin(expr) => self.visit_anti_join_expr(*expr, ctx), - RelExpr::FixedPointIter(expr) => self.visit_fixed_point_iter_expr(*expr, ctx), + RelExpr::Source(expr) => self.visit_source_expr(expr, ctx), + RelExpr::Output(expr) => self.visit_output_expr(expr, ctx), + RelExpr::Alias(expr) => self.visit_alias_expr(expr, ctx), + RelExpr::Distinct(expr) => self.visit_distinct_expr(expr, ctx), + RelExpr::Union(expr) => self.visit_union_expr(expr, ctx), + RelExpr::Difference(expr) => self.visit_difference_expr(expr, ctx), + RelExpr::Selection(expr) => self.visit_selection_expr(expr, ctx), + RelExpr::Projection(expr) => self.visit_projection_expr(expr, ctx), + RelExpr::CartesianProduct(expr) => self.visit_cartesian_product_expr(expr, ctx), + RelExpr::EquiJoin(expr) => self.visit_equi_join_expr(expr, ctx), + RelExpr::MultiWayEquiJoin(expr) => self.visit_multi_way_equi_join_expr(expr, ctx), + RelExpr::AntiJoin(expr) => self.visit_anti_join_expr(expr, ctx), + RelExpr::FixedPointIter(expr) => self.visit_fixed_point_iter_expr(expr, ctx), + } + } + fn visit_source_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_output_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_alias_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_distinct_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_union_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_difference_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_selection_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_projection_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_cartesian_product_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_equi_join_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_multi_way_equi_join_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_anti_join_expr(&mut self, expr: Box, ctx: C) -> T; + fn visit_fixed_point_iter_expr(&mut self, expr: Box, ctx: C) -> T; +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::host::expr::VarExpr; + + /// A stand-in relation operand. [`MultiWayEquiJoinExpr::validate`] only ever + /// counts these, so their content is irrelevant. + fn relations(count: usize) -> Vec { + (0..count) + .map(|idx| Expr::from(VarExpr::new(format!("r{idx}")))) + .collect() + } + + fn join_variable(name: &str, occurrences: &[RelationIdx]) -> JoinVariable { + JoinVariable { + name: name.to_string(), + occurrences: occurrences + .iter() + .map(|relation| (*relation, Expr::from(VarExpr::new(name)))) + .collect(), + } + } + + #[test] + fn accepts_a_join_variable_shared_by_two_relations() { + let joined = + MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0, 1])], None) + .expect("A variable bound by two relations is a join variable"); + assert_eq!(joined.on.len(), 1); + assert_eq!(joined.on_exprs().count(), 2); + } + + #[test] + fn accepts_an_empty_join_condition_as_a_cartesian_product() { + let joined = MultiWayEquiJoinExpr::new(relations(3), vec![], None) + .expect("An empty join condition is a cartesian product, not an error"); + assert!(joined.on.is_empty()); + } + + #[test] + fn rejects_fewer_than_two_relations() { + for count in 0..2 { + assert!( + MultiWayEquiJoinExpr::new(relations(count), vec![], None).is_err(), + "A join over {count} relation(s) should be rejected" + ); } } - fn visit_source_expr(&mut self, expr: SourceExpr, ctx: C) -> T; - fn visit_output_expr(&mut self, expr: OutputExpr, ctx: C) -> T; - fn visit_alias_expr(&mut self, expr: AliasExpr, ctx: C) -> T; - fn visit_distinct_expr(&mut self, expr: DistinctExpr, ctx: C) -> T; - fn visit_union_expr(&mut self, expr: UnionExpr, ctx: C) -> T; - fn visit_difference_expr(&mut self, expr: DifferenceExpr, ctx: C) -> T; - fn visit_selection_expr(&mut self, expr: SelectionExpr, ctx: C) -> T; - fn visit_projection_expr(&mut self, expr: ProjectionExpr, ctx: C) -> T; - fn visit_cartesian_product_expr(&mut self, expr: CartesianProductExpr, ctx: C) -> T; - fn visit_equi_join_expr(&mut self, expr: EquiJoinExpr, ctx: C) -> T; - fn visit_anti_join_expr(&mut self, expr: AntiJoinExpr, ctx: C) -> T; - fn visit_fixed_point_iter_expr(&mut self, expr: FixedPointIterExpr, ctx: C) -> T; -} - -impl MemAddr for RelExpr {} -impl MemAddr for SourceExpr {} -impl MemAddr for OutputExpr {} -impl MemAddr for AliasExpr {} -impl MemAddr for DistinctExpr {} -impl MemAddr for UnionExpr {} -impl MemAddr for DifferenceExpr {} -impl MemAddr for SelectionExpr {} -impl MemAddr for ProjectionExpr {} -impl MemAddr for CartesianProductExpr {} -impl MemAddr for EquiJoinExpr {} -impl MemAddr for AntiJoinExpr {} -impl MemAddr for ThetaJoinExpr {} -impl MemAddr for FixedPointIterExpr {} + + #[test] + fn rejects_a_single_occurrence_because_it_constrains_nothing() { + // The whole point of the `on` representation: a variable bound by only + // one relation is not an equality class. It reaches the output through + // that relation's schema instead, which is why rejecting it here is safe + // and keeps `on.is_empty()` an exact cartesian-product test. + let error = MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0])], None) + .expect_err("A single occurrence must not be representable"); + assert!(error.to_string().contains("occurrence")); + } + + #[test] + fn rejects_an_out_of_bounds_relation_index() { + assert!( + MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0, 2])], None) + .is_err() + ); + } + + #[test] + fn rejects_a_variable_occurring_twice_in_one_relation() { + // Such a repetition is a local equality condition on that one relation, + // so it belongs in a `SelectionExpr` beneath the join. + assert!( + MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0, 0])], None) + .is_err() + ); + } + + #[test] + fn rejects_two_join_variables_claiming_the_same_output_name() { + assert!( + MultiWayEquiJoinExpr::new( + relations(3), + vec![join_variable("x", &[0, 1]), join_variable("x", &[1, 2])], + None + ) + .is_err() + ); + } + + #[test] + fn normalizes_occurrences_into_relation_order() { + // Plans have to be reproducible: the occurrence order must not depend on + // the order the producer happened to discover the occurrences in. + let joined = + MultiWayEquiJoinExpr::new(relations(3), vec![join_variable("x", &[2, 0, 1])], None) + .expect("Out-of-order occurrences are normalized, not rejected"); + let order: Vec = joined.on[0] + .occurrences + .iter() + .map(|(relation, _)| *relation) + .collect(); + assert_eq!(order, vec![0, 1, 2]); + } + + #[test] + fn validate_agrees_with_new_on_hand_assembled_joins() { + // The fields are public, so a hand-built or rewritten plan can violate + // the invariants; the resolver relies on `validate` catching that. + let malformed = MultiWayEquiJoinExpr { + relations: relations(2), + on: vec![join_variable("x", &[0])], + attributes: None, + }; + assert!(malformed.validate().is_err()); + } +} diff --git a/packages/coln-query/src/relational/incremental/dbsp/wrapper.rs b/packages/coln-query/src/relational/incremental/dbsp/wrapper.rs index 6a44b404..15445e04 100644 --- a/packages/coln-query/src/relational/incremental/dbsp/wrapper.rs +++ b/packages/coln-query/src/relational/incremental/dbsp/wrapper.rs @@ -2,11 +2,9 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::relational::relation::Relation; - -use super::super::super::relation::{ - RelationData, RelationSchema, SchemaTuple, TupleKey, TupleValue, -}; +use crate::api::deltas::ZRow; +use crate::relational::incremental::schema::{SchemaTuple, StreamSchema, TupleKey}; +use crate::relational::relation::{self, Relation, RelationData, RelationRef, TupleValue}; use cli_table::{Cell, Style, Table, format::Justify}; pub use dbsp::{ DBSPHandle as DbspHandle, Error as DbspError, NestedCircuit, RootCircuit, Runtime, ZWeight, @@ -180,10 +178,34 @@ impl IntoIterator for &'_ StreamWrapper { } } -/// A DBSP stream is the DBSP backend's concrete relation representation. This is -/// the single point where the DBSP runtime plugs into the backend-neutral -/// [`Relation`] envelope. -impl RelationData for StreamWrapper { +/// A stream plus the schema its `(TupleKey, TupleValue)` pairs are laid out by: +/// the DBSP backend's concrete relation representation, and the single point +/// where the DBSP runtime plugs into the backend-neutral [`Relation`] envelope. +/// +/// The schema rides *here*, next to the stream, rather than in [`Relation`]: +/// keying a relation is a DBSP requirement (`OrdIndexedZSet`), and the schema +/// changes as operators build the circuit, so each derived stream carries the +/// schema its own rows have. The pair is what every DBSP operator recovers via +/// [`as_dbsp`](AsDbspRelation::as_dbsp). +#[derive(Clone)] +pub struct DbspRelation { + schema: StreamSchema, + stream: StreamWrapper, +} + +impl DbspRelation { + pub fn new(schema: StreamSchema, stream: StreamWrapper) -> Self { + Self { schema, stream } + } + pub fn schema(&self) -> &StreamSchema { + &self.schema + } + pub fn stream(&self) -> &StreamWrapper { + &self.stream + } +} + +impl RelationData for DbspRelation { fn as_any(&self) -> &dyn Any { self } @@ -192,6 +214,37 @@ impl RelationData for StreamWrapper { } } +impl Display for DbspRelation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}", self.schema) + } +} + +impl Debug for DbspRelation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}", self.schema) + } +} + +/// Recover the DBSP backend's own relation from the type-erased envelope the +/// host layer passes around. Every DBSP operator starts here. +pub trait AsDbspRelation { + fn as_dbsp(&self) -> &DbspRelation; +} + +impl AsDbspRelation for Relation { + fn as_dbsp(&self) -> &DbspRelation { + self.downcast_ref::() + } +} + +/// A fresh [`RelationRef`] over `stream` and the schema its rows have. The DBSP +/// backend's counterpart to [`relation::new_relation`], which takes the pair +/// pre-assembled. +pub fn new_relation(schema: StreamSchema, stream: StreamWrapper) -> RelationRef { + relation::new_relation(DbspRelation::new(schema, stream)) +} + #[derive(Default, Debug, Clone)] pub struct DbspInputs { inputs: HashMap, @@ -203,8 +256,8 @@ impl DbspInputs { inputs: HashMap::from_iter(inputs), } } - pub fn get(&self, name: &str) -> Option<&DbspInput> { - self.inputs.get(name) + pub fn get>(&self, name: Q) -> Option<&DbspInput> { + self.inputs.get(name.as_ref()) } pub fn take(&mut self, name: &str) -> Option { self.inputs.remove(name) @@ -216,18 +269,18 @@ impl DbspInputs { #[derive(Clone)] pub struct DbspInput { - schema: RelationSchema, + schema: StreamSchema, handle: OrdIndexedStreamInputHandle, } impl DbspInput { - pub fn new(schema: RelationSchema, handle: OrdIndexedStreamInputHandle) -> Self { + pub fn new(schema: StreamSchema, handle: OrdIndexedStreamInputHandle) -> Self { Self { schema, handle } } /// Feed a batch of value tuples (with z-weights) into this input. The tuple /// key is derived from the value by picking the schema's key fields, so /// callers only supply the value — matching the neutral `Runtime::feed`. - pub fn feed(&self, rows: impl IntoIterator) { + pub fn feed(&self, rows: impl IntoIterator) { let tuple_names: Vec = self.schema.tuple.field_names(&None).collect(); let key_indices: Vec = self .schema @@ -242,11 +295,13 @@ impl DbspInput { .collect(); let mut batch = rows .into_iter() - .map(|(value, weight)| { + .map(|row_delta| { + let zweight = row_delta.zweight(); + let row = row_delta.into_row(); let key = TupleKey { - data: key_indices.iter().map(|&i| value.data[i].clone()).collect(), + data: key_indices.iter().map(|&i| row.data[i].clone()).collect(), }; - Tup2(key, Tup2(value, weight)) + Tup2(key, Tup2(row, zweight)) }) .collect(); self.handle.append(&mut batch); @@ -266,44 +321,65 @@ impl Debug for DbspInput { pub struct DbspOutput { handle: OrdIndexedStreamOutputHandle, - schema: RelationSchema, + schema: StreamSchema, } impl DbspOutput { - pub fn new(schema: RelationSchema, handle: OrdIndexedStreamOutputHandle) -> Self { - Self { schema, handle } - } - pub fn to_batch(&self) -> DbspOutputBatch<'_> { - let inner = self.handle.concat().iter().collect::>(); - DbspOutputBatch { - schema: &self.schema, - inner, + pub fn drain(&self) -> DbspOutputDelta { + // This can already be iterated and saved into a collection, e.g., a Vector. + // Yet, I believe this does not guarantee that each (TupleKey, TupleValue) + // pair is unique but instead could appear multiple times with different + // zweights which would need to be accumulated for each + // (TupleKey, TupleValue) pair. + let delta: SpineSnapshot> = self.handle.concat(); + // Therefore, we play it safe and consolidate here, which guarantees that + // each (TupleKey, TupleValue) pair is unique with its accumulated zweight. + // If at some point, the accumulation should happen through a custom data + // structure, this step may be omitted for performance reasons. + let delta: OrdIndexedZSet = delta.consolidate(); + DbspOutputDelta { + schema: self.schema.clone(), + delta, } } } +impl std::fmt::Debug for DbspOutput { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DbspOutput") + .field("schema", &self.schema) + .finish_non_exhaustive() + } +} + impl From<&Relation> for DbspOutput { fn from(relation: &Relation) -> Self { - let schema = relation.schema.clone(); - let handle = relation.downcast_ref::().output(); - Self { schema, handle } + let relation = relation.as_dbsp(); + Self { + schema: relation.schema().clone(), + handle: relation.stream().output(), + } } } -pub struct DbspOutputBatch<'a> { - schema: &'a RelationSchema, - inner: Vec<(TupleKey, TupleValue, ZWeight)>, +#[derive(Debug, Clone)] +pub struct DbspOutputDelta { + schema: StreamSchema, + delta: OrdIndexedZSet, } -impl DbspOutputBatch<'_> { +impl DbspOutputDelta { const JUSTIFICATION: Justify = Justify::Right; + pub fn schema(&self) -> &StreamSchema { + &self.schema + } pub fn as_table(&self) -> impl Display { - self.inner + self.delta .iter() .map(|(key, tuple, weight)| { iter::once(weight.to_string().cell().justify(Self::JUSTIFICATION)).chain( - SchemaTuple::new(&self.schema.tuple, tuple) + SchemaTuple::new(&self.schema.tuple, &tuple) .fields() .map(|attribute| attribute.to_string().cell().justify(Self::JUSTIFICATION)) .collect::>(), @@ -319,7 +395,7 @@ impl DbspOutputBatch<'_> { .expect("Table error") } pub fn as_debug_table(&self) -> impl Display { - self.inner + self.delta .iter() .map(|(key, tuple, weight)| { // We ensure that the key and tuple data lengths match the @@ -328,7 +404,7 @@ impl DbspOutputBatch<'_> { debug_assert!(tuple.data.len() == self.schema.tuple.full_len()); iter::once(weight.to_string().cell().justify(Self::JUSTIFICATION)) .chain( - SchemaTuple::new(&self.schema.key, key) + SchemaTuple::new(&self.schema.key, &key) .all_fields() .map(|attribute| { attribute.to_string().cell().justify(Self::JUSTIFICATION) @@ -336,7 +412,7 @@ impl DbspOutputBatch<'_> { .collect::>(), ) .chain( - SchemaTuple::new(&self.schema.tuple, tuple) + SchemaTuple::new(&self.schema.tuple, &tuple) .all_fields() .map(|attribute| { attribute.to_string().cell().justify(Self::JUSTIFICATION) @@ -364,30 +440,42 @@ impl DbspOutputBatch<'_> { .display() .expect("Table error") } - pub fn as_data(&self) -> impl Iterator { - self.inner + /// Outputs only the visible columns of the output. + fn as_data(&self) -> impl Iterator { + self.delta.iter().map(|(_key, tuple, zweight)| { + let tuple: TupleValue = SchemaTuple::new(&self.schema.tuple, &tuple) + .fields() + .cloned() + .collect(); + (zweight, tuple) + }) + } + /// Unlike [`as_data`](Self::as_data), this Includes hidden/inactive + /// columns in its output. + fn as_debug_data(&self) -> impl Iterator { + self.delta .iter() - .map(|(_key, tuple, weight)| (*weight, tuple)) + .map(|(_key, tuple, zweight)| (zweight, tuple)) } - pub fn as_zset(&self) -> OrdZSet { + pub fn as_zrows(&self) -> impl Iterator { + self.as_data() + .filter_map(|(zweight, tuple)| ZRow::new(zweight, tuple)) + } + pub fn as_debug_zrows(&self) -> impl Iterator { + self.as_debug_data() + .filter_map(|(zweight, tuple)| ZRow::new(zweight, tuple)) + } + pub fn to_zset(&self) -> OrdZSet { let keys = self - .inner - .iter() - .map(|(_key, tuple, weight)| { - let tuple: TupleValue = SchemaTuple::new(&self.schema.tuple, tuple) - .fields() - .cloned() - .collect(); - Tup2(tuple, *weight) - }) + .as_data() + .map(|(zweight, tuple)| Tup2(tuple, zweight)) .collect::>(); OrdZSet::from_keys((), keys) } - pub fn as_debug_zset(&self) -> OrdZSet { + pub fn to_debug_zset(&self) -> OrdZSet { let keys = self - .inner - .iter() - .map(|(_key, tuple, weight)| Tup2(tuple.clone(), *weight)) + .as_debug_data() + .map(|(zweight, tuple)| Tup2(tuple, zweight)) .collect::>(); OrdZSet::from_keys((), keys) } diff --git a/packages/coln-query/src/relational/incremental/interpreter.rs b/packages/coln-query/src/relational/incremental/interpreter.rs index 5f8855b3..122b791d 100644 --- a/packages/coln-query/src/relational/incremental/interpreter.rs +++ b/packages/coln-query/src/relational/incremental/interpreter.rs @@ -2,30 +2,32 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use super::super::relation::{ - Relation, RelationRef, SchemaTuple, TupleKey, TupleValue, new_relation, -}; +use super::super::relation::{Relation, RelationRef, TupleValue}; use super::operators::{ coalesce::coalesce_helper, projection::{ProjectionStrategy, projection_helper}, reindex::reindex_helper, }; -use crate::relational::RelationSchema; +use super::schema::{DbspTupleContext, SchemaTuple, StreamSchema, TupleKey}; +use crate::relational::expr::MultiWayEquiJoinExpr; use crate::relational::incremental::dbsp::{ - DbspInput, OrdIndexedStreamInputHandle, new_ord_indexed_stream, + AsDbspRelation, DbspInput, DbspRelation, OrdIndexedStreamInputHandle, new_ord_indexed_stream, + new_relation, }; +use crate::relational::schema::TableSchema; use crate::{ error::BuildError, host::variable::{Value, VariableSlot}, host::{ expr::{Expr, VarExpr}, interpreter::{EvalResult, HostInterpreter, InterpreterContext, assert_type, is_truthy}, - stmt::Stmt, + walk::{Node, pre_order}, }, + relational::catalog::SourceSchemas, relational::expr::{ AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, FixedPointIterExpr, OutputExpr, OutputKind, ProjectionExpr, RelExpr, RelExprVisitor, - SelectionExpr, SinkId, SourceExpr, UnionExpr, + SelectionExpr, SinkId, SourceExpr, SourceId, UnionExpr, }, relational::incremental::dbsp::{ DbspError, DbspInputs, DbspOutput, NestedCircuit, OrdIndexedNestedStream, RootCircuit, @@ -36,115 +38,25 @@ use crate::{ use std::collections::HashMap; use std::{cell::Ref, rc::Rc}; -// TODO: As I expect such traversals to be more common, implement iterators in -// at least execution order (post-order traversal), and possibly other orderings, -// and simply filter on the iterator.. - -/// Collect every [`SourceExpr`] reachable from `stmts`. -/// -/// The DBSP backend uses this to wire the root inputs of sources referenced -/// inside a [`FixedPointIterExpr`] step *before* entering the `recursive` -/// block: DBSP forbids adding a root input once a nested circuit is under -/// construction (the input node would not belong to the root scope). Unlike the -/// main interpretation walk this only inspects the tree, so keeping it separate -/// is cheap and avoids threading circuit state through a discovery pass. -fn collect_source_exprs<'a>(stmts: &'a [Stmt], out: &mut Vec<&'a SourceExpr>) { - fn walk_stmt<'a>(stmt: &'a Stmt, out: &mut Vec<&'a SourceExpr>) { - match stmt { - Stmt::Var(stmt) => { - if let Some(expr) = &stmt.initializer { - walk_expr(expr, out); - } - } - Stmt::Expr(stmt) => walk_expr(&stmt.expr, out), - Stmt::Block(stmt) => stmt.stmts.iter().for_each(|stmt| walk_stmt(stmt, out)), - } - } - fn walk_expr<'a>(expr: &'a Expr, out: &mut Vec<&'a SourceExpr>) { - match expr { - Expr::Literal(_) | Expr::Var(_) => {} - Expr::Tuple(expr) => expr.elements.iter().for_each(|expr| walk_expr(expr, out)), - Expr::GetIndex(expr) => { - walk_expr(&expr.target, out); - walk_expr(&expr.index, out); - } - Expr::Grouping(expr) => walk_expr(&expr.expr, out), - Expr::Binary(expr) => { - walk_expr(&expr.left, out); - walk_expr(&expr.right, out); - } - Expr::Unary(expr) => walk_expr(&expr.operand, out), - Expr::Assign(expr) => walk_expr(&expr.value, out), - Expr::Call(expr) => { - walk_expr(&expr.callee, out); - expr.arguments.iter().for_each(|arg| walk_expr(arg, out)); - } - Expr::Function(expr) => expr.body.stmts.iter().for_each(|stmt| walk_stmt(stmt, out)), - Expr::Relational(expr) => walk_rel(expr, out), - } - } - fn walk_rel<'a>(rel: &'a RelExpr, out: &mut Vec<&'a SourceExpr>) { - match rel { - RelExpr::Source(source) => out.push(source), - RelExpr::Output(expr) => walk_expr(&expr.relation, out), - RelExpr::Alias(expr) => walk_expr(&expr.relation, out), - RelExpr::Distinct(expr) => walk_expr(&expr.relation, out), - RelExpr::Union(expr) => expr.relations.iter().for_each(|rel| walk_expr(rel, out)), - RelExpr::Difference(expr) => { - walk_expr(&expr.left, out); - walk_expr(&expr.right, out); - } - RelExpr::Selection(expr) => { - walk_expr(&expr.relation, out); - walk_expr(&expr.condition, out); - } - RelExpr::Projection(expr) => { - walk_expr(&expr.relation, out); - expr.attributes.iter().for_each(|(_, e)| walk_expr(e, out)); - } - RelExpr::CartesianProduct(expr) => walk_equi_join(&expr.inner, out), - RelExpr::EquiJoin(expr) => walk_equi_join(expr, out), - RelExpr::AntiJoin(expr) => { - walk_expr(&expr.left, out); - walk_expr(&expr.right, out); - expr.on.iter().for_each(|(l, r)| { - walk_expr(l, out); - walk_expr(r, out); - }); - } - RelExpr::FixedPointIter(expr) => { - walk_expr(&expr.accumulator.1, out); - expr.step.stmts.iter().for_each(|stmt| walk_stmt(stmt, out)); - } - } - } - fn walk_equi_join<'a>(expr: &'a EquiJoinExpr, out: &mut Vec<&'a SourceExpr>) { - walk_expr(&expr.left, out); - walk_expr(&expr.right, out); - expr.on.iter().for_each(|(l, r)| { - walk_expr(l, out); - walk_expr(r, out); - }); - if let Some(attributes) = &expr.attributes { - attributes.iter().for_each(|(_, e)| walk_expr(e, out)); - } - } - stmts.iter().for_each(|stmt| walk_stmt(stmt, out)); -} - -type Sources = HashMap; +type Sources = HashMap; struct Source { - schema: RelationSchema, + schema: StreamSchema, handle: OrdIndexedStreamInputHandle, stream: StreamWrapper, } impl Source { - fn from_source_expr(source_expr: &SourceExpr, root_circuit: &mut RootCircuit) -> Self { + /// Wire a fresh input stream for the relation `table` describes. The table + /// schema comes from the plan's catalog, since a [`SourceExpr`] leaf only + /// names its relation. This is the one place it is turned into a live + /// stream and, with it, the one place a table's declared key(s) become the + /// single key this circuit's `OrdIndexedZSet` is indexed by (see + /// [`StreamSchema::from`]). + fn new(table: &TableSchema, root_circuit: &mut RootCircuit) -> Self { let (stream, handle) = new_ord_indexed_stream(root_circuit); Source { - schema: source_expr.schema.clone(), + schema: StreamSchema::from(table), handle, stream: StreamWrapper::from(stream), } @@ -153,7 +65,10 @@ impl Source { impl From<&Source> for Relation { fn from(source: &Source) -> Self { - Relation::new(source.schema.clone(), source.stream.clone()) + Relation::new(DbspRelation::new( + source.schema.clone(), + source.stream.clone(), + )) } } @@ -164,7 +79,7 @@ impl From for DbspInputs { .into_iter() // We drop the stream in `source` because it is !Send to be able // to cross the `DbspRuntime::init_circuit` boundary. - .map(|(name, source)| (name, DbspInput::new(source.schema, source.handle))), + .map(|(name, source)| (name.0, DbspInput::new(source.schema, source.handle))), ) } } @@ -178,7 +93,7 @@ enum ImportKey { /// An outer relation reached by variable, keyed by its resolved slot. Var(VariableSlot), /// A [`SourceExpr`] leaf, keyed by source name. - Source(String), + Source(SourceId), } /// State that only exists while walking a [`FixedPointIterExpr`] step body, @@ -199,9 +114,9 @@ pub struct DbspInterpreter { /// The scalar engine driven on the per-tuple hot path (selection conditions, /// projection attributes, join keys). engine: E, - /// Live input streams by source name, populated lazily the first time each - /// [`SourceExpr`] leaf is visited. Serves both deduplication (one stream per - /// source, however many leaves reference it) and binding. + /// Every input stream the plan needs, keyed by the [`SourceId`] its + /// [`SourceExpr`] leaves name it by. Wired once in [`new`](Self::new) from + /// the schemas the pipeline resolved, so a leaf is *bound* here. sources: Sources, /// Output read handles collected while walking the plan, one per /// [`OutputExpr`] tap, in plan order. The backend drains these after @@ -216,11 +131,29 @@ pub struct DbspInterpreter { } impl DbspInterpreter { - pub fn new(root_circuit: RootCircuit, engine: E) -> Self { + /// Wires one root input stream per source the plan names, up front. + /// + /// Eagerly rather than on first visit, because DBSP refuses a root input + /// once a nested circuit is under construction: wiring everything before + /// interpretation starts is what lets a [`FixedPointIterExpr`] step body + /// reach a source at all, without a pre-pass hoisting that body's sources + /// out by hand. `sources` came from a walk of this same plan, so the set is + /// the same one lazy wiring would have reached. + /// + /// Sorted by [`SourceId`], so that circuit construction does not inherit the + /// iteration order of a [`HashMap`]. + pub fn new(root_circuit: RootCircuit, engine: E, sources: SourceSchemas) -> Self { + let mut root_circuit = root_circuit; + let mut wired = Sources::with_capacity(sources.len()); + let mut sources: Vec<_> = sources.into_iter().collect(); + sources.sort_by(|(left, _), (right, _)| left.as_str().cmp(right.as_str())); + for (id, table) in sources { + wired.insert(id, Source::new(&table, &mut root_circuit)); + } Self { root_circuit, engine, - sources: Sources::new(), + sources: wired, sinks: Vec::new(), step: None, } @@ -232,7 +165,7 @@ impl DbspInterpreter { fn bridge_import( &mut self, key: ImportKey, - schema: RelationSchema, + schema: StreamSchema, root_stream: &StreamWrapper, ) -> RelationRef { let step = self @@ -298,20 +231,16 @@ impl HostInterpreter for DbspInterpreter { return Ok(value); }; let relation = Rc::clone(relation); - let is_root = matches!( - relation.borrow().downcast_ref::(), - StreamWrapper::Root(_) - ); + let is_root = matches!(relation.borrow().as_dbsp().stream(), StreamWrapper::Root(_)); if !is_root { return Ok(value); } let borrowed = relation.borrow(); - let schema = borrowed.schema.clone(); - let root = borrowed.downcast_ref::(); + let borrowed = borrowed.as_dbsp(); Ok(Value::Relation(self.bridge_import( ImportKey::Var(resolved), - schema, - root, + borrowed.schema().clone(), + borrowed.stream(), ))) } } @@ -320,25 +249,23 @@ impl RelExprVisitor> for DbspInterpreter { fn visit_source_expr(&mut self, expr: &SourceExpr, ctx: VisitorCtx) -> ExprVisitorResult { - // Wire a fresh root input the first time we meet a source, reusing it - // for every later leaf naming the same source. - if !self.sources.contains_key(expr.as_id()) { - let source = Source::from_source_expr(expr, &mut self.root_circuit); - self.sources.insert(expr.to_id(), source); - } - // Snapshot the root stream + schema, dropping the borrow on `sources` - // before any `&mut self` call below. + // Every source the plan names was wired in `new`, so this only binds the + // leaf to its stream. Exactly one stream per source, however many leaves name + // it. Snapshot stream and schema to drop the borrow on `sources` before + // any `&mut self` call below. let (schema, root_stream) = { - let source = self - .sources - .get(expr.as_id()) - .expect("source just wired above"); + let source = self.sources.get(&expr.id).ok_or_else(|| { + BuildError::new(format!( + "Source '{}' was not among the sources this plan was resolved to", + expr.id + )) + })?; (source.schema.clone(), source.stream.clone()) }; // Inside a fixed-point step the source is an outer relation and must be // `delta0`'d into the nested circuit, just like an outer variable. let relation = if self.step.is_some() { - self.bridge_import(ImportKey::Source(expr.to_id()), schema, &root_stream) + self.bridge_import(ImportKey::Source(expr.id.clone()), schema, &root_stream) } else { new_relation(schema, root_stream) }; @@ -369,10 +296,10 @@ impl RelExprVisitor> .map(coalesce_helper)?; let relation_ref = relation.borrow(); - let distincted = relation_ref.downcast_ref::().distinct(); + let distincted = relation_ref.as_dbsp().stream().distinct(); Ok(Value::Relation(new_relation( - relation_ref.schema.clone(), + relation_ref.as_dbsp().schema().clone(), distincted, ))) } @@ -395,13 +322,15 @@ impl RelExprVisitor> .split_first() .expect("Resolver has *not* done its job and ensured that there are at least two operands to a union!"); - let unioned = first.downcast_ref::().sum( - others - .iter() - .map(|relation| relation.downcast_ref::()), - ); + let unioned = first + .as_dbsp() + .stream() + .sum(others.iter().map(|relation| relation.as_dbsp().stream())); - Ok(Value::Relation(new_relation(first.schema.clone(), unioned))) + Ok(Value::Relation(new_relation( + first.as_dbsp().schema().clone(), + unioned, + ))) } fn visit_difference_expr( @@ -421,11 +350,12 @@ impl RelExprVisitor> let left_ref = left.borrow(); let differenced = left_ref - .downcast_ref::() - .minus(right.borrow().downcast_ref::()); + .as_dbsp() + .stream() + .minus(right.borrow().as_dbsp().stream()); Ok(Value::Relation(new_relation( - left_ref.schema.clone(), + left_ref.as_dbsp().schema().clone(), differenced, ))) } @@ -444,10 +374,12 @@ impl RelExprVisitor> .expect("Condition compilation error"); let environment = ctx.environment.clone(); let selected = relation_ref - .downcast_ref::() + .as_dbsp() + .stream() .filter(move |(_key, tuple)| { // No need to run resolver here, already resolved! - let schema = &relation_clone.borrow().schema; + let borrowed = relation_clone.borrow(); + let schema = borrowed.as_dbsp().schema(); let environment = &mut environment.clone(); let mut new_ctx = InterpreterContext::new(environment); new_ctx.extend_tuple_ctx(&None, &schema.tuple, tuple); @@ -458,7 +390,7 @@ impl RelExprVisitor> }); Ok(Value::Relation(new_relation( - relation_ref.schema.select(), + relation_ref.as_dbsp().schema().select(), selected, ))) } @@ -476,12 +408,13 @@ impl RelExprVisitor> let (schema, projected) = match projection_helper(&expr.attributes) { ProjectionStrategy::Projection(projection) => { let (schema, projection) = - projection.prepare(&relation_ref.schema, self.engine.clone()); - let projected = relation_ref.downcast_ref::().map_index({ + projection.prepare(relation_ref.as_dbsp().schema(), self.engine.clone()); + let projected = relation_ref.as_dbsp().stream().map_index({ let relation_clone = Rc::clone(&relation); let environment = ctx.environment.clone(); move |(key, tuple)| { - let schema = &relation_clone.borrow().schema; + let borrowed = relation_clone.borrow(); + let schema = borrowed.as_dbsp().schema(); let environment = &mut environment.clone(); let mut new_ctx = InterpreterContext::new(environment); new_ctx.extend_tuple_ctx(&None, &schema.tuple, tuple); @@ -491,8 +424,8 @@ impl RelExprVisitor> (schema, projected) } ProjectionStrategy::Pick(pick) => { - let schema = pick.prepare(&relation_ref.schema); - let picked = relation_ref.downcast_ref::().clone(); + let schema = pick.prepare(relation_ref.as_dbsp().schema()); + let picked = relation_ref.as_dbsp().stream().clone(); (schema, picked) } }; @@ -540,8 +473,9 @@ impl RelExprVisitor> let joined_schema = left .borrow() - .schema - .join(&right.borrow().schema, key_fields); + .as_dbsp() + .schema() + .join(right.borrow().as_dbsp().schema(), key_fields); let (schema, projection) = match expr .attributes @@ -565,8 +499,9 @@ impl RelExprVisitor> let right_rel = Rc::clone(&right); let environment = ctx.environment.clone(); move |key: &TupleKey, left: &TupleValue, right: &TupleValue| { - let left_schema = &left_rel.borrow().schema; - let right_schema = &right_rel.borrow().schema; + let (left_borrow, right_borrow) = (left_rel.borrow(), right_rel.borrow()); + let left_schema = left_borrow.as_dbsp().schema(); + let right_schema = right_borrow.as_dbsp().schema(); let joined_tuple: TupleValue = SchemaTuple::new(&left_schema.tuple, left) .join(&SchemaTuple::new(&right_schema.tuple, right)) .collect(); @@ -586,6 +521,23 @@ impl RelExprVisitor> Ok(Value::Relation(new_relation(schema, joined))) } + fn visit_multi_way_equi_join_expr( + &mut self, + expr: &MultiWayEquiJoinExpr, + ctx: VisitorCtx<'_, '_>, + ) -> ExprVisitorResult { + // Unreachable through the pipeline: a DBSP circuit joins two streams at + // a time, which is why `DbspBackend::lower` folds every multi way join + // into a chain of binary ones before the plan reaches this interpreter. + // Getting here means the plan skipped that stage. + unimplemented!( + "Multi way equi joins are not supported by DBSP. \ + `DbspBackend::lower` (see `relational::incremental::lowering`) folds them \ + into a sequence of binary equi joins; run the plan through `Pipeline` \ + rather than building a circuit from an unlowered plan." + ) + } + fn visit_anti_join_expr(&mut self, expr: &AntiJoinExpr, ctx: VisitorCtx) -> ExprVisitorResult { let left = self .visit_expr(&expr.left, ctx) @@ -613,8 +565,9 @@ impl RelExprVisitor> let anti_joined_schema = left .borrow() - .schema - .anti_join(&right.borrow().schema, key_fields); + .as_dbsp() + .schema() + .anti_join(right.borrow().as_dbsp().schema(), key_fields); let anti_joined = left_indexed.anti_join_index(&right_indexed); Ok(Value::Relation(new_relation( @@ -645,24 +598,24 @@ impl RelExprVisitor> let (accumulator_init, schema) = { let accumulator = accumulator.borrow(); - ( - accumulator.downcast_ref::().clone(), - accumulator.schema.clone(), - ) + let accumulator = accumulator.as_dbsp(); + (accumulator.stream().clone(), accumulator.schema().clone()) }; - // Wire the root inputs of any source referenced inside the step *before* - // entering `recursive`: DBSP forbids adding a root input once a nested - // circuit is under construction. Deduped against sources already wired - // elsewhere; inside the step each is `delta0`'d like any outer relation. - let mut step_sources = Vec::new(); - collect_source_exprs(&expr.step.stmts, &mut step_sources); - for source_expr in step_sources { - if !self.sources.contains_key(source_expr.as_id()) { - let source = Source::from_source_expr(source_expr, &mut self.root_circuit); - self.sources.insert(source_expr.to_id(), source); - } - } + // A source referenced inside the step needs no special handling: DBSP + // forbids adding a root input once a nested circuit is under + // construction, but `new` wired every one of them before interpretation + // began. Inside the step each is `delta0`'d like any outer relation. + + // DBSP does not allow outputting a stream attached to a child circuit. + pre_order(&expr.step.stmts) + .filter_map(Node::as_output) + .next() + .map_or(Ok(()), |_output_expr| { + Err(BuildError::new( + "a fix point's step body must not contain output expressions", + )) + })?; // A build error raised while walking the step body cannot be returned // through `recursive`'s closure (its error channel is DBSP's @@ -687,9 +640,10 @@ impl RelExprVisitor> // value `delta0`'d in, plus the recursive feedback. delta0 // does not alter the schema. let accumulator = accumulator_rel.borrow(); - let schema = accumulator.schema.clone(); + let accumulator = accumulator.as_dbsp(); + let schema = accumulator.schema().clone(); let accumulator = accumulator - .downcast_ref::() + .stream() .delta0(&nested_for_setup) .plus(&acc_for_step.into()); environment.define_var(new_relation(schema, accumulator)); @@ -705,11 +659,7 @@ impl RelExprVisitor> BuildError::new("Fixed point iteration body did not return a value.") })?; let relation = assert_type!(value, Value::Relation).map(coalesce_helper)?; - Ok(relation - .borrow() - .downcast_ref::() - .expect_nested() - .clone()) + Ok(relation.borrow().as_dbsp().stream().expect_nested().clone()) }); match stream { Ok(stream) => Ok(stream), diff --git a/packages/coln-query/src/relational/incremental/lowering.rs b/packages/coln-query/src/relational/incremental/lowering.rs new file mode 100644 index 00000000..7de92046 --- /dev/null +++ b/packages/coln-query/src/relational/incremental/lowering.rs @@ -0,0 +1,475 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! The DBSP backend's [lowering](crate::relational::Backend::lower) pass: it +//! folds every [`MultiWayEquiJoinExpr`] into a left-deep chain of binary +//! [`EquiJoinExpr`]s, which is the only join shape a DBSP circuit can be built +//! from. +//! +//! Expressed as a single [`TransformationRule`] over the shared +//! [rewriter](crate::optimizer::rewrite), so what lives here is the translation +//! itself and nothing about how the plan is walked. The rule is bottom-up, +//! which is what keeps [`fold`] simple: by the time a join is offered, +//! its operands (and any multi-way join nested in them) are already lowered, +//! so the fold can move them into the chain as they are. +//! +//! # Why a pass and not a step of the interpreter +//! +//! The two join nodes are not the same operator at two arities. A +//! [`MultiWayEquiJoinExpr`] is *variable-oriented*: its condition is a set of +//! equality classes, each with one output name, which is what a worst-case +//! optimal join algorithm iterates and what coln's FLIR naturally lowers to. An +//! [`EquiJoinExpr`] is *pair-oriented*: an ordered left and right, joined on +//! arbitrary pairs of expressions, both sides surviving into the output. Going +//! from the first to the second means *choosing a join order* and *naming the +//! accumulated key*, so it is a real translation between two vocabularies — +//! worth doing once, over a plan that can be printed and tested, rather than +//! inside the circuit-building fold of +//! [`DbspInterpreter`](super::interpreter::DbspInterpreter). +//! +//! Running before the [`Resolver`](crate::host::resolver) is what lets this pass +//! mint nodes freely: the [`VarExpr`]s it creates for the accumulated join keys +//! are resolved along with everything else afterwards. +//! +//! # How the fold names its keys +//! +//! Step `k` of the fold joins the accumulated relation with the next relation +//! in the order. For a [`JoinVariable`] bound by the incoming relation and by +//! some relation already in the accumulator, the pair is +//! `(VarExpr(variable.name), )`: the left +//! side addresses the accumulator's single surviving copy of that column, which +//! [`StreamSchema::join`](super::schema::StreamSchema::join) guarantees +//! is the one contributed by the first relation to bind it. +//! +//! That copy is only addressable by [`JoinVariable::name`] if the *first* +//! occurrence in fold order is a plain pick of a column already carrying that +//! name — which is exactly what the FLIR lowering emits, since it projects every +//! atom onto the names of the variables it binds. When it is not (an aliased +//! pick, a computed expression, or occurrences that disagree on a name), the +//! accumulated relation has no such column and the pass reports it instead of +//! silently building a circuit that joins on the wrong thing. Renaming into +//! agreement is the job of whoever builds the join, as documented on +//! [`JoinVariable::name`]. + +use crate::{ + error::{LoweringError, RewriteError}, + host::{ + QueryIr, + expr::{Expr, VarExpr}, + walk::{self, Node}, + }, + optimizer::rewrite::{RewriteDriver, Rewritten, TransformationRule}, + relational::expr::{ + EquiJoinExpr, JoinVariable, MultiWayEquiJoinExpr, RelExpr, RelKind, RelationIdx, + }, +}; +use std::collections::HashMap; + +/// Rewrite `plan` so that no [`MultiWayEquiJoinExpr`] remains, leaving a plan +/// the DBSP backend can build a circuit from. See the [module docs](self). +pub fn fold_multi_way_joins(plan: QueryIr) -> Result { + let lowered = RewriteDriver::new(vec![Box::new(MultiWayJoinFold)]).run(plan)?; + + // Checked, not assumed. What this stage owes the interpreter is a + // post-condition — "no multi-way join survives" — and one the pass verifies + // is worth more than one its rule scheduling merely implies. The + // alternative is discovering the gap as an `unimplemented!()` part-way + // through building a circuit. + let survived = walk::pre_order(&lowered) + .filter_map(Node::as_rel) + .any(|rel_expr| rel_expr.kind() == RelKind::MultiWayEquiJoin); + if survived { + return Err(LoweringError::new( + "A multi way equi join survived lowering, so the DBSP backend has no way to build \ + a circuit for it. Either the fold declined it or a later rule put one back", + )); + } + + Ok(lowered) +} + +/// Replaces one [`MultiWayEquiJoinExpr`] with a chain of binary joins. +struct MultiWayJoinFold; + +impl TransformationRule for MultiWayJoinFold { + fn name(&self) -> &'static str { + "multi-way-join-fold" + } + + /// The one node kind this translation is about. Everything that decides + /// *how* a join folds — the order, whether its keys can be named — needs + /// the join's contents, so it lives in [`fold`] rather than here. + fn interest(&self) -> &'static [RelKind] { + &[RelKind::MultiWayEquiJoin] + } + + fn apply(&mut self, node: RelExpr) -> Result { + match node { + RelExpr::MultiWayEquiJoin(join) => fold(*join).map(Rewritten::Changed), + other => Ok(Rewritten::Unchanged(other)), + } + } +} + +/// The order the relations of a join are folded into the chain in, as positions +/// into [`MultiWayEquiJoinExpr::relations`]. Contract: a permutation of +/// `0..relations.len()`. +/// +/// This is the seam a cost-based optimizer plugs into, and the reason it is a +/// step of its own: *which* order is fastest is a cardinality question, and +/// therefore an optimizer's business, while turning a chosen order into a chain +/// of binary joins is mechanical and belongs here. With nothing to inform the +/// choice yet, the order the plan already states is as good a guess as any. +fn join_order(join: &MultiWayEquiJoinExpr) -> Vec { + (0..join.relations.len()).collect() +} + +/// Folds one join, whose operands the [rewriter](crate::optimizer::rewrite) +/// has already lowered on the way down. +fn fold(join: MultiWayEquiJoinExpr) -> Result { + // The fields are public, so this plan may have been assembled or rewritten + // by hand; the fold relies on the invariants (two relations or more, in + // bounds and pairwise distinct occurrences) rather than re-deriving them. + join.validate()?; + + let order = join_order(&join); + let position: HashMap = order + .iter() + .enumerate() + .map(|(position, relation)| (*relation, position)) + .collect(); + + let MultiWayEquiJoinExpr { + relations, + on, + attributes, + } = join; + let mut keys = keys_by_relation(on, &position)?; + // Taken one by one, in fold order, rather than in relation order. + let mut relations: Vec> = relations.into_iter().map(Some).collect(); + let mut take = |relation: RelationIdx| { + relations[relation] + .take() + .expect("A join order names every relation exactly once") + }; + + // The last relation is split off rather than folded with the others so that + // the projection can operate on the outermost join alone: an inner step that + // projected would drop columns a later one still has to join on or include + // columns that a later one still has to produce. + let folded = order + .split_last() + .map(|(last, order)| { + let mut order = order.iter(); + let first = order + .next() + .expect("A validated join has at least two relations"); + let acc = order.fold(take(*first), |acc, relation| { + Expr::from(EquiJoinExpr { + left: acc, + right: take(*relation), + // No keys for this relation makes the step a cartesian + // product, which is what an empty `on` already means for an + // `EquiJoinExpr`. So the chain stays one node kind instead + // of switching to `CartesianProductExpr`, a newtype over it. + on: keys.remove(relation).unwrap_or_default(), + attributes: None, + }) + }); + Expr::from(EquiJoinExpr { + left: acc, + right: take(*last), + on: keys.remove(last).unwrap_or_default(), + attributes, + }) + }) + .expect("A validated join has at least two relations"); + + Ok(folded) +} + +/// Distributes the join condition over the fold: which key pairs each relation +/// contributes when it enters the chain. +/// +/// A variable's first occurrence *in fold order* produces no pair — it only +/// carries the column into the accumulator, so there is nothing to constrain it +/// against yet. Every later occurrence is compared against that copy by name. +fn keys_by_relation( + on: Vec, + position: &HashMap, +) -> Result>, RewriteError> { + let mut keys: HashMap> = HashMap::new(); + + for variable in on { + let JoinVariable { + name, + mut occurrences, + } = variable; + // `MultiWayEquiJoinExpr` normalizes into relation order, which is not + // the order the relations enter the chain in. + occurrences.sort_by_key(|(relation, _)| position[relation]); + + let mut occurrences = occurrences.into_iter(); + let (carrier, carried) = occurrences + .next() + .expect("A validated join variable has at least two occurrences"); + if !matches!(&carried, Expr::Var(var) if var.name == name) { + return Err(RewriteError::new(format!( + "Cannot lower join variable '{name}': the first relation to bind it \ + (relation {carrier}) does not do so as a plain pick of a column named \ + '{name}', so the accumulated relation of the binary join chain has no \ + such column to compare later occurrences against. Project the relation \ + onto '{name}' beneath the join" + ))); + } + + for (relation, occurrence) in occurrences { + keys.entry(relation) + .or_default() + .push((Expr::from(VarExpr::new(name.clone())), occurrence)); + } + } + + Ok(keys) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + host::{ + expr::{Literal, LiteralExpr}, + stmt::{ExprStmt, Stmt, VarStmt}, + }, + relational::expr::{OutputExpr, OutputKind, SinkId}, + }; + + fn relation(name: &str) -> Expr { + Expr::from(VarExpr::new(name)) + } + + fn relations(count: usize) -> Vec { + (0..count).map(|idx| relation(&format!("r{idx}"))).collect() + } + + /// A join variable bound by `occurrences`, each as the plain column pick the + /// FLIR lowering emits. + fn join_variable(name: &str, occurrences: &[RelationIdx]) -> JoinVariable { + JoinVariable { + name: name.to_string(), + occurrences: occurrences + .iter() + .map(|relation| (*relation, Expr::from(VarExpr::new(name)))) + .collect(), + } + } + + fn lower(expr: Expr) -> Result { + let lowered = fold_multi_way_joins(QueryIr::new(vec![Stmt::from(ExprStmt { expr })]))?; + match lowered.into_stmts().pop() { + Some(Stmt::Expr(stmt)) => Ok(stmt.expr), + other => panic!("Lowering a single expression statement yielded {other:?}"), + } + } + + fn assert_equi_join(expr: &Expr) -> &EquiJoinExpr { + match expr { + Expr::Relational(RelExpr::EquiJoin(join)) => join, + other => { + panic!("Expected a relational expression containing an equi join, got {other:?}") + } + } + } + + /// The pair of variable names an equi join compares on, which is all the + /// tests need to see of a key. + fn key_names(join: &EquiJoinExpr) -> Vec<(String, String)> { + let name = |expr: &Expr| match expr { + Expr::Var(var) => var.name.clone(), + other => panic!("Expected a plain column pick, got {other:?}"), + }; + join.on + .iter() + .map(|(left, right)| (name(left), name(right))) + .collect() + } + + #[test] + fn folds_a_three_way_join_into_a_left_deep_chain() { + let joined = MultiWayEquiJoinExpr::new( + relations(3), + vec![join_variable("x", &[0, 1]), join_variable("y", &[1, 2])], + None, + ) + .expect("A well-formed three way join"); + + let lowered = lower(Expr::from(joined)).expect("Plain column picks are lowerable"); + + // ((r0 ⋈x r1) ⋈y r2) + let outer = assert_equi_join(&lowered); + assert_eq!(outer.right, relation("r2")); + assert_eq!(key_names(outer), vec![("y".to_string(), "y".to_string())]); + + let inner = assert_equi_join(&outer.left); + assert_eq!(inner.left, relation("r0")); + assert_eq!(inner.right, relation("r1")); + assert_eq!(key_names(inner), vec![("x".to_string(), "x".to_string())]); + } + + #[test] + fn compares_a_later_occurrence_against_the_accumulated_copy() { + // `x` is bound by all three relations. The chain may only ever compare + // the incoming relation against the accumulator, never against a + // relation that is already inside it. + let joined = + MultiWayEquiJoinExpr::new(relations(3), vec![join_variable("x", &[0, 1, 2])], None) + .expect("A well-formed three way join"); + + let lowered = lower(Expr::from(joined)).expect("Plain column picks are lowerable"); + + let outer = assert_equi_join(&lowered); + assert_eq!(key_names(outer), vec![("x".to_string(), "x".to_string())]); + assert_eq!(outer.right, relation("r2")); + assert_eq!( + key_names(assert_equi_join(&outer.left)), + vec![("x".to_string(), "x".to_string())] + ); + } + + #[test] + fn keeps_the_projection_on_the_outermost_join() { + // An inner step has no business projecting: it would drop columns a + // later step still has to join on. + let attributes = vec![("z".to_string(), Expr::from(VarExpr::new("x")))]; + let joined = MultiWayEquiJoinExpr::new( + relations(3), + vec![join_variable("x", &[0, 1]), join_variable("y", &[1, 2])], + Some(attributes.clone()), + ) + .expect("A well-formed three way join"); + + let lowered = lower(Expr::from(joined)).expect("Plain column picks are lowerable"); + + let outer = assert_equi_join(&lowered); + assert_eq!(outer.attributes, Some(attributes)); + assert_eq!(assert_equi_join(&outer.left).attributes, None); + } + + #[test] + fn folds_an_empty_join_condition_into_a_chain_of_products() { + let joined = MultiWayEquiJoinExpr::new(relations(3), vec![], None) + .expect("An empty join condition is a cartesian product"); + + let lowered = lower(Expr::from(joined)).expect("A product needs no names"); + + let outer = assert_equi_join(&lowered); + assert!(outer.on.is_empty()); + assert!(assert_equi_join(&outer.left).on.is_empty()); + } + + #[test] + fn lowers_a_join_nested_inside_another_one() { + // The rule is bottom-up, so the inner join is already a chain by the + // time the outer one is offered — one round, no scheduling involved. + let inner = + MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0, 1])], None) + .expect("A well-formed two way join"); + let outer = MultiWayEquiJoinExpr::new( + vec![Expr::from(inner), relation("r2")], + vec![join_variable("x", &[0, 1])], + None, + ) + .expect("A join over a join"); + + let lowered = lower(Expr::from(outer)).expect("Plain column picks are lowerable"); + + let outer = assert_equi_join(&lowered); + assert_eq!(outer.right, relation("r2")); + let inner = assert_equi_join(&outer.left); + assert_eq!(inner.left, relation("r0")); + assert_eq!(inner.right, relation("r1")); + } + + #[test] + fn lowers_a_join_beneath_an_unrelated_operator() { + let joined = + MultiWayEquiJoinExpr::new(relations(2), vec![join_variable("x", &[0, 1])], None) + .expect("A well-formed two way join"); + let tapped = Expr::from(OutputExpr { + relation: Expr::from(joined), + id: SinkId::from("out"), + kind: OutputKind::Channel, + }); + + let lowered = lower(tapped).expect("Plain column picks are lowerable"); + + match &lowered { + Expr::Relational(rel) => match rel { + RelExpr::Output(output) => { + assert_eq!( + key_names(assert_equi_join(&output.relation)), + vec![("x".to_string(), "x".to_string())] + ); + } + other => panic!("Expected the output tap to survive, got {other:?}"), + }, + other => panic!("Expected a relational expression, got {other:?}"), + } + } + + #[test] + fn rejects_a_join_variable_the_accumulator_cannot_name() { + // `x` is carried into the accumulator by an expression rather than by a + // column named `x`, so there is nothing for the second occurrence to be + // compared against. Renaming beneath the join is the caller's job. + let joined = MultiWayEquiJoinExpr::new( + relations(2), + vec![JoinVariable { + name: "x".to_string(), + occurrences: vec![ + (0, Expr::from(LiteralExpr::from(1u64))), + (1, Expr::from(VarExpr::new("x"))), + ], + }], + None, + ) + .expect("Validation does not constrain the occurrence expressions"); + + let error = lower(Expr::from(joined)).expect_err("An unnameable key must not be lowered"); + assert!(error.message.contains("'x'"), "{}", error.message); + } + + #[test] + fn rejects_a_malformed_join() { + // Hand-assembled, so it never went through `new`. + let joined = MultiWayEquiJoinExpr { + relations: relations(2), + on: vec![join_variable("x", &[0])], + attributes: None, + }; + assert!(lower(Expr::from(joined)).is_err()); + } + + #[test] + fn leaves_a_plan_without_multi_way_joins_untouched() { + let plan = QueryIr::new(vec![Stmt::from(VarStmt { + name: "joined".to_string(), + initializer: Some(Expr::from(EquiJoinExpr { + left: relation("r0"), + right: relation("r1"), + on: vec![(Expr::from(VarExpr::new("x")), Expr::from(VarExpr::new("y")))], + attributes: Some(vec![( + "z".to_string(), + Expr::from(LiteralExpr { + value: Literal::Uint(1), + }), + )]), + })), + })]); + + let lowered = fold_multi_way_joins(plan.clone()).expect("Nothing to lower"); + + assert_eq!(lowered, plan); + } +} diff --git a/packages/coln-query/src/relational/incremental/mod.rs b/packages/coln-query/src/relational/incremental/mod.rs index e0800768..bcd86fa1 100644 --- a/packages/coln-query/src/relational/incremental/mod.rs +++ b/packages/coln-query/src/relational/incremental/mod.rs @@ -5,14 +5,16 @@ //! A [DBSP](`dbsp`) powered incremental [`Backend`], that is [`DbspBackend`], //! and [`Runtime`], which is [`DbspRuntime`]. -use super::relation::TupleValue; use super::{Backend, Runtime}; -use crate::error::{BuildError, RuntimeError}; +use crate::api::deltas::ZRow; +use crate::error::{BuildError, LoweringError, RuntimeError}; +use crate::relational::incremental::dbsp::DbspOutputDelta; use crate::{ - api::deltas::ZWeight, - host::{HostInterpreter, InterpreterContext, resolver::ResolvedCode, variable::Environment}, + host::{ + HostInterpreter, InterpreterContext, QueryIr, resolver::ResolvedCode, variable::Environment, + }, relational::{ - Delta, + catalog::SourceSchemas, expr::{OutputKind, SinkId, SourceId}, }, scalarial::{RowScalarEngine, TreeWalk}, @@ -26,7 +28,9 @@ use std::{ pub mod dbsp; pub mod interpreter; +pub mod lowering; pub mod operators; +pub mod schema; /// The incremental backend: compiles the plan into a standing DBSP circuit. /// @@ -55,15 +59,31 @@ impl Backend for DbspBackend { type Runtime = DbspRuntime; type Error = BuildError; - fn build(self, threads: NonZeroUsize, plan: ResolvedCode) -> Result { + /// A DBSP circuit joins two streams at a time, so every + /// [`MultiWayEquiJoinExpr`](crate::relational::expr::MultiWayEquiJoinExpr) + /// has to become a chain of binary ones before + /// [`build`](Self::build) walks the plan. See [`lowering`]. + fn lower(&self, plan: QueryIr) -> Result { + lowering::fold_multi_way_joins(plan) + } + + fn build( + self, + threads: NonZeroUsize, + plan: ResolvedCode, + sources: SourceSchemas, + ) -> Result { let engine = self.scalar_engine; let (handle, (inputs, outputs)) = CircuitRuntime::init_circuit(threads, move |root_circuit| { // The plan is already resolved, so we interpret directly (no - // resolver pass here) with a fresh environment. + // resolver pass here) with a fresh environment. Cloning the + // schemas is what makes this closure `Send + 'static`, which + // DBSP requires because it runs it once per worker thread. let mut environment = Environment::default(); let mut ctx = InterpreterContext::new(&mut environment); - let mut interpreter = DbspInterpreter::new(root_circuit.clone(), engine.clone()); + let mut interpreter = + DbspInterpreter::new(root_circuit.clone(), engine.clone(), sources.clone()); // Walk the plan for its side effects: each `SourceExpr` leaf // wires a fresh input stream (deduplicated by name) and each // `OutputExpr` tap wires an output read handle. The plan's final @@ -106,7 +126,9 @@ impl Backend for DbspBackend { } /// A standing DBSP circuit plus its input feed handles (by [`SourceId`]) and -/// output read handles (by [`SinkId`]). Yields per-transaction [`Delta`]s. +/// output read handles (by [`SinkId`]). Yields per-transaction +/// [`DbspOutputDelta`]s. +#[derive(Debug)] pub struct DbspRuntime { handle: DbspHandle, inputs: DbspInputs, @@ -120,24 +142,19 @@ pub struct DbspRuntime { } impl Runtime for DbspRuntime { - type Output = Delta; + type Output = DbspOutputDelta; type Error = RuntimeError; fn feed( &mut self, source: &SourceId, - rows: impl IntoIterator, - ) -> Result<(), Self::Error> { - self.inputs.get(source.as_str()).map_or_else( - || { - Err(RuntimeError::new(format!( - "tried to feed unknown source '{}'", - source.as_str() - ))) - }, + rows: impl IntoIterator, + ) -> Result { + self.inputs.get(source).map_or_else( + || Ok(false), |input| { - let _: () = input.feed(rows); - Ok(()) + input.feed(rows); + Ok(true) }, ) } @@ -148,15 +165,15 @@ impl Runtime for DbspRuntime { // debugging. This drains the handle, which is why CLI taps are not // readable via `output`. for (id, output) in &self.cli_outputs { - let batch = output.to_batch(); - println!("output '{}':\n{}", id.as_str(), batch.as_debug_table()); + let output = output.drain(); + println!("output '{}':\n{}", id.as_str(), output.as_debug_table()); } Ok(()) } - fn output(&self, out: &SinkId) -> Result { + fn output(&self, out: &SinkId) -> Result { match self.outputs.get(out) { - Some(output) => Ok(Delta(output.to_batch().as_debug_zset())), + Some(output) => Ok(output.drain()), // Distinguish an unknown name from a print-only CLI tap so the error // points at the actual mistake. None if self.cli_outputs.iter().any(|(id, _)| id == out) => { @@ -171,4 +188,8 @@ impl Runtime for DbspRuntime { ))), } } + + fn list_outputs(&self) -> impl Iterator { + self.outputs.keys() + } } diff --git a/packages/coln-query/src/relational/incremental/operators/coalesce.rs b/packages/coln-query/src/relational/incremental/operators/coalesce.rs index f8eee15a..7fbf5916 100644 --- a/packages/coln-query/src/relational/incremental/operators/coalesce.rs +++ b/packages/coln-query/src/relational/incremental/operators/coalesce.rs @@ -2,8 +2,9 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use super::super::super::relation::{RelationRef, SchemaTuple, TupleKey, TupleValue, new_relation}; -use super::StreamWrapper; +use super::super::super::relation::{RelationRef, TupleValue}; +use super::super::dbsp::{AsDbspRelation, new_relation}; +use super::super::schema::{SchemaTuple, TupleKey}; use std::rc::Rc; /// If the schema is not coalesced, this helper will compact the tuple key and @@ -14,22 +15,19 @@ use std::rc::Rc; pub fn coalesce_helper(relation: RelationRef) -> RelationRef { let relation_ref = relation.borrow(); - if relation_ref.schema.is_coalesced() { + if relation_ref.as_dbsp().schema().is_coalesced() { drop(relation_ref); return relation; } - let schema = relation_ref.schema.coalesce(); - let coalesced = relation_ref.downcast_ref::().map_index({ + let schema = relation_ref.as_dbsp().schema().coalesce(); + let coalesced = relation_ref.as_dbsp().stream().map_index({ let relation = Rc::clone(&relation); move |(key, tuple)| { let relation_ref = relation.borrow(); - let key: TupleKey = SchemaTuple::new(&relation_ref.schema.key, key) - .coalesce() - .collect(); - let tuple: TupleValue = SchemaTuple::new(&relation_ref.schema.tuple, tuple) - .coalesce() - .collect(); + let schema = relation_ref.as_dbsp().schema(); + let key: TupleKey = SchemaTuple::new(&schema.key, key).coalesce().collect(); + let tuple: TupleValue = SchemaTuple::new(&schema.tuple, tuple).coalesce().collect(); (key, tuple) } }); diff --git a/packages/coln-query/src/relational/incremental/operators/projection.rs b/packages/coln-query/src/relational/incremental/operators/projection.rs index 039a6af2..4f72f8ce 100644 --- a/packages/coln-query/src/relational/incremental/operators/projection.rs +++ b/packages/coln-query/src/relational/incremental/operators/projection.rs @@ -2,7 +2,8 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use super::super::super::relation::{RelationSchema, Tuple, TupleKey, TupleValue}; +use super::super::super::relation::{Tuple, TupleValue}; +use super::super::schema::{StreamSchema, TupleKey}; use crate::{ host::InterpreterContext, host::expr::Expr, @@ -41,10 +42,10 @@ impl ProjectionHelper { } pub fn prepare( self, - schema: &RelationSchema, + schema: &StreamSchema, engine: E, ) -> ( - RelationSchema, + StreamSchema, impl Fn(InterpreterContext) -> (TupleKey, TupleValue) + Clone + use, ) { let schema = schema.project(self.attributes); @@ -93,7 +94,7 @@ impl<'a> PickHelper<'a> { .collect(); Self { attributes } } - pub fn prepare(&self, schema: &RelationSchema) -> RelationSchema { + pub fn prepare(&self, schema: &StreamSchema) -> StreamSchema { schema.pick(&self.attributes) } } diff --git a/packages/coln-query/src/relational/incremental/operators/reindex.rs b/packages/coln-query/src/relational/incremental/operators/reindex.rs index 940f22f0..9c53a946 100644 --- a/packages/coln-query/src/relational/incremental/operators/reindex.rs +++ b/packages/coln-query/src/relational/incremental/operators/reindex.rs @@ -2,7 +2,9 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use super::super::super::relation::{RelationRef, SchemaTuple, TupleKey}; +use super::super::super::relation::RelationRef; +use super::super::dbsp::AsDbspRelation; +use super::super::schema::{DbspTupleContext, SchemaTuple, TupleKey}; use super::StreamWrapper; use super::projection::is_pickable; use crate::{ @@ -30,7 +32,7 @@ pub fn reindex_helper( .enumerate() .map(|(idx, _)| format!("anonym_field_{idx}")) .collect(); - let indexed = relation_ref.downcast_ref::().map_index({ + let indexed = relation_ref.as_dbsp().stream().map_index({ let relation = Rc::clone(relation); // Compile each key expression once, off the per-tuple hot path. let programs = on @@ -41,7 +43,8 @@ pub fn reindex_helper( .expect("Key expression compilation error"); let environment = environment.clone(); move |(_key, tuple)| { - let schema = &relation.borrow().schema; + let relation_ref = relation.borrow(); + let schema = relation_ref.as_dbsp().schema(); let environment = &mut environment.clone(); let mut new_ctx = InterpreterContext::new(environment); new_ctx.extend_tuple_ctx(&None, &schema.tuple, tuple); @@ -69,11 +72,12 @@ pub fn reindex_helper( .clone() }) .collect(); - let indexed = relation_ref.downcast_ref::().map_index({ + let indexed = relation_ref.as_dbsp().stream().map_index({ let key_field_picks = key_field_picks.clone(); let relation = Rc::clone(relation); move |(_key, tuple)| { - let key: TupleKey = SchemaTuple::new(&relation.borrow().schema.tuple, tuple) + let relation_ref = relation.borrow(); + let key: TupleKey = SchemaTuple::new(&relation_ref.as_dbsp().schema().tuple, tuple) .pick(key_field_picks.as_slice()) .collect(); (key, tuple.clone()) diff --git a/packages/coln-query/src/relational/incremental/schema.rs b/packages/coln-query/src/relational/incremental/schema.rs new file mode 100644 index 00000000..1c413dd3 --- /dev/null +++ b/packages/coln-query/src/relational/incremental/schema.rs @@ -0,0 +1,468 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! The DBSP backend's *physical* schema: how a relation's rows are actually laid +//! out for a circuit to compute on. +//! +//! A DBSP relation is an `OrdIndexedZSet`, so every +//! relation is *keyed* — and every operator that changes the shape of a row has +//! to say what the key and the value of its output are. That is what +//! [`StreamSchema`] tracks, and why it lives here rather than beside the +//! backend-neutral [`TableSchema`]: "this relation is indexed by these columns" +//! is a statement about `OrdIndexedZSet`, not about a table. A table may declare +//! several candidate keys or none at all; [`from`](StreamSchema::from) is where +//! that becomes the one key DBSP indexes by. +//! +//! [`TupleSchema`]'s inactive-field bookkeeping ([`FieldInfo::active`]) is +//! physical for the same reason: a projection marks columns dropped without +//! rebuilding every tuple, and the actual compaction happens later, at a point +//! where the circuit needs coalesced rows (see +//! [`coalesce_helper`](super::operators::coalesce::coalesce_helper)). Positions +//! in a `Vec` are the thing being tracked, so the names of a +//! relation's columns and their indexes only mean something together. + +use super::super::relation::{Tuple, TupleValue}; +use crate::{ + host::interpreter::InterpreterContext, + relational::schema::{Column, TableSchema}, + scalarial::ScalarTypedValue, +}; +use dbsp::{never_none, never_roaring_filter}; +use std::{ + collections::HashSet, + fmt::{self, Debug, Display}, +}; + +#[derive( + Clone, + Default, + Debug, + Eq, + PartialEq, + Ord, + PartialOrd, + Hash, + size_of::SizeOf, + rkyv::Archive, + rkyv::Serialize, + rkyv::Deserialize, +)] +#[archive_attr(derive(Ord, Eq, PartialEq, PartialOrd))] +pub struct TupleKey { + pub data: Vec, +} + +never_none!(TupleKey); +never_roaring_filter!(TupleKey); + +impl> FromIterator for TupleKey { + fn from_iter>(iter: I) -> Self { + Self { + data: iter.into_iter().map(|v| v.into()).collect(), + } + } +} + +impl Tuple for TupleKey { + fn data_at(&self, index: usize) -> &ScalarTypedValue { + &self.data[index] + } + fn data(&self) -> impl Iterator { + self.data.iter() + } +} + +impl Display for TupleKey { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}", self.data_to_string()) + } +} + +pub struct SchemaTuple<'a, T> { + schema: &'a TupleSchema, + tuple: &'a T, +} + +impl<'a, T: Tuple> SchemaTuple<'a, T> { + pub fn new(schema: &'a TupleSchema, tuple: &'a T) -> Self { + Self { schema, tuple } + } + pub fn fields(&self) -> impl Iterator { + self.schema + .active_fields() + .map(|(index, info)| self.tuple.data_at(index)) + } + pub fn all_fields(&self) -> impl Iterator { + self.schema + .all_fields() + .map(|(index, _info)| self.tuple.data_at(index)) + } + pub fn named_fields( + &self, + alias: &Option, + ) -> impl Iterator { + self.schema + .active_fields() + .map(|(index, info)| (info.name(alias), self.tuple.data_at(index).clone())) + } + pub fn coalesce(&self) -> impl Iterator { + self.schema + .active_fields() + .map(|(index, info)| self.tuple.data_at(index).clone()) + } + pub fn pick(&self, fields: &[String]) -> impl Iterator { + self.schema.active_fields().filter_map(|(index, info)| { + if fields.contains(&info.name) { + Some(self.tuple.data_at(index).clone()) + } else { + None + } + }) + } + pub fn join(&self, other: &Self) -> impl Iterator { + self.fields().chain(other.fields()).cloned() + } +} + +impl Debug for SchemaTuple<'_, TupleValue> { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_list() + .entries(self.schema.active_fields().map(|(index, info)| { + format!("{}: {}", info.name(&None), self.tuple.data_at(index)) + })) + .finish() + } +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FieldInfo { + /// The field's name. + name: String, + /// Whether the field is active, that is, not eliminated by, e.g., + /// a projection. + active: bool, + // Maybe add type information here, too. +} + +impl FieldInfo { + fn new(name: String) -> Self { + Self { name, active: true } + } + fn name(&self, alias: &Option) -> String { + let name = alias + .as_ref() + .map(|alias| format!("{}.{}", alias, self.name)) + .unwrap_or_else(|| self.name.clone()); + if self.active { + name + } else { + format!("{name}*") + } + } +} + +type Index = usize; + +#[derive(Clone, PartialEq, Eq)] +pub struct TupleSchema { + fields: Vec, +} + +impl TupleSchema { + pub fn new>(fields: impl IntoIterator) -> Self { + Self { + fields: fields + .into_iter() + .map(|name| FieldInfo::new(name.into())) + .collect(), + } + } + pub fn empty() -> Self { + Self { fields: vec![] } + } + /// Only the active fields are included in the count. + pub fn len(&self) -> usize { + self.fields.iter().filter(|info| info.active).count() + } + pub fn is_empty(&self) -> bool { + self.fields.is_empty() + } + /// Includes the active and inactive fields in the count. + pub fn full_len(&self) -> usize { + self.fields.len() + } + fn is_coalesced(&self) -> bool { + !self.fields.iter().any(|info| !info.active) + } + fn coalesce(&self) -> Self { + self.fields + .iter() + .filter(|info| info.active) + .cloned() + .collect() + } + fn active_fields(&self) -> impl Iterator { + self.fields + .iter() + .enumerate() + .filter(|(_index, info)| info.active) + } + fn all_fields(&self) -> impl Iterator { + self.fields.iter().enumerate() + } + pub fn field_names(&self, alias: &Option) -> impl Iterator { + self.active_fields().map(|(_index, info)| info.name(alias)) + } + pub fn all_field_names(&self, alias: &Option) -> impl Iterator { + self.all_fields().map(|(_index, info)| info.name(alias)) + } + fn select(&self) -> Self { + self.clone() + } + /// We mark all fields as inactive, that is, we forget about them. + fn forget(&self) -> Self { + self.fields + .iter() + .map(|info| FieldInfo { + name: info.name.clone(), + active: false, + }) + .collect() + } + /// In contrast to the `project` method, this method does not remove fields + /// from the schema but marks them as inactive, thereby not coalescing the + /// schema and the order of fields. Optionally, you can rename a field by + /// providing an alias/new name/target name as a second element. + fn pick(&self, fields: &Vec<(&String, Option<&String>)>) -> Self { + // For keeping track of duplicated field names. + let mut active = HashSet::with_capacity(fields.len()); + // Don't use active_fields() here because the tuple is not coalesced + // but we only allow picking from the set of active fields though. + self.all_fields() + .map(|(_index, info)| { + // We do not reactivate already inactive fields. + if !info.active { + return info.clone(); + } + if let Some((source_name, target_name)) = + fields.iter().find(|field| *field.0 == info.name) + { + let name = target_name.cloned().unwrap_or_else(|| info.name.clone()); + if !active.contains(&name) { + active.insert(name.clone()); + return FieldInfo::new(name); // Field is active by constructor. + } else { + // We have a duplicated field name, so we mark it as inactive. + return FieldInfo { + name, + active: false, + }; + } + } + // Field is not in the list of fields to pick, so we mark it as inactive. + FieldInfo { + name: info.name.clone(), + active: false, + } + }) + .collect() + } + /// In case of a full projection, we coalesce the schema and remove inactive + /// fields. The order is also redefined according to the projection. + fn project(&self, fields: Vec) -> Self { + fields.into_iter().collect() + } + fn join(&self, other: &Self) -> Self { + let self_active_field_table: HashSet<&String> = + self.active_fields().map(|(_, info)| &info.name).collect(); + // We mark every active field of `other` as inactive if it is + // shadowed by an active field of `self` with the same name. + let other_fields = other.active_fields().map(|(_, info)| { + let mut info = info.clone(); + if self_active_field_table.contains(&info.name) { + info.active = false; + } + info + }); + self.active_fields() + .map(|(_index, info)| info.clone()) + .chain(other_fields) + .collect() + } + fn fields_to_string<'a>( + &self, + fields: impl Iterator, + with_extra: bool, + ) -> String { + let fields = fields + .map(|(_, info)| info.name(&None)) + .collect::>() + .join("|"); + format!("|{fields}|") + } +} + +impl FromIterator for TupleSchema { + fn from_iter>(iter: I) -> Self { + Self { + fields: iter.into_iter().collect(), + } + } +} + +impl FromIterator for TupleSchema { + fn from_iter>(iter: I) -> Self { + Self { + fields: iter.into_iter().map(FieldInfo::new).collect(), + } + } +} + +impl Debug for TupleSchema { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}", self.fields_to_string(self.all_fields(), true)) + } +} + +impl Display for TupleSchema { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}", self.fields_to_string(self.active_fields(), false)) + } +} + +/// The physical schema of a keyed DBSP relation: which fields make up its +/// `TupleKey`, which make up its `TupleValue`, and at which index each sits. +/// +/// Unlike a [`RelationType`](crate::relational::relation::RelationType), +/// this exists at runtime: An operator needs it to read a field of the row +/// in hand by name. +/// Unlike a [`TableSchema`], it describes one *stream* rather than one table: +/// every operator that reshapes rows derives a new one (see +/// [`select`](Self::select), [`join`](Self::join), …), so its +/// [`name`](Self::name) is a transformation trace rather than an identity. +#[derive(Clone, Debug)] +pub struct StreamSchema { + /// Not a real name to reference the relation but more like a transformation + /// trace. Real names are handled by variable names. + pub name: String, + pub key: TupleSchema, + pub tuple: TupleSchema, +} + +impl StreamSchema { + pub fn is_coalesced(&self) -> bool { + self.key.is_coalesced() && self.tuple.is_coalesced() + } + pub fn coalesce(&self) -> Self { + Self { + name: format!("[{}-coalesced]", self.name), + key: self.key.coalesce(), + tuple: self.tuple.coalesce(), + } + } + /// Just clones the current schema, as selections do not alter the schema. + pub fn select(&self) -> Self { + Self { + name: format!("[{}-selected]", self.name), + key: self.key.clone(), + tuple: self.tuple.clone(), + } + } + pub fn pick(&self, fields: &Vec<(&String, Option<&String>)>) -> Self { + Self { + name: format!("[{}-picked]", self.name), + // To keep the `ProjectionExpr`'s semantics consistent, + // we erase the key here, too, as we do for the full projection below. + key: self.key.forget(), + tuple: self.tuple.pick(fields), + } + } + pub fn project(&self, fields: Vec) -> Self { + Self { + name: format!("[{}-projected]", self.name), + key: TupleSchema::empty(), + tuple: self.tuple.project(fields), + } + } + pub fn join(&self, other: &Self, key_fields: impl IntoIterator) -> Self { + Self { + name: format!("[{}-{}-joined]", self.name, other.name), + key: key_fields.into_iter().collect(), + tuple: self.tuple.join(&other.tuple), + } + } + pub fn anti_join(&self, other: &Self, key_fields: impl IntoIterator) -> Self { + // We do not need to store the key in the schema, as it is not used + // in the anti-join. + Self { + name: format!("{}-{}-anti-joined", self.name, other.name), + key: key_fields.into_iter().collect(), + tuple: self.tuple.clone(), + } + } +} + +impl Display for StreamSchema { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "", self.name) + } +} + +impl PartialEq for StreamSchema { + fn eq(&self, other: &Self) -> bool { + self.key == other.key && self.tuple == other.tuple + } +} + +impl Eq for StreamSchema {} + +/// Where a table's key(s) become *the* key a DBSP circuit indexes by. +/// +/// A [`TableSchema`] may declare several candidate keys, or none; an +/// `OrdIndexedZSet` has exactly one. We take the first declared key. For a +/// FLIR base table that is the row id, the only key guaranteed to be unique, +/// and fall back to the empty key, which indexes every row under the same +/// (empty) key, for a relation that declares none. +impl From<&TableSchema> for StreamSchema { + fn from(table: &TableSchema) -> Self { + let key = table + .primary_keys() + .next() + .map(|primary_key| TupleSchema::new(primary_key.map(Column::name))) + .unwrap_or_else(TupleSchema::empty); + Self { + name: table.name().to_string(), + key, + tuple: TupleSchema::new(table.columns().iter().map(Column::name)), + } + } +} + +/// Bind the fields of one physical row as scalar variables, so that a host +/// scalar fragment (a selection condition, a projection attribute, a join key) +/// can name a column of the row currently being processed. +/// +/// An extension of the host's [`InterpreterContext`] rather than a method on it: +/// the host offers the variable map, but only a backend knows how a row of its +/// own is laid out, which is exactly what a [`TupleSchema`] describes. +pub trait DbspTupleContext { + fn extend_tuple_ctx( + &mut self, + alias: &Option, + schema: &TupleSchema, + tuple: &T, + ); +} + +impl DbspTupleContext for InterpreterContext<'_> { + fn extend_tuple_ctx( + &mut self, + alias: &Option, + schema: &TupleSchema, + tuple: &T, + ) { + self.tuple_vars + .extend(SchemaTuple::new(schema, tuple).named_fields(alias)); + } +} diff --git a/packages/coln-query/src/relational/mod.rs b/packages/coln-query/src/relational/mod.rs index 3d462812..061790c3 100644 --- a/packages/coln-query/src/relational/mod.rs +++ b/packages/coln-query/src/relational/mod.rs @@ -8,43 +8,42 @@ //! Two traits split the concern: //! //! - [`Backend`] — the last *compile* step: a [`ResolvedCode`] plan → a runnable -//! artifact. One impl per execution strategy -//! ([`DbspBackend`](incremental::DbspBackend) incremental, +//! artifact, plus the [`lower`](Backend::lower) pass that first restricts the +//! plan to the operators this backend can execute. One impl per execution +//! strategy ([`DbspBackend`](incremental::DbspBackend) incremental, //! [`BatchBackend`](batch::BatchBackend) eager). //! - [`Runtime`] — the runnable artifact: feed input changes, advance, read //! results. This is where incremental vs batch actually differ — DBSP's -//! `commit` runs one incremental transaction and yields per-step [`Delta`]s; -//! the batch engine recomputes from the accumulated inputs and yields -//! [`Snapshot`]s. +//! `commit` runs one incremental transaction and yields per-step +//! [`DbspOutputDelta`](incremental::dbsp::DbspOutputDelta)s; the batch engine +//! recomputes from the accumulated inputs and yields +//! [`Snapshot`](batch::Snapshot)s. //! -//! Sources are **not** passed in: the plan is self-describing. Every extensional -//! input is a [`SourceExpr`](crate::relational::expr::SourceExpr) leaf, -//! so [`Backend::build`] discovers and wires them from the plan itself. +//! Which extensional inputs to wire comes from the plan itself, since every one +//! of them is a [`SourceExpr`](crate::relational::expr::SourceExpr) leaf. What +//! those leaves *are* arrives alongside as [`SourceSchemas`] — a leaf names its +//! relation without describing it, so a relation the plan references `N` times +//! is described once. The pipeline resolves that against the program's +//! [`Catalog`](catalog::Catalog) before calling [`Backend::build`]. pub mod batch; +pub mod catalog; pub mod expr; pub mod incremental; pub mod relation; - -use std::num::NonZeroUsize; +pub mod schema; use crate::{ - error::{BuildError, RuntimeError}, - host::resolver::ResolvedCode, - relational::expr::{SinkId, SourceId}, + api::deltas::ZRow, + error::{BuildError, LoweringError, RuntimeError}, + host::{QueryIr, resolver::ResolvedCode}, + relational::{ + catalog::SourceSchemas, + expr::{SinkId, SourceId}, + }, }; -use incremental::dbsp::{OrdZSet, ZWeight}; -pub use relation::{RelationRef, RelationSchema, RelationType, TupleKey, TupleValue}; - -/// A change to a result relation since the last [`Runtime::commit`] — a Z-set of -/// ±weighted rows. The natural output of an incremental backend. -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct Delta(pub OrdZSet); - -/// The full current state of a result relation. The natural output of a batch -/// backend. -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct Snapshot(pub OrdZSet); +pub use relation::{RelationRef, TupleValue}; +use std::num::NonZeroUsize; /// The last compile step: a resolved plan → a runnable computation. One impl per /// execution strategy; the plan and (row) scalar evaluation are shared. @@ -52,31 +51,89 @@ pub trait Backend { type Runtime: Runtime; type Error: Into; - fn build(self, threads: NonZeroUsize, plan: ResolvedCode) - -> Result; + /// Rewrite the plan into the operator vocabulary this backend can execute. + /// The default keeps it as it is, for a backend that supports the full + /// [`RelExpr`](expr::RelExpr) vocabulary natively. + /// + /// Runs after logical optimization and, crucially, *before* resolution, so + /// a pass may freely mint nodes: the [`Resolver`](crate::host::resolver) + /// has not assigned variable slots yet, and [`ResolvedCode`] exists to + /// promise [`build`](Self::build) that they have been. + /// + /// This is **not** an [`Optimizer`](crate::optimizer::Optimizer), even + /// though both are semantics-preserving rewrites of the plan. An optimizer + /// is chosen independently of the backend and may always decline to do + /// anything; a lowering is mandatory, and skipping it hands the backend a + /// node it cannot compile. Correctness must not depend on which optimizer + /// the pipeline was configured with. + /// + /// Takes `&self` rather than `self` so it can run before + /// [`build`](Self::build) consumes the backend. + fn lower(&self, plan: QueryIr) -> Result { + Ok(plan) + } + + /// `sources` describes the plan's [`SourceExpr`](expr::SourceExpr) leaves, + /// which name their relations without describing them. + /// + /// A backend receives the *resolved* schemas rather than the + /// [`Catalog`](catalog::Catalog) they came from: the pipeline consults the + /// catalog once ([`resolve_sources`](catalog::resolve_sources)) and hands + /// the result on, so a backend never has to reach back into a frontend's + /// data structure — and, for an incremental one, could not, since the owned + /// schemas have to cross into a `Send + 'static` circuit constructor. + fn build( + self, + threads: NonZeroUsize, + plan: ResolvedCode, + sources: SourceSchemas, + ) -> Result; } /// A runnable computation. `feed` stages input changes, `commit` advances the /// computation, `output` reads a result. Incremental and batch backends differ /// only in [`Runtime::Output`] and in how `commit` honors it. pub trait Runtime { - /// The natural result form: [`Delta`] (incremental) or [`Snapshot`] (batch). + /// The natural result form: + /// [`DbspOutputDelta`](incremental::dbsp::DbspOutputDelta) (incremental) or + /// [`Snapshot`](batch::Snapshot) (batch). type Output; /// A runtime error. - type Error: Into; + type Error: Into + std::fmt::Debug; - /// Stage input value tuples (with z-weights) for a named source. The tuple - /// key is derived from the source schema, so only values are supplied. + /// Stage input rows (with z-weights) for a named source. The tuple + /// key is derived from the source schema, so only the row is supplied. + /// Returns `Ok(true)` if the input source is known and data has been fed. + /// Otherwise, it returns `Ok(false)`. + #[must_use = "Do not miss a missed update"] fn feed( &mut self, source: &SourceId, - rows: impl IntoIterator, - ) -> Result<(), Self::Error>; + rows: impl IntoIterator, + ) -> Result; + /// Advance the computation over everything fed since the last commit. fn commit(&mut self) -> Result<(), Self::Error>; + /// Read a result relation by the name of the [`OutputExpr`](expr::OutputExpr) - /// tap that produced it. Errors if no *readable* output carries that name — + /// tap that produced it. Errors if no *readable* output carries that name, /// e.g. the name is unknown, or it belongs to a print-only - /// [`OutputKind::Cli`](expr::OutputKind::Cli) tap. + /// [`OutputKind::Cli`](expr::OutputKind::Cli) tap. Contains all changes + /// since the last call to [`commit`](Self::commit). fn output(&self, out: &SinkId) -> Result; + /// List all [`OutputExpr`](expr::OutputExpr) by their name (a [`SinkId`]). + fn list_outputs(&self) -> impl Iterator; + /// Get an iterator over all known [`outputs`](Self::output). A shortcut for + /// inquiring all outputs (through calling [`output`](Self::output) for + /// all valid [`SinkId`]s) for new results after a call to + /// [`commit`](Self::commit). + fn all_outputs(&self) -> impl Iterator { + self.list_outputs().map(|sink_id| { + ( + sink_id, + self.output(sink_id) + .expect("list_outputs() impl must only return valid sink ids"), + ) + }) + } } diff --git a/packages/coln-query/src/relational/relation.rs b/packages/coln-query/src/relational/relation.rs index 797619c3..afc61437 100644 --- a/packages/coln-query/src/relational/relation.rs +++ b/packages/coln-query/src/relational/relation.rs @@ -2,13 +2,13 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::scalarial::ScalarTypedValue; -use crate::{error::SyntaxError, scalarial::ScalarType}; +use crate::relational::schema::TableSchema; +use crate::scalarial::{ScalarType, ScalarTypedValue}; use dbsp::{never_none, never_roaring_filter}; use std::{ any::Any, cell::RefCell, - collections::{HashMap, HashSet}, + collections::HashMap, fmt::{self, Debug, Display}, rc::Rc, }; @@ -27,64 +27,8 @@ pub trait Tuple: FromIterator { .data() .map(|field| field.to_string()) .collect::>() - .join(" | "); - format!("| {fields} |") - } -} - -pub struct SchemaTuple<'a, T> { - schema: &'a TupleSchema, - tuple: &'a T, -} - -impl<'a, T: Tuple> SchemaTuple<'a, T> { - pub fn new(schema: &'a TupleSchema, tuple: &'a T) -> Self { - Self { schema, tuple } - } - pub fn fields(&self) -> impl Iterator { - self.schema - .active_fields() - .map(|(index, info)| self.tuple.data_at(index)) - } - pub fn all_fields(&self) -> impl Iterator { - self.schema - .all_fields() - .map(|(index, _info)| self.tuple.data_at(index)) - } - pub fn named_fields( - &self, - alias: &Option, - ) -> impl Iterator { - self.schema - .active_fields() - .map(|(index, info)| (info.name(alias), self.tuple.data_at(index).clone())) - } - pub fn coalesce(&self) -> impl Iterator { - self.schema - .active_fields() - .map(|(index, info)| self.tuple.data_at(index).clone()) - } - pub fn pick(&self, fields: &[String]) -> impl Iterator { - self.schema.active_fields().filter_map(|(index, info)| { - if fields.contains(&info.name) { - Some(self.tuple.data_at(index).clone()) - } else { - None - } - }) - } - pub fn join(&self, other: &Self) -> impl Iterator { - self.fields().chain(other.fields()).cloned() - } -} - -impl Debug for SchemaTuple<'_, TupleValue> { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_list() - .entries(self.schema.active_fields().map(|(index, info)| { - format!("{}: {}", info.name(&None), self.tuple.data_at(index)) - })) - .finish() + .join("|"); + format!("[{fields}]") } } @@ -149,51 +93,6 @@ impl Display for TupleValue { } } -#[derive( - Clone, - Default, - Debug, - Eq, - PartialEq, - Ord, - PartialOrd, - Hash, - size_of::SizeOf, - rkyv::Archive, - rkyv::Serialize, - rkyv::Deserialize, -)] -#[archive_attr(derive(Ord, Eq, PartialEq, PartialOrd))] -pub struct TupleKey { - pub data: Vec, -} - -never_none!(TupleKey); -never_roaring_filter!(TupleKey); - -impl> FromIterator for TupleKey { - fn from_iter>(iter: I) -> Self { - Self { - data: iter.into_iter().map(|v| v.into()).collect(), - } - } -} - -impl Tuple for TupleKey { - fn data_at(&self, index: usize) -> &ScalarTypedValue { - &self.data[index] - } - fn data(&self) -> impl Iterator { - self.data.iter() - } -} - -impl Display for TupleKey { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.data_to_string()) - } -} - /// Currently unused. #[derive(Debug, Hash, Eq, PartialEq, Clone, PartialOrd, Ord)] struct Identifier { @@ -209,22 +108,25 @@ impl Display for Identifier { /// Convenience type alias for a reference to a [`Relation`]. pub type RelationRef = Rc>; -pub fn new_relation(schema: RelationSchema, inner: R) -> RelationRef { - Rc::new(RefCell::new(Relation::new(schema, inner))) +pub fn new_relation(inner: R) -> RelationRef { + Rc::new(RefCell::new(Relation::new(inner))) } -/// The backend-neutral payload of a [`Relation`] — a type-erased envelope over a -/// backend's concrete relation representation (a DBSP `StreamWrapper`, a batch -/// Z-set, …). +/// The backend-neutral payload of a [`Relation`]: A type-erased envelope over a +/// backend's concrete relation representation (a +/// [`DbspRelation`](super::incremental::dbsp::DbspRelation), a batch Z-set, …). /// -/// This is layer 3 of the multi-backend split ("the only place `StreamWrapper` -/// vs a batch Z-set actually differs"). It carries **no algebra**: relational -/// operations live in each backend's +/// It carries **no algebra**: relational operations live in each backend's /// [`RelExprVisitor`](crate::relational::expr::RelExprVisitor), which recovers /// its own concrete type via [`Relation::downcast_ref`]. Keeping this trait /// algebra-free is what stops any backend's operator vocabulary from leaking /// into the host `Value` type. -pub trait RelationData: Any { +/// +/// [`Display`] and [`Debug`] are required because the host layer prints +/// relations ([`Value::Relation`](crate::host::variable::Value)) without knowing +/// what one is: how a relation describes itself is the one thing a backend has +/// to tell the layer above, and its schema is what it says. +pub trait RelationData: Any + Display + Debug { fn as_any(&self) -> &dyn Any; fn clone_box(&self) -> Box; } @@ -235,306 +137,21 @@ impl Clone for Box { } } -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct FieldInfo { - /// The field's name. - name: String, - /// Whether the field is active, that is, not eliminated by, e.g., a - /// projection. - active: bool, - // Maybe add type information here, too. -} - -impl FieldInfo { - fn new(name: String) -> Self { - Self { name, active: true } - } - fn name(&self, alias: &Option) -> String { - let name = alias - .as_ref() - .map(|alias| format!("{}.{}", alias, self.name)) - .unwrap_or_else(|| self.name.clone()); - if self.active { - name - } else { - format!("{name}*") - } - } -} - -type Index = usize; - -#[derive(Clone, PartialEq, Eq)] -pub struct TupleSchema { - fields: Vec, -} - -impl TupleSchema { - pub fn new>(fields: impl IntoIterator) -> Self { - Self { - fields: fields - .into_iter() - .map(|name| FieldInfo::new(name.into())) - .collect(), - } - } - pub fn empty() -> Self { - Self { fields: vec![] } - } - /// Only the active fields are included in the count. - pub fn len(&self) -> usize { - self.fields.iter().filter(|info| info.active).count() - } - pub fn is_empty(&self) -> bool { - self.fields.is_empty() - } - /// Includes the active and inactive fields in the count. - pub fn full_len(&self) -> usize { - self.fields.len() - } - fn is_coalesced(&self) -> bool { - !self.fields.iter().any(|info| !info.active) - } - fn coalesce(&self) -> Self { - self.fields - .iter() - .filter(|info| info.active) - .cloned() - .collect() - } - fn active_fields(&self) -> impl Iterator { - self.fields - .iter() - .enumerate() - .filter(|(_index, info)| info.active) - } - fn all_fields(&self) -> impl Iterator { - self.fields.iter().enumerate() - } - pub fn field_names(&self, alias: &Option) -> impl Iterator { - self.active_fields().map(|(_index, info)| info.name(alias)) - } - pub fn all_field_names(&self, alias: &Option) -> impl Iterator { - self.all_fields().map(|(_index, info)| info.name(alias)) - } - fn select(&self) -> Self { - self.clone() - } - /// We mark all fields as inactive, that is, we forget about them. - fn forget(&self) -> Self { - self.fields - .iter() - .map(|info| FieldInfo { - name: info.name.clone(), - active: false, - }) - .collect() - } - /// In contrast to the `project` method, this method does not remove fields - /// from the schema but marks them as inactive, thereby not coalescing the - /// schema and the order of fields. Optionally, you can rename a field by - /// providing an alias/new name/target name as a second element. - fn pick(&self, fields: &Vec<(&String, Option<&String>)>) -> Self { - // For keeping track of duplicated field names. - let mut active = HashSet::with_capacity(fields.len()); - // Don't use active_fields() here because the tuple is not coalesced - // but we only allow picking from the set of active fields though. - self.all_fields() - .map(|(_index, info)| { - // We do not reactivate already inactive fields. - if !info.active { - return info.clone(); - } - if let Some((source_name, target_name)) = - fields.iter().find(|field| *field.0 == info.name) - { - let name = target_name.cloned().unwrap_or_else(|| info.name.clone()); - if !active.contains(&name) { - active.insert(name.clone()); - return FieldInfo::new(name); // Field is active by constructor. - } else { - // We have a duplicated field name, so we mark it as inactive. - return FieldInfo { - name, - active: false, - }; - } - } - // Field is not in the list of fields to pick, so we mark it as inactive. - FieldInfo { - name: info.name.clone(), - active: false, - } - }) - .collect() - } - /// In case of a full projection, we coalesce the schema and remove inactive - /// fields. The order is also redefined according to the projection. - fn project(&self, fields: Vec) -> Self { - fields.into_iter().collect() - } - fn join(&self, other: &Self) -> Self { - let self_active_field_table: HashSet<&String> = - self.active_fields().map(|(_, info)| &info.name).collect(); - // We mark every active field of `other` as inactive if it is - // shadowed by an active field of `self` with the same name. - let other_fields = other.active_fields().map(|(_, info)| { - let mut info = info.clone(); - if self_active_field_table.contains(&info.name) { - info.active = false; - } - info - }); - self.active_fields() - .map(|(_index, info)| info.clone()) - .chain(other_fields) - .collect() - } - fn fields_to_string<'a>( - &self, - fields: impl Iterator, - with_extra: bool, - ) -> String { - let fields = fields - .map(|(_, info)| info.name(&None)) - .collect::>() - .join(" | "); - format!("| {fields} |") - } -} - -impl FromIterator for TupleSchema { - fn from_iter>(iter: I) -> Self { - Self { - fields: iter.into_iter().collect(), - } - } -} - -impl FromIterator for TupleSchema { - fn from_iter>(iter: I) -> Self { - Self { - fields: iter.into_iter().map(FieldInfo::new).collect(), - } - } -} - -impl Debug for TupleSchema { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.fields_to_string(self.all_fields(), true)) - } -} - -impl Display for TupleSchema { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.fields_to_string(self.active_fields(), false)) - } -} - -/// A [`Relation`]'s schema is a set of fields and we store the index of each. -/// Unlike [`RelationType`], this exists at runtime to, e.g., be able to index -/// fields by name. -#[derive(Clone, Debug)] -pub struct RelationSchema { - /// Not a real name to reference the relation but more like a transformation - /// trace. Real names are handled by variable names. - pub name: String, - pub key: TupleSchema, - pub tuple: TupleSchema, -} - -impl RelationSchema { - pub fn new>( - name: T, - tuple_fields: impl IntoIterator, - key_fields: impl IntoIterator, - ) -> Result { - Ok(Self { - name: name.into(), - key: TupleSchema::new(key_fields), - tuple: TupleSchema::new(tuple_fields), - }) - } - pub fn is_coalesced(&self) -> bool { - self.key.is_coalesced() && self.tuple.is_coalesced() - } - pub fn coalesce(&self) -> Self { - Self { - name: format!("[{}-coalesced]", self.name), - key: self.key.coalesce(), - tuple: self.tuple.coalesce(), - } - } - /// Just clones the current schema, as selections do not alter the schema. - pub fn select(&self) -> Self { - Self { - name: format!("[{}-selected]", self.name), - key: self.key.clone(), - tuple: self.tuple.clone(), - } - } - pub fn pick(&self, fields: &Vec<(&String, Option<&String>)>) -> Self { - Self { - name: format!("[{}-picked]", self.name), - // To keep the `ProjectionExpr`'s semantics consistent, - // we erase the key here, too, as we do for the full projection below. - key: self.key.forget(), - tuple: self.tuple.pick(fields), - } - } - pub fn project(&self, fields: Vec) -> Self { - Self { - name: format!("[{}-projected]", self.name), - key: TupleSchema::empty(), - tuple: self.tuple.project(fields), - } - } - pub fn join(&self, other: &Self, key_fields: impl IntoIterator) -> Self { - Self { - name: format!("[{}-{}-joined]", self.name, other.name), - key: key_fields.into_iter().collect(), - tuple: self.tuple.join(&other.tuple), - } - } - pub fn anti_join(&self, other: &Self, key_fields: impl IntoIterator) -> Self { - // We do not need to store the key in the schema, as it is not used - // in the anti-join. - Self { - name: format!("{}-{}-anti-joined", self.name, other.name), - key: key_fields.into_iter().collect(), - tuple: self.tuple.clone(), - } - } -} - -impl Display for RelationSchema { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "", self.name) - } -} - -impl PartialEq for RelationSchema { - fn eq(&self, other: &Self) -> bool { - self.key == other.key && self.tuple == other.tuple - } -} - -impl Eq for RelationSchema {} - +/// A relation flowing through a plan, as the host layer sees it: an opaque +/// handle. Everything about *how* the relation is represented, including the +/// physical schema its rows are laid out by, belongs to the backend and lives +/// inside [`RelationData`], because a schema that says "key columns" already +/// says which backend is running. #[derive(Clone)] pub struct Relation { - /// The schema of the relation. We need to track it on a per-relation basis - /// because it may change during execution. - pub schema: RelationSchema, /// The backend's concrete relation, type-erased. Access it from within a /// backend via [`Self::downcast_ref`]. inner: Box, } impl Relation { - pub fn new(schema: RelationSchema, inner: R) -> Self { + pub fn new(inner: R) -> Self { Self { - schema, inner: Box::new(inner), } } @@ -551,18 +168,20 @@ impl Relation { impl Display for Relation { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.schema) + write!(f, "{}", self.inner) } } impl Debug for Relation { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.schema) + write!(f, "{:?}", self.inner) } } -/// Unlike [`RelationSchema`], this does not exist at runtime but is only used -/// during static analysis. +/// The set of columns a plan's expressions may name at a given point, with each +/// column's type. Unlike a [`TableSchema`], this does not exist at runtime but +/// is only used during static analysis. Unlike a backend's physical schema, +/// it says nothing about keys or field order. #[derive(Debug, Clone, PartialEq, Eq)] pub struct RelationType { fields: HashMap, @@ -587,6 +206,8 @@ impl Display for RelationType { } impl RelationType { + // TODO: Maybe the relation type should be position-aware and allow for + // duplicated columns sharing the same name.. pub fn join(self, other: Self) -> Self { // We start with other to have duplicate fields' types be taken from self. let mut fields = other.fields; @@ -657,13 +278,12 @@ impl<'a> FromIterator<(&'a String, ScalarType)> for RelationType { } } -impl From<&RelationSchema> for RelationType { - fn from(value: &RelationSchema) -> Self { +impl From<&TableSchema> for RelationType { + fn from(value: &TableSchema) -> Self { let fields = value - .tuple - .field_names(&None) - // Until we have proper types in Datalog, we assign the dummy null type. - .map(|name| (name, ScalarType::Null)) + .columns() + .iter() + .map(|column| (column.name().to_string(), column.scalar_type())) .collect(); Self { fields } } diff --git a/packages/coln-query/src/relational/schema.rs b/packages/coln-query/src/relational/schema.rs new file mode 100644 index 00000000..2c0f01e7 --- /dev/null +++ b/packages/coln-query/src/relational/schema.rs @@ -0,0 +1,207 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! What a table *is*, said in a way no backend and no frontend owns. +//! +//! This is the middle of three schema vocabularies coln-query speaks: +//! +//! 1. A **frontend** schema knows how the table got here. coln's FLIR frontend +//! has `BaseTableSchema`, which carries one column view per engine (compiler, +//! store, query) and the index translations between them. It converts *into* +//! the schema in this module. +//! 2. The **neutral** [`TableSchema`] here: named, typed columns in one order, +//! plus the (compound) primary key(s) over them. This is what a +//! [`Catalog`](super::catalog::Catalog) answers with, what the type resolver +//! reads, and what a [`Backend`](super::Backend) is handed. +//! 3. A **backend** schema knows how the table is physically laid out for +//! execution. The DBSP backend has +//! [`StreamSchema`](super::incremental::schema::StreamSchema), whose keyed +//! `(TupleKey, TupleValue)` shape follows from `OrdIndexedZSet` rather than +//! from anything about the table itself. Each backend converts the neutral +//! schema into its own and keeps it inside its own relation representation. +//! +//! Layer 2 is deliberately the poorest of the three: a table has columns, +//! types, and key(s), and a relation may have several candidate keys or none. +//! Which one — if any — becomes a physical index is layer 3's decision. + +use crate::{ + relational::expr::{SinkId, SourceId}, + scalarial::ScalarType, +}; +use std::fmt::{self, Display}; + +/// An identifier that uniquely identifies a table (globally across the store). +#[derive(Eq, PartialEq, Hash, Debug, Clone)] +pub struct EntityRef { + inner: String, +} + +impl EntityRef { + pub fn id(&self) -> &str { + &self.inner + } +} + +impl Display for EntityRef { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.inner.fmt(f) + } +} + +/// The other direction of the same identity: a plan's +/// [`OutputExpr`](crate::relational::expr::OutputExpr) leaf names its derived +/// view by a [`SinkId`] built from that table's name, so a `SinkId` +/// and the `TableRef` of the table it names hold the same string. +/// This is what lets a [`Catalog`](super::catalog::Catalog) lookup reach a +/// `TableRef`-keyed map. +impl From<&SinkId> for EntityRef { + fn from(value: &SinkId) -> Self { + EntityRef { + inner: value.0.clone(), + } + } +} + +/// The other direction of the same identity: a plan's +/// [`SourceExpr`](crate::relational::expr::SourceExpr) leaf names its base +/// table by a [`SourceId`] built from that table's name, so a `SourceId` +/// and the `TableRef` of the table it names hold the same string. +/// This is what lets a [`Catalog`](super::catalog::Catalog) lookup reach a +/// `TableRef`-keyed map. +impl From<&SourceId> for EntityRef { + fn from(value: &SourceId) -> Self { + EntityRef { + inner: value.0.clone(), + } + } +} + +impl From<&str> for EntityRef { + fn from(value: &str) -> Self { + EntityRef { + inner: value.to_string(), + } + } +} + +impl From for EntityRef { + fn from(value: String) -> Self { + EntityRef { inner: value } + } +} + +impl From<&EntityRef> for SourceId { + fn from(value: &EntityRef) -> Self { + SourceId(value.inner.clone()) + } +} + +/// The backend-neutral and frontend-neutral description of one table: +/// Its columns and the key(s) over them. See the [module docs](self) for the +/// layer this sits in. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct TableSchema { + /// The table's unique identifier/name. + name: EntityRef, + /// All fields of the table in their physical order. + columns: Vec, + /// The list of (possibly compound) primary keys into the table, specified + /// as indexes into the schema's [`columns`](Self::columns). + /// + /// Plural and possibly empty, unlike the single key a keyed backend needs: + /// picking one of these (or synthesizing one) is that backend's business. + primary_keys: Vec>, +} + +impl TableSchema { + pub fn new(name: EntityRef, columns: Vec, primary_keys: Vec>) -> Self { + debug_assert!( + primary_keys + .iter() + .flatten() + .all(|idx| *idx < columns.len()), + "primary key of table {name} indexes a column it does not have" + ); + Self { + name, + columns, + primary_keys, + } + } + pub fn name(&self) -> &EntityRef { + &self.name + } + pub fn columns(&self) -> &[Column] { + &self.columns + } + /// The (compound) primary key(s) expressed in [`columns`](Self::columns). + pub fn primary_keys(&self) -> impl Iterator> { + self.primary_keys.iter().map(|primary_key| { + primary_key.iter().map(|idx| { + self.columns + .get(*idx) + .expect("primary key indexes a column of its own schema") + }) + }) + } + /// Everything but the name: the typed columns and the key(s) over them, as + /// `(a: uint, b: iint) key(a)`. What a plan printer wants, since a source + /// leaf has already named the relation by the time its schema is rendered. + pub fn shape(&self) -> impl Display { + Shape(self) + } +} + +impl Display for TableSchema { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}{}", self.name, self.shape()) + } +} + +struct Shape<'a>(&'a TableSchema); + +impl Display for Shape<'_> { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let joined = |names: Vec| names.join(", "); + let columns = joined(self.0.columns.iter().map(Column::to_string).collect()); + write!(f, "({columns})")?; + self.0.primary_keys().try_for_each(|primary_key| { + let stringified = joined( + primary_key + .map(|column| column.name().to_string()) + .collect(), + ); + write!(f, " key({stringified})") + }) + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Column { + /// The column's name. + name: String, + /// Ihe column's (scalar) type. + scalar_type: ScalarType, +} + +impl Column { + pub fn new, T: Into>(name: N, scalar_type: T) -> Self { + Self { + name: name.into(), + scalar_type: scalar_type.into(), + } + } + pub fn name(&self) -> &str { + &self.name + } + pub fn scalar_type(&self) -> ScalarType { + self.scalar_type + } +} + +impl Display for Column { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}: {}", self.name, self.scalar_type) + } +} diff --git a/packages/coln-query/src/test_helper.rs b/packages/coln-query/src/test_helper.rs deleted file mode 100644 index 8e8fdf84..00000000 --- a/packages/coln-query/src/test_helper.rs +++ /dev/null @@ -1,394 +0,0 @@ -// SPDX-FileCopyrightText: 2026 Coln contributors -// -// SPDX-License-Identifier: Apache-2.0 OR MIT - -//! This module is only available if compiling with the `test` feature. -//! It provides helpers for testing and benchmarking purposes. - -use crate::{ - relational::{RelationSchema, TupleKey, TupleValue, incremental::dbsp::ZWeight}, - scalarial::ScalarTypedValue, -}; -use std::fmt::Debug; - -/// Convenience: turn input entities (with per-row z-weights) into the value rows -/// that [`crate::relational::Runtime::feed`] expects. -pub fn rows( - entities: impl IntoIterator, -) -> Vec<(TupleValue, ZWeight)> { - entities - .into_iter() - .map(|(entity, weight)| (entity.into(), weight)) - .collect() -} - -/// Convenience: turn input entities into value rows all carrying `weight`. -pub fn rows_with_weight( - entities: impl IntoIterator, - weight: ZWeight, -) -> Vec<(TupleValue, ZWeight)> { - entities - .into_iter() - .map(|entity| (entity.into(), weight)) - .collect() -} - -pub trait InputEntity: Into + Into + Clone + Debug { - fn schema() -> RelationSchema; -} - -#[derive(Clone, Debug)] -pub struct Person { - person_id: u64, - name: String, - age: u64, - profession_id: u64, -} - -impl InputEntity for Person { - fn schema() -> RelationSchema { - RelationSchema::new( - "person", - ["person_id", "name", "age", "profession_id"], - ["person_id"], - ) - .expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(person: Person) -> Self { - TupleKey { - data: vec![ScalarTypedValue::Uint(person.person_id)], - } - } -} - -impl From for TupleValue { - fn from(person: Person) -> Self { - TupleValue { - data: vec![ - ScalarTypedValue::Uint(person.person_id), - ScalarTypedValue::String(person.name), - ScalarTypedValue::Uint(person.age), - ScalarTypedValue::Uint(person.profession_id), - ], - } - } -} - -#[derive(Clone, Debug)] -pub struct Profession { - profession_id: u64, - name: String, -} - -impl InputEntity for Profession { - fn schema() -> RelationSchema { - RelationSchema::new("profession", ["profession_id", "name"], ["profession_id"]) - .expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(profession: Profession) -> Self { - TupleKey { - data: vec![ScalarTypedValue::Uint(profession.profession_id)], - } - } -} - -impl From for TupleValue { - fn from(profession: Profession) -> Self { - TupleValue { - data: vec![ - ScalarTypedValue::Uint(profession.profession_id), - ScalarTypedValue::String(profession.name), - ], - } - } -} - -pub fn person_profession_data() -> [(Vec, Vec); 1] { - [( - vec![ - Person { - person_id: 0, - name: "Alice".to_string(), - age: 20, - profession_id: 0, - }, - Person { - person_id: 1, - name: "Bob".to_string(), - age: 30, - profession_id: 1, - }, - Person { - person_id: 2, - name: "Charlie".to_string(), - age: 40, - profession_id: 0, - }, - ], - vec![ - Profession { - profession_id: 0, - name: "Engineer".to_string(), - }, - Profession { - profession_id: 1, - name: "Doctor".to_string(), - }, - ], - )] -} - -#[derive(Copy, Clone, Debug)] -pub struct PlainRelation { - a: u64, - b: u64, - c: u64, -} - -impl PlainRelation { - pub fn new(a: u64, b: u64, c: u64) -> Self { - Self { a, b, c } - } - const STEPS: usize = 1; - pub fn test_data_1() -> [Vec; Self::STEPS] { - [vec![ - PlainRelation::new(1, 2, 3), - PlainRelation::new(4, 5, 6), - PlainRelation::new(7, 8, 9), - ]] - } - pub fn test_data_2() -> [Vec; Self::STEPS] { - [vec![PlainRelation::new(1, 2, 3)]] - } - pub fn test_data_3() -> [Vec; Self::STEPS] { - [vec![PlainRelation::new(4, 5, 6)]] - } -} - -impl InputEntity for PlainRelation { - fn schema() -> RelationSchema { - RelationSchema::new("plain", ["a", "b", "c"], []).expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(fact: PlainRelation) -> Self { - TupleKey { data: vec![] } - } -} - -impl From for TupleValue { - fn from(fact: PlainRelation) -> Self { - TupleValue { - data: vec![ - ScalarTypedValue::Uint(fact.a), - ScalarTypedValue::Uint(fact.b), - ScalarTypedValue::Uint(fact.c), - ], - } - } -} - -#[derive(Copy, Clone, Debug)] -pub struct Edge { - from: u64, - to: u64, - weight: u64, - active: bool, -} - -impl Edge { - pub fn new(from: u64, to: u64, weight: u64) -> Self { - Self { - from, - to, - weight, - active: true, - } - } -} - -impl InputEntity for Edge { - fn schema() -> RelationSchema { - RelationSchema::new("edges", ["from", "to", "weight", "active"], ["from", "to"]) - .expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(edge: Edge) -> Self { - TupleKey { - data: vec![ - ScalarTypedValue::Uint(edge.from), - ScalarTypedValue::Uint(edge.to), - ], - } - } -} - -impl From for TupleValue { - fn from(edge: Edge) -> Self { - TupleValue { - data: vec![ - ScalarTypedValue::Uint(edge.from), - ScalarTypedValue::Uint(edge.to), - ScalarTypedValue::Uint(edge.weight), - ScalarTypedValue::Bool(edge.active), - ], - } - } -} - -#[derive(Copy, Clone, Debug)] -pub struct SetOp { - rep_id: u64, - ctr: u64, - key: u64, - value: u64, -} - -impl SetOp { - pub fn new(rep_id: u64, ctr: u64, key: u64, value: u64) -> Self { - Self { - rep_id, - ctr, - key, - value, - } - } -} - -impl InputEntity for SetOp { - fn schema() -> RelationSchema { - RelationSchema::new("set", ["RepId", "Ctr", "Key", "Value"], ["RepId", "Ctr"]) - .expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(set_op: SetOp) -> Self { - TupleKey::from_iter([set_op.rep_id, set_op.ctr]) - } -} - -impl From for TupleValue { - fn from(set_op: SetOp) -> Self { - TupleValue::from_iter([set_op.rep_id, set_op.ctr, set_op.key, set_op.value]) - } -} - -#[derive(Copy, Clone, Debug)] -pub struct PredRel { - from_rep_id: u64, - from_ctr: u64, - to_rep_id: u64, - to_ctr: u64, -} - -impl PredRel { - pub fn new(from_rep_id: u64, from_ctr: u64, to_rep_id: u64, to_ctr: u64) -> Self { - Self { - from_rep_id, - from_ctr, - to_rep_id, - to_ctr, - } - } -} - -impl InputEntity for PredRel { - fn schema() -> RelationSchema { - RelationSchema::new( - "pred", - ["FromRepId", "FromCtr", "ToRepId", "ToCtr"], - ["FromRepId", "FromCtr", "ToRepId", "ToCtr"], - ) - .expect("Correct schema definition") - } -} - -impl From for TupleKey { - fn from(pred_rel: PredRel) -> Self { - TupleKey::from_iter([ - pred_rel.from_rep_id, - pred_rel.from_ctr, - pred_rel.to_rep_id, - pred_rel.to_ctr, - ]) - } -} - -impl From for TupleValue { - fn from(pred_rel: PredRel) -> Self { - TupleValue::from_iter([ - pred_rel.from_rep_id, - pred_rel.from_ctr, - pred_rel.to_rep_id, - pred_rel.to_ctr, - ]) - } -} - -/// This function returns test data for an operation history of the MVR CRDT -/// store. The history is as follows. -/// The notation is `set__(, )`. -/// -/// 1. step (just one root operation setting register with key 1 to value 1): -/// -/// ```text -/// set_0_0(1, 1) -/// ``` -/// -/// 2. step (concurrent writes by replica 0 and 1): -/// -/// ```text -/// ---> set_0_1(1, 2) -/// set_0_0(1, 1) -/// ---> set_1_0(1, 3) -/// ``` -/// -/// 3. step (replica 1 does a "merge" operation overwriting the previous -/// conflict): -/// -/// ```text -/// ---> set_0_1(1, 2) -/// set_0_0(1, 1) ---> set_1_2(1, 4) -/// ---> set_1_0(1, 3) -/// ``` -/// -/// 4. step (replica 0 overwrites a not-yet delivered operation): -/// -/// ```text -/// ---> set_0_1(1, 2) -/// set_0_0(1, 1) ---> set_1_2(1, 4) ---> missing ---> set_0_4(1, 6) -/// ---> set_1_0(1, 3) -/// ``` -/// -/// 5. step (replica 0's missing operation arrives): -/// -/// ```text -/// ---> set_0_1(1, 2) -/// set_0_0(1, 1) ---> set_1_2(1, 4) ---> set_0_3(1, 5) ---> set_0_4(1, 6) -/// ---> set_1_0(1, 3) -/// ``` -pub fn mvr_store_operation_history() -> [(Vec, Vec); 5] { - [ - (vec![], vec![SetOp::new(0, 0, 1, 1)]), - ( - vec![PredRel::new(0, 0, 0, 1), PredRel::new(0, 0, 1, 0)], - vec![SetOp::new(0, 1, 1, 2), SetOp::new(1, 0, 1, 3)], - ), - ( - vec![PredRel::new(0, 1, 1, 2), PredRel::new(1, 0, 1, 2)], - vec![SetOp::new(1, 2, 1, 4)], - ), - (vec![PredRel::new(0, 3, 0, 4)], vec![SetOp::new(0, 4, 1, 6)]), - (vec![PredRel::new(1, 2, 0, 3)], vec![SetOp::new(0, 3, 1, 5)]), - ] -} diff --git a/packages/coln-query/src/test_utils.rs b/packages/coln-query/src/test_utils.rs new file mode 100644 index 00000000..7fac9b24 --- /dev/null +++ b/packages/coln-query/src/test_utils.rs @@ -0,0 +1,886 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +//! This module is only available if compiling with the `test` feature. +//! It provides helpers for testing and benchmarking purposes. + +use crate::{ + api::deltas::ZRow, + host::QueryIr, + program::QueryProgram, + relational::{ + TupleValue, + catalog::{Catalog, SourceSchemas}, + expr::SourceId, + incremental::{dbsp::ZWeight, schema::TupleKey}, + schema::{Column, EntityRef, TableSchema}, + }, + scalarial::{ScalarType, ScalarTypedValue}, +}; +use std::borrow::Cow; +use std::fmt::Debug; + +/// Builders for FLIR by hand, for the realms no `.json` fixture covers — a +/// monitored rule today, since coln-compiler does not emit one yet. +/// +/// These are the [`ir`](coln_flir_rs::ir) halves only. A caller that also needs +/// the *lowering's* +/// own vocabulary (`FriendlyVar` and friends, to drive +/// [`FlirProgram`](crate::api::query::FlirProgram) internals directly) keeps that +/// next to those tests; what lives here is the part every such test states the +/// same way. +pub mod flir { + use coln_flir_rs::ir; + + /// A base table `name` whose columns are given as `(name, type)` pairs, in + /// order, and which declares no primary key of its own. + pub fn table_entry(name: &str, columns: Vec<(&str, ir::ColType)>) -> ir::TableEntry { + ir::TableEntry { + path: ir::Path::from(name), + table: ir::Schema { + entity_variant: ir::EntityVariant::Table, + columns: columns + .into_iter() + .map(|(name, col_type)| ir::ColumnEntry { + path: ir::Path::from(name), + col_type, + }) + .collect(), + primary_key: None, + }, + } + } + + /// A rule `antecedents => consequents` of the given [variant](ir::RuleVariant), + /// over the given variables. + /// + /// Both sides are [`Prop`](ir::Prop)s rather than [`Atom`](ir::Atom)s, so a + /// side may carry conditions as well as atoms — which is what a rule needs to + /// say anything more interesting than "these relations join". See + /// [`atom_props`] for the atoms-only case. + pub fn rule_entry<'a>( + name: &str, + variant: ir::RuleVariant, + vars: impl IntoIterator, + antecedents: Vec, + consequents: Vec, + ) -> ir::RuleEntry { + let (var_names, var_types) = vars + .into_iter() + .map(|(name, col_type)| (ir::Path::from(name), col_type)) + .unzip(); + ir::RuleEntry { + path: ir::Path::from(name), + rule: ir::Rule { + rule_variant: variant, + var_names, + var_types, + antecedents, + consequents, + }, + } + } + + /// One side of a rule that is nothing but atoms. + pub fn atom_props(atoms: Vec) -> Vec { + atoms.into_iter().map(atom_prop).collect() + } + + pub fn atom_prop(atom: ir::Atom) -> ir::Prop { + ir::Prop::Atom { atom } + } + + /// A condition as one side of a rule sees it. [`equality`] on its own is + /// what [`ConjunctiveQuery`](crate::api::query) holds, before the split into + /// atoms and conditions has happened. + pub fn eq_prop(equality: ir::Equality) -> ir::Prop { + ir::Prop::Eq { equality } + } + + /// An atom over `entity`, optionally binding its row id, and binding or + /// constraining the columns named by index in `values`. + pub fn atom( + entity: &str, + row_id: Option, + values: Vec<(ir::ColumnIdx, ir::Term)>, + ) -> ir::Atom { + ir::Atom { + entity: ir::Path::from(entity), + row_id, + values: values + .into_iter() + .map(|(column, term)| ir::ValueEntry { column, term }) + .collect(), + } + } + + pub fn builtin_int() -> ir::ColType { + ir::ColType::BuiltinTy { + builtin_ty: ir::BuiltinTy::BuiltinInt, + } + } + + pub fn row_id_of(entity: &str) -> ir::ColType { + ir::ColType::RowId { + path: ir::Path::from(entity), + } + } + + pub fn var_term(index: ir::VarIdx) -> ir::Term { + ir::Term::Var { index } + } + + pub fn lit_term(value: i64) -> ir::Term { + ir::Term::Lit { + lit: ir::Lit::Int { + value: value.try_into().unwrap(), + }, + } + } + + pub fn equality(left: ir::Term, right: ir::Term) -> ir::Equality { + ir::Equality { left, right } + } +} + +/// A realm with a single *monitored* rule, so that the one path +/// [`TxOutcome::SoftViolationsDelta`](crate::api::transaction::TxOutcome::SoftViolationsDelta) +/// describes can actually be driven. +/// +/// The rule reads `t(a = x) => t(a = x) and x == 1`, which the lowering turns +/// into `AntiJoin(t(x), σ(x == 1) t(x))`: its violations are exactly the rows of +/// `t` whose `a` is not `1`. Trivial to violate, and trivial to repair again, +/// which is the pair of transactions the monitored semantics turn on. +pub mod monitored_flir { + use super::flir; + use crate::{relational::TupleValue, scalarial::ScalarTypedValue}; + use coln_flir_rs::ir; + + /// The one base table. + pub const TABLE: &str = "t"; + /// The monitored rule, and so also the sink its violations arrive on. + pub const RULE: &str = "m"; + /// The only value of `a` the rule tolerates. + pub const PERMITTED: i64 = 1; + + pub fn realm() -> ir::FlatRealm { + // The single variable `x`, bound to column 0 (`a`) on both sides. + let x = || vec![(0, flir::var_term(0))]; + ir::FlatRealm { + tables: vec![flir::table_entry(TABLE, vec![("a", flir::builtin_int())])], + rules: vec![flir::rule_entry( + RULE, + ir::RuleVariant::Monitored, + [("x", flir::builtin_int())], + flir::atom_props(vec![flir::atom(TABLE, None, x())]), + vec![ + flir::atom_prop(flir::atom(TABLE, None, x())), + flir::eq_prop(flir::equality(flir::var_term(0), flir::lit_term(PERMITTED))), + ], + )], + } + } + + /// One row of [`TABLE`]: the implicit row id, which reaches the query engine + /// as its two halves, followed by the declared column `a`. + pub fn row(row_id_hash: u64, row_id_ctr: u64, a: i64) -> TupleValue { + [ + ScalarTypedValue::from(row_id_hash), + ScalarTypedValue::from(row_id_ctr), + ScalarTypedValue::from(a), + ] + .into_iter() + .collect() + } +} + +pub mod graph_flir { + use crate::{ + api::deltas::{StoreDelta, TableDelta, ZRow}, + relational::TupleValue, + scalarial::ScalarTypedValue, + }; + use coln_flir_rs::ir; + + pub trait JsonFlir { + const FILENAME: &'static str; + + fn load(&self) -> ir::FlatRealm { + coln_flir_rs::test_utils::load_theory_from_json(Self::FILENAME) + } + } + + pub struct GraphFlir { + hash: u64, + ctr: u64, + store_delta: StoreDelta, + } + + impl GraphFlir { + pub fn init() -> Self { + Self { + hash: 0, + ctr: 0, + store_delta: StoreDelta::empty(), + } + } + pub fn epoch(&self) -> u64 { + self.hash + } + pub fn ctr(&self) -> u64 { + self.ctr + } + pub fn next_epoch(&mut self) -> StoreDelta { + self.hash += 1; + self.ctr = 0; + std::mem::take(&mut self.store_delta) + } + pub fn next_ctr(&mut self) -> u64 { + let ctr = self.ctr; + self.ctr += 1; + ctr + } + fn with_zweight(zweight: i64, row: TupleValue) -> ZRow { + ZRow::new(zweight, row).expect("non-zero zweight") + } + pub fn insert_vertex(&mut self) -> Vertex { + let vertex = Vertex::new(self.hash, self.next_ctr()); + self.insert_to_table_delta(&vertex); + vertex + } + pub fn insert_edge(&mut self, from: &Vertex, to: &Vertex) -> Edge { + let edge = Edge::with_vertices(self.hash, self.next_ctr(), from, to); + self.insert_to_table_delta(&edge); + edge + } + pub fn insert_raw_edge(&mut self, edge: Edge) -> Edge { + self.insert_to_table_delta(&edge); + edge + } + fn insert_to_table_delta(&mut self, entry: &T) { + // Maybe improve by collecting all vertices of this epoch in a single + // table delta but maybe it's good to test this not-so-pretty code + // path as well.. + let table_delta = + TableDelta::new(&T::ir_path(), vec![Self::with_zweight(1, entry.to_row())]); + self.store_delta.extend(Some(table_delta)); + } + } + + impl JsonFlir for GraphFlir { + const FILENAME: &'static str = "Graph.json"; + } + + pub trait Entity { + const NAME: &'static str; + + fn ir_path() -> ir::Path { + ir::Path::from(Self::NAME) + } + + fn to_row(&self) -> TupleValue; + + fn row_id(&self) -> &RowId; + } + + pub struct RowId { + hash: u64, + ctr: u64, + } + + impl RowId { + pub fn hash(&self) -> u64 { + self.hash + } + pub fn ctr(&self) -> u64 { + self.ctr + } + } + + pub struct Vertex { + row_id: RowId, + } + + impl Vertex { + fn new(hash: u64, ctr: u64) -> Vertex { + Self { + row_id: RowId { hash, ctr }, + } + } + } + + impl Entity for Vertex { + const NAME: &'static str = "Graph.V"; + + fn to_row(&self) -> TupleValue { + [ + ScalarTypedValue::from(self.row_id.hash()), + ScalarTypedValue::from(self.row_id.ctr()), + ] + .into_iter() + .collect() + } + + fn row_id(&self) -> &RowId { + &self.row_id + } + } + + pub struct Edge { + row_id: RowId, + from_hash: u64, + from_ctr: u64, + to_hash: u64, + to_ctr: u64, + } + + impl Edge { + pub fn with_vertices(hash: u64, ctr: u64, from: &Vertex, to: &Vertex) -> Edge { + Edge::new( + hash, + ctr, + from.row_id.hash(), + from.row_id.ctr(), + to.row_id.hash(), + to.row_id.ctr(), + ) + } + pub fn new( + hash: u64, + ctr: u64, + from_hash: u64, + from_ctr: u64, + to_hash: u64, + to_ctr: u64, + ) -> Edge { + Self { + row_id: RowId { hash, ctr }, + from_hash, + from_ctr, + to_hash, + to_ctr, + } + } + } + + impl Entity for Edge { + const NAME: &'static str = "Graph.E"; + + fn to_row(&self) -> TupleValue { + [ + ScalarTypedValue::from(self.row_id.hash()), + ScalarTypedValue::from(self.row_id.ctr()), + ScalarTypedValue::from(self.from_hash), + ScalarTypedValue::from(self.from_ctr), + ScalarTypedValue::from(self.to_hash), + ScalarTypedValue::from(self.to_ctr), + ] + .into_iter() + .collect() + } + + fn row_id(&self) -> &RowId { + &self.row_id + } + } +} + +/// Assemble a [`TableSchema`] the way a test states one: named and typed columns +/// in physical order, plus the names of the columns forming its primary key +/// (empty for a relation that declares none). +pub fn table_schema<'a>( + name: &str, + columns: impl IntoIterator, + key: impl IntoIterator, +) -> TableSchema { + let columns: Vec = columns + .into_iter() + .map(|(name, scalar_type)| Column::new(name, scalar_type)) + .collect(); + let key: Vec = key + .into_iter() + .map(|key_column| { + columns + .iter() + .position(|column| column.name() == key_column) + .unwrap_or_else(|| panic!("key column '{key_column}' is not a column of '{name}'")) + }) + .collect(); + let primary_keys = if key.is_empty() { vec![] } else { vec![key] }; + TableSchema::new(EntityRef::from(name), columns, primary_keys) +} + +/// A [`QueryProgram`] assembled by hand: the plan under test, plus the schemas +/// of the relations its [`SourceExpr`](crate::relational::expr::SourceExpr) +/// leaves name. +pub struct TestProgram { + code: QueryIr, + sources: SourceSchemas, +} + +impl TestProgram { + /// `schemas` are keyed by their own [`name`](TableSchema::name), which is + /// the name a [`SourceExpr` leaf](crate::relational::expr::SourceExpr) + /// refers to them. + pub fn new(code: impl Into, schemas: impl IntoIterator) -> Self { + Self { + code: code.into(), + sources: schemas + .into_iter() + .map(|schema| (SourceId::from(schema.name()), schema)) + .collect(), + } + } +} + +/// A test states its schemas outright, so they are already in resolved form and +/// this delegates to [`SourceSchemas`]' own [`Catalog`] impl, the +/// [`Cow::Borrowed`] side, where a stored schema is lent rather than built. +impl Catalog for TestProgram { + fn source_schema(&self, id: &SourceId) -> Option> { + self.sources.source_schema(id) + } +} + +impl QueryProgram for TestProgram { + fn code(&self) -> &QueryIr { + &self.code + } + + fn take_code(&mut self) -> QueryIr { + std::mem::take(&mut self.code) + } +} + +/// Convenience: turn input entities (with per-row z-weights) into the value rows +/// that [`crate::relational::Runtime::feed`] expects. +pub fn rows(entities: impl IntoIterator) -> Vec { + entities + .into_iter() + .map(|(entity, weight)| ZRow::new(weight, entity.into()).expect("non-zero zweight")) + .collect() +} + +/// Convenience: turn input entities into value rows all carrying `weight`. +pub fn rows_with_weight( + entities: impl IntoIterator, + weight: ZWeight, +) -> Vec { + entities + .into_iter() + .map(|entity| ZRow::new(weight, entity.into()).expect("non-zero zweight")) + .collect() +} + +pub trait InputRel: Into + Into + Clone + Debug { + fn schema() -> TableSchema; + + /// The name a plan's [`SourceExpr`](crate::relational::expr::SourceExpr) + /// leaf refers to this relation by. Derived from the schema's name, so a + /// leaf and the [`TestProgram`] catalog entry describing it cannot drift + /// apart. + fn id() -> SourceId { + SourceId::from(Self::schema().name()) + } +} + +#[derive(Clone, Debug)] +pub struct PersonRel { + person_id: u64, + name: String, + age: u64, + profession_id: u64, +} + +impl InputRel for PersonRel { + fn schema() -> TableSchema { + table_schema( + "person", + [ + ("person_id", ScalarType::Uint), + ("name", ScalarType::String), + ("age", ScalarType::Uint), + ("profession_id", ScalarType::Uint), + ], + ["person_id"], + ) + } +} + +impl From for TupleKey { + fn from(person: PersonRel) -> Self { + TupleKey { + data: vec![ScalarTypedValue::Uint(person.person_id)], + } + } +} + +impl From for TupleValue { + fn from(person: PersonRel) -> Self { + TupleValue { + data: vec![ + ScalarTypedValue::Uint(person.person_id), + ScalarTypedValue::String(person.name), + ScalarTypedValue::Uint(person.age), + ScalarTypedValue::Uint(person.profession_id), + ], + } + } +} + +#[derive(Clone, Debug)] +pub struct ProfessionRel { + profession_id: u64, + name: String, +} + +impl InputRel for ProfessionRel { + fn schema() -> TableSchema { + table_schema( + "profession", + [ + ("profession_id", ScalarType::Uint), + ("name", ScalarType::String), + ], + ["profession_id"], + ) + } +} + +impl From for TupleKey { + fn from(profession: ProfessionRel) -> Self { + TupleKey { + data: vec![ScalarTypedValue::Uint(profession.profession_id)], + } + } +} + +impl From for TupleValue { + fn from(profession: ProfessionRel) -> Self { + TupleValue { + data: vec![ + ScalarTypedValue::Uint(profession.profession_id), + ScalarTypedValue::String(profession.name), + ], + } + } +} + +pub fn person_profession_data() -> [(Vec, Vec); 1] { + [( + vec![ + PersonRel { + person_id: 0, + name: "Alice".to_string(), + age: 20, + profession_id: 0, + }, + PersonRel { + person_id: 1, + name: "Bob".to_string(), + age: 30, + profession_id: 1, + }, + PersonRel { + person_id: 2, + name: "Charlie".to_string(), + age: 40, + profession_id: 0, + }, + ], + vec![ + ProfessionRel { + profession_id: 0, + name: "Engineer".to_string(), + }, + ProfessionRel { + profession_id: 1, + name: "Doctor".to_string(), + }, + ], + )] +} + +#[derive(Copy, Clone, Debug)] +pub struct PlainRel { + a: u64, + b: u64, + c: u64, +} + +impl PlainRel { + pub fn new(a: u64, b: u64, c: u64) -> Self { + Self { a, b, c } + } + const STEPS: usize = 1; + pub fn test_data_1() -> [Vec; Self::STEPS] { + [vec![ + PlainRel::new(1, 2, 3), + PlainRel::new(4, 5, 6), + PlainRel::new(7, 8, 9), + ]] + } + pub fn test_data_2() -> [Vec; Self::STEPS] { + [vec![PlainRel::new(1, 2, 3)]] + } + pub fn test_data_3() -> [Vec; Self::STEPS] { + [vec![PlainRel::new(4, 5, 6)]] + } +} + +impl InputRel for PlainRel { + fn schema() -> TableSchema { + table_schema( + "plain", + [ + ("a", ScalarType::Uint), + ("b", ScalarType::Uint), + ("c", ScalarType::Uint), + ], + [], + ) + } +} + +impl From for TupleKey { + fn from(fact: PlainRel) -> Self { + TupleKey { data: vec![] } + } +} + +impl From for TupleValue { + fn from(fact: PlainRel) -> Self { + TupleValue { + data: vec![ + ScalarTypedValue::Uint(fact.a), + ScalarTypedValue::Uint(fact.b), + ScalarTypedValue::Uint(fact.c), + ], + } + } +} + +#[derive(Copy, Clone, Debug)] +pub struct EdgeRel { + from: u64, + to: u64, + weight: u64, + active: bool, +} + +impl EdgeRel { + pub fn new(from: u64, to: u64, weight: u64) -> Self { + Self { + from, + to, + weight, + active: true, + } + } +} + +impl InputRel for EdgeRel { + fn schema() -> TableSchema { + table_schema( + "edges", + [ + ("from", ScalarType::Uint), + ("to", ScalarType::Uint), + ("weight", ScalarType::Uint), + ("active", ScalarType::Bool), + ], + ["from", "to"], + ) + } +} + +impl From for TupleKey { + fn from(edge: EdgeRel) -> Self { + TupleKey { + data: vec![ + ScalarTypedValue::Uint(edge.from), + ScalarTypedValue::Uint(edge.to), + ], + } + } +} + +impl From for TupleValue { + fn from(edge: EdgeRel) -> Self { + TupleValue { + data: vec![ + ScalarTypedValue::Uint(edge.from), + ScalarTypedValue::Uint(edge.to), + ScalarTypedValue::Uint(edge.weight), + ScalarTypedValue::Bool(edge.active), + ], + } + } +} + +#[derive(Copy, Clone, Debug)] +pub struct SetRel { + rep_id: u64, + ctr: u64, + key: u64, + value: u64, +} + +impl SetRel { + pub fn new(rep_id: u64, ctr: u64, key: u64, value: u64) -> Self { + Self { + rep_id, + ctr, + key, + value, + } + } +} + +impl InputRel for SetRel { + fn schema() -> TableSchema { + table_schema( + "set", + [ + ("RepId", ScalarType::Uint), + ("Ctr", ScalarType::Uint), + ("Key", ScalarType::Uint), + ("Value", ScalarType::Uint), + ], + ["RepId", "Ctr"], + ) + } +} + +impl From for TupleKey { + fn from(set_op: SetRel) -> Self { + TupleKey::from_iter([set_op.rep_id, set_op.ctr]) + } +} + +impl From for TupleValue { + fn from(set_op: SetRel) -> Self { + TupleValue::from_iter([set_op.rep_id, set_op.ctr, set_op.key, set_op.value]) + } +} + +#[derive(Copy, Clone, Debug)] +pub struct PredRel { + from_rep_id: u64, + from_ctr: u64, + to_rep_id: u64, + to_ctr: u64, +} + +impl PredRel { + pub fn new(from_rep_id: u64, from_ctr: u64, to_rep_id: u64, to_ctr: u64) -> Self { + Self { + from_rep_id, + from_ctr, + to_rep_id, + to_ctr, + } + } +} + +impl InputRel for PredRel { + fn schema() -> TableSchema { + table_schema( + "pred", + [ + ("FromRepId", ScalarType::Uint), + ("FromCtr", ScalarType::Uint), + ("ToRepId", ScalarType::Uint), + ("ToCtr", ScalarType::Uint), + ], + ["FromRepId", "FromCtr", "ToRepId", "ToCtr"], + ) + } +} + +impl From for TupleKey { + fn from(pred_rel: PredRel) -> Self { + TupleKey::from_iter([ + pred_rel.from_rep_id, + pred_rel.from_ctr, + pred_rel.to_rep_id, + pred_rel.to_ctr, + ]) + } +} + +impl From for TupleValue { + fn from(pred_rel: PredRel) -> Self { + TupleValue::from_iter([ + pred_rel.from_rep_id, + pred_rel.from_ctr, + pred_rel.to_rep_id, + pred_rel.to_ctr, + ]) + } +} + +/// This function returns test data for an operation history of the MVR CRDT +/// store. The history is as follows. +/// The notation is `set__(, )`. +/// +/// 1. step (just one root operation setting register with key 1 to value 1): +/// +/// ```text +/// set_0_0(1, 1) +/// ``` +/// +/// 2. step (concurrent writes by replica 0 and 1): +/// +/// ```text +/// ---> set_0_1(1, 2) +/// set_0_0(1, 1) +/// ---> set_1_0(1, 3) +/// ``` +/// +/// 3. step (replica 1 does a "merge" operation overwriting the previous +/// conflict): +/// +/// ```text +/// ---> set_0_1(1, 2) +/// set_0_0(1, 1) ---> set_1_2(1, 4) +/// ---> set_1_0(1, 3) +/// ``` +/// +/// 4. step (replica 0 overwrites a not-yet delivered operation): +/// +/// ```text +/// ---> set_0_1(1, 2) +/// set_0_0(1, 1) ---> set_1_2(1, 4) ---> missing ---> set_0_4(1, 6) +/// ---> set_1_0(1, 3) +/// ``` +/// +/// 5. step (replica 0's missing operation arrives): +/// +/// ```text +/// ---> set_0_1(1, 2) +/// set_0_0(1, 1) ---> set_1_2(1, 4) ---> set_0_3(1, 5) ---> set_0_4(1, 6) +/// ---> set_1_0(1, 3) +/// ``` +pub fn mvr_store_operation_history() -> [(Vec, Vec); 5] { + [ + (vec![], vec![SetRel::new(0, 0, 1, 1)]), + ( + vec![PredRel::new(0, 0, 0, 1), PredRel::new(0, 0, 1, 0)], + vec![SetRel::new(0, 1, 1, 2), SetRel::new(1, 0, 1, 3)], + ), + ( + vec![PredRel::new(0, 1, 1, 2), PredRel::new(1, 0, 1, 2)], + vec![SetRel::new(1, 2, 1, 4)], + ), + ( + vec![PredRel::new(0, 3, 0, 4)], + vec![SetRel::new(0, 4, 1, 6)], + ), + ( + vec![PredRel::new(1, 2, 0, 3)], + vec![SetRel::new(0, 3, 1, 5)], + ), + ] +} diff --git a/packages/coln-query/src/typing/mod.rs b/packages/coln-query/src/typing/mod.rs index 227d3332..d734102c 100644 --- a/packages/coln-query/src/typing/mod.rs +++ b/packages/coln-query/src/typing/mod.rs @@ -5,4 +5,3 @@ //! A future/potential type checking pass. Mostly unused today. pub mod type_resolver; -pub use type_resolver::ExprType; diff --git a/packages/coln-query/src/typing/type_resolver.rs b/packages/coln-query/src/typing/type_resolver.rs index c2ae37c9..ca72d275 100644 --- a/packages/coln-query/src/typing/type_resolver.rs +++ b/packages/coln-query/src/typing/type_resolver.rs @@ -9,19 +9,22 @@ use crate::{ AssignExpr, BinaryExpr, CallExpr, Expr, ExprVisitor, FunctionExpr, GetIndexExpr, GroupingExpr, Literal, LiteralExpr, TupleExpr, UnaryExpr, VarExpr, }, + function::FunctionType, operator::Operator, resolver::ScopeStack, stmt::{BlockStmt, ExprStmt, Stmt, StmtVisitor, VarStmt}, tuple::TupleType, }, - relational::expr::{ - AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, EquiJoinExpr, - FixedPointIterExpr, OutputExpr, ProjectionExpr, RelExpr, RelExprVisitor, SelectionExpr, - SourceExpr, UnionExpr, + relational::{ + catalog::Catalog, + expr::{ + AliasExpr, AntiJoinExpr, CartesianProductExpr, DifferenceExpr, DistinctExpr, + EquiJoinExpr, FixedPointIterExpr, MultiWayEquiJoinExpr, OutputExpr, ProjectionExpr, + RelExpr, RelExprVisitor, SelectionExpr, SourceExpr, UnionExpr, + }, + relation::RelationType, }, -}; -pub use crate::{ - host::function::FunctionType, relational::relation::RelationType, scalarial::ScalarType, + scalarial::ScalarType, }; use std::collections::HashMap; use std::fmt; @@ -39,13 +42,24 @@ macro_rules! assert_type { } /// This is "get me the type of this part of the AST but don't type check it". -#[derive(Default)] -pub struct TypeResolver {} +pub struct TypeResolver<'a> { + /// What the plan's [`SourceExpr`] leaves name: a leaf carries only a + /// [`SourceId`](crate::relational::expr::SourceId), so this is where the + /// type of an extensional relation comes from. + /// + /// It sits on the resolver rather than in [`TypeResolverContext`] because + /// the context is per-traversal scope state, while the catalog is fixed for + /// the whole run. + catalog: &'a dyn Catalog, +} -impl TypeResolver { - pub fn resolve<'a>( +impl<'a> TypeResolver<'a> { + pub fn new(catalog: &'a dyn Catalog) -> Self { + Self { catalog } + } + pub fn resolve<'b>( &mut self, - stmts: impl IntoIterator, + stmts: impl IntoIterator, ctx: VisitorCtx, ) -> Result, SyntaxError> { // Ensure we have a global scope before resolving. @@ -64,9 +78,9 @@ impl TypeResolver { pub fn resolve_stmt(&mut self, stmt: &Stmt, ctx: VisitorCtx) -> VisitorResult { self.visit_stmt(stmt, ctx) } - fn visit_stmts<'a>( + fn visit_stmts<'b>( &mut self, - stmts: impl IntoIterator, + stmts: impl IntoIterator, ctx: VisitorCtx, ) -> Result, SyntaxError> { stmts @@ -152,7 +166,7 @@ impl TypeResolverContext<'_> { } } -impl TypeResolver { +impl TypeResolver<'_> { /// A helper method to deal with projections. fn visit_projection_attributes( &mut self, @@ -179,7 +193,7 @@ impl TypeResolver { } } -impl ExprVisitor> for TypeResolver { +impl ExprVisitor> for TypeResolver<'_> { fn visit_literal_expr(&mut self, expr: &LiteralExpr, ctx: VisitorCtx) -> VisitorResult { Ok(ExprType::from(&expr.value)) } @@ -286,9 +300,16 @@ impl ExprVisitor> for TypeResolver { } } -impl RelExprVisitor> for TypeResolver { +impl RelExprVisitor> for TypeResolver<'_> { fn visit_source_expr(&mut self, expr: &SourceExpr, ctx: VisitorCtx) -> VisitorResult { - Ok(ExprType::Relation(RelationType::from(&expr.schema))) + // The leaf names its relation; the catalog says what that relation is. + let schema = self.catalog.source_schema(&expr.id).ok_or_else(|| { + SyntaxError::new(format!( + "Source '{}' is not described by the catalog this plan is typed against", + expr.id + )) + })?; + Ok(ExprType::Relation(RelationType::from(schema.as_ref()))) } fn visit_output_expr(&mut self, expr: &OutputExpr, ctx: VisitorCtx) -> VisitorResult { @@ -308,7 +329,7 @@ impl RelExprVisitor> for TypeResolver { let first = expr .relations .first() - .ok_or_else(|| SyntaxError::new("Union expr with only no operands"))?; + .ok_or_else(|| SyntaxError::new("Union expr with no operands"))?; self.visit_expr(first, ctx) } @@ -349,6 +370,28 @@ impl RelExprVisitor> for TypeResolver { self.visit_projection_attributes(joined, expr.attributes.as_ref(), ctx) } + fn visit_multi_way_equi_join_expr( + &mut self, + expr: &MultiWayEquiJoinExpr, + ctx: VisitorCtx<'_, '_>, + ) -> VisitorResult { + let mut relations = expr.relations.iter(); + let first_relation_type = relations + .next() + .ok_or_else(|| SyntaxError::new("Multi way equi join with no operands")) + .and_then(|first_relation| { + self.visit_expr(first_relation, ctx) + .and_then(|expr_type| assert_type!(expr_type, ExprType::Relation)) + })?; + let joined = relations.try_fold(first_relation_type, |fold, relation| { + let relation_type = self + .visit_expr(relation, ctx) + .and_then(|expr_type| assert_type!(expr_type, ExprType::Relation))?; + Ok(fold.join(relation_type)) + })?; + self.visit_projection_attributes(joined, expr.attributes.as_ref(), ctx) + } + fn visit_anti_join_expr(&mut self, expr: &AntiJoinExpr, ctx: VisitorCtx) -> VisitorResult { self.visit_expr(&expr.left, ctx) } @@ -375,7 +418,7 @@ impl From<&Literal> for ExprType { } } -impl StmtVisitor> for TypeResolver { +impl StmtVisitor> for TypeResolver<'_> { fn visit_var_stmt(&mut self, stmt: &VarStmt, ctx: VisitorCtx) -> VisitorResult { todo!(); } diff --git a/packages/coln-query/src/util.rs b/packages/coln-query/src/util.rs deleted file mode 100644 index 31ff81c6..00000000 --- a/packages/coln-query/src/util.rs +++ /dev/null @@ -1,51 +0,0 @@ -// SPDX-FileCopyrightText: 2026 Coln contributors -// -// SPDX-License-Identifier: Apache-2.0 OR MIT - -use crate::host::variable::VariableSlot; - -pub trait MemAddr { - /// Make sure that the value is not moved in memory! - fn mem_addr(&self) -> usize { - self as *const Self as *const () as usize - } -} - -pub trait Resolvable { - fn set_resolved(&mut self, resolved: VariableSlot); -} - -pub trait Named { - fn name(&self) -> &str; -} - -/// An AST node identifier. -/// Can be its address in memory if using a pointer-based AST -/// or its index if using a flattened AST. -#[derive(Eq, PartialEq, Hash, Clone, Copy, Debug)] -pub struct NodeRef(usize); - -impl From for NodeRef { - fn from(index: usize) -> Self { - Self(index) - } -} - -impl From<&T> for NodeRef { - fn from(addr: &T) -> Self { - Self(addr.mem_addr()) - } -} - -#[macro_export] -macro_rules! impl_from_auto_box { - ($enum:ty, $(($variant:path, $expr:ty)),*) => { - $( - impl From<$expr> for $enum { - fn from(value: $expr) -> Self { - $variant(Box::new(value)) - } - } - )* - } -} diff --git a/packages/coln-query/src/utils.rs b/packages/coln-query/src/utils.rs new file mode 100644 index 00000000..d83b2dd3 --- /dev/null +++ b/packages/coln-query/src/utils.rs @@ -0,0 +1,28 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +/// Generates the two ways of putting a node payload into its enum: from the +/// bare payload (which has to allocate) and from an already boxed one (which +/// reuses that allocation). +/// +/// The boxed impl is what makes an owned rewriting pass cheap: the `…VisitorOwn` +/// families hand out `Box` precisely so that a node the pass leaves +/// alone can go back into its enum without a round trip through the allocator. +#[macro_export] +macro_rules! impl_from_auto_box { + ($enum:ty, $(($variant:path, $expr:ty)),*) => { + $( + impl From<$expr> for $enum { + fn from(value: $expr) -> Self { + $variant(Box::new(value)) + } + } + impl From> for $enum { + fn from(value: Box<$expr>) -> Self { + $variant(value) + } + } + )* + } +} diff --git a/packages/coln-rpc/Cargo.toml b/packages/coln-rpc/Cargo.toml new file mode 100644 index 00000000..92860ee2 --- /dev/null +++ b/packages/coln-rpc/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "coln-rpc" +edition = "2024" +version.workspace = true +authors.workspace = true +description.workspace = true +homepage.workspace = true +repository.workspace = true + +[dependencies] +coln-flir-rs = { version = "0.1.0", path = "../coln-flir-rs" } +coln-store = { version = "0.1.0", path = "../coln-store" } +serde.workspace = true +serde_json.workspace = true +specta.workspace = true +specta-serde = "0.0.12" +specta-typescript = { version = "0.0.12", features = ["serde"] } + +[lib] +path = "src/rust/lib.rs" + +[[bin]] +name = "export" +path = "scripts/export.rs" diff --git a/packages/coln-rpc/scripts/export.rs b/packages/coln-rpc/scripts/export.rs new file mode 100644 index 00000000..e78616df --- /dev/null +++ b/packages/coln-rpc/scripts/export.rs @@ -0,0 +1,18 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +use coln_rpc::api::*; +use specta::Types; +use specta_typescript::Typescript; + +fn main() { + let mut types = Types::default(); + + types.register_mut::(); + types.register_mut::(); + + Typescript::default() + .export_to("./types.ts", &types, specta_serde::Format) + .unwrap(); +} diff --git a/packages/coln-rpc/src/rust/api.rs b/packages/coln-rpc/src/rust/api.rs new file mode 100644 index 00000000..ea2291e4 --- /dev/null +++ b/packages/coln-rpc/src/rust/api.rs @@ -0,0 +1,59 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +use coln_flir_rs::ir; +use serde::{Deserialize, Serialize}; +use specta::Type; +// use coln_store::id_packer::IdPacker; +use coln_store::{table::WireValue, txn::TxnWireRowId}; + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum WhereClause { + PrimaryKey { values: Vec }, + ExceptRowId { values: Vec }, + Generic { values_at: Vec<(u32, WireValue)> }, +} + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum SelectClause { + All { columns: Vec }, + One { columns: Vec }, + Existence, +} + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum Query { + SelectWhere { + table_name: ir::Path, + select: SelectClause, + r#where: WhereClause, + }, +} + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum QueryResponse { + All { tuples: Vec> }, + One { values: Vec }, + Present, + Absent, +} + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum Mutation { + Insert { + table_name: ir::Path, + columns: Vec, + }, +} + +#[derive(Type, Serialize, Deserialize)] +#[serde(tag = "type")] +pub enum MutationResponse { + Id(TxnWireRowId), +} diff --git a/packages/coln-rpc/src/rust/lib.rs b/packages/coln-rpc/src/rust/lib.rs new file mode 100644 index 00000000..00fc022c --- /dev/null +++ b/packages/coln-rpc/src/rust/lib.rs @@ -0,0 +1,5 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +pub mod api; diff --git a/packages/coln-rpc/src/rust/state_machine.rs b/packages/coln-rpc/src/rust/state_machine.rs new file mode 100644 index 00000000..c6932f3d --- /dev/null +++ b/packages/coln-rpc/src/rust/state_machine.rs @@ -0,0 +1,3 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT diff --git a/packages/coln-rpc/src/typescript/store.ts b/packages/coln-rpc/src/typescript/store.ts new file mode 100644 index 00000000..78fbe104 --- /dev/null +++ b/packages/coln-rpc/src/typescript/store.ts @@ -0,0 +1,20 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +import {WhereClause, WireRowId, Value, Path, CommitHash} from "./types.js" + +export type WireValue = Value + +export type WireTuple = WireValue[] + +export interface Store { + commit(): CommitHash + abort(): null + + all(query: WhereClause, select: [number]): [WireTuple] + one(query: WhereClause, select: [number]): WireTuple + exists(query: WhereClause): boolean + + add(table_name: Path, values: WireTuple): WireRowId +} diff --git a/packages/coln-rpc/src/typescript/types.ts b/packages/coln-rpc/src/typescript/types.ts new file mode 100644 index 00000000..36db2d9f --- /dev/null +++ b/packages/coln-rpc/src/typescript/types.ts @@ -0,0 +1,28 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT + +// This file has been generated by Specta. Do not edit this file manually. +export type CommitHash = string; + +export type Path = string[][]; + +export type Query = { type: "SelectWhere"; table_name: Path; select: SelectClause; where: WhereClause }; + +export type QueryResponse = { type: "All"; tuples: Value[][] } | { type: "One"; values: Value[] } | { type: "Present" } | { type: "Absent" }; + +export type SelectClause = { type: "All"; columns: number[] } | { type: "One"; columns: number[] } | { type: "Existence" }; + +export type Value = I | number | string; + +export type WhereClause = { type: "PrimaryKey"; values: Value[] } | { type: "ExceptRowId"; values: Value[] } | { type: "Generic"; values_at: ([number, Value])[] }; + +/** + * The unique id that identifies each row in a table. + * + * It is managed by the database and read-only for the user. + */ +export type WireRowId = { + commit: CommitHash, + counter: number, +}; diff --git a/packages/coln-store/Cargo.toml b/packages/coln-store/Cargo.toml index bbf4d804..d415a66b 100644 --- a/packages/coln-store/Cargo.toml +++ b/packages/coln-store/Cargo.toml @@ -25,7 +25,7 @@ coln-flir-rs = { path = "../coln-flir-rs" } csv = "1.4.0" coln-query = { path = "../coln-query", optional = true } ena = "0.14.4" -hex = "0.4.3" +hex = { version = "0.4.3", features = ["serde"] } hexane = "1.0.0-alpha.5" js-sys = { version = "0.3.83", optional = true } leb128 = "0.2.6" @@ -42,6 +42,7 @@ tracing-subscriber = { workspace = true, features = [ "env-filter", "fmt", ], optional = true } +specta.workspace = true [dev-dependencies] diff --git a/packages/coln-store/docs/primitives.md b/packages/coln-store/docs/primitives.md index e17f889d..cde6cf35 100644 --- a/packages/coln-store/docs/primitives.md +++ b/packages/coln-store/docs/primitives.md @@ -51,7 +51,7 @@ binding ```rust struct RowView { row_id: RowId, - values: Vec, + values: Vec, } Store::scan_table(table_path) -> Option> diff --git a/packages/coln-store/src/commit/hash.rs b/packages/coln-store/src/commit/hash.rs index 26532c29..3d9b50f5 100644 --- a/packages/coln-store/src/commit/hash.rs +++ b/packages/coln-store/src/commit/hash.rs @@ -2,13 +2,21 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT +use serde::{Deserialize, Serialize}; +use specta::Type; use std::fmt; /// The number of bytes in a commit hash. pub(crate) const HASH_SIZE: usize = 32; -#[derive(Copy, Clone, Eq, PartialEq, Ord, PartialOrd, Hash, Debug)] -pub struct CommitHash(pub [u8; HASH_SIZE]); +#[derive( + Copy, Clone, Eq, PartialEq, Ord, PartialOrd, Hash, Debug, Serialize, Deserialize, Type, +)] +pub struct CommitHash( + #[serde(with = "hex::serde")] + #[specta(type = String)] + pub [u8; HASH_SIZE], +); impl CommitHash { pub(crate) fn as_bytes(&self) -> &[u8] { diff --git a/packages/coln-store/src/commit/mod.rs b/packages/coln-store/src/commit/mod.rs index ec90dbaf..ff4fbe0f 100644 --- a/packages/coln-store/src/commit/mod.rs +++ b/packages/coln-store/src/commit/mod.rs @@ -29,7 +29,7 @@ use crate::{ ir::Path, op::Op, table::{TableMeta, TableOid}, - txn::{PendingOp, RowRef, TxnCellValue}, + txn::{PendingOp, TxnWireRowId, TxnWireValue}, }; /// A commit: canonical payload bytes, content hash, and parsed metadata. @@ -211,7 +211,7 @@ fn collect_op_hashes(pending: &[PendingOp], hash_mapper: &mut HashMapper) { for op in pending { let PendingOp::Add { values, .. } = op; for value in values { - if let TxnCellValue::Id(RowRef::Existing(row_id)) = value { + if let TxnWireValue::Id(TxnWireRowId::Existing(row_id)) = value { hash_mapper.insert(row_id.commit); } } @@ -228,8 +228,8 @@ mod tests { use crate::commit::chunk::{Chunk, hash}; use crate::commit::hash::HASH_SIZE; use crate::ir::{BuiltinTy, ColType, ColumnEntry, EntityVariant, Path, TableEntry}; - use crate::table::{RowId, TableMeta, TableOid}; - use crate::txn::{RowRef, TempRowId}; + use crate::table::{TableMeta, TableOid, WireRowId}; + use crate::txn::{TempRowId, TxnWireRowId}; fn zero_hash() -> CommitHash { CommitHash([0u8; HASH_SIZE]) @@ -415,7 +415,7 @@ mod tests { fn decode_data_preserves_payload_metadata_and_ops() { let dep = zero_hash(); let deps = vec![dep]; - let rid = RowId { + let rid = WireRowId { commit: dep, counter: 7, }; @@ -423,15 +423,15 @@ mod tests { PendingOp::Add { row_id: TempRowId(0), table: 0, - values: vec![1i64.into()], + values: vec![1i32.into()], }, PendingOp::Add { row_id: TempRowId(1), table: 1, values: vec![ - TxnCellValue::Id(RowRef::Existing(rid)), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), - TxnCellValue::Str("x".into()), + TxnWireValue::Id(TxnWireRowId::Existing(rid)), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), + TxnWireValue::Str("x".into()), ], }, ]; @@ -606,22 +606,22 @@ mod tests { let dep = zero_hash(); let deps = vec![dep]; let author = Author::foo(); - let rid = RowId { + let rid = WireRowId { commit: dep, counter: 7, }; let op0 = PendingOp::Add { row_id: TempRowId(0), table: 0, - values: vec![1i64.into()], + values: vec![1i32.into()], }; let op1 = PendingOp::Add { row_id: TempRowId(1), table: 1, values: vec![ - TxnCellValue::Id(RowRef::Existing(rid)), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), - TxnCellValue::Str("x".into()), + TxnWireValue::Id(TxnWireRowId::Existing(rid)), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), + TxnWireValue::Str("x".into()), ], }; let pending = vec![op0, op1]; @@ -650,22 +650,22 @@ mod tests { let dep = zero_hash(); let deps = vec![dep]; let author = Author::foo(); - let rid = RowId { + let rid = WireRowId { commit: dep, counter: 7, }; let op0 = PendingOp::Add { row_id: TempRowId(0), table: 0, - values: vec![1i64.into()], + values: vec![1i32.into()], }; let op1 = PendingOp::Add { row_id: TempRowId(1), table: 1, values: vec![ - TxnCellValue::Id(RowRef::Existing(rid)), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), - TxnCellValue::Str("x".into()), + TxnWireValue::Id(TxnWireRowId::Existing(rid)), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), + TxnWireValue::Str("x".into()), ], }; let pending = vec![op0, op1]; @@ -689,16 +689,16 @@ mod tests { fn other_hashes_contain_right_hashes() { let ha = CommitHash([1u8; HASH_SIZE]); let hb = CommitHash([2u8; HASH_SIZE]); - let rid_a = RowId { + let rid_a = WireRowId { commit: ha, counter: 0, }; - let rid_b = RowId { + let rid_b = WireRowId { commit: hb, counter: 3, }; // also point to ha - let rid_a_later = RowId { + let rid_a_later = WireRowId { commit: ha, counter: 99, }; @@ -707,16 +707,16 @@ mod tests { row_id: TempRowId(0), table: 0, values: vec![ - TxnCellValue::Id(RowRef::Existing(rid_a)), - TxnCellValue::Id(RowRef::Existing(rid_a)), + TxnWireValue::Id(TxnWireRowId::Existing(rid_a)), + TxnWireValue::Id(TxnWireRowId::Existing(rid_a)), ], }; let op1 = PendingOp::Add { row_id: TempRowId(1), table: 0, values: vec![ - TxnCellValue::Id(RowRef::Existing(rid_b)), - TxnCellValue::Id(RowRef::Existing(rid_a_later)), + TxnWireValue::Id(TxnWireRowId::Existing(rid_b)), + TxnWireValue::Id(TxnWireRowId::Existing(rid_a_later)), ], }; let commit = Commit::from_commit_data( diff --git a/packages/coln-store/src/commit/pst.rs b/packages/coln-store/src/commit/pst.rs index 3c2e923e..0b198cd1 100644 --- a/packages/coln-store/src/commit/pst.rs +++ b/packages/coln-store/src/commit/pst.rs @@ -100,7 +100,7 @@ mod tests { use crate::commit::hash::{CommitHash, HASH_SIZE}; use crate::commit::wire::CommitData; use crate::ir::{FlatRealm, Path, Schema, TableEntry}; - use crate::table::CellValue; + use crate::table::WireValue; fn int_schema() -> Schema { Schema { @@ -207,7 +207,7 @@ mod tests { let mut store = Store::try_from_ir(int_theory()).expect("store"); let table = Path::from("T"); let mut txn = store.transaction(); - txn.add(&table, vec![99_i64.into()]).expect("add row"); + txn.add(&table, vec![99_i32]).expect("add row"); txn.commit().expect("commit"); let expected = store @@ -347,7 +347,7 @@ mod tests { let root = store.commits().root_commit().expect("root").hash(); let table = Path::from("T"); let mut txn = store.transaction(); - txn.add(&table, vec![99_i64.into()]).expect("add row"); + txn.add(&table, vec![99_i32]).expect("add row"); let commit = txn.commit().expect("commit"); let bytes = encode_store(&store).unwrap(); @@ -355,7 +355,7 @@ mod tests { let restored_table = restored.table_at(&table).expect("table"); assert_eq!(restored_table.row_count(), 1); - assert_eq!(restored_table.cell_at(0, 0), Some(CellValue::Int(99))); + assert_eq!(restored_table.cell_at(0, 0), Some(WireValue::Int(99))); assert_eq!(restored_table.row_id_at(0).expect("row id").commit, commit); assert_eq!( restored.commits().parents_of(&commit), diff --git a/packages/coln-store/src/commit/wire/data.rs b/packages/coln-store/src/commit/wire/data.rs index 2fcec98b..e16afb95 100644 --- a/packages/coln-store/src/commit/wire/data.rs +++ b/packages/coln-store/src/commit/wire/data.rs @@ -20,8 +20,8 @@ use crate::{ }, ir::{BuiltinTy, ColType, Path, Schema}, op::OP_KIND_ADD, - table::{RowId, TableMeta, TableOid}, - txn::{PendingOp, RowRef, TempRowId, TxnCellValue}, + table::{TableMeta, TableOid, WireRowId}, + txn::{PendingOp, TempRowId, TxnWireRowId, TxnWireValue}, }; // TODO change this to i32 when we support it as a column type @@ -205,28 +205,28 @@ where /// encode the row_ref column, which might be a pending id or a already resolved id fn encode_txn_row_ref_column( - values: &[TxnCellValue], + values: &[TxnWireValue], hash_mapper: &HashMapper, ) -> Result, CodecError> { let mut hash_indices: Vec = Vec::with_capacity(values.len()); let mut counters = Vec::with_capacity(values.len()); for value in values { - let TxnCellValue::Id(row_ref) = value else { + let TxnWireValue::Id(row_ref) = value else { return Err(CodecError::SchemaError(format!( "expected row reference, got {value:?}" ))); }; match row_ref { - RowRef::Existing(RowId { commit, counter }) => { + TxnWireRowId::Existing(WireRowId { commit, counter }) => { let hash_index = hash_mapper.index(*commit).ok_or_else(|| { CodecError::SchemaError(format!("missing commit hash in dictionary: {commit}")) })? as i64; hash_indices.push(hash_index); counters.push(*counter); } - RowRef::Pending(temp_id) => { + TxnWireRowId::Pending(temp_id) => { hash_indices.push(LOCAL_COMMIT_HASH_INDEX); counters.push(temp_id.counter()); } @@ -243,7 +243,7 @@ fn encode_txn_row_ref_column( } fn encode_txn_prim_value_column( - values: &[TxnCellValue], + values: &[TxnWireValue], prim: &BuiltinTy, ) -> Result, CodecError> { let mut value_bytes = Vec::new(); @@ -268,7 +268,7 @@ fn encode_txn_prim_value_column( /// Columnar encode for one schema column of transaction cell values. fn encode_txn_value_column( - values: &[TxnCellValue], + values: &[TxnWireValue], col_type: &ColType, hash_mapper: &HashMapper, ) -> Result, CodecError> { @@ -290,7 +290,7 @@ fn encode_op_group( ops: &[&PendingOp], hash_mapper: &HashMapper, ) -> Result, CodecError> { - let mut rows: Vec<&[TxnCellValue]> = Vec::with_capacity(ops.len()); + let mut rows: Vec<&[TxnWireValue]> = Vec::with_capacity(ops.len()); for op in ops { let PendingOp::Add { table: op_table, @@ -462,7 +462,7 @@ where struct DecodedOpGroup { table: TableOid, - rows: Vec>, + rows: Vec>, } fn decode_op_group<'s, F>( @@ -534,7 +534,7 @@ fn decode_txn_value_column( data: &[u8], col_type: &ColType, hashes: &[CommitHash], -) -> Result, CodecError> { +) -> Result, CodecError> { match col_type { ColType::RowId { .. } => decode_txn_row_ref_column(data, hashes), ColType::BuiltinTy { builtin_ty } => decode_txn_prim_value_column(data, builtin_ty), @@ -544,7 +544,7 @@ fn decode_txn_value_column( fn decode_txn_row_ref_column( data: &[u8], hashes: &[CommitHash], -) -> Result, CodecError> { +) -> Result, CodecError> { let mut pos = 0usize; let hash_index_blob = commit_leb128::read_len_prefixed_bytes(data, &mut pos, "txn row-ref hash-index column")?; @@ -572,14 +572,14 @@ fn decode_txn_row_ref_column( .zip(counters) .map(|(hash_index, counter)| { if hash_index == LOCAL_COMMIT_HASH_INDEX { - Ok(TxnCellValue::Id(RowRef::Pending(TempRowId(counter)))) + Ok(TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(counter)))) } else { let commit = hashes.get(hash_index as usize).copied().ok_or_else(|| { CodecError::DataFormatError(format!( "txn row-ref hash index {hash_index} out of bounds" )) })?; - Ok(TxnCellValue::Id(RowRef::Existing(RowId { + Ok(TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit, counter, }))) @@ -591,7 +591,7 @@ fn decode_txn_row_ref_column( fn decode_txn_prim_value_column( data: &[u8], prim: &BuiltinTy, -) -> Result, CodecError> { +) -> Result, CodecError> { let mut pos = 0usize; let meta_blob = commit_leb128::read_len_prefixed_bytes(data, &mut pos, "txn prim value meta column")?; @@ -641,8 +641,8 @@ mod tests { use crate::commit::hash::HASH_SIZE; use crate::commit::wire::prim::ValueType; use crate::ir::{BuiltinTy, ColType, ColumnEntry, EntityVariant, Path, Schema}; - use crate::table::RowId; - use crate::txn::{RowRef, TempRowId}; + use crate::table::WireRowId; + use crate::txn::{TempRowId, TxnWireRowId}; #[test] fn txn_row_ref_column_round_trips_existing_and_pending_refs() { @@ -652,16 +652,16 @@ mod tests { hash_mapper.insert(ha); hash_mapper.insert(hb); let values = vec![ - TxnCellValue::Id(RowRef::Existing(RowId { + TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: ha, counter: 7, })), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), - TxnCellValue::Id(RowRef::Existing(RowId { + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), + TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: hb, counter: 11, })), - TxnCellValue::Id(RowRef::Pending(TempRowId(2))), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(2))), ]; let encoded = encode_txn_row_ref_column(&values, &hash_mapper).expect("encode row refs"); @@ -673,7 +673,7 @@ mod tests { #[test] fn txn_row_ref_column_rejects_non_ref_values() { - let err = encode_txn_row_ref_column(&[TxnCellValue::Int(42)], &HashMapper::new()) + let err = encode_txn_row_ref_column(&[TxnWireValue::Int(42)], &HashMapper::new()) .expect_err("int is not a row ref"); assert!(matches!(err, CodecError::SchemaError(_))); @@ -681,7 +681,7 @@ mod tests { #[test] fn txn_row_ref_column_rejects_unmapped_existing_hashes() { - let value = TxnCellValue::Id(RowRef::Existing(RowId { + let value = TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: CommitHash([9u8; HASH_SIZE]), counter: 1, })); @@ -696,7 +696,7 @@ mod tests { let col = ColType::BuiltinTy { builtin_ty: BuiltinTy::BuiltinInt, }; - let values = vec![1i64.into(), 2i64.into(), (-3i64).into()]; + let values = vec![1i32.into(), 2i32.into(), (-3i32).into()]; let encoded = encode_txn_value_column(&values, &col, &HashMapper::new()).expect("encode int col"); let decoded = decode_txn_value_column(&encoded, &col, &[]).expect("decode int col"); @@ -722,15 +722,15 @@ mod tests { }; // Span every leb byte-length boundary, including the signed extremes, // so the declared meta length must agree with the signed-leb encoding. - let values: Vec = vec![ - 0i64.into(), - (-1i64).into(), - 1i64.into(), - 127i64.into(), - 128i64.into(), - (-128i64).into(), - i64::MIN.into(), - i64::MAX.into(), + let values: Vec = vec![ + 0i32.into(), + (-1i32).into(), + 1i32.into(), + 127i32.into(), + 128i32.into(), + (-128i32).into(), + i32::MIN.into(), + i32::MAX.into(), ]; let encoded = encode_txn_value_column(&values, &col, &HashMapper::new()).expect("encode int col"); @@ -746,14 +746,14 @@ mod tests { // Zero-length values interleaved with non-empty ones: the value blob // is shorter than the row count, so offsets must advance by zero. - let values: Vec = vec!["".into(), "x".into(), "".into()]; + let values: Vec = vec!["".into(), "x".into(), "".into()]; let encoded = encode_txn_value_column(&values, &col, &HashMapper::new()).expect("encode str col"); let decoded = decode_txn_value_column(&encoded, &col, &[]).expect("decode str col"); assert_eq!(decoded, values); // Zero rows: empty meta and value blobs must round-trip to an empty column. - let empty: Vec = vec![]; + let empty: Vec = vec![]; let encoded = encode_txn_value_column(&empty, &col, &HashMapper::new()).expect("encode empty col"); let decoded = decode_txn_value_column(&encoded, &col, &[]).expect("decode empty col"); @@ -786,11 +786,11 @@ mod tests { path: Path::from("T.E"), }; let values = vec![ - TxnCellValue::Id(RowRef::Existing(RowId { + TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: ha, counter: 1, })), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), ]; let encoded = encode_txn_value_column(&values, &col, &hash_mapper).expect("encode entity col"); @@ -805,7 +805,7 @@ mod tests { builtin_ty: BuiltinTy::BuiltinInt, }; let err = encode_txn_value_column( - &[TxnCellValue::Str("nope".into())], + &[TxnWireValue::Str("nope".into())], &col, &HashMapper::new(), ) @@ -848,9 +848,9 @@ mod tests { row_id: TempRowId(0), table: table_oid, values: vec![ - 1i64.into(), + 1i32.into(), "a".into(), - TxnCellValue::Id(RowRef::Existing(RowId { + TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: ha, counter: 7, })), @@ -860,9 +860,9 @@ mod tests { row_id: TempRowId(1), table: table_oid, values: vec![ - 2i64.into(), + 2i32.into(), "b".into(), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), ], }, ]; @@ -888,16 +888,16 @@ mod tests { let decoded = decode_txn_value_column(blob, &col_entry.col_type, hash_mapper.hashes()) .expect("decode col"); match index { - 0 => assert_eq!(decoded, vec![1i64.into(), 2i64.into()]), + 0 => assert_eq!(decoded, vec![1i32.into(), 2i32.into()]), 1 => assert_eq!(decoded, vec!["a".into(), "b".into()]), 2 => assert_eq!( decoded, vec![ - TxnCellValue::Id(RowRef::Existing(RowId { + TxnWireValue::Id(TxnWireRowId::Existing(WireRowId { commit: ha, counter: 7, })), - TxnCellValue::Id(RowRef::Pending(TempRowId(0))), + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId(0))), ] ), _ => unreachable!(), @@ -986,7 +986,7 @@ mod tests { PendingOp::Add { row_id: TempRowId(0), table: 0, - values: vec![1i64.into()], + values: vec![1i32.into()], }, PendingOp::Add { row_id: TempRowId(1), @@ -996,7 +996,7 @@ mod tests { PendingOp::Add { row_id: TempRowId(2), table: 0, - values: vec![2i64.into()], + values: vec![2i32.into()], }, ]; @@ -1099,7 +1099,7 @@ mod tests { PendingOp::Add { row_id: TempRowId(0), table: 0, - values: vec![1i64.into()], + values: vec![1i32.into()], }, PendingOp::Add { row_id: TempRowId(1), @@ -1109,7 +1109,7 @@ mod tests { PendingOp::Add { row_id: TempRowId(2), table: 0, - values: vec![2i64.into()], + values: vec![2i32.into()], }, ]; diff --git a/packages/coln-store/src/commit/wire/prim.rs b/packages/coln-store/src/commit/wire/prim.rs index 02bf174f..076c5696 100644 --- a/packages/coln-store/src/commit/wire/prim.rs +++ b/packages/coln-store/src/commit/wire/prim.rs @@ -8,7 +8,7 @@ use coln_flir_rs::ir::{self, BuiltinTy}; use hexane::{Codec, PackError, lebsize}; use crate::commit::leb128 as commit_leb128; -use crate::{commit::error::CodecError, txn::TxnCellValue}; +use crate::{commit::error::CodecError, txn::TxnWireValue}; /// Number of low bits reserved for the [`ValueType`] code in a [`ValueMeta`]. const TYPE_CODE_BITS: u32 = 5; @@ -145,25 +145,25 @@ impl hexane::PrefixValue for ValueMeta { /// Writes to the buffer `out` the encoded data, and returns the corresponding /// `ValueMeta` representation pub(crate) fn encode_prim_value( - value: &TxnCellValue, + value: &TxnWireValue, prim: &BuiltinTy, out: &mut Vec, ) -> Result { match prim { BuiltinTy::BuiltinInt => { - let TxnCellValue::Int(i) = value else { + let TxnWireValue::Int(i) = value else { return Err(CodecError::SchemaError(format!( "expected int, got {value:?}" ))); }; - leb128::write::signed(out, *i) + leb128::write::signed(out, *i as i64) .map_err(|e| CodecError::DataFormatError(e.to_string()))?; - Ok(ValueMeta::new(ValueType::Leb, lebsize(*i) as usize)) + Ok(ValueMeta::new(ValueType::Leb, lebsize(*i as i64) as usize)) } BuiltinTy::BuiltinStr => { - let TxnCellValue::Str(s) = value else { + let TxnWireValue::Str(s) = value else { return Err(CodecError::SchemaError(format!( "expected string, got {value:?}" ))); @@ -180,7 +180,7 @@ pub(crate) fn decode_prim_value( meta: ValueMeta, prim: &BuiltinTy, bytes: &[u8], -) -> Result { +) -> Result { let ty = meta.type_code(); if !ty.is_valid_for(prim) { return Err(CodecError::SchemaError(format!( @@ -198,12 +198,12 @@ pub(crate) fn decode_prim_value( "trailing bytes in leb value".into(), )); } - Ok(TxnCellValue::Int(i)) + Ok(TxnWireValue::Int(i as i32)) } ValueType::String => { let s = std::str::from_utf8(bytes) .map_err(|_| CodecError::DataFormatError("value column: invalid utf-8".into()))?; - Ok(TxnCellValue::Str(s.to_owned())) + Ok(TxnWireValue::Str(s.to_owned())) } other => Err(CodecError::DataFormatError(format!( "unsupported value type code {other:?}" diff --git a/packages/coln-store/src/commit/wire/root.rs b/packages/coln-store/src/commit/wire/root.rs index b449cfcd..e9847add 100644 --- a/packages/coln-store/src/commit/wire/root.rs +++ b/packages/coln-store/src/commit/wire/root.rs @@ -16,6 +16,8 @@ pub(crate) fn deserialize_root(data: &[u8]) -> Result { #[cfg(test)] mod tests { + use coln_flir_rs::ir::Equality; + use super::*; use crate::ir::{ Atom, BuiltinTy, ColType, ColumnEntry, EntityVariant, Path, Prop, Rule, RuleEntry, @@ -76,8 +78,10 @@ mod tests { }, }], consequents: vec![Prop::Eq { - left: Term::Var { index: 0 }, - right: Term::Var { index: 0 }, + equality: Equality { + left: Term::Var { index: 0 }, + right: Term::Var { index: 0 }, + }, }], }, } diff --git a/packages/coln-store/src/id_packer.rs b/packages/coln-store/src/id_packer.rs index 8c5cb487..a2d00078 100644 --- a/packages/coln-store/src/id_packer.rs +++ b/packages/coln-store/src/id_packer.rs @@ -5,7 +5,7 @@ use crate::commit::hash_dict::HashMapper; use crate::op::Op; use crate::rollback::Rollback; -use crate::table::{CellValue, PackedCell, PackedOp, PackedRowId, RowId}; +use crate::table::{PackedOp, PackedRowId, PackedValue, WireRowId, WireValue}; /// A packer doing dictionary encoding while supporting rollbacks. #[derive(Debug)] @@ -26,7 +26,7 @@ impl IdPacker { } /// Packs `id`, interning its commit hash if it is new. - pub(crate) fn pack_row_id(&mut self, id: RowId) -> PackedRowId { + pub(crate) fn pack_row_id(&mut self, id: WireRowId) -> PackedRowId { PackedRowId { commit_idx: self.dict.insert(id.commit), counter: id.counter, @@ -36,15 +36,15 @@ impl IdPacker { /// Packs `id` without interning its commit hash. /// /// Returns `None` when the commit hash has not already been interned. - pub(crate) fn lookup_row_id(&self, id: RowId) -> Option { + pub(crate) fn lookup_row_id(&self, id: WireRowId) -> Option { Some(PackedRowId { commit_idx: self.dict.index(id.commit)?, counter: id.counter, }) } - pub(crate) fn unpack_row_id(&self, id: PackedRowId) -> RowId { - RowId { + pub(crate) fn unpack_row_id(&self, id: PackedRowId) -> WireRowId { + WireRowId { commit: self .dict .hash_at(id.commit_idx) @@ -53,11 +53,11 @@ impl IdPacker { } } - pub(crate) fn pack_cell(&mut self, value: CellValue) -> PackedCell { + pub(crate) fn pack_cell(&mut self, value: WireValue) -> PackedValue { match value { - CellValue::Id(id) => PackedCell::Id(self.pack_row_id(id)), - CellValue::Int(value) => PackedCell::Int(value), - CellValue::Str(value) => PackedCell::Str(value), + WireValue::Id(id) => PackedValue::Id(self.pack_row_id(id)), + WireValue::Int(value) => PackedValue::Int(value), + WireValue::Str(value) => PackedValue::Str(value), } } @@ -77,11 +77,11 @@ impl IdPacker { /// Packs a cell without modifying the dictionary. /// /// Returns `None` when an ID cell's commit hash has not been interned. - pub(crate) fn try_pack_cell(&self, value: &CellValue) -> Option { + pub(crate) fn try_pack_cell(&self, value: &WireValue) -> Option { Some(match value { - CellValue::Id(id) => PackedCell::Id(self.lookup_row_id(*id)?), - CellValue::Int(value) => PackedCell::Int(*value), - CellValue::Str(value) => PackedCell::Str(value.clone()), + WireValue::Id(id) => PackedValue::Id(self.lookup_row_id(*id)?), + WireValue::Int(value) => PackedValue::Int(*value), + WireValue::Str(value) => PackedValue::Str(value.clone()), }) } @@ -122,8 +122,8 @@ mod tests { use super::*; use crate::commit::hash::CommitHash; - fn row_id(byte: u8, counter: u32) -> RowId { - RowId { + fn row_id(byte: u8, counter: u32) -> WireRowId { + WireRowId { commit: CommitHash([byte; 32]), counter, } diff --git a/packages/coln-store/src/lib.rs b/packages/coln-store/src/lib.rs index 6b32bb1c..0b134c92 100644 --- a/packages/coln-store/src/lib.rs +++ b/packages/coln-store/src/lib.rs @@ -14,5 +14,6 @@ pub mod solver; pub mod store; pub mod table; pub mod txn; +pub mod value; use coln_flir_rs::ir; diff --git a/packages/coln-store/src/op.rs b/packages/coln-store/src/op.rs index 81bea0fe..91dc5196 100644 --- a/packages/coln-store/src/op.rs +++ b/packages/coln-store/src/op.rs @@ -2,16 +2,16 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::table::{CellValue, RowId, TableOid}; +use crate::table::{TableOid, WireRowId, WireValue}; pub const OP_KIND_ADD: u32 = 0; #[derive(Debug, PartialEq, Eq, Clone)] pub enum Op { Add { - row_id: RowId, + row_id: WireRowId, table: TableOid, - values: Vec, + values: Vec, }, // Delete { // row_id: RowId, @@ -20,7 +20,7 @@ pub enum Op { } impl Op { - pub fn id(&self) -> RowId { + pub fn id(&self) -> WireRowId { match self { Op::Add { row_id, .. } => *row_id, } diff --git a/packages/coln-store/src/repl/exe/mod.rs b/packages/coln-store/src/repl/exe/mod.rs index f90fa71a..3ba464d7 100644 --- a/packages/coln-store/src/repl/exe/mod.rs +++ b/packages/coln-store/src/repl/exe/mod.rs @@ -12,17 +12,22 @@ use std::{ use anyhow::{Context, Result, anyhow, bail}; -use crate::repl::{ - Session, ShellMode, Step, - parse::coln::{self, BatchAssignment, parse_cell_value, parse_cell_value_batch}, - parse::{ColnCommand, MetaCommand, SqlCommand}, -}; use crate::{ commit::pst::{decode_store, encode_store}, ir::{BuiltinTy, ColType, ColumnEntry, FlatRealm}, store::Store, - table::{RowId, TableRef}, - txn::{TempRowId, TxnCellValue}, + table::{TableRef, WireRowId}, + txn::{TempRowId, TxnWireValue}, +}; +use crate::{ + repl::{ + Session, ShellMode, Step, + parse::{ + ColnCommand, MetaCommand, SqlCommand, + coln::{self, BatchAssignment, parse_cell_value, parse_cell_value_batch}, + }, + }, + txn::TxnWireRowId, }; fn help_text(mode: ShellMode) -> String { @@ -387,7 +392,7 @@ pub fn add_rows( store: &mut Store, table_name: &str, raw_rows: &[Vec], -) -> Result> { +) -> Result> { let table_path = crate::ir::Path::from(table_name); let oid = store .resolve_table(&table_path) @@ -474,7 +479,7 @@ pub fn run_transact(store: &mut Store, assignments: &[BatchAssignment]) -> Resul Ok(message) } -fn parse_txn_values(table: TableRef<'_>, raw_values: &[String]) -> Result> { +fn parse_txn_values(table: TableRef<'_>, raw_values: &[String]) -> Result> { let expected = table.schema().columns.len(); if raw_values.len() != expected { bail!( @@ -488,10 +493,10 @@ fn parse_txn_values(table: TableRef<'_>, raw_values: &[String]) -> Result Result { + .map(|(idx, column)| -> Result { let raw = &raw_values[idx]; parse_cell_value(&column.col_type, raw) - .map(Into::into) + .map(|v| v.map_owned(TxnWireRowId::Existing)) .map_err(|message| anyhow!("column {idx}: {message}")) }) .collect() diff --git a/packages/coln-store/src/repl/parse/coln.rs b/packages/coln-store/src/repl/parse/coln.rs index 41f62e5a..91a8f74d 100644 --- a/packages/coln-store/src/repl/parse/coln.rs +++ b/packages/coln-store/src/repl/parse/coln.rs @@ -7,8 +7,8 @@ use std::collections::HashMap; use crate::{ commit::hash::{CommitHash, HASH_SIZE}, ir::{BuiltinTy, ColType}, - table::{CellValue, RowId}, - txn::{TempRowId, TxnCellValue}, + table::{WireRowId, WireValue}, + txn::{TempRowId, TxnWireRowId, TxnWireValue}, }; /// Parse failure for a single cell inside a `begin batch` block (before column index is known). @@ -94,24 +94,24 @@ pub(crate) fn parse_cell_value_batch( col_type: &ColType, raw: &str, bindings: &HashMap, -) -> Result { +) -> Result { match col_type { ColType::RowId { .. } => { if raw.starts_with('#') { - parse_cell_value(col_type, raw).map(Into::into) + parse_cell_value(col_type, raw).map(|v| v.map_owned(TxnWireRowId::Existing)) } else if is_binding_ident(raw) { let id = bindings .get(raw) .copied() .ok_or_else(|| ParserError::UnknownBinding(raw.to_string()))?; - Ok(TxnCellValue::from(id)) + Ok(TxnWireValue::Id(TxnWireRowId::Pending(id))) } else { Err(ParserError::InvalidValue(format!( "expected entity id like #: or a binding name, got {raw}" ))) } } - _ => parse_cell_value(col_type, raw).map(Into::into), + _ => parse_cell_value(col_type, raw).map(|v| v.map_owned(TxnWireRowId::Existing)), } } @@ -308,20 +308,20 @@ fn split_add_row_tokens(row_src: &str) -> Vec { out } -pub(crate) fn parse_cell_value(col_type: &ColType, raw: &str) -> Result { +pub(crate) fn parse_cell_value(col_type: &ColType, raw: &str) -> Result { match col_type { - ColType::RowId { .. } => parse_row_id(raw).map(CellValue::Id), + ColType::RowId { .. } => parse_row_id(raw).map(WireValue::Id), ColType::BuiltinTy { builtin_ty } => match builtin_ty { BuiltinTy::BuiltinInt => raw - .parse::() - .map(CellValue::Int) + .parse::() + .map(WireValue::Int) .map_err(|_| ParserError::InvalidValue(format!("invalid int: {raw}"))), - BuiltinTy::BuiltinStr => Ok(CellValue::Str(raw.to_string())), + BuiltinTy::BuiltinStr => Ok(WireValue::Str(raw.to_string())), }, } } -fn parse_row_id(raw: &str) -> Result { +fn parse_row_id(raw: &str) -> Result { let Some(rest) = raw.strip_prefix('#') else { return invalid_input_err("expected entity id like #:"); }; @@ -339,7 +339,7 @@ fn parse_row_id(raw: &str) -> Result { let counter = counter_raw .parse::() .map_err(|_| ParserError::InvalidValue(format!("invalid entity id: {raw}")))?; - Ok(RowId { + Ok(WireRowId { commit: CommitHash(hash), counter, }) diff --git a/packages/coln-store/src/rowing/mod.rs b/packages/coln-store/src/rowing/mod.rs index 1ae7762a..a6c18f16 100644 --- a/packages/coln-store/src/rowing/mod.rs +++ b/packages/coln-store/src/rowing/mod.rs @@ -146,12 +146,12 @@ impl Rowing { #[cfg(test)] mod tests { use crate::commit::hash::CommitHash; - use crate::table::RowId; + use crate::table::WireRowId; use super::*; - fn row_id(byte: u8) -> RowId { - RowId { + fn row_id(byte: u8) -> WireRowId { + WireRowId { commit: CommitHash([byte; 32]), counter: 0, } diff --git a/packages/coln-store/src/rowing/uf.rs b/packages/coln-store/src/rowing/uf.rs index 20be0b12..92f243f4 100644 --- a/packages/coln-store/src/rowing/uf.rs +++ b/packages/coln-store/src/rowing/uf.rs @@ -4,7 +4,7 @@ use ena::unify::{InPlaceUnificationTable, UnifyKey, UnifyValue}; -use crate::table::RowId; +use crate::table::WireRowId; pub(super) type UnionFind = InPlaceUnificationTable; @@ -12,7 +12,7 @@ pub(super) type UnionFind = InPlaceUnificationTable; pub(super) struct NodeId(u32); impl UnifyKey for NodeId { - type Value = RowId; + type Value = WireRowId; fn index(&self) -> u32 { self.0 @@ -27,7 +27,7 @@ impl UnifyKey for NodeId { } } -impl UnifyValue for RowId { +impl UnifyValue for WireRowId { type Error = ena::unify::NoError; fn unify_values(value1: &Self, value2: &Self) -> Result { diff --git a/packages/coln-store/src/solver/bind.rs b/packages/coln-store/src/solver/bind.rs index 6330cdbf..74500a9c 100644 --- a/packages/coln-store/src/solver/bind.rs +++ b/packages/coln-store/src/solver/bind.rs @@ -11,13 +11,13 @@ use crate::{ matcher::term_matches, }, store::Store, - table::{CellValue, RowId, TableRef}, + table::{TableRef, WireRowId, WireValue}, }; #[derive(Debug, PartialEq, Eq, Clone)] pub enum BoundValue { - RId(RowId), - Cell(CellValue), + RId(WireRowId), + Cell(WireValue), } pub type Binding = Vec>; @@ -72,9 +72,9 @@ fn match_atom_row( pub fn eval_term(binding: &Binding, term: &CompTerm) -> Option { match term { CompTerm::Var(slot) => binding.get(*slot).and_then(|v| v.clone()), - CompTerm::Lit(ir::Lit::Int { value }) => Some(BoundValue::Cell(CellValue::Int(*value))), + CompTerm::Lit(ir::Lit::Int { value }) => Some(BoundValue::Cell(WireValue::Int(*value))), CompTerm::Lit(ir::Lit::String { value }) => { - Some(BoundValue::Cell(CellValue::Str(value.clone()))) + Some(BoundValue::Cell(WireValue::Str(value.clone()))) } } } @@ -143,6 +143,8 @@ pub fn bind_rule(store: &Store, rule: &CompRule) -> Vec { #[cfg(test)] mod tests { + use coln_flir_rs::ir::Equality; + use super::*; use crate::{ ir::{ @@ -150,7 +152,7 @@ mod tests { RuleVariant, Schema, }, solver::compile::compile_rule, - table::CellValue, + table::WireValue, }; fn int_ty() -> ColType { @@ -203,21 +205,9 @@ mod tests { .expect("create table"); let mut txn = store.transaction(); - txn.add( - &path, - vec![CellValue::Int(1).into(), CellValue::Int(2).into()], - ) - .expect("insert row"); - txn.add( - &path, - vec![CellValue::Int(2).into(), CellValue::Int(3).into()], - ) - .expect("insert row"); - txn.add( - &path, - vec![CellValue::Int(9).into(), CellValue::Int(4).into()], - ) - .expect("insert row"); + txn.add(&path, vec![1i32, 2i32]).expect("insert row"); + txn.add(&path, vec![2i32, 3i32]).expect("insert row"); + txn.add(&path, vec![9i32, 4i32]).expect("insert row"); txn.commit().expect("commit rows"); let rule = enforced_rule( @@ -273,14 +263,14 @@ mod tests { let bindings = bind_rule(&store, &compiled); assert_eq!(bindings.len(), 1); - assert_eq!(bindings[0][0], Some(BoundValue::Cell(CellValue::Int(1)))); - assert_eq!(bindings[0][1], Some(BoundValue::Cell(CellValue::Int(2)))); - assert_eq!(bindings[0][2], Some(BoundValue::Cell(CellValue::Int(3)))); + assert_eq!(bindings[0][0], Some(BoundValue::Cell(WireValue::Int(1)))); + assert_eq!(bindings[0][1], Some(BoundValue::Cell(WireValue::Int(2)))); + assert_eq!(bindings[0][2], Some(BoundValue::Cell(WireValue::Int(3)))); } /// Build a single-column `Int` table `T` populated with the supplied /// values, returning the store. - fn store_with_int_column(values: &[i64]) -> (Store, Path) { + fn store_with_int_column(values: &[i32]) -> (Store, Path) { let path = Path::from("T"); let mut store = Store::new(); store @@ -288,8 +278,7 @@ mod tests { .expect("create table"); let mut txn = store.transaction(); for value in values { - txn.add(&path, vec![CellValue::Int(*value).into()]) - .expect("insert row"); + txn.add(&path, vec![(*value)]).expect("insert row"); } txn.commit().expect("commit rows"); (store, path) @@ -324,8 +313,10 @@ mod tests { }, }, ir::Prop::Eq { - left: ir::Term::Var { index: 0 }, - right: ir::Term::Var { index: 1 }, + equality: Equality { + left: ir::Term::Var { index: 0 }, + right: ir::Term::Var { index: 1 }, + }, }, ], vec![ir::Prop::Atom { @@ -368,9 +359,11 @@ mod tests { }, }, ir::Prop::Eq { - left: ir::Term::Var { index: 0 }, - right: ir::Term::Lit { - lit: ir::Lit::Int { value: 2 }, + equality: Equality { + left: ir::Term::Var { index: 0 }, + right: ir::Term::Lit { + lit: ir::Lit::Int { value: 2 }, + }, }, }, ], @@ -390,6 +383,6 @@ mod tests { let bindings = bind_rule(&store, &compiled); assert_eq!(bindings.len(), 1); - assert_eq!(bindings[0][0], Some(BoundValue::Cell(CellValue::Int(2)))); + assert_eq!(bindings[0][0], Some(BoundValue::Cell(WireValue::Int(2)))); } } diff --git a/packages/coln-store/src/solver/compile.rs b/packages/coln-store/src/solver/compile.rs index ab0e02bb..9067d9b9 100644 --- a/packages/coln-store/src/solver/compile.rs +++ b/packages/coln-store/src/solver/compile.rs @@ -14,9 +14,9 @@ pub enum CompileError { #[error("unsupported term")] UnsupportedTerm, #[error("invalid var index: {index} var_count {var_count}")] - InvalidVarIndex { index: i64, var_count: usize }, + InvalidVarIndex { index: u64, var_count: usize }, #[error("invalid column index {column}")] - InvalidColumnIndex { column: i64 }, + InvalidColumnIndex { column: u64 }, } /// A rule lowered into a small execution-oriented rule form. @@ -100,7 +100,7 @@ pub fn compile_rule(rule_entry: &RuleEntry) -> Result { let antecedent = compile_props(&rule_entry.rule.antecedents, var_count)?; let consequent = compile_props(&rule_entry.rule.consequents, var_count)?; - let rule_variant = rule_entry.rule.rule_variant.clone(); + let rule_variant = rule_entry.rule.rule_variant; let mut seen = HashSet::new(); let mut tables = Vec::new(); @@ -148,9 +148,9 @@ fn compile_props(props: &[Prop], var_count: usize) -> Result Result { match prop { Prop::Atom { atom } => Ok(CompProp::Atom(compile_atom(atom, var_count)?)), - Prop::Eq { left, right } => Ok(CompProp::Eq(CompEq { - left: compile_term(left, var_count)?, - right: compile_term(right, var_count)?, + Prop::Eq { equality } => Ok(CompProp::Eq(CompEq { + left: compile_term(&equality.left, var_count)?, + right: compile_term(&equality.right, var_count)?, })), } } @@ -188,17 +188,13 @@ fn compile_atom(atom: &Atom, var_count: usize) -> Result fn compile_term(term: &Term, var_count: usize) -> Result { match term { Term::Var { index } => { - let index = usize::try_from(*index).map_err(|_| CompileError::InvalidVarIndex { - index: *index, - var_count, - })?; - if index >= var_count { + if *index >= var_count as u64 { return Err(CompileError::InvalidVarIndex { - index: index as i64, + index: *index, var_count, }); } - Ok(CompTerm::Var(index)) + Ok(CompTerm::Var(*index as usize)) } Term::Lit { lit } => Ok(CompTerm::Lit(lit.clone())), } @@ -336,6 +332,8 @@ fn var_name(index: usize) -> String { #[cfg(test)] mod tests { + use coln_flir_rs::ir::Equality; + use super::*; use crate::ir::{BuiltinTy, ColType, Path, Rule, RuleEntry, RuleVariant}; @@ -527,8 +525,10 @@ mod tests { "T.eq_antecedent", vec![int_ty(), int_ty()], vec![Prop::Eq { - left: Term::Var { index: 0 }, - right: Term::Var { index: 1 }, + equality: Equality { + left: Term::Var { index: 0 }, + right: Term::Var { index: 1 }, + }, }], vec![Prop::Atom { atom: Atom { @@ -567,8 +567,10 @@ mod tests { }, }], vec![Prop::Eq { - left: Term::Var { index: 0 }, - right: Term::Var { index: 1 }, + equality: Equality { + left: Term::Var { index: 0 }, + right: Term::Var { index: 1 }, + }, }], ); @@ -621,8 +623,10 @@ mod tests { }, }, Prop::Eq { - left: Term::Var { index: 0 }, - right: Term::Var { index: 1 }, + equality: Equality { + left: Term::Var { index: 0 }, + right: Term::Var { index: 1 }, + }, }, ], ); diff --git a/packages/coln-store/src/solver/matcher.rs b/packages/coln-store/src/solver/matcher.rs index ca2e9d84..3b102042 100644 --- a/packages/coln-store/src/solver/matcher.rs +++ b/packages/coln-store/src/solver/matcher.rs @@ -10,15 +10,15 @@ use crate::{ bind::{Binding, BoundValue}, compile::CompTerm, }, - table::CellValue, + table::WireValue, }; /// Check if an _already_ bound variable slot matches value. pub(crate) fn boundvar_matches(binding: &Binding, slot: usize, value: &BoundValue) -> bool { match binding.get(slot) { Some(Some(bound)) => match (bound, value) { - (BoundValue::RId(a), BoundValue::Cell(CellValue::Id(b))) - | (BoundValue::Cell(CellValue::Id(a)), BoundValue::RId(b)) => { + (BoundValue::RId(a), BoundValue::Cell(WireValue::Id(b))) + | (BoundValue::Cell(WireValue::Id(a)), BoundValue::RId(b)) => { debug!(bound=?bound, value = ?value, "matching"); a == b } @@ -36,10 +36,10 @@ pub(crate) fn term_matches(binding: &Binding, term: &CompTerm, value: &BoundValu match term { CompTerm::Var(slot) => boundvar_matches(binding, *slot, value), CompTerm::Lit(ir::Lit::Int { value: expected }) => { - *value == BoundValue::Cell(CellValue::Int(*expected)) + *value == BoundValue::Cell(WireValue::Int(*expected)) } CompTerm::Lit(ir::Lit::String { value: expected }) => { - *value == BoundValue::Cell(CellValue::Str(expected.clone())) + *value == BoundValue::Cell(WireValue::Str(expected.clone())) } } } diff --git a/packages/coln-store/src/solver/validate.rs b/packages/coln-store/src/solver/validate.rs index 5e2518c5..a8e01eed 100644 --- a/packages/coln-store/src/solver/validate.rs +++ b/packages/coln-store/src/solver/validate.rs @@ -14,7 +14,7 @@ use crate::{ matcher::term_matches, }, store::Store, - table::CellValue, + table::WireValue, }; /// Why a rule was violated at a given binding. @@ -93,8 +93,8 @@ pub fn consequent_eq_holds(binding: &Binding, eq: &CompEq) -> bool { }; match (&l, &r) { // Row ids and entity cells refer to the same identity when equal. - (BoundValue::RId(a), BoundValue::Cell(CellValue::Id(b))) - | (BoundValue::Cell(CellValue::Id(a)), BoundValue::RId(b)) => a == b, + (BoundValue::RId(a), BoundValue::Cell(WireValue::Id(b))) + | (BoundValue::Cell(WireValue::Id(a)), BoundValue::RId(b)) => a == b, _ => l == r, } } @@ -148,6 +148,8 @@ pub fn check_rule(store: &Store, rule: &CompRule) -> Result<(), Box ColType { @@ -244,10 +246,8 @@ mod tests { .expect("create target table"); let mut txn = store.transaction(); - txn.add(&source, vec![CellValue::Int(7).into()]) - .expect("insert source row"); - txn.add(&target, vec![CellValue::Int(7).into()]) - .expect("insert target row"); + txn.add(&source, vec![7i32]).expect("insert source row"); + txn.add(&target, vec![7i32]).expect("insert target row"); txn.commit().expect("commit matching rows"); let rule = enforced_rule( @@ -341,11 +341,8 @@ mod tests { let compiled = compile_rule(&rule).expect("compile rule"); let mut txn = store.transaction(); - txn.add( - &link, - vec![CellValue::Int(10).into(), CellValue::Int(20).into()], - ) - .expect("insert referencing row"); + txn.add(&link, vec![10i32, 20i32]) + .expect("insert referencing row"); txn.commit().expect("commit referencing row"); let violation = check_rule(&store, &compiled).expect_err("missing referenced rows"); @@ -357,16 +354,14 @@ mod tests { assert_eq!( violation.binding, vec![ - Some(BoundValue::Cell(CellValue::Int(10))), - Some(BoundValue::Cell(CellValue::Int(20))), + Some(BoundValue::Cell(WireValue::Int(10))), + Some(BoundValue::Cell(WireValue::Int(20))), ] ); let mut txn = store.transaction(); - txn.add(&left, vec![CellValue::Int(10).into()]) - .expect("insert left row"); - txn.add(&right, vec![CellValue::Int(20).into()]) - .expect("insert right row"); + txn.add(&left, vec![10i32]).expect("insert left row"); + txn.add(&right, vec![20i32]).expect("insert right row"); txn.commit().expect("commit referenced rows"); assert!(check_rule(&store, &compiled).is_ok()); @@ -380,8 +375,7 @@ mod tests { .create_table(t.clone(), int_schema(&["c0", "c1"])) .expect("create table"); let mut txn = store.transaction(); - txn.add(&t, vec![CellValue::Int(5).into(), CellValue::Int(5).into()]) - .expect("insert row"); + txn.add(&t, vec![5i32, 5i32]).expect("insert row"); txn.commit().expect("commit row"); let rule = enforced_rule( @@ -404,8 +398,10 @@ mod tests { }, }], vec![ir::Prop::Eq { - left: ir::Term::Var { index: 0 }, - right: ir::Term::Var { index: 1 }, + equality: Equality { + left: ir::Term::Var { index: 0 }, + right: ir::Term::Var { index: 1 }, + }, }], ); @@ -421,8 +417,7 @@ mod tests { .create_table(t.clone(), int_schema(&["c0", "c1"])) .expect("create table"); let mut txn = store.transaction(); - txn.add(&t, vec![CellValue::Int(1).into(), CellValue::Int(2).into()]) - .expect("insert row"); + txn.add(&t, vec![1i32, 2i32]).expect("insert row"); txn.commit().expect("commit row"); let rule = enforced_rule( @@ -445,8 +440,10 @@ mod tests { }, }], vec![ir::Prop::Eq { - left: ir::Term::Var { index: 0 }, - right: ir::Term::Var { index: 1 }, + equality: Equality { + left: ir::Term::Var { index: 0 }, + right: ir::Term::Var { index: 1 }, + }, }], ); diff --git a/packages/coln-store/src/store/mod.rs b/packages/coln-store/src/store/mod.rs index dd1729dd..f6ef104a 100644 --- a/packages/coln-store/src/store/mod.rs +++ b/packages/coln-store/src/store/mod.rs @@ -22,10 +22,11 @@ use crate::solver::validate::RuleViolation; use crate::solver::{self}; use crate::store::error::{CommitApplyError, StoreError}; use crate::table::{ - CellValue, RowId, RowView, Table, TableMeta, TableOid, TableRef, TableSnapshot, ValidationError, + RowView, Table, TableMeta, TableOid, TableRef, TableSnapshot, ValidationError, WireRowId, + WireValue, }; use crate::txn::{OwnedTransaction, Transaction}; -use crate::{op::Op, txn::RowHandle}; +use crate::{op::Op, txn::TxnLiveRowId}; #[derive(Debug)] pub struct Store { @@ -168,13 +169,13 @@ impl Store { Ok(serde_json::to_string(&realm).map_err(CodecError::from)?) } - pub(crate) fn canonical_row_id(&self, row_id: RowId) -> Option { + pub(crate) fn canonical_row_id(&self, row_id: WireRowId) -> Option { let packed = self.id_packer.lookup_row_id(row_id)?; let canonical = self.rowing.canonical_id(&packed, &self.id_packer); Some(self.id_packer.unpack_row_id(canonical)) } - pub fn row_by_handle(&self, table: &ir::Path, row_handle: RowHandle) -> Option { + pub fn row_by_handle(&self, table: &ir::Path, row_handle: TxnLiveRowId) -> Option { let row_id = row_handle.row_id().ok()?; let con_rowid = self.canonical_row_id(row_id)?; // replace the rowid in the row_handle so it stays canonical @@ -187,7 +188,7 @@ impl Store { // This function will canonicalise the row_id on read, but will not change it // See `row_by_handle` which will actually canonicalise the handle. // We need both because the TS FFI does not deal with handles. - pub fn row_by_id(&self, table: &ir::Path, row_id: RowId) -> Option { + pub fn row_by_id(&self, table: &ir::Path, row_id: WireRowId) -> Option { let row_id = self.canonical_row_id(row_id)?; self.table_at(table) .and_then(|table| table.row_at(table.row_position(row_id)?)) @@ -574,7 +575,7 @@ impl Store { // TODO also need to validate that ids in op is referring to an existing id fn validate_commit_ops(&self, ops: &[Op]) -> Result<(), StoreError> { - let mut pending_pk: HashMap>> = HashMap::new(); + let mut pending_pk: HashMap>> = HashMap::new(); for op in ops { let Op::Add { table, values, .. } = op; diff --git a/packages/coln-store/src/store/tests.rs b/packages/coln-store/src/store/tests.rs index 2570dd0a..8f0d172b 100644 --- a/packages/coln-store/src/store/tests.rs +++ b/packages/coln-store/src/store/tests.rs @@ -118,10 +118,10 @@ fn single_int_store() -> Store { store } -fn commit_int(store: &mut Store, value: i64) -> CommitHash { +fn commit_int(store: &mut Store, value: i32) -> CommitHash { let path = Path::from("T"); let mut tx = store.transaction(); - tx.add(&path, vec![value.into()]).expect("add row"); + tx.add(&path, vec![value]).expect("add row"); tx.commit().expect("commit row") } @@ -211,10 +211,8 @@ mod transactions { .expect("create table"); let mut txn = store.transaction(); - txn.add(&path, vec![CellValue::Int(1).into()]) - .expect("first add"); - txn.add(&path, vec![CellValue::Int(2).into()]) - .expect("second add"); + txn.add(&path, vec![1i32]).expect("first add"); + txn.add(&path, vec![2i32]).expect("second add"); txn.commit().expect("commit"); @@ -243,10 +241,8 @@ mod transactions { let err = { let mut txn = store.transaction(); - txn.add(&path, vec![CellValue::Int(1).into()]) - .expect("first add"); - txn.add(&Path::from("missing"), vec![CellValue::Int(2).into()]) - .unwrap_err() + txn.add(&path, vec![1i32]).expect("first add"); + txn.add(&Path::from("missing"), vec![2i32]).unwrap_err() }; assert!(matches!( @@ -275,10 +271,8 @@ mod transactions { .expect("create table"); let mut txn = store.transaction(); - txn.add(&path, vec![CellValue::Int(1).into()]) - .expect("first add"); - txn.add(&path, vec![CellValue::Int(1).into()]) - .expect("second add"); + txn.add(&path, vec![1i32]).expect("first add"); + txn.add(&path, vec![1i32]).expect("second add"); let err = txn.commit().unwrap_err(); assert!(matches!( @@ -307,13 +301,12 @@ mod transactions { .expect("create table"); let mut txn = store.transaction(); - txn.add(&path, vec![CellValue::Int(42).into()]) - .expect("add"); + txn.add(&path, vec![42i32]).expect("add"); txn.commit().expect("commit"); let t = store.table_at(&path).expect("T"); assert_eq!(t.row_count(), 1); - assert_eq!(t.cell_at(0, 0), Some(CellValue::Int(42))); + assert_eq!(t.cell_at(0, 0), Some(42i32.into())); } #[test] @@ -324,11 +317,7 @@ mod transactions { let packed_id_count = store.id_packer.len(); let mut txn = store.transaction(); - txn.add( - &link, - vec![CellValue::Int(10).into(), CellValue::Int(20).into()], - ) - .expect("add"); + txn.add(&link, vec![10i32, 20i32]).expect("add"); let err = txn.commit().unwrap_err(); assert!(matches!(err, StoreError::Rule(_))); @@ -343,8 +332,7 @@ mod transactions { let store = Store::try_from_ir(theory).expect("theory"); let mut tx = OwnedTransaction::new(store); - tx.add(&link, vec![10_i64.into(), 20_i64.into()]) - .expect("add"); + tx.add(&link, vec![10_i32, 20_i32]).expect("add"); let (err, recovered) = tx.commit().unwrap_err(); assert!(matches!(err, StoreError::Rule(_))); @@ -377,8 +365,8 @@ mod query { .expect("known table") .collect::>(), vec![RowView { - row_id: RowId { commit, counter: 0 }, - values: vec![CellValue::Int(42)], + row_id: WireRowId { commit, counter: 0 }, + values: vec![42i32.into()], }] ); } @@ -388,26 +376,29 @@ mod query { let path = Path::from("T"); let mut store = single_int_store(); let commit = commit_int(&mut store, 42); - let row_id = RowId { commit, counter: 0 }; + let row_id = WireRowId { commit, counter: 0 }; assert_eq!( store.row_by_id(&path, row_id), Some(RowView { row_id, - values: vec![CellValue::Int(42)], + values: vec![42i32.into()], }) ); - assert_eq!(store.row_by_id(&path, RowId { commit, counter: 1 }), None); + assert_eq!( + store.row_by_id(&path, WireRowId { commit, counter: 1 }), + None + ); assert_eq!(store.row_by_id(&Path::from("missing"), row_id), None); } } mod rowing { use super::*; - use crate::txn::TxnValue; + use crate::txn::TxnLiveValue; - fn row_id_from(commit_byte: u8, counter: u32) -> RowId { - RowId { + fn row_id_from(commit_byte: u8, counter: u32) -> WireRowId { + WireRowId { commit: CommitHash([commit_byte; 32]), counter, } @@ -501,7 +492,7 @@ mod rowing { store } - fn add_op(store: &Store, table: &str, rid: RowId, values: Vec) -> Op { + fn add_op(store: &Store, table: &str, rid: WireRowId, values: Vec) -> Op { Op::Add { row_id: rid, table: store @@ -524,7 +515,7 @@ mod rowing { &store, "Term", t_high, - vec![CellValue::Int(7)], + vec![WireValue::Int(7)], )]) .unwrap(); @@ -536,23 +527,23 @@ mod rowing { &store, "Plus", plus, - vec![CellValue::Id(t_high), CellValue::Id(t_high)], + vec![WireValue::Id(t_high), WireValue::Id(t_high)], ), - add_op(&store, "Note", note, vec![CellValue::Id(t_high)]), + add_op(&store, "Note", note, vec![WireValue::Id(t_high)]), ]) .unwrap(); // A smaller equal term swaps the class canonical from t_high to t_low. let t_low = row_id_from(1, 0); store - .apply_ops_and_rebuild(vec![add_op(&store, "Term", t_low, vec![CellValue::Int(7)])]) + .apply_ops_and_rebuild(vec![add_op(&store, "Term", t_low, vec![WireValue::Int(7)])]) .unwrap(); // The stored row is now t_low; the stale id t_high resolves to it. let term_path = Path::from("Term"); let term_view = Some(RowView { row_id: t_low, - values: vec![CellValue::Int(7)], + values: vec![WireValue::Int(7)], }); assert_eq!(store.row_by_id(&term_path, t_low), term_view); assert_eq!(store.row_by_id(&term_path, t_high), term_view); @@ -564,14 +555,14 @@ mod rowing { store.row_by_id(&Path::from("Plus"), plus), Some(RowView { row_id: plus, - values: vec![CellValue::Id(t_low), CellValue::Id(t_low)], + values: vec![WireValue::Id(t_low), WireValue::Id(t_low)], }) ); assert_eq!( store.row_by_id(&Path::from("Note"), note), Some(RowView { row_id: note, - values: vec![CellValue::Id(t_low)], + values: vec![WireValue::Id(t_low)], }) ); } @@ -595,19 +586,19 @@ mod rowing { let dup = row_id_from(4, 0); store .apply_ops_and_rebuild(vec![ - add_op(&store, "Term", t_low, vec![CellValue::Int(7)]), - add_op(&store, "Term", t_high, vec![CellValue::Int(7)]), + add_op(&store, "Term", t_low, vec![WireValue::Int(7)]), + add_op(&store, "Term", t_high, vec![WireValue::Int(7)]), add_op( &store, "Plus", keep, - vec![CellValue::Id(t_high), CellValue::Id(t_high)], + vec![WireValue::Id(t_high), WireValue::Id(t_high)], ), add_op( &store, "Plus", dup, - vec![CellValue::Id(t_high), CellValue::Id(t_high)], + vec![WireValue::Id(t_high), WireValue::Id(t_high)], ), ]) .expect("duplicates merge rather than failing the commit"); @@ -619,7 +610,7 @@ mod rowing { // The surviving row keeps the canonical id and names canonical children. assert_eq!(plus[0].row_id, keep); - assert_eq!(plus[0].values, [CellValue::Id(t_low), CellValue::Id(t_low)]); + assert_eq!(plus[0].values, [WireValue::Id(t_low), WireValue::Id(t_low)]); // Both stale ids still resolve to what replaced them. let plus_path = Path::from("Plus"); assert_eq!( @@ -642,15 +633,15 @@ mod rowing { let plus = row_id_from(3, 0); store .apply_ops_and_rebuild(vec![ - add_op(&store, "Term", t_low, vec![CellValue::Int(7)]), - add_op(&store, "Term", u_low, vec![CellValue::Int(8)]), - add_op(&store, "Term", t_high, vec![CellValue::Int(7)]), - add_op(&store, "Term", u_high, vec![CellValue::Int(8)]), + add_op(&store, "Term", t_low, vec![WireValue::Int(7)]), + add_op(&store, "Term", u_low, vec![WireValue::Int(8)]), + add_op(&store, "Term", t_high, vec![WireValue::Int(7)]), + add_op(&store, "Term", u_high, vec![WireValue::Int(8)]), add_op( &store, "Plus", plus, - vec![CellValue::Id(t_high), CellValue::Id(u_high)], + vec![WireValue::Id(t_high), WireValue::Id(u_high)], ), ]) .expect("duplicates merge rather than failing the commit"); @@ -661,7 +652,7 @@ mod rowing { store.row_by_id(&Path::from("Plus"), plus), Some(RowView { row_id: plus, - values: vec![CellValue::Id(t_low), CellValue::Id(u_low)], + values: vec![WireValue::Id(t_low), WireValue::Id(u_low)], }) ); } @@ -675,23 +666,29 @@ mod rowing { let mult_path = Path::from("Mult"); let mut txn = store.transaction(); - let t7 = txn.add(&term_path, vec![TxnValue::Int(7)]).unwrap(); - let t8 = txn.add(&term_path, vec![TxnValue::Int(8)]).unwrap(); + let t7 = txn.add(&term_path, vec![TxnLiveValue::Int(7)]).unwrap(); + let t8 = txn.add(&term_path, vec![TxnLiveValue::Int(8)]).unwrap(); let tp = txn - .add(&plus_path, vec![TxnValue::Id(t7), TxnValue::Id(t8)]) - .unwrap(); - txn.add(&mult_path, vec![TxnValue::Id(tp.clone()), TxnValue::Id(tp)]) + .add(&plus_path, vec![TxnLiveValue::Id(t7), TxnLiveValue::Id(t8)]) .unwrap(); + txn.add( + &mult_path, + vec![TxnLiveValue::Id(tp.clone()), TxnLiveValue::Id(tp)], + ) + .unwrap(); txn.commit().unwrap(); let mut txn2 = store.transaction(); - let t7 = txn2.add(&term_path, vec![TxnValue::Int(7)]).unwrap(); - let t8 = txn2.add(&term_path, vec![TxnValue::Int(8)]).unwrap(); + let t7 = txn2.add(&term_path, vec![TxnLiveValue::Int(7)]).unwrap(); + let t8 = txn2.add(&term_path, vec![TxnLiveValue::Int(8)]).unwrap(); let tp = txn2 - .add(&plus_path, vec![TxnValue::Id(t7), TxnValue::Id(t8)]) - .unwrap(); - txn2.add(&mult_path, vec![TxnValue::Id(tp.clone()), TxnValue::Id(tp)]) + .add(&plus_path, vec![TxnLiveValue::Id(t7), TxnLiveValue::Id(t8)]) .unwrap(); + txn2.add( + &mult_path, + vec![TxnLiveValue::Id(tp.clone()), TxnLiveValue::Id(tp)], + ) + .unwrap(); txn2.commit().unwrap(); let terms: Vec = store.scan_table(&term_path).unwrap().collect(); @@ -704,10 +701,10 @@ mod rowing { assert_eq!(plus.len(), 1); assert_eq!(mult.len(), 1); - let term_id = |value: i64| { + let term_id = |value: i32| { let matching: Vec<&RowView> = terms .iter() - .filter(|row| row.values == [CellValue::Int(value)]) + .filter(|row| row.values == [value.into()]) .collect(); assert_eq!(matching.len(), 1, "exactly one Term({value})"); matching[0].row_id @@ -717,10 +714,10 @@ mod rowing { // Each surviving row references the canonical id of its children, not the // duplicate the second commit allocated for them. - assert_eq!(plus[0].values, [CellValue::Id(t7), CellValue::Id(t8)]); + assert_eq!(plus[0].values, [WireValue::Id(t7), WireValue::Id(t8)]); assert_eq!( mult[0].values, - [CellValue::Id(plus[0].row_id), CellValue::Id(plus[0].row_id)] + [WireValue::Id(plus[0].row_id), WireValue::Id(plus[0].row_id)] ); } @@ -738,11 +735,17 @@ mod rowing { let f = Path::from("F"); let mut first = store.transaction(); - let t1 = first.add(&term, vec![TxnValue::Int(1)]).expect("Term(1)"); - let t2 = first.add(&term, vec![TxnValue::Int(2)]).expect("Term(2)"); - first.add(&term, vec![TxnValue::Int(3)]).expect("Term(3)"); + let t1 = first + .add(&term, vec![TxnLiveValue::Int(1)]) + .expect("Term(1)"); + let t2 = first + .add(&term, vec![TxnLiveValue::Int(2)]) + .expect("Term(2)"); + first + .add(&term, vec![TxnLiveValue::Int(3)]) + .expect("Term(3)"); first - .add(&f, vec![TxnValue::Id(t1), TxnValue::Id(t2)]) + .add(&f, vec![TxnLiveValue::Id(t1), TxnLiveValue::Id(t2)]) .expect("F(Term1, Term2)"); first.commit().expect("x is mapped only once"); @@ -756,10 +759,14 @@ mod rowing { // the two Term(1) rows canonicalise onto one id and F's x cell is // rewritten, which is why the check cannot live in the pre-apply pass. let mut second = store.transaction(); - let t1_again = second.add(&term, vec![TxnValue::Int(1)]).expect("Term(1)"); - let t4 = second.add(&term, vec![TxnValue::Int(4)]).expect("Term(4)"); + let t1_again = second + .add(&term, vec![TxnLiveValue::Int(1)]) + .expect("Term(1)"); + let t4 = second + .add(&term, vec![TxnLiveValue::Int(4)]) + .expect("Term(4)"); second - .add(&f, vec![TxnValue::Id(t1_again), TxnValue::Id(t4)]) + .add(&f, vec![TxnLiveValue::Id(t1_again), TxnLiveValue::Id(t4)]) .expect("F(Term1, Term4)"); let err = second.commit().unwrap_err(); @@ -857,7 +864,7 @@ mod commits { assert!(pending.is_empty()); let table = restored.table_at(&Path::from("T")).expect("table"); - assert_eq!(table.cell_at(0, 0), Some(CellValue::Int(99))); + assert_eq!(table.cell_at(0, 0), Some(WireValue::Int(99))); assert_eq!(restored.heads(), vec![commit]); } @@ -872,7 +879,7 @@ mod commits { let table = target.table_at(&Path::from("T")).expect("table"); assert_eq!(table.row_count(), 1); - assert_eq!(table.cell_at(0, 0), Some(CellValue::Int(99))); + assert_eq!(table.cell_at(0, 0), Some(WireValue::Int(99))); assert_eq!(table.row_id_at(0).expect("row id").commit, commit); assert_eq!(target.heads(), source.heads()); } @@ -890,8 +897,8 @@ mod commits { let table = target.table_at(&Path::from("T")).expect("table"); assert_eq!(table.row_count(), 2); - assert_eq!(table.cell_at(0, 0), Some(CellValue::Int(1))); - assert_eq!(table.cell_at(1, 0), Some(CellValue::Int(2))); + assert_eq!(table.cell_at(0, 0), Some(WireValue::Int(1))); + assert_eq!(table.cell_at(1, 0), Some(WireValue::Int(2))); assert_eq!(target.heads(), source.heads()); } @@ -1005,8 +1012,8 @@ mod commits { let table = target.table_at(&Path::from("T")).expect("table"); assert_eq!(table.row_count(), 2); - assert_eq!(table.cell_at(0, 0), Some(CellValue::Int(1))); - assert_eq!(table.cell_at(1, 0), Some(CellValue::Int(2))); + assert_eq!(table.cell_at(0, 0), Some(1i32.into())); + assert_eq!(table.cell_at(1, 0), Some(2i32.into())); assert_eq!(target.heads(), vec![second]); } } diff --git a/packages/coln-store/src/table/cell.rs b/packages/coln-store/src/table/cell.rs index c347501f..48d47667 100644 --- a/packages/coln-store/src/table/cell.rs +++ b/packages/coln-store/src/table/cell.rs @@ -2,24 +2,29 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT +use serde::{Deserialize, Serialize}; +use specta::Type; use std::fmt; use crate::column_map::ColIndex; use crate::commit::hash::CommitHash; use crate::ir::{BuiltinTy, ColType}; +use crate::value::Value; use super::ValidationError; /// The unique id that identifies each row in a table. /// /// It is managed by the database and read-only for the user. -#[derive(Copy, Clone, Eq, PartialEq, Ord, PartialOrd, Debug, Hash)] -pub struct RowId { +#[derive( + Copy, Clone, Eq, PartialEq, Ord, PartialOrd, Debug, Hash, Serialize, Deserialize, Type, +)] +pub struct WireRowId { pub commit: CommitHash, pub counter: u32, } -impl fmt::Display for RowId { +impl fmt::Display for WireRowId { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { for byte in &self.commit.0[..6] { write!(f, "{byte:02x}")?; @@ -28,13 +33,7 @@ impl fmt::Display for RowId { } } -/// One cell in columnar storage: an entity id or a primitive value. -#[derive(Debug, Clone, PartialOrd, PartialEq, Eq, Hash)] -pub enum CellValue { - Id(RowId), - Int(i64), - Str(String), -} +pub type WireValue = Value; #[derive(Debug, PartialEq, Eq, Clone, Copy)] pub enum CellKind { @@ -57,12 +56,12 @@ impl From<&ColType> for CellKind { } } -impl From<&CellValue> for CellKind { - fn from(value: &CellValue) -> Self { +impl From<&WireValue> for CellKind { + fn from(value: &WireValue) -> Self { match value { - CellValue::Id(_) => CellKind::RowId, - CellValue::Int(_) => CellKind::Int, - CellValue::Str(_) => CellKind::Str, + WireValue::Id(_) => CellKind::RowId, + WireValue::Int(_) => CellKind::Int, + WireValue::Str(_) => CellKind::Str, } } } @@ -77,7 +76,7 @@ impl fmt::Display for CellKind { } } -impl CellValue { +impl WireValue { pub(super) fn matches_schema( &self, col_type: &ColType, @@ -97,12 +96,12 @@ impl CellValue { } } -impl fmt::Display for CellValue { +impl fmt::Display for WireValue { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { - CellValue::Id(id) => write!(f, "#{id}"), - CellValue::Int(value) => write!(f, "{value}"), - CellValue::Str(value) => write!(f, "{value:?}"), + WireValue::Id(id) => write!(f, "#{id}"), + WireValue::Int(value) => write!(f, "{value}"), + WireValue::Str(value) => write!(f, "{value:?}"), } } } @@ -166,9 +165,4 @@ impl ColIndex for PackedRowId { } } -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) enum PackedCell { - Id(PackedRowId), - Int(i64), - Str(String), -} +pub type PackedValue = Value; diff --git a/packages/coln-store/src/table/col.rs b/packages/coln-store/src/table/col.rs index 35d9af9f..0bb5d479 100644 --- a/packages/coln-store/src/table/col.rs +++ b/packages/coln-store/src/table/col.rs @@ -7,7 +7,7 @@ use std::ops::Range; use crate::id_packer::IdPacker; -use super::{CellKind, CellValue, PackedCell, PackedRowId}; +use super::{CellKind, PackedRowId, PackedValue, WireValue}; /// Columnar storage for [`PackedRowId`]s, split into two parallel columns. /// @@ -90,11 +90,11 @@ impl IdColumn { } /// One column of typed storage. The variant is fixed by the schema column type. -/// Each id is 8 bytes instead of a 40-byte [`CellValue`]. +/// Each id is 8 bytes instead of a 40-byte [`WireValue`]. #[derive(Debug, Clone)] pub(super) enum Column { Id(IdColumn), - Int(hexane::Column), + Int(hexane::Column), // TODO: change to i32 Str(hexane::Column), } @@ -111,11 +111,11 @@ impl Column { /// /// Panics on a type mismatch, which `Table::validate_insert` rules out /// before rows reach storage. - pub(super) fn insert(&mut self, row: usize, value: PackedCell) { + pub(super) fn insert(&mut self, row: usize, value: PackedValue) { match (self, value) { - (Column::Id(cells), PackedCell::Id(id)) => cells.insert(row, id), - (Column::Int(cells), PackedCell::Int(value)) => cells.insert(row, value), - (Column::Str(cells), PackedCell::Str(value)) => cells.insert(row, value), + (Column::Id(cells), PackedValue::Id(id)) => cells.insert(row, id), + (Column::Int(cells), PackedValue::Int(value)) => cells.insert(row, value as i64), + (Column::Str(cells), PackedValue::Str(value)) => cells.insert(row, value), (column, value) => panic!( "cell type mismatch: column stores {:?}, got {value:?}", CellKind::from(&*column) @@ -131,29 +131,31 @@ impl Column { } } - pub(super) fn get(&self, row: usize, packer: &IdPacker) -> Option { + pub(super) fn get(&self, row: usize, packer: &IdPacker) -> Option { match self { Column::Id(cells) => cells .get(row) - .map(|id| CellValue::Id(packer.unpack_row_id(id))), - Column::Int(cells) => cells.get(row).map(CellValue::Int), - Column::Str(cells) => cells.get(row).map(|s| CellValue::Str(s.to_owned())), + .map(|id| WireValue::Id(packer.unpack_row_id(id))), + Column::Int(cells) => cells.get(row).map(|i| WireValue::Int(i as i32)), + Column::Str(cells) => cells.get(row).map(|s| WireValue::Str(s.to_owned())), } } - pub(super) fn get_packed(&self, row: usize) -> Option { + pub(super) fn get_packed(&self, row: usize) -> Option { match self { - Column::Id(cells) => cells.get(row).map(PackedCell::Id), - Column::Int(cells) => cells.get(row).map(PackedCell::Int), - Column::Str(cells) => cells.get(row).map(|s| PackedCell::Str(s.to_owned())), + Column::Id(cells) => cells.get(row).map(PackedValue::Id), + Column::Int(cells) => cells.get(row).map(|i| PackedValue::Int(i as i32)), + Column::Str(cells) => cells.get(row).map(|s| PackedValue::Str(s.to_owned())), } } - pub(super) fn scope_to_value(&self, value: &PackedCell, range: Range) -> Range { + pub(super) fn scope_to_value(&self, value: &PackedValue, range: Range) -> Range { match (self, value) { - (Column::Id(column), PackedCell::Id(value)) => column.scope_to_value(*value, range), - (Column::Int(column), PackedCell::Int(value)) => column.scope_to_value(*value, range), - (Column::Str(column), PackedCell::Str(value)) => { + (Column::Id(column), PackedValue::Id(value)) => column.scope_to_value(*value, range), + (Column::Int(column), PackedValue::Int(value)) => { + column.scope_to_value(*value as i64, range) + } + (Column::Str(column), PackedValue::Str(value)) => { column.scope_to_value(value.as_str(), range) } (column, value) => panic!( diff --git a/packages/coln-store/src/table/index.rs b/packages/coln-store/src/table/index.rs index 840e8f14..f31a188a 100644 --- a/packages/coln-store/src/table/index.rs +++ b/packages/coln-store/src/table/index.rs @@ -15,7 +15,7 @@ use std::ops::Range; use crate::ir::Schema; -use super::{CellKind, Column, IdColumn, PackedCell, PackedRowId}; +use super::{CellKind, Column, IdColumn, PackedRowId, PackedValue}; pub(crate) type IndexId = usize; @@ -57,7 +57,7 @@ impl TableIndex { } } - pub(super) fn insert(&mut self, key: Vec, value: PackedRowId) { + pub(super) fn insert(&mut self, key: Vec, value: PackedRowId) { let key_range = self.scope_key(&key); let value_range = self.values.scope_to_value(value, key_range); let position = value_range.end; @@ -68,7 +68,7 @@ impl TableIndex { self.values.insert(position, value); } - pub(super) fn remove(&mut self, key: &[PackedCell], value: PackedRowId) { + pub(super) fn remove(&mut self, key: &[PackedValue], value: PackedRowId) { let key_range = self.scope_key(key); let value_range = self.values.scope_to_value(value, key_range); for position in value_range.rev() { @@ -79,15 +79,15 @@ impl TableIndex { } } - pub(super) fn get(&self, key: &[PackedCell]) -> impl Iterator { + pub(super) fn get(&self, key: &[PackedValue]) -> impl Iterator { self.scope_key(key).map(|position| self.values.at(position)) } - pub(super) fn contains_key(&self, key: &[PackedCell]) -> bool { + pub(super) fn contains_key(&self, key: &[PackedValue]) -> bool { self.get(key).next().is_some() } - fn scope_key(&self, key: &[PackedCell]) -> Range { + fn scope_key(&self, key: &[PackedValue]) -> Range { debug_assert_eq!( key.len(), self.keys.len(), @@ -136,16 +136,16 @@ mod tests { let rows = [(5, packed(1)), (1, packed(2)), (9, packed(3))]; for (key, row_id) in rows { - index.insert(vec![PackedCell::Int(key)], row_id); + index.insert(vec![PackedValue::Int(key)], row_id); } for (key, row_id) in rows { assert_eq!( - index.get(&[PackedCell::Int(key)]).collect::>(), + index.get(&[PackedValue::Int(key)]).collect::>(), vec![row_id] ); } - assert!(!index.contains_key(&[PackedCell::Int(3)])); + assert!(!index.contains_key(&[PackedValue::Int(3)])); } /// If there are multiple keys of the same value, then `get` returns an iterator @@ -156,13 +156,13 @@ mod tests { let first = packed(1); let second = packed(2); let third = packed(3); - index.insert(vec![PackedCell::Int(5)], third); - index.insert(vec![PackedCell::Int(7)], packed(4)); - index.insert(vec![PackedCell::Int(5)], first); - index.insert(vec![PackedCell::Int(5)], second); + index.insert(vec![PackedValue::Int(5)], third); + index.insert(vec![PackedValue::Int(7)], packed(4)); + index.insert(vec![PackedValue::Int(5)], first); + index.insert(vec![PackedValue::Int(5)], second); assert_eq!( - index.get(&[PackedCell::Int(5)]).collect::>(), + index.get(&[PackedValue::Int(5)]).collect::>(), vec![first, second, third] ); } @@ -175,16 +175,16 @@ mod tests { let first = packed(1); let second = packed(2); let other = packed(3); - index.insert(vec![PackedCell::Int(5)], second); - index.insert(vec![PackedCell::Int(7)], other); - index.insert(vec![PackedCell::Int(5)], first); + index.insert(vec![PackedValue::Int(5)], second); + index.insert(vec![PackedValue::Int(7)], other); + index.insert(vec![PackedValue::Int(5)], first); - index.remove(&[PackedCell::Int(5)], first); - index.remove(&[PackedCell::Int(5)], second); + index.remove(&[PackedValue::Int(5)], first); + index.remove(&[PackedValue::Int(5)], second); - assert_eq!(index.get(&[PackedCell::Int(5)]).next(), None); + assert_eq!(index.get(&[PackedValue::Int(5)]).next(), None); assert_eq!( - index.get(&[PackedCell::Int(7)]).collect::>(), + index.get(&[PackedValue::Int(7)]).collect::>(), vec![other] ); } @@ -196,20 +196,20 @@ mod tests { let first = packed(1); let second = packed(2); let other = packed(3); - index.insert(vec![PackedCell::Int(5)], second); - index.insert(vec![PackedCell::Int(7)], other); - index.insert(vec![PackedCell::Int(5)], first); + index.insert(vec![PackedValue::Int(5)], second); + index.insert(vec![PackedValue::Int(7)], other); + index.insert(vec![PackedValue::Int(5)], first); - index.remove(&[PackedCell::Int(4)], first); - index.remove(&[PackedCell::Int(5)], packed(9)); - index.remove(&[PackedCell::Int(4)], first); + index.remove(&[PackedValue::Int(4)], first); + index.remove(&[PackedValue::Int(5)], packed(9)); + index.remove(&[PackedValue::Int(4)], first); assert_eq!( - index.get(&[PackedCell::Int(5)]).collect::>(), + index.get(&[PackedValue::Int(5)]).collect::>(), vec![first, second] ); assert_eq!( - index.get(&[PackedCell::Int(7)]).collect::>(), + index.get(&[PackedValue::Int(7)]).collect::>(), vec![other] ); assert_eq!(index.values.len(), 3); @@ -242,7 +242,7 @@ mod tests { ]; for ((c1, c0), row_id) in entries { - index.insert(vec![PackedCell::Int(c1), PackedCell::Int(c0)], row_id); + index.insert(vec![PackedValue::Int(c1), PackedValue::Int(c0)], row_id); } let stored = (0..index.values.len()) @@ -257,11 +257,11 @@ mod tests { assert_eq!( stored, vec![ - (PackedCell::Int(0), PackedCell::Int(9), packed(5)), - (PackedCell::Int(1), PackedCell::Int(1), packed(1)), - (PackedCell::Int(1), PackedCell::Int(1), packed(2)), - (PackedCell::Int(1), PackedCell::Int(2), packed(4)), - (PackedCell::Int(2), PackedCell::Int(0), packed(3)), + (PackedValue::Int(0), PackedValue::Int(9), packed(5)), + (PackedValue::Int(1), PackedValue::Int(1), packed(1)), + (PackedValue::Int(1), PackedValue::Int(1), packed(2)), + (PackedValue::Int(1), PackedValue::Int(2), packed(4)), + (PackedValue::Int(2), PackedValue::Int(0), packed(3)), ] ); } diff --git a/packages/coln-store/src/table/mod.rs b/packages/coln-store/src/table/mod.rs index 5db41566..4a6100ce 100644 --- a/packages/coln-store/src/table/mod.rs +++ b/packages/coln-store/src/table/mod.rs @@ -9,7 +9,7 @@ pub mod sorted; pub mod table_ref; mod undo; -pub use cell::{CellKind, CellValue, RowId}; +pub use cell::{CellKind, WireRowId, WireValue}; pub use table_ref::TableRef; use std::collections::{HashMap, HashSet}; @@ -24,7 +24,7 @@ use crate::table::index::{IndexId, IndexMeta, TableIndex}; use crate::table::undo::UndoOp; use crate::txn::TxnId; -pub(crate) use self::cell::{PackedCell, PackedRowId}; +pub(crate) use self::cell::{PackedRowId, PackedValue}; use self::col::{Column, IdColumn}; pub type TableOid = usize; @@ -42,7 +42,7 @@ pub(crate) struct TableMeta<'a> { pub(crate) enum PackedOp { Add { row_id: PackedRowId, - values: Vec, + values: Vec, }, Delete { row_id: PackedRowId, @@ -75,7 +75,7 @@ pub enum ValidationError { #[error("row handle belongs to a different transaction: current {current:?}, got {got:?}")] TxnIdMismatch { current: TxnId, got: TxnId }, #[error("invalid row handle: {reason}")] - InvalidRowHandle { reason: String }, + InvalidTxnLiveRowId { reason: String }, #[error("invalid index id passed {index}")] InvalidIndex { index: u64 }, #[error("invalid index key for index {index}: expected {expected} values, got {got}")] @@ -91,8 +91,8 @@ pub enum ValidationError { /// Public facing row value #[derive(Debug, Clone, PartialEq, Eq)] pub struct RowView { - pub row_id: RowId, - pub values: Vec, + pub row_id: WireRowId, + pub values: Vec, } type ColName = ir::Path; @@ -216,7 +216,7 @@ impl Table { } /// Row id at a given physical row index. - pub(crate) fn row_id_at(&self, row_idx: usize, packer: &IdPacker) -> Option { + pub(crate) fn row_id_at(&self, row_idx: usize, packer: &IdPacker) -> Option { self.row_ids .get(row_idx) .map(|packed| packer.unpack_row_id(packed)) @@ -229,7 +229,7 @@ impl Table { row_idx: usize, col_idx: usize, packer: &IdPacker, - ) -> Option { + ) -> Option { self.cols .get(col_idx) .and_then(|col| col.get(row_idx, packer)) @@ -262,7 +262,7 @@ impl Table { #[derive(Debug, PartialEq, Eq)] pub struct SeekKey { pub(crate) column: usize, - pub(crate) value: CellValue, + pub(crate) value: WireValue, } impl Table { @@ -279,7 +279,7 @@ impl Table { /// O(N * log S) as first find out the index from the row_id, and then do a /// lookup on each column - pub(crate) fn packed_row_at(&self, row_id: PackedRowId) -> Option> { + pub(crate) fn packed_row_at(&self, row_id: PackedRowId) -> Option> { let row_idx = self.row_ids.position(row_id).ok()?; (0..self.schema.columns.len()) .map(|col_idx| { @@ -298,7 +298,7 @@ impl Table { &self, key: &[SeekKey], id_packer: &IdPacker, - ) -> Result, ValidationError> { + ) -> Result, ValidationError> { if let Some(column) = key .iter() .map(|part| part.column) @@ -325,9 +325,9 @@ impl Table { pub(crate) fn index_seek( &self, index: IndexId, - key: &[CellValue], + key: &[WireValue], id_packer: &IdPacker, - ) -> Result, ValidationError> { + ) -> Result, ValidationError> { let table_index = self .indexes .get(index) @@ -359,7 +359,7 @@ impl Table { pub(crate) fn index_seek_packed( &self, index: IndexId, - key: &[PackedCell], + key: &[PackedValue], ) -> Result, ValidationError> { let table_index = self .indexes @@ -388,7 +388,7 @@ impl Table { pub(crate) fn index_lookup( &self, index: IndexId, - key: &[CellValue], + key: &[WireValue], id_packer: &IdPacker, ) -> Result { Ok(self.index_seek(index, key, id_packer)?.next().is_some()) @@ -412,7 +412,7 @@ impl Table { /// Checks schema and primary-key constraints against rows already stored. pub(crate) fn validate_insert( &self, - values: &[CellValue], + values: &[WireValue], dict: &IdPacker, ) -> Result<(), ValidationError> { // duplicated as txn::add(), but this is cheap enough we can afford to @@ -454,7 +454,7 @@ impl Table { /// Values at primary-key columns for this row. /// A primary key definition would occur in tables that do not end up in Query /// An empty primary key means the table would have at most one row. - pub fn primary_key_values(&self, values: &[CellValue]) -> Option> { + pub fn primary_key_values(&self, values: &[WireValue]) -> Option> { self.schema.primary_key.as_ref().and_then(|pk| { if pk.is_empty() { Some(Vec::new()) @@ -627,7 +627,7 @@ impl Table { } let new_row_id = rowing.canonical_id(&old_row_id, id_packer); - let old_cells: Vec = self + let old_cells: Vec = self .cols .iter() .map(|column| { @@ -663,14 +663,14 @@ impl Table { /// Rewrite every id cell to its canonical id, leaving other cells alone. fn canonicalise_cells( - values: &[PackedCell], + values: &[PackedValue], rowing: &Rowing, id_packer: &IdPacker, - ) -> Vec { + ) -> Vec { values .iter() .map(|cell| match cell { - PackedCell::Id(id) => PackedCell::Id(rowing.canonical_id(id, id_packer)), + PackedValue::Id(id) => PackedValue::Id(rowing.canonical_id(id, id_packer)), other => other.clone(), }) .collect() @@ -678,12 +678,12 @@ impl Table { /// ids referred by this row. #[expect(dead_code)] - fn referenced_ids(values: &[PackedCell]) -> impl Iterator { + fn referenced_ids(values: &[PackedValue]) -> impl Iterator { values .iter() .enumerate() .filter_map(|(i, cell)| match cell { - PackedCell::Id(id) if !values[..i].contains(cell) => Some(*id), + PackedValue::Id(id) if !values[..i].contains(cell) => Some(*id), _ => None, }) } @@ -697,7 +697,7 @@ impl Table { /// Only does primary key check, but no other validation. pub(super) fn insert_row( &mut self, - values: Vec, + values: Vec, row_id: PackedRowId, rowing: &mut Rowing, ) -> Result<(), ValidationError> { @@ -741,7 +741,7 @@ impl Table { } /// Place a row in columnar storage and every index, with no validation - fn insert_packed(&mut self, values: Vec, row_id: PackedRowId) { + fn insert_packed(&mut self, values: Vec, row_id: PackedRowId) { debug_assert_eq!(values.len(), self.schema.columns.len()); for index in &mut self.indexes { @@ -763,7 +763,7 @@ impl Table { } /// Take a row out of columnar storage and every index, returning its cells - fn remove_packed(&mut self, row_id: PackedRowId) -> Vec { + fn remove_packed(&mut self, row_id: PackedRowId) -> Vec { let row_idx = self .row_ids .position(row_id) @@ -797,7 +797,7 @@ impl Table { values } - fn project_index_key(index: &TableIndex, values: &[PackedCell]) -> Vec { + fn project_index_key(index: &TableIndex, values: &[PackedValue]) -> Vec { index .key_cols() .iter() diff --git a/packages/coln-store/src/table/sorted.rs b/packages/coln-store/src/table/sorted.rs index 71605784..ce6fe0e2 100644 --- a/packages/coln-store/src/table/sorted.rs +++ b/packages/coln-store/src/table/sorted.rs @@ -133,7 +133,7 @@ impl Store { } impl<'a> SortedTableSnapshot for SortedTable<'a> { - type Value = table::CellValue; + type Value = table::WireValue; /// Number of columns, including rowid column fn arity(&self) -> usize { diff --git a/packages/coln-store/src/table/table_ref.rs b/packages/coln-store/src/table/table_ref.rs index 670dea7b..51ed6929 100644 --- a/packages/coln-store/src/table/table_ref.rs +++ b/packages/coln-store/src/table/table_ref.rs @@ -6,7 +6,7 @@ use crate::id_packer::IdPacker; use crate::ir; use crate::ir::Schema; use crate::table::index::{IndexId, IndexMeta}; -use crate::table::{CellValue, RowId, RowView, SeekKey, Table, TableOid, ValidationError}; +use crate::table::{RowView, SeekKey, Table, TableOid, ValidationError, WireRowId, WireValue}; /// A [`Table`] together with the store-wide hash dictionary, for read-only /// access. This is what [`Store`](crate::store::Store) accessors hand out, so @@ -42,11 +42,11 @@ impl<'a> TableRef<'a> { self.table.cols.len() + 1 } - pub fn row_id_at(self, row_idx: usize) -> Option { + pub fn row_id_at(self, row_idx: usize) -> Option { self.table.row_id_at(row_idx, self.id_packer) } - pub fn cell_at(self, row_idx: usize, col_idx: usize) -> Option { + pub fn cell_at(self, row_idx: usize, col_idx: usize) -> Option { self.table.cell_at(row_idx, col_idx, self.id_packer) } @@ -54,7 +54,7 @@ impl<'a> TableRef<'a> { self.table.row_at(row_idx, self.id_packer) } - pub fn row_position(self, row_id: RowId) -> Option { + pub fn row_position(self, row_id: WireRowId) -> Option { let row_id = self.id_packer.lookup_row_id(row_id)?; self.table.row_idx(row_id) } @@ -67,15 +67,15 @@ impl<'a> TableRef<'a> { self.table.indexes_meta() } - pub fn seek(self, key: &[SeekKey]) -> Result, ValidationError> { + pub fn seek(self, key: &[SeekKey]) -> Result, ValidationError> { self.table.seek(key, self.id_packer) } pub fn index_seek( self, index: IndexId, - key: &[CellValue], - ) -> Result, ValidationError> { + key: &[WireValue], + ) -> Result, ValidationError> { self.table.index_seek(index, key, self.id_packer) } @@ -83,7 +83,7 @@ impl<'a> TableRef<'a> { self.table.lookup(key, self.id_packer) } - pub fn index_lookup(self, index: IndexId, key: &[CellValue]) -> Result { + pub fn index_lookup(self, index: IndexId, key: &[WireValue]) -> Result { self.table.index_lookup(index, key, self.id_packer) } @@ -95,11 +95,11 @@ impl<'a> TableRef<'a> { self.table.validate_column_count(got) } - pub fn validate_insert(self, values: &[CellValue]) -> Result<(), ValidationError> { + pub fn validate_insert(self, values: &[WireValue]) -> Result<(), ValidationError> { self.table.validate_insert(values, self.id_packer) } - pub fn primary_key_values(self, values: &[CellValue]) -> Option> { + pub fn primary_key_values(self, values: &[WireValue]) -> Option> { self.table.primary_key_values(values) } } diff --git a/packages/coln-store/src/table/tests.rs b/packages/coln-store/src/table/tests.rs index 77fcd4c7..1ec7c5fc 100644 --- a/packages/coln-store/src/table/tests.rs +++ b/packages/coln-store/src/table/tests.rs @@ -8,15 +8,15 @@ use crate::ir::{self, Path}; use crate::ir::{BuiltinTy, ColType}; use crate::op::Op; -fn test_row_id(counter: u32) -> RowId { - RowId { +fn test_row_id(counter: u32) -> WireRowId { + WireRowId { commit: CommitHash([0; 32]), counter, } } -fn row_id_from(commit_byte: u8, counter: u32) -> RowId { - RowId { +fn row_id_from(commit_byte: u8, counter: u32) -> WireRowId { + WireRowId { commit: CommitHash([commit_byte; 32]), counter, } @@ -66,11 +66,11 @@ impl TestTable { self.table.row_count() } - fn row_id_at(&self, row_idx: usize) -> Option { + fn row_id_at(&self, row_idx: usize) -> Option { self.table.row_id_at(row_idx, &self.dict) } - fn cell_at(&self, row_idx: usize, col_idx: usize) -> Option { + fn cell_at(&self, row_idx: usize, col_idx: usize) -> Option { self.table.cell_at(row_idx, col_idx, &self.dict) } @@ -78,17 +78,17 @@ impl TestTable { self.table.row_at(row_idx, &self.dict) } - fn row_position(&self, row_id: RowId) -> Option { + fn row_position(&self, row_id: WireRowId) -> Option { let row_id = self.dict.lookup_row_id(row_id)?; self.table.row_idx(row_id) } /// Rows the table records as referring to `child`, in row id order. - fn referring_rows(&self, child: RowId) -> Vec { + fn referring_rows(&self, child: WireRowId) -> Vec { let Some(child) = self.dict.lookup_row_id(child) else { return Vec::new(); }; - let mut rows: Vec = self + let mut rows: Vec = self .table .rebuild_index .get(&child) @@ -102,11 +102,11 @@ impl TestTable { rows } - fn validate_insert(&self, values: &[CellValue]) -> Result<(), ValidationError> { + fn validate_insert(&self, values: &[WireValue]) -> Result<(), ValidationError> { self.table.validate_insert(values, &self.dict) } - fn insert_row(&mut self, values: Vec, row_id: RowId) { + fn insert_row(&mut self, values: Vec, row_id: WireRowId) { let row_id = self.dict.pack_row_id(row_id); let values = values .into_iter() @@ -132,7 +132,7 @@ impl TestTable { self.table.stage_update(PackedOp::Add { row_id, values }); } - fn stage_delete(&mut self, row_id: RowId) { + fn stage_delete(&mut self, row_id: WireRowId) { let row_id = self.dict.pack_row_id(row_id); self.table.stage_update(PackedOp::Delete { row_id }); } @@ -177,25 +177,25 @@ fn rollback_removes_applied_rows_and_index_entries() { let existing = test_row_id(0); let first_added = test_row_id(1); let second_added = test_row_id(2); - tbl.insert_row(vec![CellValue::Int(1)], existing); + tbl.insert_row(vec![WireValue::Int(1)], existing); let snapshot = tbl.table.snapshot(); tbl.stage_update(Op::Add { row_id: first_added, table: 0, - values: vec![CellValue::Int(2)], + values: vec![WireValue::Int(2)], }); tbl.stage_update(Op::Add { row_id: second_added, table: 0, - values: vec![CellValue::Int(3)], + values: vec![WireValue::Int(3)], }); tbl.apply_staged_ops() .expect("the added rows have distinct keys"); assert_eq!(tbl.row_count(), 3); assert_eq!( - tbl.validate_insert(&[CellValue::Int(2)]), + tbl.validate_insert(&[WireValue::Int(2)]), Err(ValidationError::DuplicatePrimaryKey) ); @@ -205,8 +205,8 @@ fn rollback_removes_applied_rows_and_index_entries() { assert_eq!(tbl.row_id_at(0), Some(existing)); assert_eq!(tbl.row_position(first_added), None); assert_eq!(tbl.row_position(second_added), None); - assert!(tbl.validate_insert(&[CellValue::Int(2)]).is_ok()); - assert!(tbl.validate_insert(&[CellValue::Int(3)]).is_ok()); + assert!(tbl.validate_insert(&[WireValue::Int(2)]).is_ok()); + assert!(tbl.validate_insert(&[WireValue::Int(3)]).is_ok()); assert!(tbl.table.undo_log.is_none()); } @@ -221,8 +221,8 @@ fn staged_delete_removes_row_and_undo_restores_it() { ); let kept = test_row_id(0); let removed = test_row_id(1); - tbl.insert_row(vec![CellValue::Int(1)], kept); - tbl.insert_row(vec![CellValue::Int(2)], removed); + tbl.insert_row(vec![WireValue::Int(1)], kept); + tbl.insert_row(vec![WireValue::Int(2)], removed); let snapshot = tbl.table.snapshot(); tbl.stage_delete(removed); @@ -232,16 +232,16 @@ fn staged_delete_removes_row_and_undo_restores_it() { assert_eq!(tbl.row_count(), 1); assert_eq!(tbl.row_position(removed), None); // The primary key index gave up the key, so it is free to reuse. - assert!(tbl.validate_insert(&[CellValue::Int(2)]).is_ok()); + assert!(tbl.validate_insert(&[WireValue::Int(2)]).is_ok()); tbl.table.rollback(snapshot); assert_eq!(tbl.row_count(), 2); assert_eq!(tbl.row_position(kept), Some(0)); let idx = tbl.row_position(removed).expect("row is restored"); - assert_eq!(tbl.cell_at(idx, 0), Some(CellValue::Int(2))); + assert_eq!(tbl.cell_at(idx, 0), Some(WireValue::Int(2))); assert_eq!( - tbl.validate_insert(&[CellValue::Int(2)]), + tbl.validate_insert(&[WireValue::Int(2)]), Err(ValidationError::DuplicatePrimaryKey) ); } @@ -259,8 +259,8 @@ fn rebuild_index_tracks_rows_referring_to_an_id() { let pair = test_row_id(0); let doubled = test_row_id(1); - tbl.insert_row(vec![CellValue::Id(a), CellValue::Id(b)], pair); - tbl.insert_row(vec![CellValue::Id(a), CellValue::Id(a)], doubled); + tbl.insert_row(vec![WireValue::Id(a), WireValue::Id(b)], pair); + tbl.insert_row(vec![WireValue::Id(a), WireValue::Id(a)], doubled); // `doubled` refers to `a` twice but is recorded against it once, so a // rebuild pass restages it once rather than deleting it twice. @@ -290,7 +290,7 @@ fn full_rebuild_rewrites_stale_id_cells() { let canonical_child = row_id_from(1, 0); let stale_child = row_id_from(2, 0); let owner = test_row_id(0); - tbl.insert_row(vec![CellValue::Id(stale_child)], owner); + tbl.insert_row(vec![WireValue::Id(stale_child)], owner); let stale_child = tbl.dict.lookup_row_id(stale_child).unwrap(); let canonical_child = tbl.dict.pack_row_id(canonical_child); @@ -302,7 +302,7 @@ fn full_rebuild_rewrites_stale_id_cells() { assert_eq!(tbl.row_count(), 1); assert_eq!(tbl.row_id_at(0), Some(owner)); - assert_eq!(tbl.cell_at(0, 0), Some(CellValue::Id(row_id_from(1, 0)))); + assert_eq!(tbl.cell_at(0, 0), Some(WireValue::Id(row_id_from(1, 0)))); } #[test] @@ -311,8 +311,8 @@ fn full_rebuild_collapses_a_displaced_row_onto_its_canonical_row() { TestTable::with_structural_index(Path::from("term"), int_schema(&["value"], None)); let canonical = row_id_from(1, 0); let displaced = row_id_from(2, 0); - tbl.insert_row(vec![CellValue::Int(7)], displaced); - tbl.insert_row(vec![CellValue::Int(7)], canonical); + tbl.insert_row(vec![WireValue::Int(7)], displaced); + tbl.insert_row(vec![WireValue::Int(7)], canonical); tbl.rowing.apply_unions(&tbl.dict); tbl.table.rebuild_full(&tbl.rowing, &tbl.dict); @@ -320,7 +320,7 @@ fn full_rebuild_collapses_a_displaced_row_onto_its_canonical_row() { assert_eq!(tbl.row_count(), 1); assert_eq!(tbl.row_id_at(0), Some(canonical)); - assert_eq!(tbl.cell_at(0, 0), Some(CellValue::Int(7))); + assert_eq!(tbl.cell_at(0, 0), Some(WireValue::Int(7))); } /// Rollback replays the undo log through the same insert path, so the @@ -332,7 +332,7 @@ fn rollback_restores_rebuild_index_entries() { let a = row_id_from(1, 0); let b = row_id_from(1, 1); let row = test_row_id(0); - tbl.insert_row(vec![CellValue::Id(a), CellValue::Id(b)], row); + tbl.insert_row(vec![WireValue::Id(a), WireValue::Id(b)], row); let snapshot = tbl.table.snapshot(); tbl.stage_delete(row); @@ -356,7 +356,7 @@ fn commit_snapshot_keeps_rows_and_discards_undo_log() { tbl.stage_update(Op::Add { row_id, table: 0, - values: vec![CellValue::Int(7)], + values: vec![WireValue::Int(7)], }); tbl.apply_staged_ops() .expect("a table without a primary key accepts the row"); @@ -376,7 +376,7 @@ fn rollback_discards_updates_staged_after_snapshot() { tbl.stage_update(Op::Add { row_id: test_row_id(0), table: 0, - values: vec![CellValue::Int(7)], + values: vec![WireValue::Int(7)], }); tbl.table.rollback(snapshot); @@ -399,10 +399,10 @@ fn empty_primary_key_rejects_second_row() { }; let mut tbl = TestTable::new(Path::from("singleton"), schema); - tbl.insert_row(vec![CellValue::Int(0)], test_row_id(0)); + tbl.insert_row(vec![WireValue::Int(0)], test_row_id(0)); assert_eq!(tbl.row_count(), 1); - let values1 = vec![CellValue::Int(1)]; + let values1 = vec![WireValue::Int(1)]; let err = tbl.validate_insert(&values1).unwrap_err(); assert_eq!(err, ValidationError::DuplicatePrimaryKey); assert_eq!(tbl.row_count(), 1); @@ -432,7 +432,7 @@ fn row_read_helpers_return_row_id_and_cells() { let row_id = test_row_id(0); tbl.insert_row( - vec![CellValue::Int(7), CellValue::Str("x".to_string())], + vec![WireValue::Int(7), WireValue::Str("x".to_string())], row_id, ); @@ -440,19 +440,19 @@ fn row_read_helpers_return_row_id_and_cells() { tbl.row_at(0), Some(RowView { row_id, - values: vec![CellValue::Int(7), CellValue::Str("x".to_string())], + values: vec![WireValue::Int(7), WireValue::Str("x".to_string())], }) ); assert_eq!(tbl.row_id_at(0), Some(row_id)); - assert_eq!(tbl.cell_at(0, 0), Some(CellValue::Int(7))); - assert_eq!(tbl.cell_at(0, 1), Some(CellValue::Str("x".to_string()))); + assert_eq!(tbl.cell_at(0, 0), Some(WireValue::Int(7))); + assert_eq!(tbl.cell_at(0, 1), Some(WireValue::Str("x".to_string()))); let packed = tbl .dict .lookup_row_id(row_id) .expect("insert packed the row id"); assert_eq!( tbl.table.packed_row_at(packed), - Some(vec![PackedCell::Int(7), PackedCell::Str("x".to_string())]) + Some(vec![PackedValue::Int(7), PackedValue::Str("x".to_string())]) ); assert_eq!(tbl.row_at(1), None); assert_eq!(tbl.row_id_at(1), None); @@ -471,14 +471,14 @@ fn packed_row_ids_round_trip_across_commits() { (row_id_from(1, 2), row_id_from(2, 1), row_id_from(3, 7)), ]; for (rid, src, dst) in rows { - tbl.insert_row(vec![CellValue::Id(src), CellValue::Id(dst)], rid); + tbl.insert_row(vec![WireValue::Id(src), WireValue::Id(dst)], rid); } for (rid, src, dst) in rows { let idx = tbl.row_position(rid).expect("row is stored"); assert_eq!(tbl.row_id_at(idx), Some(rid)); - assert_eq!(tbl.cell_at(idx, 0), Some(CellValue::Id(src))); - assert_eq!(tbl.cell_at(idx, 1), Some(CellValue::Id(dst))); + assert_eq!(tbl.cell_at(idx, 0), Some(WireValue::Id(src))); + assert_eq!(tbl.cell_at(idx, 1), Some(WireValue::Id(dst))); } // Four distinct commit hashes, each interned exactly once. @@ -511,10 +511,10 @@ fn rows_stay_sorted_by_row_id() { (row_id_from(1, 2), 4), ]; for (rid, v) in rows { - tbl.insert_row(vec![CellValue::Int(v)], rid); + tbl.insert_row(vec![WireValue::Int(v)], rid); } - let stored: Vec = (0..tbl.row_count()) + let stored: Vec = (0..tbl.row_count()) .map(|idx| tbl.row_id_at(idx).expect("row id")) .collect(); assert_eq!( @@ -531,7 +531,7 @@ fn rows_stay_sorted_by_row_id() { // Cells moved together with their row ids. for (rid, v) in rows { let idx = tbl.row_position(rid).expect("row is stored"); - assert_eq!(tbl.cell_at(idx, 0), Some(CellValue::Int(v))); + assert_eq!(tbl.cell_at(idx, 0), Some(WireValue::Int(v))); } // Absent ids: known commit with unused counter, and unknown commit. @@ -549,19 +549,19 @@ fn primary_key_detects_duplicates_in_id_columns() { let src = row_id_from(3, 7); tbl.insert_row( - vec![CellValue::Id(src), CellValue::Id(row_id_from(4, 8))], + vec![WireValue::Id(src), WireValue::Id(row_id_from(4, 8))], row_id_from(1, 0), ); - let duplicate = vec![CellValue::Id(src), CellValue::Id(row_id_from(4, 9))]; + let duplicate = vec![WireValue::Id(src), WireValue::Id(row_id_from(4, 9))]; assert_eq!( tbl.validate_insert(&duplicate), Err(ValidationError::DuplicatePrimaryKey) ); let unseen_commit = vec![ - CellValue::Id(row_id_from(9, 7)), - CellValue::Id(row_id_from(4, 8)), + WireValue::Id(row_id_from(9, 7)), + WireValue::Id(row_id_from(4, 8)), ]; assert!(tbl.validate_insert(&unseen_commit).is_ok()); } @@ -591,19 +591,19 @@ fn multi_column_primary_key_checks_all_columns() { let rows = [(3, 1), (1, 2), (1, 1), (2, 1), (2, 2)]; for (i, (a, b)) in rows.into_iter().enumerate() { - let values = vec![CellValue::Int(a), CellValue::Int(b), CellValue::Int(0)]; + let values = vec![WireValue::Int(a), WireValue::Int(b), WireValue::Int(0)]; tbl.validate_insert(&values).expect("unique pair"); tbl.insert_row(values, test_row_id(i as u32)); } for (a, b) in rows { - let dup = vec![CellValue::Int(a), CellValue::Int(b), CellValue::Int(9)]; + let dup = vec![WireValue::Int(a), WireValue::Int(b), WireValue::Int(9)]; assert_eq!( tbl.validate_insert(&dup), Err(ValidationError::DuplicatePrimaryKey) ); } - let fresh = vec![CellValue::Int(3), CellValue::Int(2), CellValue::Int(0)]; + let fresh = vec![WireValue::Int(3), WireValue::Int(2), WireValue::Int(0)]; assert!(tbl.validate_insert(&fresh).is_ok()); } @@ -623,17 +623,17 @@ fn string_primary_key_detects_duplicates() { let mut tbl = TestTable::new(Path::from("named"), schema); for (i, name) in ["b", "a", "c"].into_iter().enumerate() { - let values = vec![CellValue::Str(name.to_string())]; + let values = vec![WireValue::Str(name.to_string())]; tbl.validate_insert(&values).expect("unique name"); tbl.insert_row(values, test_row_id(i as u32)); } assert_eq!( - tbl.validate_insert(&[CellValue::Str("a".to_string())]), + tbl.validate_insert(&[WireValue::Str("a".to_string())]), Err(ValidationError::DuplicatePrimaryKey) ); assert!( - tbl.validate_insert(&[CellValue::Str("d".to_string())]) + tbl.validate_insert(&[WireValue::Str("d".to_string())]) .is_ok() ); } @@ -677,7 +677,7 @@ fn pk_insert_benchmark() { let start = std::time::Instant::now(); for i in 0..n { let row_id = test_row_id(i as u32); - let values = vec![CellValue::Id(row_id), CellValue::Int(i)]; + let values = vec![WireValue::Id(row_id), WireValue::Int(i)]; tbl.validate_insert(&values).expect("keys are unique"); tbl.insert_row(values, row_id); } @@ -690,25 +690,25 @@ fn pk_insert_benchmark() { fn table_performs_index_lookup() { let schema = int_schema(&["indexed", "plain"], Some(&["indexed"])); let mut tbl = TestTable::new(Path::from("lookup"), schema); - tbl.insert_row(vec![CellValue::Int(7), CellValue::Int(70)], test_row_id(0)); - tbl.insert_row(vec![CellValue::Int(8), CellValue::Int(80)], test_row_id(1)); + tbl.insert_row(vec![WireValue::Int(7), WireValue::Int(70)], test_row_id(0)); + tbl.insert_row(vec![WireValue::Int(8), WireValue::Int(80)], test_row_id(1)); let index = tbl.table.primary_index().expect("primary-key index"); assert_eq!( tbl.table - .index_lookup(index, &[CellValue::Int(7)], &tbl.dict), + .index_lookup(index, &[WireValue::Int(7)], &tbl.dict), Ok(true) ); assert_eq!( tbl.table - .index_lookup(index, &[CellValue::Int(9)], &tbl.dict), + .index_lookup(index, &[WireValue::Int(9)], &tbl.dict), Ok(false) ); assert_eq!( tbl.table.lookup( &[SeekKey { column: 1, - value: CellValue::Int(80), + value: WireValue::Int(80), }], &tbl.dict, ), @@ -718,7 +718,7 @@ fn table_performs_index_lookup() { tbl.table.lookup( &[SeekKey { column: 1, - value: CellValue::Int(90), + value: WireValue::Int(90), }], &tbl.dict, ), @@ -734,7 +734,7 @@ fn table_index_lookup_non_existing_index() { let tbl = TestTable::new(Path::from("lookup"), schema); assert_eq!( - tbl.table.index_lookup(99, &[CellValue::Int(7)], &tbl.dict), + tbl.table.index_lookup(99, &[WireValue::Int(7)], &tbl.dict), Err(ValidationError::InvalidIndex { index: 99 }) ); } @@ -749,7 +749,7 @@ fn table_index_lookup_incorrect_key() { assert_eq!( tbl.table - .index_lookup(index, &[CellValue::Int(7), CellValue::Int(8)], &tbl.dict,), + .index_lookup(index, &[WireValue::Int(7), WireValue::Int(8)], &tbl.dict,), Err(ValidationError::InvalidIndexKey { index, expected: 1, @@ -768,14 +768,14 @@ fn table_index_non_index_give_same_results() { let mut tbl = TestTable::new(Path::from("lookup"), schema); for value in [7, 8] { let row_id = test_row_id(tbl.row_count() as u32); - tbl.insert_row(vec![CellValue::Int(value), CellValue::Int(value)], row_id); + tbl.insert_row(vec![WireValue::Int(value), WireValue::Int(value)], row_id); } let index = tbl.table.primary_index().expect("primary-key index"); for value in [7, 9] { let indexed = tbl .table - .index_seek(index, &[CellValue::Int(value)], &tbl.dict) + .index_seek(index, &[WireValue::Int(value)], &tbl.dict) .expect("valid index lookup") .collect::>(); let scanned = tbl @@ -783,7 +783,7 @@ fn table_index_non_index_give_same_results() { .seek( &[SeekKey { column: 1, - value: CellValue::Int(value), + value: WireValue::Int(value), }], &tbl.dict, ) @@ -816,11 +816,11 @@ fn debug_dumps_rows() { let mut tbl = TestTable::new(Path::from("debug.table"), schema); tbl.insert_row( - vec![CellValue::Int(7), CellValue::Str("x".to_string())], + vec![WireValue::Int(7), WireValue::Str("x".to_string())], test_row_id(0), ); tbl.insert_row( - vec![CellValue::Int(8), CellValue::Str("y".to_string())], + vec![WireValue::Int(8), WireValue::Str("y".to_string())], test_row_id(1), ); diff --git a/packages/coln-store/src/table/undo.rs b/packages/coln-store/src/table/undo.rs index f673ce3d..cae196c5 100644 --- a/packages/coln-store/src/table/undo.rs +++ b/packages/coln-store/src/table/undo.rs @@ -2,7 +2,7 @@ // // SPDX-License-Identifier: Apache-2.0 OR MIT -use crate::table::{PackedCell, PackedRowId}; +use crate::table::{PackedRowId, PackedValue}; // Undo operations for each variant in the `Op` enum. #[derive(Debug)] @@ -12,6 +12,6 @@ pub(super) enum UndoOp { }, // undo add means delete the row_id UndoDelete { row_id: PackedRowId, - values: Vec, + values: Vec, }, // undo delete means add back the row } diff --git a/packages/coln-store/src/txn/inner.rs b/packages/coln-store/src/txn/inner.rs index f75b6920..15e5564a 100644 --- a/packages/coln-store/src/txn/inner.rs +++ b/packages/coln-store/src/txn/inner.rs @@ -11,7 +11,9 @@ use crate::{ commit::{Commit, author::Author, hash::CommitHash, wire::CommitData}, store::{Store, error::StoreError}, table::ValidationError, - txn::{PendingOp, RowHandle, TempRowId, TxnCellValue, TxnId, TxnValue, timestamp::Timestamp}, + txn::{ + PendingOp, TempRowId, TxnId, TxnLiveRowId, TxnLiveValue, TxnWireValue, timestamp::Timestamp, + }, }; static NEXT_TX_ID: AtomicU64 = AtomicU64::new(1); @@ -27,7 +29,7 @@ pub(crate) struct TxnInner { timestamp: Timestamp, message: Option, tx_id: TxnId, - pending_handles: Vec, + pending_handles: Vec, } impl TxnInner { @@ -51,7 +53,7 @@ impl TxnInner { &mut self, store: &Store, table: &ir::Path, - values: Vec, + values: Vec, ) -> Result { let t = store.table_at(table).ok_or(ValidationError::UnknownTable { path: table.clone(), @@ -66,18 +68,18 @@ impl TxnInner { Ok(temp_id) } - pub(super) fn add( + pub(super) fn add>( &mut self, store: &Store, table: &ir::Path, - values: Vec, - ) -> Result { + values: Vec, + ) -> Result { let txn_values = values .into_iter() - .map(|v| v.to_txn_cell_value(self.tx_id)) - .collect::, _>>()?; + .map(|v| v.into().to_txn_cell_value(self.tx_id)) + .collect::, _>>()?; let temp_id = self.add_cell_values(store, table, txn_values)?; - let handle = RowHandle::from_pending(self.tx_id, temp_id.0); + let handle = TxnLiveRowId::from_pending(self.tx_id, temp_id.0); self.pending_handles.push(handle.clone()); Ok(handle) } @@ -88,12 +90,12 @@ impl TxnInner { &mut self, store: &Store, table: &ir::Path, - values: Vec, + values: Vec, ) -> Result { self.add_cell_values(store, table, values) } - fn invalidate_handles(pending_handles: Vec, reason: &str) { + fn invalidate_handles(pending_handles: Vec, reason: &str) { pending_handles .into_iter() .for_each(|h| h.invalidate(reason)); @@ -102,7 +104,7 @@ impl TxnInner { /// Finalize handles to the id the store actually kept: a row that was /// deduplicated against an existing class finalizes to that class's /// canonical id, not to the never-stored raw id. - fn finalize_handles(pending_handles: Vec, h: CommitHash, store: &Store) { + fn finalize_handles(pending_handles: Vec, h: CommitHash, store: &Store) { pending_handles.into_iter().for_each(|handle| { handle.finalize(h, |rid| store.canonical_row_id(rid).unwrap_or(rid)) }); diff --git a/packages/coln-store/src/txn/mod.rs b/packages/coln-store/src/txn/mod.rs index f28b8222..15ce5f80 100644 --- a/packages/coln-store/src/txn/mod.rs +++ b/packages/coln-store/src/txn/mod.rs @@ -14,8 +14,8 @@ use crate::{ }; use inner::TxnInner; -pub(crate) use row_handle::{PendingOp, RowRef, TempRowId, TxnCellValue}; -pub use row_handle::{RowHandle, TxnId, TxnValue}; +pub(crate) use row_handle::{PendingOp, TempRowId}; +pub use row_handle::{TxnId, TxnLiveRowId, TxnLiveValue, TxnWireRowId, TxnWireValue, empty_row}; pub struct Transaction<'a> { inner: TxnInner, @@ -31,13 +31,11 @@ impl<'a> Transaction<'a> { } } - // TODO this API is a bit awkward to use, clients have to call .into() all - // the time on their values - pub fn add( + pub fn add>( &mut self, table: &ir::Path, - values: Vec, - ) -> Result { + values: Vec, + ) -> Result { self.inner.add(self.store, table, values) } @@ -46,7 +44,7 @@ impl<'a> Transaction<'a> { pub(crate) fn add_internal( &mut self, table: &ir::Path, - values: Vec, + values: Vec, ) -> Result { self.inner.add_internal(self.store, table, values) } @@ -75,11 +73,11 @@ impl OwnedTransaction { } } - pub fn add( + pub fn add>( &mut self, table: &ir::Path, - values: Vec, - ) -> Result { + values: Vec, + ) -> Result { self.inner.add(&self.store, table, values) } @@ -100,9 +98,11 @@ impl OwnedTransaction { #[cfg(test)] mod tests { + use super::*; use crate::ir::{BuiltinTy, ColType, ColumnEntry, EntityVariant, Path, Schema}; - use crate::table::{CellValue, ValidationError}; + use crate::table::{ValidationError, WireValue}; + use crate::txn::row_handle::empty_row; fn table_schema(columns: Vec, primary_key: Option>) -> Schema { Schema { @@ -138,7 +138,7 @@ mod tests { .expect("create table"); let mut tx = OwnedTransaction::new(store); - tx.add(&path, vec![42_i64.into()]).expect("add"); + tx.add(&path, vec![42_i32]).expect("add"); let (_hash, committed) = tx.commit().expect("commit"); assert_eq!(committed.table_at(&path).expect("T").row_count(), 1); @@ -154,15 +154,13 @@ mod tests { .expect("create table"); let mut tx = OwnedTransaction::new(store); - let err = tx - .add(&Path::from("missing"), vec![1_i64.into()]) - .unwrap_err(); + let err = tx.add(&Path::from("missing"), vec![1_i32]).unwrap_err(); assert!(matches!( err, StoreError::Validation(ValidationError::UnknownTable { .. }) )); - let err = tx.add(&path, vec![1_i64.into(), 2_i64.into()]).unwrap_err(); + let err = tx.add(&path, vec![1_i32, 2_i32]).unwrap_err(); assert!(matches!( err, StoreError::Validation(ValidationError::ColumnCount { .. }) @@ -185,8 +183,9 @@ mod tests { .expect("create edges table"); let mut tx = store.transaction(); - let node_temp = tx.add(&nodes, vec![]).expect("add node"); - tx.add(&edges, vec![node_temp.into()]).expect("add edge"); + let node_temp = tx.add(&nodes, empty_row()).expect("add node"); + tx.add(&edges, vec![TxnLiveValue::Id(node_temp)]) + .expect("add edge"); let commit = tx.commit().expect("commit"); let node_id = store @@ -201,7 +200,7 @@ mod tests { assert_eq!(node_id.counter, 0); assert_eq!(edge_id.commit, commit); assert_eq!(edge_id.counter, 1); - assert_eq!(edge.cell_at(0, 0), Some(CellValue::Id(node_id))); + assert_eq!(edge.cell_at(0, 0), Some(WireValue::Id(node_id))); } #[test] @@ -220,18 +219,19 @@ mod tests { .expect("create edges table"); let mut tx = store.transaction(); - let node = tx.add(&nodes, vec![]).expect("add node"); + let node = tx.add(&nodes, empty_row()).expect("add node"); let first_commit = tx.commit().expect("commit node"); let node_id = node.row_id().expect("node handle finalized"); assert_eq!(node_id.commit, first_commit); let mut tx = store.transaction(); - tx.add(&edges, vec![node.into()]).expect("add edge"); + tx.add(&edges, vec![TxnLiveValue::Id(node)]) + .expect("add edge"); tx.commit().expect("commit edge"); let edge = store.table_at(&edges).expect("Edges"); - assert_eq!(edge.cell_at(0, 0), Some(CellValue::Id(node_id))); + assert_eq!(edge.cell_at(0, 0), Some(WireValue::Id(node_id))); } #[test] @@ -249,20 +249,20 @@ mod tests { ) .expect("create edges table"); let mut tx = store.transaction(); - let node = tx.add(&nodes, vec![]).expect("add node"); + let node = tx.add(&nodes, empty_row()).expect("add node"); tx.abort(); let err = node.row_id().expect_err("abort invalidates handle"); assert!(matches!( err, - StoreError::Validation(ValidationError::InvalidRowHandle { .. }) + StoreError::Validation(ValidationError::InvalidTxnLiveRowId { .. }) )); let mut tx = store.transaction(); let err = tx - .add(&edges, vec![node.into()]) + .add(&edges, vec![TxnLiveValue::Id(node)]) .expect_err("aborted handle cannot be reused"); assert!(matches!( err, - StoreError::Validation(ValidationError::InvalidRowHandle { .. }) + StoreError::Validation(ValidationError::InvalidTxnLiveRowId { .. }) )); assert_eq!(store.table_at(&nodes).expect("Nodes").row_count(), 0); } @@ -283,8 +283,9 @@ mod tests { .expect("create edges table"); let mut tx = store.transaction(); - let node = tx.add(&nodes, vec![]).expect("add first node"); - tx.add(&nodes, vec![]).expect("add duplicate singleton row"); + let node = tx.add(&nodes, empty_row()).expect("add first node"); + tx.add(&nodes, empty_row()) + .expect("add duplicate singleton row"); let err = tx .commit() .expect_err("duplicate singleton row should fail"); @@ -296,16 +297,16 @@ mod tests { let err = node.row_id().expect_err("failed commit invalidates handle"); assert!(matches!( err, - StoreError::Validation(ValidationError::InvalidRowHandle { .. }) + StoreError::Validation(ValidationError::InvalidTxnLiveRowId { .. }) )); let mut tx = store.transaction(); let err = tx - .add(&edges, vec![node.into()]) + .add(&edges, vec![TxnLiveValue::Id(node)]) .expect_err("invalid handle cannot be reused"); assert!(matches!( err, - StoreError::Validation(ValidationError::InvalidRowHandle { .. }) + StoreError::Validation(ValidationError::InvalidTxnLiveRowId { .. }) )); } @@ -324,13 +325,13 @@ mod tests { store.set_structural_index_for_test(&term, true); let mut tx = store.transaction(); - let first = tx.add(&term, vec![7_i64.into()]).expect("add first term"); + let first = tx.add(&term, vec![7_i32]).expect("add first term"); tx.commit().expect("commit first term"); // Structurally equal row: deduplicates into the first row's class. // Which id wins the merge depends on the commit hash ordering. let mut tx = store.transaction(); - let second = tx.add(&term, vec![7_i64.into()]).expect("add equal term"); + let second = tx.add(&term, vec![7_i32]).expect("add equal term"); tx.commit().expect("commit equal term"); let stored = store @@ -364,8 +365,7 @@ mod tests { let root = store.commits().root_commit().expect("root commit").hash(); let mut tx = store.transaction(); - tx.add(&path, vec![CellValue::Int(1).into()]) - .expect("add first row"); + tx.add(&path, vec![1i32]).expect("add first row"); let first = tx.commit().expect("first commit"); assert!(store.commits().contains(&first)); @@ -376,8 +376,7 @@ mod tests { ); let mut tx = store.transaction(); - tx.add(&path, vec![CellValue::Int(2).into()]) - .expect("add second row"); + tx.add(&path, vec![2i32]).expect("add second row"); let second = tx.commit().expect("second commit"); assert!(store.commits().contains(&second)); diff --git a/packages/coln-store/src/txn/row_handle.rs b/packages/coln-store/src/txn/row_handle.rs index 59692b37..80e75c06 100644 --- a/packages/coln-store/src/txn/row_handle.rs +++ b/packages/coln-store/src/txn/row_handle.rs @@ -4,11 +4,15 @@ use std::{cell::RefCell, rc::Rc}; +use serde::{Deserialize, Serialize}; +use specta::Type; + use crate::{ commit::hash::CommitHash, op::Op, store::error::StoreError, - table::{CellValue, RowId, TableOid, ValidationError}, + table::{TableOid, ValidationError, WireRowId, WireValue}, + value::Value, }; #[derive(Copy, Clone, Debug, Eq, PartialEq, Hash)] @@ -31,30 +35,36 @@ impl From for TxnId { } #[derive(Clone, Debug)] -enum RowHandleState { +enum TxnLiveRowIdState { Pending { tx_id: TxnId, counter: u32 }, - Existing(RowId), + Existing(WireRowId), Invalid(String), } -/// A RowHandle abstracts away the difference between a pending id and an existing +/// A TxnLiveRowId abstracts away the difference between a pending id and an existing /// rowid. It is a reference counted handle that can be shared and will be automatically /// converted from a temp rowid to a rowid on successful commit. #[derive(Clone, Debug)] -pub struct RowHandle { +pub struct TxnLiveRowId { // ? Arc - state: Rc>, + state: Rc>, +} + +impl From for TxnLiveValue { + fn from(i: TxnLiveRowId) -> TxnLiveValue { + TxnLiveValue::Id(i) + } } -impl RowHandle { - pub fn row_id(&self) -> Result { +impl TxnLiveRowId { + pub fn row_id(&self) -> Result { match &*self.state.borrow() { - RowHandleState::Existing(row_id) => Ok(*row_id), - RowHandleState::Pending { .. } => Err(ValidationError::InvalidRowHandle { + TxnLiveRowIdState::Existing(row_id) => Ok(*row_id), + TxnLiveRowIdState::Pending { .. } => Err(ValidationError::InvalidTxnLiveRowId { reason: "row handle is still pending".to_string(), } .into()), - RowHandleState::Invalid(reason) => Err(ValidationError::InvalidRowHandle { + TxnLiveRowIdState::Invalid(reason) => Err(ValidationError::InvalidTxnLiveRowId { reason: reason.clone(), } .into()), @@ -65,134 +75,102 @@ impl RowHandle { #[doc(hidden)] pub fn pending_ids(&self) -> Result<(u64, u32), StoreError> { match *self.state.borrow() { - RowHandleState::Pending { tx_id, counter } => Ok((tx_id.as_u64(), counter)), - _ => Err(ValidationError::InvalidRowHandle { + TxnLiveRowIdState::Pending { tx_id, counter } => Ok((tx_id.as_u64(), counter)), + _ => Err(ValidationError::InvalidTxnLiveRowId { reason: "not txn id on existing ids or invalid handles".to_string(), } .into()), } } - pub(crate) fn canonicalise(&self, new_row_id: RowId) -> Result<(), StoreError> { + pub(crate) fn canonicalise(&self, new_row_id: WireRowId) -> Result<(), StoreError> { let mut state = self.state.borrow_mut(); match &*state { - RowHandleState::Existing(..) => { - *state = RowHandleState::Existing(new_row_id); + TxnLiveRowIdState::Existing(..) => { + *state = TxnLiveRowIdState::Existing(new_row_id); Ok(()) } - _ => Err(ValidationError::InvalidRowHandle { + _ => Err(ValidationError::InvalidTxnLiveRowId { reason: "cannot replace row id on a non finalised rowhandle".to_string(), } .into()), } } - pub(crate) fn to_txn_cell_value(&self, current_tx: TxnId) -> Result { + pub(crate) fn to_txn_cell_value(&self, current_tx: TxnId) -> Result { match &*self.state.borrow() { - RowHandleState::Existing(row_id) => Ok(TxnCellValue::Id(RowRef::Existing(*row_id))), - RowHandleState::Pending { tx_id, counter } if *tx_id == current_tx => { - Ok(TxnCellValue::Id(RowRef::Pending(TempRowId::from(*counter)))) + TxnLiveRowIdState::Existing(row_id) => { + Ok(TxnWireValue::Id(TxnWireRowId::Existing(*row_id))) } - RowHandleState::Pending { tx_id, .. } => Err(ValidationError::TxnIdMismatch { + TxnLiveRowIdState::Pending { tx_id, counter } if *tx_id == current_tx => Ok( + TxnWireValue::Id(TxnWireRowId::Pending(TempRowId::from(*counter))), + ), + TxnLiveRowIdState::Pending { tx_id, .. } => Err(ValidationError::TxnIdMismatch { current: current_tx, got: *tx_id, } .into()), - RowHandleState::Invalid(reason) => Err(ValidationError::InvalidRowHandle { + TxnLiveRowIdState::Invalid(reason) => Err(ValidationError::InvalidTxnLiveRowId { reason: reason.clone(), } .into()), } } - pub(crate) fn finalize(&self, commit: CommitHash, resolve: impl Fn(RowId) -> RowId) { + pub(crate) fn finalize(&self, commit: CommitHash, resolve: impl Fn(WireRowId) -> WireRowId) { let mut state = self.state.borrow_mut(); - if let RowHandleState::Pending { counter, .. } = *state { - *state = RowHandleState::Existing(resolve(RowId { commit, counter })); + if let TxnLiveRowIdState::Pending { counter, .. } = *state { + *state = TxnLiveRowIdState::Existing(resolve(WireRowId { commit, counter })); } } pub(crate) fn invalidate(&self, reason: &str) { - *self.state.borrow_mut() = RowHandleState::Invalid(reason.into()); + *self.state.borrow_mut() = TxnLiveRowIdState::Invalid(reason.into()); } #[doc(hidden)] pub fn from_pending(tx_id: TxnId, counter: u32) -> Self { - let state = Rc::new(RefCell::new(RowHandleState::Pending { tx_id, counter })); - RowHandle { state } + let state = Rc::new(RefCell::new(TxnLiveRowIdState::Pending { tx_id, counter })); + TxnLiveRowId { state } } #[doc(hidden)] - pub fn from_existing(row_id: RowId) -> Self { - let state = Rc::new(RefCell::new(RowHandleState::Existing(row_id))); - RowHandle { state } + pub fn from_existing(row_id: WireRowId) -> Self { + let state = Rc::new(RefCell::new(TxnLiveRowIdState::Existing(row_id))); + TxnLiveRowId { state } } } -#[derive(Clone)] -pub enum TxnValue { - Id(RowHandle), - Int(i64), - Str(String), -} +pub type TxnLiveValue = Value; -impl TxnValue { - pub(crate) fn to_txn_cell_value(&self, current_tx: TxnId) -> Result { +impl TxnLiveValue { + pub(crate) fn to_txn_cell_value(&self, current_tx: TxnId) -> Result { match self { - TxnValue::Id(handle) => handle.to_txn_cell_value(current_tx), - TxnValue::Int(value) => Ok(TxnCellValue::Int(*value)), - TxnValue::Str(value) => Ok(TxnCellValue::Str(value.clone())), + TxnLiveValue::Id(handle) => handle.to_txn_cell_value(current_tx), + TxnLiveValue::Int(value) => Ok(TxnWireValue::Int(*value)), + TxnLiveValue::Str(value) => Ok(TxnWireValue::Str(value.clone())), } } } -impl From for TxnValue { - fn from(value: RowHandle) -> Self { - TxnValue::Id(value) +impl From for TxnLiveValue { + fn from(value: WireValue) -> Self { + value.map_owned(TxnLiveRowId::from_existing) } } -impl From for TxnValue { - fn from(value: RowId) -> Self { - TxnValue::Id(RowHandle::from_existing(value)) - } -} - -impl From for TxnValue { - fn from(value: i64) -> Self { - TxnValue::Int(value) - } -} - -impl From for TxnValue { - fn from(value: String) -> Self { - TxnValue::Str(value) - } -} - -impl From<&str> for TxnValue { - fn from(value: &str) -> Self { - TxnValue::Str(value.to_owned()) - } -} - -impl From for TxnValue { - fn from(value: CellValue) -> Self { - match value { - CellValue::Id(id) => TxnValue::Id(RowHandle::from_existing(id)), - CellValue::Int(value) => TxnValue::Int(value), - CellValue::Str(value) => TxnValue::Str(value), - } - } +pub fn empty_row() -> Vec { + Vec::new() } /// A temporary row ID that is valid only within a transaction. -#[derive(Copy, Clone, Debug, Eq, PartialEq)] -pub(crate) struct TempRowId(pub(crate) u32); +#[derive(Copy, Clone, Debug, Eq, PartialEq, Serialize, Deserialize, Type)] +#[serde(transparent)] +pub struct TempRowId(pub u32); impl TempRowId { - pub(crate) fn resolve(self, commit: CommitHash) -> RowId { - RowId { + pub(crate) fn resolve(self, commit: CommitHash) -> WireRowId { + WireRowId { commit, counter: self.0, } @@ -210,97 +188,35 @@ impl From for TempRowId { } /// A reference to an existing row or a pending row in the current transaction. -#[derive(Copy, Clone, Debug, Eq, PartialEq)] -pub(crate) enum RowRef { - Existing(RowId), +#[derive(Copy, Clone, Debug, Eq, PartialEq, Serialize, Deserialize, Type)] +#[serde(tag = "type")] +pub enum TxnWireRowId { + Existing(WireRowId), Pending(TempRowId), } -impl RowRef { - fn resolve(&self, commit: CommitHash) -> RowId { +impl TxnWireRowId { + fn resolve(&self, commit: CommitHash) -> WireRowId { match self { - RowRef::Existing(row_id) => *row_id, - RowRef::Pending(temp_id) => temp_id.resolve(commit), + TxnWireRowId::Existing(row_id) => *row_id, + TxnWireRowId::Pending(temp_id) => temp_id.resolve(commit), } } } -impl From for RowRef { - fn from(value: RowId) -> Self { - RowRef::Existing(value) +impl From for TxnWireRowId { + fn from(value: WireRowId) -> Self { + TxnWireRowId::Existing(value) } } -impl From for RowRef { +impl From for TxnWireRowId { fn from(value: TempRowId) -> Self { - RowRef::Pending(value) + TxnWireRowId::Pending(value) } } -// TODO should clean this up, who uses txncellvalue and it should have a better name -/// The internal transaction representation derived from `TxnValue`. -#[derive(Clone, Debug, Eq, PartialEq)] -pub(crate) enum TxnCellValue { - Id(RowRef), - Int(i64), - Str(String), -} - -impl TxnCellValue { - fn resolve(&self, commit: CommitHash) -> CellValue { - match self { - TxnCellValue::Id(row_ref) => CellValue::Id(row_ref.resolve(commit)), - TxnCellValue::Int(value) => CellValue::Int(*value), - TxnCellValue::Str(value) => CellValue::Str(value.clone()), - } - } -} - -impl From for TxnCellValue { - fn from(value: RowRef) -> Self { - TxnCellValue::Id(value) - } -} - -impl From for TxnCellValue { - fn from(value: RowId) -> Self { - TxnCellValue::Id(RowRef::Existing(value)) - } -} - -impl From for TxnCellValue { - fn from(value: TempRowId) -> Self { - TxnCellValue::Id(RowRef::Pending(value)) - } -} - -impl From for TxnCellValue { - fn from(value: i64) -> Self { - TxnCellValue::Int(value) - } -} - -impl From for TxnCellValue { - fn from(value: String) -> Self { - TxnCellValue::Str(value) - } -} - -impl From<&str> for TxnCellValue { - fn from(value: &str) -> Self { - TxnCellValue::Str(value.to_owned()) - } -} - -impl From for TxnCellValue { - fn from(value: CellValue) -> Self { - match value { - CellValue::Id(id) => TxnCellValue::Id(RowRef::Existing(id)), - CellValue::Int(value) => TxnCellValue::Int(value), - CellValue::Str(value) => TxnCellValue::Str(value), - } - } -} +pub type TxnWireValue = Value; /// An operation staged within a transaction. #[derive(Clone, Debug, Eq, PartialEq)] @@ -308,7 +224,7 @@ pub(crate) enum PendingOp { Add { row_id: TempRowId, table: TableOid, - values: Vec, + values: Vec, }, } @@ -322,7 +238,10 @@ impl PendingOp { } => Op::Add { row_id: row_id.resolve(commit), table: *table, - values: values.iter().map(|value| value.resolve(commit)).collect(), + values: values + .iter() + .map(|value| value.map(|i| i.resolve(commit))) + .collect(), }, } } diff --git a/packages/coln-store/src/value.rs b/packages/coln-store/src/value.rs new file mode 100644 index 00000000..aab17f67 --- /dev/null +++ b/packages/coln-store/src/value.rs @@ -0,0 +1,49 @@ +// SPDX-FileCopyrightText: 2026 Coln contributors +// +// SPDX-License-Identifier: Apache-2.0 OR MIT +use serde::{Deserialize, Serialize}; +use specta::Type; + +#[derive(Clone, Debug, Eq, PartialEq, Hash, PartialOrd, Ord, Serialize, Deserialize, Type)] +#[serde(untagged)] +pub enum Value { + Id(I), + Int(i32), + Str(String), +} + +impl Value { + pub fn map J>(&self, f: F) -> Value { + match self { + Value::Id(i) => Value::Id(f(i)), + Value::Int(i) => Value::Int(*i), + Value::Str(s) => Value::Str(s.clone()), + } + } + + pub fn map_owned J>(self, f: F) -> Value { + match self { + Value::Id(i) => Value::Id(f(i)), + Value::Int(i) => Value::Int(i), + Value::Str(s) => Value::Str(s), + } + } +} + +impl From for Value { + fn from(value: i32) -> Self { + Value::Int(value) + } +} + +impl From for Value { + fn from(value: String) -> Self { + Value::Str(value) + } +} + +impl From<&str> for Value { + fn from(value: &str) -> Self { + Value::Str(value.into()) + } +} diff --git a/packages/coln-store/tests/test_path.rs b/packages/coln-store/tests/test_path.rs index 3c2bbb30..07b92584 100644 --- a/packages/coln-store/tests/test_path.rs +++ b/packages/coln-store/tests/test_path.rs @@ -6,10 +6,11 @@ use std::{collections::BTreeSet, path::PathBuf, sync::Once}; use coln_flir_rs::ir::{self, FlatRealm, Path}; use coln_store::{ - commit::hash::CommitHash, - commit::pst, + commit::{hash::CommitHash, pst}, store::{Store, error::StoreError}, - table::{CellValue, RowId}, + table::{WireRowId, WireValue}, + txn::empty_row, + value::Value, }; use tracing_subscriber::EnvFilter; @@ -48,13 +49,13 @@ fn add_basic_data_to_path(store: &mut Store) -> Result<(), StoreError> { let ge = Path::from("Path.G.E"); let mut tx = store.transaction(); - let gid1 = tx.add(&graphs, vec![])?; - let gid2 = tx.add(&graphs, vec![])?; - tx.add(&g0, vec![gid2.clone().into()])?; - tx.add(&g1, vec![gid2.clone().into()])?; - let v1 = tx.add(&gv, vec![gid1.clone().into()])?; - let v2 = tx.add(&gv, vec![gid1.clone().into()])?; - tx.add(&ge, vec![gid1.into(), v1.into(), v2.into()])?; + let gid1 = tx.add(&graphs, empty_row())?; + let gid2 = tx.add(&graphs, empty_row())?; + tx.add(&g0, vec![gid2.clone()])?; + tx.add(&g1, vec![gid2.clone()])?; + let v1 = tx.add(&gv, vec![gid1.clone()])?; + let v2 = tx.add(&gv, vec![gid1.clone()])?; + tx.add(&ge, vec![gid1, v1, v2])?; tx.commit()?; Ok(()) @@ -63,14 +64,16 @@ fn add_basic_data_to_path(store: &mut Store) -> Result<(), StoreError> { fn add_vertex_to_graph(store: &mut Store, graph_row: usize) -> Result { let graphs = Path::from("Path.Graphs"); let gv = Path::from("Path.G.V"); - let graph = store - .table_at(&graphs) - .expect("Path.Graphs table") - .row_id_at(graph_row) - .expect("graph row"); + let graph = Value::Id( + store + .table_at(&graphs) + .expect("Path.Graphs table") + .row_id_at(graph_row) + .expect("graph row"), + ); let mut tx = store.transaction(); - tx.add(&gv, vec![graph.into()])?; + tx.add(&gv, vec![graph])?; tx.commit() } @@ -80,16 +83,18 @@ fn add_extra_edge_to_first_graph(store: &mut Store) -> Result SedimentreeId { SedimentreeId::new(root_hash(store).0) } -fn row_values(store: &Store) -> BTreeSet<(CommitHash, u32, i64)> { +fn row_values(store: &Store) -> BTreeSet<(CommitHash, u32, i32)> { let table = store.table_at(&Path::from("T")).expect("T table"); (0..table.row_count()) .map(|row| { let id = table.row_id_at(row).expect("row id"); let value = match table.cell_at(row, 0).expect("cell") { - CellValue::Int(value) => value, + WireValue::Int(value) => value, other => panic!("expected int cell, got {other:?}"), }; (id.commit, id.counter, value) @@ -79,9 +79,9 @@ fn row_values(store: &Store) -> BTreeSet<(CommitHash, u32, i64)> { .collect() } -fn add_row(store: &mut Store, value: i64) -> Result> { +fn add_row(store: &mut Store, value: i32) -> Result> { let mut tx = store.transaction(); - tx.add(&Path::from("T"), vec![value.into()])?; + tx.add(&Path::from("T"), vec![value])?; tx.commit().map_err(Into::into) } @@ -197,7 +197,7 @@ async fn subduction_sync_coln_chunks() -> Result<(), Box> { type TestTransport = MessageTransport; let (left_sd, _left_handler, left_listener, left_manager) = - SubductionBuilder::<_, _, _, _, _, 256>::new() + SubductionBuilder::<_, _, _, _, _>::new() .signer(MemorySigner::from_bytes(&[1; 32])) .storage(MemoryStorage::new(), Arc::new(OpenPolicy)) .spawner(ChannelTokioSpawn) @@ -205,7 +205,7 @@ async fn subduction_sync_coln_chunks() -> Result<(), Box> { .build::(); let (right_sd, _right_handler, right_listener, right_manager) = - SubductionBuilder::<_, _, _, _, _, 256>::new() + SubductionBuilder::<_, _, _, _, _>::new() .signer(MemorySigner::from_bytes(&[2; 32])) .storage(MemoryStorage::new(), Arc::new(OpenPolicy)) .spawner(ChannelTokioSpawn) @@ -330,7 +330,7 @@ async fn subduction_websocket_sync_coln_chunks() -> Result<(), Box> { let server_peer_id = PeerId::from(server_signer.verifying_key()); let (server_sd, _server_handler, server_listener, server_manager) = - SubductionBuilder::<_, _, _, _, _, 256>::new() + SubductionBuilder::<_, _, _, _, _>::new() .signer(server_signer) .storage(MemoryStorage::new(), Arc::new(OpenPolicy)) .spawner(TrackedTokioSpawn::default()) @@ -352,7 +352,7 @@ async fn subduction_websocket_sync_coln_chunks() -> Result<(), Box> { let left_signer = MemorySigner::from_bytes(&[11; 32]); let (left_sd, _left_handler, left_listener, left_manager) = - SubductionBuilder::<_, _, _, _, _, 256>::new() + SubductionBuilder::<_, _, _, _, _>::new() .signer(left_signer.clone()) .storage(MemoryStorage::new(), Arc::new(OpenPolicy)) .spawner(WebSocketTokioSpawn) @@ -364,7 +364,7 @@ async fn subduction_websocket_sync_coln_chunks() -> Result<(), Box> { let right_signer = MemorySigner::from_bytes(&[12; 32]); let (right_sd, _right_handler, right_listener, right_manager) = - SubductionBuilder::<_, _, _, _, _, 256>::new() + SubductionBuilder::<_, _, _, _, _>::new() .signer(right_signer.clone()) .storage(MemoryStorage::new(), Arc::new(OpenPolicy)) .spawner(WebSocketTokioSpawn)