From 666b944a90059ace10b50359d42700ec2fb07c70 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sat, 19 Sep 2026 19:55:16 +0300 Subject: [PATCH 01/51] fix(group): bound the tied groups the radix top-N selection keeps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The native top-N of the radix path kept every group at or beyond the threshold. Groups strictly beyond it number fewer than N, but the groups AT the threshold can be nearly all of them: a count-per-group top-10 over a near-unique key has a threshold of 1, every group ties it, and the "kept superset" is the whole grouping, heap-sorted by first row — a comparison sort over 100M pairs for a query whose answer is ten rows. The emitted prefix is the tied groups' first-seen order, so only the N tied groups with the smallest first row can ever be taken. The selection now keeps the strictly-better groups plus exactly those N, through a bounded max-heap, and sorts at most 2N entries. The rows emitted are the ones the full kept set produced. Tests: the pinned kept count of the radix native top-N case becomes N; a near-unique two-key grouping whose threshold every group ties, in both directions and under a row selection, against the full grouping. Co-Authored-By: Claude Fable 5.1 --- src/ops/agg_engine.c | 87 ++++++++++++++++++++++--- test/rfl/group/emit_filter_v2_route.rfl | 25 +++++++ test/test_agg_contract.c | 8 ++- 3 files changed, 109 insertions(+), 11 deletions(-) diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index e030b13c..8f414c61 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -3724,20 +3724,91 @@ agg_radix_select_topn(ray_pool_t* pool, const agg_radix_part_t* parts, uint32_t int64_t kept = 0; for (uint32_t p = 0; p < n_parts; p++) kept += c.kept[p]; ray_free_raw(cand); ray_free_raw(cand_n); - agg_radix_order_t* sel = ray_alloc_raw((size_t)(kept > 0 ? kept : 1) * sizeof(agg_radix_order_t)); - int64_t* fr = ray_alloc_raw((size_t)(kept > 0 ? kept : 1) * sizeof(int64_t)); + /* The kept set is every group at or beyond the threshold. Groups + * strictly beyond it number fewer than N by construction, but the groups + * AT the threshold can be nearly all of them — a count-per-group top-10 + * over a near-unique key has a threshold of 1 and every group ties it — + * and sorting that set is a comparison sort over the whole grouping. + * The emitted prefix is the tied groups' first-seen order, so only the + * N tied groups with the smallest first_row can ever be taken: keep + * exactly those, through a bounded max-heap, and the sort below runs + * over at most 2N entries. Without a threshold every group is kept + * (the take exceeds the group count) and the set is small anyway. */ + int64_t take = c.have_thr ? ef->top_count_take : 0; + int64_t sel_cap = c.have_thr ? 2 * take : kept; + if (sel_cap <= 0) sel_cap = 1; + agg_radix_order_t* sel = ray_alloc_raw((size_t)sel_cap * sizeof(agg_radix_order_t)); + int64_t* fr = ray_alloc_raw((size_t)sel_cap * sizeof(int64_t)); if (!sel || !fr) { ray_free_raw(sel); ray_free_raw(fr); ray_free_raw(base); ray_free_raw(vals); ray_free_raw(keep); *rc = 1; return NULL; } int64_t k = 0; - for (uint32_t p = 0; p < n_parts; p++) - for (int64_t gg = 0; gg < parts[p].ng; gg++) - if (keep[base[p] + gg]) { - sel[k].idx = ((int64_t)p << 32) | (uint32_t)gg; - fr[k] = parts[p].first_row[gg]; - k++; + if (!c.have_thr) { + for (uint32_t p = 0; p < n_parts; p++) + for (int64_t gg = 0; gg < parts[p].ng; gg++) + if (keep[base[p] + gg]) { + sel[k].idx = ((int64_t)p << 32) | (uint32_t)gg; + fr[k] = parts[p].first_row[gg]; + k++; + } + } else { + /* strict groups fill sel[0..k); the tied heap lives in sel[take..) */ + agg_radix_order_t* hp = sel + take; + int64_t* hk = fr + take; + int64_t hn = 0; + bool overflow = false; + for (uint32_t p = 0; p < n_parts && !overflow; p++) { + const double* pv = vals + base[p]; + const uint8_t* pk = keep + base[p]; + const int64_t* pfr = parts[p].first_row; + for (int64_t gg = 0; gg < parts[p].ng; gg++) { + if (!pk[gg]) continue; + int64_t payload = ((int64_t)p << 32) | (uint32_t)gg; + if (pv[gg] != c.thr) { + if (k >= take) { overflow = true; break; } + sel[k].idx = payload; fr[k] = pfr[gg]; k++; + continue; + } + int64_t first = pfr[gg]; + if (hn < take) { + int64_t i = hn++; + hk[i] = first; hp[i].idx = payload; + while (i > 0) { /* sift up */ + int64_t par = (i - 1) / 2; + if (hk[par] >= hk[i]) break; + int64_t tk = hk[par]; hk[par] = hk[i]; hk[i] = tk; + int64_t tp = hp[par].idx; hp[par].idx = hp[i].idx; hp[i].idx = tp; + i = par; + } + } else if (first < hk[0]) { + hk[0] = first; hp[0].idx = payload; + int64_t i = 0; + for (;;) { /* sift down */ + int64_t l = 2 * i + 1, r = l + 1, m = i; + if (l < take && hk[l] > hk[m]) m = l; + if (r < take && hk[r] > hk[m]) m = r; + if (m == i) break; + int64_t tk = hk[m]; hk[m] = hk[i]; hk[i] = tk; + int64_t tp = hp[m].idx; hp[m].idx = hp[i].idx; hp[i].idx = tp; + i = m; + } + } } + } + if (overflow) { + /* more than N groups beyond the threshold: not a threshold this + * selection understands — let the caller take the full path */ + ray_free_raw(sel); ray_free_raw(fr); + ray_free_raw(base); ray_free_raw(vals); ray_free_raw(keep); *rc = 2; return NULL; + } + /* close the gap between the strict prefix and the tied heap */ + if (k < take) { + memmove(sel + k, hp, (size_t)hn * sizeof(agg_radix_order_t)); + memmove(fr + k, hk, (size_t)hn * sizeof(int64_t)); + } + k += hn; + } agg_sort_pairs_by_key(sel, fr, k); ray_free_raw(fr); ray_free_raw(base); ray_free_raw(vals); ray_free_raw(keep); *n_emit = k; diff --git a/test/rfl/group/emit_filter_v2_route.rfl b/test/rfl/group/emit_filter_v2_route.rfl index 54447a99..41a1ba01 100644 --- a/test/rfl/group/emit_filter_v2_route.rfl +++ b/test/rfl/group/emit_filter_v2_route.rfl @@ -109,3 +109,28 @@ (set nested2 (select {from: (select {from: (select {from: tz by: k c: (count v)}) by: c n: (count c)}) where: (> n 100)})) (count nested2) -- 1 (at (at nested2 'n) 0) -- 1000 + +;; a threshold that every group ties: a near-unique two-key grouping where +;; almost every count is 1, so the N-th largest count is 1 and the kept set +;; would be the whole grouping. The result is still the first N tied groups +;; in first-seen order, and the selection must not scale with the group count. +(set n9 300000) +(set i9 (til n9)) +(set a9 (as 'I64 (+ (* i9 2654435761) (% (* i9 7) 3)))) +(set b9 (as 'I64 (% (* i9 40503) 150000))) +(set t9 (table [a b v] (list a9 b9 (% i9 5)))) +(set r9 (select {from: t9 by: [a b] c: (count v) s: (sum v) desc: c take: 10})) +(set f9 (select {from: t9 by: [a b] c: (count v) s: (sum v)})) +(count r9) -- 10 +(all (>= (at r9 'c) (at (at (xdesc f9 'c) 'c) 9))) -- true +(== (sum (at r9 'c)) (sum (take (at (xdesc f9 'c) 'c) 10))) -- true +(all (== (at r9 'a) (take (at (select {from: f9 where: (== c (max c))}) 'a) 10))) -- true +(all (== (at r9 's) (take (at (select {from: f9 where: (== c (max c))}) 's) 10))) -- true +;; the same shape under a row selection +(set r9w (select {from: t9 by: [a b] c: (count v) desc: c take: 10 where: (!= v 0)})) +(set f9w (select {from: t9 by: [a b] c: (count v) where: (!= v 0)})) +(count r9w) -- 10 +(all (== (at r9w 'a) (take (at (select {from: f9w where: (== c (max c))}) 'a) 10))) -- true +;; ascending: the smallest counts, again first-seen among the ties +(set r9a (select {from: t9 by: [a b] c: (count v) asc: c take: 10})) +(all (== (at r9a 'a) (take (at (select {from: f9 where: (== c (min c))}) 'a) 10))) -- true diff --git a/test/test_agg_contract.c b/test/test_agg_contract.c index c05e53ea..2134755c 100644 --- a/test/test_agg_contract.c +++ b/test/test_agg_contract.c @@ -1815,9 +1815,11 @@ static test_result_t test_radix_native_topn(void) { TEST_ASSERT_EQ_I(stats.routes[AGG_ROUTE_LEGACY], 0); TEST_ASSERT_EQ_I(stats.routes[AGG_ROUTE_V2_RADIX], 1); TEST_ASSERT_TRUE(stats.topn_native); - /* counts are 1 or 2: the top-10 superset is every count-2 group (500,000 - * of 1,500,000), not the full group set */ - TEST_ASSERT_EQ_I(stats.topn_kept, 500000); + /* counts are 1 or 2, so the threshold is 2 and every count-2 group + * (500,000 of 1,500,000) ties it: the selection keeps the ten of them + * with the smallest first row — the ones the emitted first-seen order + * would take — not the whole tie set */ + TEST_ASSERT_EQ_I(stats.topn_kept, 10); TEST_ASSERT_EQ_I(ray_table_nrows(r), 10); ray_t* check = ray_eval_str( "(== (at (at (select {from:rt_t by:[k j] c:(count v) desc:c take:10}) 'c) 0) " From bc350ddc04acd8874073508d8771d8a086acc43b Mon Sep 17 00:00:00 2001 From: Anton Date: Sat, 19 Sep 2026 18:57:43 +0200 Subject: [PATCH 02/51] feat(lang): add a `while` special form MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rayfall had no way to stop iterating before the end of a sequence. Every iteration primitive — map, pmap, fold, fold-left/right, scan, scan-left/right, prior — consumes its whole input, and there was no loop construct, so a "repeat until done" loop had to be written as a fold over a fixed range whose full length was paid on every call however early the work finished. `return` did not help: it exits the lambda, not the iteration, so the fold kept calling the lambda for every remaining element. Recursion was not an alternative either, since there is no TCE and the stack tops out around 1-2k frames. (while cond body...) evaluates cond, and while it is truthy evaluates each body expression in order, then tests again. It always returns null — a statement form run for effect, and a never-taken loop has no last value to report. Zero body expressions is legal, so a condition with side effects can be the whole loop; that is the shape a drain wants, where there is no sequence to iterate over and the range was only ever scaffolding. Implemented on both evaluators, which must agree: - ray_while_fn (eval.c), registered beside `if` and `do`. - an sf_while case in the bytecode compiler emitting JMPF over a backward JMP — the first backward branch the compiler produces. The VM's op_jmp already checked for a pending interrupt when the displacement is negative (dormant until now), so a runaway loop is Ctrl-C-able for free. Compiling the body inline, rather than letting it fall through to the generic special-form path, is what keeps `return` working inside a loop: it reaches the sf_return case and unwinds the lambda instead of degrading to the tree walker's identity `return`. Neither path pushes a scope around the body. `let` binds only in the top frame (env_bind_local), so a per-iteration frame would discard loop-carried `let` state on the tree-walking path while the compiled path — whose `let` writes a function-level slot — kept it, and the two evaluators would disagree on the same source. A caller wanting a fresh frame per pass writes (while cond (do ...)), which composes. patch_jump and emit_jump_back now share write_jump_offset; an out-of-range displacement still sets c->error, dropping the lambda to the interpreter. Measured on the drain shape from the issue (4-step drain under a 4096 safety bound, release build, driver baseline subtracted): fold-left over (til 4096) 1420 us/batch, the nested 64x64 workaround 45.8 us/batch, `while` 3.8 us/batch — 374x over the original and 12x over the workaround. Closes #588 --- docs/docs/language/control-flow.md | 51 +++++++++++ docs/docs/reference/all-functions.md | 3 +- src/lang/compile.c | 60 ++++++++++++- src/lang/env.c | 2 +- src/lang/eval.c | 42 +++++++++ src/lang/eval.h | 1 + test/rfl/lang/while.rfl | 129 +++++++++++++++++++++++++++ 7 files changed, 283 insertions(+), 5 deletions(-) create mode 100644 test/rfl/lang/while.rfl diff --git a/docs/docs/language/control-flow.md b/docs/docs/language/control-flow.md index 23628d5e..8cf4d7c6 100644 --- a/docs/docs/language/control-flow.md +++ b/docs/docs/language/control-flow.md @@ -30,6 +30,57 @@ Without an else branch, `if` returns `0`: 30 ``` +## Iteration: while + +`while` evaluates `cond`, and while it is truthy evaluates each body expression +in order, then tests again. It always returns null — it is a statement form, +run for effect. + +```text +‣ (set n 5) +‣ (set total 0) +‣ (while (> n 0) (set total (+ total n)) (set n (- n 1))) +‣ total +15 +``` + +It is the only iteration form that can stop early. `map`, `fold`, `scan` and +`prior` all consume their whole input, so a "repeat until done" loop written as +a fold over a fixed range pays that range's full length on every call, however +early the work finishes. `while` stops when the condition says stop, allocates +no range, and does not recurse — so it is not bounded by the stack depth a +recursive loop would hit. + +The body may be omitted, in which case a condition with side effects is the +whole loop. That is the natural shape when there is no sequence to iterate over +in the first place: + +```text +‣ (while (drain-one-batch)) +``` + +Unlike `do`, `while` pushes no scope of its own. A `let` in the body binds in +the enclosing frame and therefore survives the iteration, which is what makes a +`let` usable as a loop variable inside a lambda: + +```lisp +((fn [n] + (let i 0) + (let acc 0) + (while (< i n) (let acc (+ acc i)) (let i (+ i 1))) + acc) 4) ; => 6 +``` + +When a fresh binding per pass is wanted instead, wrap the body in `do`, which +does push a scope: + +```lisp +(while (< i 3) (do (let tmp (* i i)) (use tmp)) (set i (+ i 1))) +``` + +A loop whose condition never goes false runs until interrupted; Ctrl-C breaks +out of one at the REPL. + ## Variable Binding: set and let `set` creates a global binding. `let` creates a local binding scoped to the enclosing `do`: diff --git a/docs/docs/reference/all-functions.md b/docs/docs/reference/all-functions.md index 85975352..fff747de 100644 --- a/docs/docs/reference/all-functions.md +++ b/docs/docs/reference/all-functions.md @@ -12,7 +12,7 @@ |---|---|---| | [Arithmetic](#arithmetic) (24) | [Comparison](#comparison) (7) | [Logic](#logic) (3) | | [Aggregation](#aggregation) (25) | [Higher-Order](#higher-order) (13) | [Collection](#collection) (41) | -| [Sorting & Ordering](#sorting) (10) | [Control Flow & Special Forms](#control) (11) | [Table Operations](#table-ops) (20) | +| [Sorting & Ordering](#sorting) (10) | [Control Flow & Special Forms](#control) (12) | [Table Operations](#table-ops) (20) | | [Query](#query) (4) | [Joins](#joins) (7) | [Pivot](#pivot) (1) | | [String](#string-ops) (11) | [Temporal](#temporal) (3) | [Type & Introspection](#type-ops) (5) | | [I/O & Output](#io) (12) | [System & Utility](#system) (15) | [Serialization](#serialization) (2) | @@ -359,6 +359,7 @@ Special forms receive their arguments unevaluated. These are the core language p | `let` | binary | special | Bind value to local variable (lexical scope) | `(let y (+ x 1))` | | `if` | variadic | special | Conditional: (if cond then else) | `(if (> x 0) "pos" "neg")` | | `do` | variadic | special | Sequential execution, returns last value | `(do (set x 1) (set y 2) (+ x y))` | +| `while` | variadic | special | Iterate while cond is truthy; returns null | `(while (> n 0) (set n (- n 1)))` | | `fn` | variadic | special | Create lambda function | `(fn [x y] (+ x y))` | | `try` | binary | special | Error handling: (try expr handler-fn-or-fallback-value) | `(try (/ 1 0) (fn [e] 0))` | | `raise` | unary | — | Throw an error with message | `(raise "bad input")` | diff --git a/src/lang/compile.c b/src/lang/compile.c index 5384ff31..a20ad839 100644 --- a/src/lang/compile.c +++ b/src/lang/compile.c @@ -183,16 +183,39 @@ static int32_t emit_jump(compiler_t *c, uint8_t opcode) { return patch_pos; } -static void patch_jump(compiler_t *c, int32_t pos) { - int32_t raw = c->code_len - pos - 2; +/* Write a jump displacement into the 2-byte operand at `pos`, which is + * relative to the instruction's end (pos + 2). Out of int16_t range sets + * c->error, aborting bytecode emission so the lambda falls back to the + * tree-walking interpreter — the same graceful degradation as the other + * compile-time bailouts. */ +static void write_jump_offset(compiler_t *c, int32_t pos, int32_t target) { + int32_t raw = target - pos - 2; if (raw > 32767 || raw < -32768) { c->error = true; return; } int16_t offset = (int16_t)raw; c->code[pos] = (uint8_t)((uint16_t)offset >> 8); c->code[pos + 1] = (uint8_t)(offset & 0xFF); } +/* Resolve a forward jump emitted earlier, now that its target is here. */ +static void patch_jump(compiler_t *c, int32_t pos) { + write_jump_offset(c, pos, c->code_len); +} + +/* Emit an unconditional jump BACKWARD to an already-emitted address — the + * mirror of patch_jump, where the target is known up front and the + * displacement comes out negative. The VM's op_jmp checks for a pending + * interrupt whenever the offset is negative, so a runaway loop built from + * this stays Ctrl-C-able. */ +static void emit_jump_back(compiler_t *c, int32_t target) { + emit(c, OP_JMP); + int32_t pos = c->code_len; + emit(c, 0); + emit(c, 0); + write_jump_offset(c, pos, target); +} + /* Cached sym IDs for special forms */ -static _Thread_local int64_t sf_set = -1, sf_let = -1, sf_if = -1, sf_do = -1, sf_fn = -1, sf_self = -1, sf_try = -1, sf_return = -1, sf_null = -1; +static _Thread_local int64_t sf_set = -1, sf_let = -1, sf_if = -1, sf_do = -1, sf_while = -1, sf_fn = -1, sf_self = -1, sf_try = -1, sf_return = -1, sf_null = -1; static _Thread_local int64_t sf_eval = -1, sf_resolve = -1; static void init_sf_syms(void) { @@ -201,6 +224,7 @@ static void init_sf_syms(void) { sf_let = ray_sym_intern("let", 3); sf_if = ray_sym_intern("if", 2); sf_do = ray_sym_intern("do", 2); + sf_while= ray_sym_intern("while", 5); sf_fn = ray_sym_intern("fn", 2); sf_self = ray_sym_intern("self", 4); sf_try = ray_sym_intern("try", 3); @@ -369,6 +393,36 @@ static void compile_list(compiler_t *c, ray_t *ast) { return; } + /* (while cond body...) — the only backward branch the compiler + * emits. Body values are discarded (OP_POP each), and the form + * yields null however many times it iterated, matching + * ray_while_fn. No scope is pushed: `let` in the body writes this + * frame's local slot, so loop-carried state survives the pass + * exactly as it does on the tree-walking path. + * + * Compiling the body inline — rather than letting this fall through + * to the generic special-form path below — is what keeps `return` + * working inside a loop: it reaches the sf_return case and unwinds + * the lambda, instead of degrading to the tree walker's identity + * `return` (issue 588). */ + if (sym_id == sf_while && n >= 2) { + int32_t top = c->code_len; + compile_expr(c, elems[1]); + /* Truthiness belongs to the materialized value, not to the + * non-NULL lazy handle containing it — same rule as sf_if. */ + emit(c, OP_FORCE); + int32_t jmpf_pos = emit_jump(c, OP_JMPF); + for (int64_t i = 2; i < n; i++) { + compile_expr(c, elems[i]); + emit(c, OP_POP); + } + emit_jump_back(c, top); + patch_jump(c, jmpf_pos); + int32_t idx = add_constant(c, RAY_NULL_OBJ); + emit_const(c, idx); + return; + } + /* (fn [params] body...) — nested lambda via dynamic eval */ if (sym_id == sf_fn && n >= 3) { if (ast_refs_locals(c, ast)) { diff --git a/src/lang/env.c b/src/lang/env.c index a4356f63..886bbd47 100644 --- a/src/lang/env.c +++ b/src/lang/env.c @@ -839,7 +839,7 @@ int32_t ray_env_list_user(int64_t* sym_ids, ray_t** vals, int32_t max_entries) { /* ---- Prefix lookup ---- */ static const char* s_keywords[] = { - "def", "do", "false", "fn", "if", "let", "set", "true", NULL + "def", "do", "false", "fn", "if", "let", "set", "true", "while", NULL }; /* Compare helper for qsort on const char* */ diff --git a/src/lang/eval.c b/src/lang/eval.c index d72841d4..defce74d 100644 --- a/src/lang/eval.c +++ b/src/lang/eval.c @@ -1926,6 +1926,47 @@ ray_t* ray_do_fn(ray_t** args, int64_t n) { return result; } +/* (while cond body...) — iterate while cond is truthy. Receives + * unevaluated args. Always returns null: it is a statement form run for + * effect, and a never-taken loop has no last value to report. + * + * Zero body expressions is legal — a condition with side effects is then + * the whole loop, which is the shape a "repeat until done" drain wants + * (issue 588): there is no sequence to iterate, so folding over a range + * was only ever scaffolding. + * + * Unlike ray_do_fn this pushes NO scope around the body. `let` binds in + * the top frame only (env_bind_local), so a per-iteration frame would + * discard loop-carried `let` state here while the compiled path — whose + * `let` writes a function-level bytecode slot — kept it, and the two + * evaluators would disagree on the same source. A caller wanting a fresh + * frame per pass writes (while cond (do ...)), which composes. + * + * No interrupt check is needed: the condition goes through ray_eval on + * every pass, whose entry guard raises `cancel` when a Ctrl-C has landed. + * The compiled form is emitted by the bytecode compiler — see compile.c. */ +ray_t* ray_while_fn(ray_t** args, int64_t n) { + if (n < 1) return ray_error("domain", "while: expected at least 1 arg (cond), got %lld", (long long)n); + for (;;) { + ray_t* cond = ray_eval(args[0]); + if (RAY_IS_ERR(cond)) return cond; + /* Materialize lazy handles before testing truthiness — the + * truthiness belongs to the value, not to the non-NULL handle + * that happens to contain it (same rule as ray_cond_fn). */ + if (ray_is_lazy(cond)) + cond = ray_lazy_materialize(cond); + if (RAY_IS_ERR(cond)) return cond; + int truthy = is_truthy(cond); + ray_release(cond); + if (!truthy) return RAY_NULL_OBJ; + for (int64_t i = 1; i < n; i++) { + ray_t* val = ray_eval(args[i]); + if (RAY_IS_ERR(val)) return val; + ray_release(val); + } + } +} + /* ══════════════════════════════════════════ * Lambda functions * ══════════════════════════════════════════ */ @@ -3149,6 +3190,7 @@ static void ray_register_builtins(void) { register_binary("let", RAY_FN_SPECIAL_FORM, ray_let_fn); register_vary("if", RAY_FN_SPECIAL_FORM, ray_cond_fn); register_vary("do", RAY_FN_SPECIAL_FORM, ray_do_fn); + register_vary("while", RAY_FN_SPECIAL_FORM, ray_while_fn); register_vary("fn", RAY_FN_SPECIAL_FORM, ray_fn); /* Aggregation builtins */ diff --git a/src/lang/eval.h b/src/lang/eval.h index d52e67b3..0630700c 100644 --- a/src/lang/eval.h +++ b/src/lang/eval.h @@ -363,6 +363,7 @@ ray_t* ray_set_fn(ray_t* name_obj, ray_t* val_expr); ray_t* ray_let_fn(ray_t* name_obj, ray_t* val_expr); ray_t* ray_cond_fn(ray_t** args, int64_t n); ray_t* ray_do_fn(ray_t** args, int64_t n); +ray_t* ray_while_fn(ray_t** args, int64_t n); ray_t* ray_fn(ray_t** args, int64_t n); ray_t* ray_raise_fn(ray_t* val); ray_t* ray_try_fn(ray_t* expr, ray_t* handler_expr); diff --git a/test/rfl/lang/while.rfl b/test/rfl/lang/while.rfl new file mode 100644 index 00000000..8b1f24ce --- /dev/null +++ b/test/rfl/lang/while.rfl @@ -0,0 +1,129 @@ +;; (while cond body...) — the language's only early-terminating loop. +;; +;; Motivation (issue 588): every iteration primitive is exhaustive, so a +;; "repeat until done" loop had to be a fold over a fixed range whose full +;; length was paid on every call. `while` stops when the condition says so. +;; +;; Two implementations must agree: the tree walker (ray_while_fn in eval.c) +;; and the bytecode compiler (the sf_while case in compile.c, which emits a +;; backward OP_JMP). Every behavioural assertion below is therefore made +;; twice where it can be — once at top level (tree walk) and once inside a +;; lambda (compiled). The `let`-as-loop-variable cases are the ones that +;; would catch the two paths drifting apart: `let` binds in the top scope +;; frame (env.c env_bind_local), so a per-iteration scope push here would +;; make loop-carried `let` state work compiled and hang interpreted. + +;; ── basic iteration ────────────────────────────────────────────────── +(set i 0) +(set n 0) +(while (< i 5) (set n (+ n 1)) (set i (+ i 1))) +i -- 5 +n -- 5 + +;; a statement form: always null, whatever the body evaluated to +(set i 0) +(nil? (while (< i 2) (set i (+ i 1)) 'discarded)) -- true + +;; zero-trip loop: body never runs, result is still null +(set x 99) +(nil? (while false (set x 1))) -- true +x -- 99 + +;; a false-from-the-start condition that is an expression, not a literal +(set g 0) +(while (> g 0) (set g (+ g 1))) +g -- 0 + +;; ── body-less form ─────────────────────────────────────────────────── +;; The condition alone drives the loop — there is no sequence to fold over, +;; which is the shape the issue's drain loop actually wanted. +(set k 0) +(while (do (set k (+ k 1)) (< k 4))) +k -- 4 + +;; ── arity ──────────────────────────────────────────────────────────── +(while) !- domain + +;; ── termination is condition-driven, not bounded by a range ────────── +;; The step runs exactly as many times as there is work. This is precisely +;; what the fold-left workaround could not do: it called the lambda for all +;; DRAIN_STEPS elements however early the drain finished. +(set work 3) +(set calls 0) +(set step (fn [] (set calls (+ calls 1)) (set work (- work 1)) (> work 0))) +(set more true) +(while more (set more (step))) +calls -- 3 +work -- 0 + +;; ── locals: `let` as loop-carried state ────────────────────────────── +;; Compiled path: `let` writes a function-level slot, so it survives the +;; iteration. If the interpreter ever pushes a per-iteration scope, the +;; same source hangs instead — which is why this is asserted both ways. +((fn [n] (let i 0) (let acc 0) (while (< i n) (let acc (+ acc i)) (let i (+ i 1))) acc) 4) -- 6 + +;; the loop condition reads a lambda PARAM (the special_form_locals hazard) +((fn [n] (let c 0) (while (< c n) (let c (+ c 1))) c) 3) -- 3 + +;; tree-walk path, same arithmetic, via globals +(set i 0) +(set acc 0) +(while (< i 4) (set acc (+ acc i)) (set i (+ i 1))) +acc -- 6 + +;; a lambda-local loop must not leak its names to the global env +((fn [] (let scratch 7) (while false (let scratch 8)) scratch)) -- 7 +scratch !- name + +;; ── `return` from inside a body exits the enclosing lambda ─────────── +;; The compiled `while` must route its body through the normal sf_return +;; path. If `while` fell back to a dynamic tree-walk eval, `return` would +;; degrade to identity and the loop would run all ten iterations. +(set hits 0) +((fn [] (let i 0) (while (< i 10) (set hits (+ hits 1)) (if (> i 2) (return 42)) (let i (+ i 1))) 7)) -- 42 +hits -- 4 + +;; without the early return the same loop runs to completion +(set hits 0) +((fn [] (let i 0) (while (< i 10) (set hits (+ hits 1)) (let i (+ i 1))) 7)) -- 7 +hits -- 10 + +;; ── errors ─────────────────────────────────────────────────────────── +;; An error raised in the body propagates out and stops the loop — note the +;; loop would otherwise never terminate, so `c` pins that it stopped on the +;; FIRST pass. `raise` reports as a bare `domain` error and carries its +;; payload on __VM->raise_val (eval.c ray_raise_fn), so the payload is +;; asserted through `try`, which is the only way to observe it. +(set c 0) +(== (try (while true (set c (+ c 1)) (raise 'boom)) (fn [e] e)) 'boom) -- true +c -- 1 + +;; an error raised in the condition propagates before the body ever runs +(set c 0) +(== (try (while (raise 'bad-cond) (set c (+ c 1))) (fn [e] e)) 'bad-cond) -- true +c -- 0 + +;; same, unwinding out of a compiled lambda +(== (try ((fn [] (while true (raise 'inner)))) (fn [e] e)) 'inner) -- true + +;; a plain (non-raise) error in the body also terminates the loop +(set c 0) +(while true (set c (+ c 1)) (fold-right +)) !- domain +c -- 1 + +;; ── nesting ────────────────────────────────────────────────────────── +(set total 0) +(set i 0) +(while (< i 3) (set j 0) (while (< j 2) (set total (+ total 1)) (set j (+ j 1))) (set i (+ i 1))) +total -- 6 + +;; nested, compiled, with let-locals on both levels +((fn [] (let i 0) (let t 0) (while (< i 3) (let j 0) (while (< j 2) (let t (+ t 1)) (let j (+ j 1))) (let i (+ i 1))) t)) -- 6 + +;; ── explicit per-iteration isolation still composes ────────────────── +;; `while` does not push a scope; `do` does. Wrapping the body in `do` +;; gives a fresh frame per pass when that is what you want. +(set out (list)) +(set i 0) +(while (< i 3) (do (let sq (* i i)) (set out (concat out sq))) (set i (+ i 1))) +out -- (list 0 1 4) From 8168f1ea8cb24e74c4b42e685da233636c30ddfb Mon Sep 17 00:00:00 2001 From: Anton Date: Sat, 19 Sep 2026 21:14:42 +0200 Subject: [PATCH 03/51] feat(lang): add `times` and `fold-while` MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The two remaining early-termination forms requested in #588, alongside the `while` that landed in #590. Neither is an unblock — `while` already covers the reporter's case — but both complete the vocabulary he asked for. `times` ------- (times n body...) runs the body exactly n times and returns null. `do` is already progn in this language, so the bounded loop could not reuse that name; overloading `do` on an integer head was rejected as genuinely ambiguous — (do 5) would have to mean either "loop five times over nothing" or "return 5", and a computed first expression that happened to be an integer would silently change meaning. The count is evaluated ONCE on entry, so a body mutating whatever produced it cannot change how many passes remain. A count of zero or less runs zero times rather than trapping; a non-integer is a type error. Compiled as a counted loop over a hidden local slot, reusing the backward branch added for `while`. ray_times_norm_fn type-checks the count and clamps a negative bound to zero on entry, which lets the per-pass test be a bare truthiness check on the counter — 0 is falsy, so no comparison call is needed per pass and a negative bound cannot run away. That helper and the decrement are pushed as constant-pool objects rather than resolved by name, so the loop's own arithmetic is unnamable from source and cannot be swapped out by a `(set - ...)` override. The counter's slot is addressed by index and its sym carries a space, so no source token can collide with it and nested `times` counters stay apart. As with `while`, the body compiles inline, so `return` unwinds the enclosing lambda from inside the loop. fold-while ---------- (fold-while pred f init xs) offers the accumulator to `pred` before each step and stops on a falsy answer, yielding the accumulator as it stands. The test precedes the first element, so a predicate false at the start returns `init` untouched. The predicate takes the accumulator rather than the element: that is the form that expresses "iterate until the running result says stop", which is the early termination actually being asked for. One deliberate divergence from ray_fold_fn: that routes its collection through unbox_vec_arg -> to_boxed_list, boxing every element up front. For a primitive whose purpose is to stop early, paying for the tail it never reaches is the cost being removed, so elements are pulled one at a time via collection_elem. A plain variadic builtin — no compiler work, since it dispatches through the normal call path. Measured, release builds: `times` 100 ns/pass against 180 ns for `while` plus a manual counter over 1e6 passes. `fold-while` stopping after three elements of a 1e6-element vector costs 3.9 us against 369 ms for the `fold-left` equivalent, which had to box and walk all million to discover it was done. Tests: test/rfl/lang/times.rfl and test/rfl/collection/fold_while.rfl, each behaviour asserted on both evaluator paths where applicable — including the count validation and negative clamp on the compiled path, which runs through entirely different code from the tree walker's check. --- docs/docs/language/control-flow.md | 42 +++++++++ docs/docs/language/functions.md | 1 + docs/docs/reference/all-functions.md | 6 +- src/lang/compile.c | 68 ++++++++++++++- src/lang/env.c | 2 +- src/lang/eval.c | 75 ++++++++++++++++ src/lang/eval.h | 3 + src/lang/internal.h | 1 + src/ops/collection.c | 48 +++++++++++ test/rfl/collection/fold_while.rfl | 75 ++++++++++++++++ test/rfl/lang/times.rfl | 124 +++++++++++++++++++++++++++ 11 files changed, 441 insertions(+), 4 deletions(-) create mode 100644 test/rfl/collection/fold_while.rfl create mode 100644 test/rfl/lang/times.rfl diff --git a/docs/docs/language/control-flow.md b/docs/docs/language/control-flow.md index 8cf4d7c6..dbe7a749 100644 --- a/docs/docs/language/control-flow.md +++ b/docs/docs/language/control-flow.md @@ -81,6 +81,35 @@ does push a scope: A loop whose condition never goes false runs until interrupted; Ctrl-C breaks out of one at the REPL. +## Bounded Iteration: times + +`times` runs the body a fixed number of times and returns null. + +```text +‣ (set n 0) +‣ (times 5 (set n (+ n 1))) +‣ n +5 +``` + +The count is evaluated **once**, on entry, so the bound is fixed however the +body mutates whatever produced it: + +```lisp +(set k 3) +(times k (set k (+ k 10))) ; runs 3 times, not forever +``` + +A count of zero or less runs the body zero times rather than raising — a bound +that computes to empty is a no-op, not an error. A non-integer count is a type +error. Like `while`, `times` pushes no scope of its own, so a `let` in the body +binds in the enclosing frame; wrap the body in `do` when a fresh binding per +pass is wanted. + +Reach for `times` when the number of passes is known up front and for `while` +when it is not. Neither allocates a sequence, so neither pays for a range that +exists only to be counted. + ## Variable Binding: set and let `set` creates a global binding. `let` creates a local binding scoped to the enclosing `do`: @@ -238,7 +267,20 @@ Lambdas are auto-mapped over vectors when called directly. Use `map` for explici ‣ (scan + [1 2 3 4 5]) ; => [1 3 6 10 15] +;; fold-while stops as soon as the running result fails the predicate, +;; leaving the rest of the collection untouched +‣ (fold-while (fn [acc] (< acc 100)) + 0 (til 1000)) +; => 105 + ;; where returns indices matching a condition ‣ (where (> (til 10) 3)) ; => [4 5 6 7 8 9] ``` + +`map`, `fold`, `scan` and `prior` all consume their whole input. `fold-while` +is the one member of the family that can stop: the accumulator is offered to +the predicate before each step, and a falsy answer ends the fold and yields the +accumulator as it stands. The test happens before the first element too, so a +predicate that is false at the start returns the initial value untouched. +Elements are pulled one at a time, so a fold that stops after three steps costs +three elements however long the collection is. diff --git a/docs/docs/language/functions.md b/docs/docs/language/functions.md index 58f578b7..943e1b78 100644 --- a/docs/docs/language/functions.md +++ b/docs/docs/language/functions.md @@ -133,6 +133,7 @@ Functions that take other functions as arguments. | `filter` | binary | Keep elements where boolean mask is true | `(filter [1 2 3 4] (> [1 2 3 4] 2))` → `[3 4]` | | `fold` | variadic | Reduce with function and initial value | `(fold + 0 [1 2 3])` → `6` | | `fold-left` | variadic | Left-associative fold | `(fold-left - 10 [1 2 3])` → `4` | +| `fold-while` | variadic | Fold that stops when the accumulator fails pred | `(fold-while (fn [a] (< a 100)) + 0 (til 1000))` → `105` | | `fold-right` | variadic | Right-associative fold | `(fold-right - 10 [1 2 3])` → `-8` | | `scan` | variadic | Running fold (returns all intermediate results) | `(scan + (enlist 1 2 3))` → `[1 3 6]` | | `scan-left` | variadic | Left-to-right running fold | `(scan-left + (enlist 1 2 3))` → `[1 3 6]` | diff --git a/docs/docs/reference/all-functions.md b/docs/docs/reference/all-functions.md index fff747de..eba0d980 100644 --- a/docs/docs/reference/all-functions.md +++ b/docs/docs/reference/all-functions.md @@ -12,7 +12,7 @@ |---|---|---| | [Arithmetic](#arithmetic) (24) | [Comparison](#comparison) (7) | [Logic](#logic) (3) | | [Aggregation](#aggregation) (25) | [Higher-Order](#higher-order) (13) | [Collection](#collection) (41) | -| [Sorting & Ordering](#sorting) (10) | [Control Flow & Special Forms](#control) (12) | [Table Operations](#table-ops) (20) | +| [Sorting & Ordering](#sorting) (10) | [Control Flow & Special Forms](#control) (13) | [Table Operations](#table-ops) (20) | | [Query](#query) (4) | [Joins](#joins) (7) | [Pivot](#pivot) (1) | | [String](#string-ops) (11) | [Temporal](#temporal) (3) | [Type & Introspection](#type-ops) (5) | | [I/O & Output](#io) (12) | [System & Utility](#system) (15) | [Serialization](#serialization) (2) | @@ -71,7 +71,7 @@ Generated from `src/lang/eval.c` in this checkout. The categorized reference bel `.db.parted.get`, `.db.parted.tables`, `.db.parted.fill`, `alter`, `print`, `.sys.gc`, `.mem.objsize`, `.mem.ts`, `.sys.timeit`, `.sys.env`, `.sys.args`, `.ipc.open`, `.ipc.handle`, `.repl.disconnect`, `.log.open`, `.log.roll`, `.log.snapshot`, `.log.sync`, `.log.close`, `.log.purge`, `quote`, `return`, `.time.now`, `.time.timer.set`, -`fold-left`, `fold-right`, `scan-left`, `scan-right`, `del`, `.sys.build`, `.sys.mem`, `.sys.prof`, +`fold-left`, `fold-while`, `fold-right`, `scan-left`, `scan-right`, `del`, `.sys.build`, `.sys.mem`, `.sys.prof`, `.sys.querylog`, `.sys.querylog.enable`, `modify`, `pivot`, `.sys.info`, `datoms`, `assert-fact`, `retract-fact`, `scan-eav`, `pull`, `rule`, `query`, `dl-program`, `dl-add-edb`, `knn`, `hnsw-build`, `ann`, `.graph.build`, `.graph.pagerank`, `.graph.connected`, `.graph.dijkstra`, `.graph.louvain`, `.graph.degree`, @@ -221,6 +221,7 @@ Functions that take other functions as arguments for mapping, folding, and filte | `pmap` | variadic | — | Parallel map (multi-threaded, returns a list) | `(pmap (fn [x] (* x x)) [1 2 3])` → `(1 4 9)` | | `fold` | variadic | — | Reduce with function and initial value | `(fold + 0 [1 2 3])` → `6` | | `fold-left` | variadic | — | Left-associative fold | `(fold-left - 10 [1 2 3])` → `4` | +| `fold-while` | variadic | — | Fold that stops when the accumulator fails pred | `(fold-while (fn [a] (< a 100)) + 0 (til 1000))` → `105` | | `fold-right` | variadic | — | Right-associative fold | `(fold-right - 10 [1 2 3])` → `-8` | | `scan` | variadic | — | Running fold (all intermediate results) | `(scan + (enlist 1 2 3))` → `[1 3 6]` | | `scan-left` | variadic | — | Left-to-right running fold | `(scan-left + (enlist 1 2 3))` → `[1 3 6]` | @@ -360,6 +361,7 @@ Special forms receive their arguments unevaluated. These are the core language p | `if` | variadic | special | Conditional: (if cond then else) | `(if (> x 0) "pos" "neg")` | | `do` | variadic | special | Sequential execution, returns last value | `(do (set x 1) (set y 2) (+ x y))` | | `while` | variadic | special | Iterate while cond is truthy; returns null | `(while (> n 0) (set n (- n 1)))` | +| `times` | variadic | special | Run body exactly n times (count evaluated once); returns null | `(times 5 (set n (+ n 1)))` | | `fn` | variadic | special | Create lambda function | `(fn [x y] (+ x y))` | | `try` | binary | special | Error handling: (try expr handler-fn-or-fallback-value) | `(try (/ 1 0) (fn [e] 0))` | | `raise` | unary | — | Throw an error with message | `(raise "bad input")` | diff --git a/src/lang/compile.c b/src/lang/compile.c index a20ad839..f37ccfb6 100644 --- a/src/lang/compile.c +++ b/src/lang/compile.c @@ -215,7 +215,7 @@ static void emit_jump_back(compiler_t *c, int32_t target) { } /* Cached sym IDs for special forms */ -static _Thread_local int64_t sf_set = -1, sf_let = -1, sf_if = -1, sf_do = -1, sf_while = -1, sf_fn = -1, sf_self = -1, sf_try = -1, sf_return = -1, sf_null = -1; +static _Thread_local int64_t sf_set = -1, sf_let = -1, sf_if = -1, sf_do = -1, sf_while = -1, sf_times = -1, sf_fn = -1, sf_self = -1, sf_try = -1, sf_return = -1, sf_null = -1; static _Thread_local int64_t sf_eval = -1, sf_resolve = -1; static void init_sf_syms(void) { @@ -225,6 +225,7 @@ static void init_sf_syms(void) { sf_if = ray_sym_intern("if", 2); sf_do = ray_sym_intern("do", 2); sf_while= ray_sym_intern("while", 5); + sf_times= ray_sym_intern("times", 5); sf_fn = ray_sym_intern("fn", 2); sf_self = ray_sym_intern("self", 4); sf_try = ray_sym_intern("try", 3); @@ -423,6 +424,71 @@ static void compile_list(compiler_t *c, ray_t *ast) { return; } + /* (times n body...) — a counted loop, desugared onto the same + * backward branch `while` uses. + * + * The count is evaluated ONCE into a hidden local slot, so the + * bound is fixed on entry however the body mutates its source. + * ray_times_norm_fn type-checks it and clamps a negative bound to + * zero, which is what lets the per-pass test be a bare truthiness + * check on the counter: 0 is falsy (is_truthy), everything else is + * truthy, so the loop needs no comparison call and a negative bound + * cannot run away. + * + * Both helpers are pushed as constant-pool objects rather than + * resolved by name, so the loop's own arithmetic is invisible to + * user code and immune to an override of `-` or `>`. The counter's + * slot is likewise addressed by index, never by name, and its + * sym contains a space so no source token can collide with it — + * which is what keeps nested `times` counters apart. + * + * As with `while`, compiling the body inline is what keeps + * `return` unwinding the enclosing lambda from inside the loop. */ + if (sym_id == sf_times && n >= 2) { + ray_t *norm_fn = ray_fn_unary("times norm", RAY_FN_NONE, ray_times_norm_fn); + ray_t *dec_fn = ray_fn_unary("times dec", RAY_FN_NONE, ray_times_dec_fn); + if (!norm_fn || RAY_IS_ERR(norm_fn) || !dec_fn || RAY_IS_ERR(dec_fn)) { + if (norm_fn && !RAY_IS_ERR(norm_fn)) ray_release(norm_fn); + if (dec_fn && !RAY_IS_ERR(dec_fn)) ray_release(dec_fn); + c->error = true; + return; + } + int32_t norm_idx = add_constant(c, norm_fn); + int32_t dec_idx = add_constant(c, dec_fn); + ray_release(norm_fn); + ray_release(dec_fn); + int32_t cslot = add_local(c, ray_sym_intern("times ctr", 9)); + if (cslot < 0 || c->error) { c->error = true; return; } + + /* counter = normalize() */ + emit_const(c, norm_idx); + compile_expr(c, elems[1]); + emit(c, OP_CALL1); + emit(c, OP_STOREENV); + emit(c, (uint8_t)cslot); + + int32_t top = c->code_len; + emit(c, OP_LOADENV); + emit(c, (uint8_t)cslot); + int32_t jmpf_pos = emit_jump(c, OP_JMPF); + for (int64_t i = 2; i < n; i++) { + compile_expr(c, elems[i]); + emit(c, OP_POP); + } + /* counter = counter - 1 */ + emit_const(c, dec_idx); + emit(c, OP_LOADENV); + emit(c, (uint8_t)cslot); + emit(c, OP_CALL1); + emit(c, OP_STOREENV); + emit(c, (uint8_t)cslot); + emit_jump_back(c, top); + patch_jump(c, jmpf_pos); + int32_t null_idx = add_constant(c, RAY_NULL_OBJ); + emit_const(c, null_idx); + return; + } + /* (fn [params] body...) — nested lambda via dynamic eval */ if (sym_id == sf_fn && n >= 3) { if (ast_refs_locals(c, ast)) { diff --git a/src/lang/env.c b/src/lang/env.c index 886bbd47..65ba600a 100644 --- a/src/lang/env.c +++ b/src/lang/env.c @@ -839,7 +839,7 @@ int32_t ray_env_list_user(int64_t* sym_ids, ray_t** vals, int32_t max_entries) { /* ---- Prefix lookup ---- */ static const char* s_keywords[] = { - "def", "do", "false", "fn", "if", "let", "set", "true", "while", NULL + "def", "do", "false", "fn", "if", "let", "set", "times", "true", "while", NULL }; /* Compare helper for qsort on const char* */ diff --git a/src/lang/eval.c b/src/lang/eval.c index defce74d..ced4f8ad 100644 --- a/src/lang/eval.c +++ b/src/lang/eval.c @@ -1967,6 +1967,79 @@ ray_t* ray_while_fn(ray_t** args, int64_t n) { } } +/* Read a loop count from an evaluated atom. Integers only: floats are + * rejected rather than truncated, because as_i64 on an F64 reads the bit + * pattern, and a silent (times 1.5 ...) meaning "once" hides a mistake + * either way. Returns 0 on success, or fills *err. */ +static int loop_count(ray_t* x, int64_t* out, const char* who, ray_t** err) { + if (ray_is_atom(x)) { + switch (x->type) { + case -RAY_I64: case -RAY_I32: case -RAY_I16: case -RAY_U8: + *out = as_i64(x); + return 0; + default: break; + } + } + *err = ray_error("type", "%s: count must be an integer, got %s", who, ray_type_name(x->type)); + return -1; +} + +/* The compiled `times` loop drives its counter through these two, held + * as constant-pool objects rather than bound in the env — so they carry + * no name a user could write, and the loop's arithmetic cannot be + * changed out from under it by a `(set - ...)` style override. + * + * Normalizing once on entry (type-check, and clamp a negative bound to + * zero) is what lets the loop test the raw counter for truthiness: 0 is + * falsy, every other count is truthy, so no comparison call is needed + * per pass and a negative bound cannot run away. */ +ray_t* ray_times_norm_fn(ray_t* x) { + int64_t reps = 0; + ray_t* err = NULL; + if (loop_count(x, &reps, "times", &err)) return err; + return make_i64(reps < 0 ? 0 : reps); +} + +ray_t* ray_times_dec_fn(ray_t* x) { + return make_i64(as_i64(x) - 1); +} + +/* (times n body...) — evaluate the body exactly n times. Receives + * unevaluated args. Always returns null, like `while`. + * + * `n` is evaluated ONCE, up front: the count is a bound fixed on entry, + * so a body that mutates whatever produced it cannot change how many + * passes remain. A count of zero or less runs the body zero times + * rather than trapping — a bound computed as empty is a no-op, not an + * error. + * + * Pushes no scope, for the reason given on ray_while_fn: `let` binds in + * the top frame only, so a per-pass frame would discard loop-carried + * state here while the compiled path kept it. + * + * The compiled form is emitted by the bytecode compiler — see compile.c. */ +ray_t* ray_times_fn(ray_t** args, int64_t n) { + if (n < 1) return ray_error("domain", "times: expected at least 1 arg (count), got %lld", (long long)n); + ray_t* cnt = ray_eval(args[0]); + if (RAY_IS_ERR(cnt)) return cnt; + if (ray_is_lazy(cnt)) + cnt = ray_lazy_materialize(cnt); + if (RAY_IS_ERR(cnt)) return cnt; + int64_t reps = 0; + ray_t* err = NULL; + int bad = loop_count(cnt, &reps, "times", &err); + ray_release(cnt); + if (bad) return err; + for (int64_t r = 0; r < reps; r++) { + for (int64_t i = 1; i < n; i++) { + ray_t* val = ray_eval(args[i]); + if (RAY_IS_ERR(val)) return val; + ray_release(val); + } + } + return RAY_NULL_OBJ; +} + /* ══════════════════════════════════════════ * Lambda functions * ══════════════════════════════════════════ */ @@ -3191,6 +3264,7 @@ static void ray_register_builtins(void) { register_vary("if", RAY_FN_SPECIAL_FORM, ray_cond_fn); register_vary("do", RAY_FN_SPECIAL_FORM, ray_do_fn); register_vary("while", RAY_FN_SPECIAL_FORM, ray_while_fn); + register_vary("times", RAY_FN_SPECIAL_FORM, ray_times_fn); register_vary("fn", RAY_FN_SPECIAL_FORM, ray_fn); /* Aggregation builtins */ @@ -3512,6 +3586,7 @@ static void ray_register_builtins(void) { /* Directional fold/scan variants */ register_vary("fold-left", RAY_FN_NONE, ray_fold_left_fn); + register_vary("fold-while", RAY_FN_NONE, ray_fold_while_fn); register_vary("fold-right", RAY_FN_NONE, ray_fold_right_fn); register_vary("scan-left", RAY_FN_NONE, ray_scan_left_fn); register_vary("scan-right", RAY_FN_NONE, ray_scan_right_fn); diff --git a/src/lang/eval.h b/src/lang/eval.h index 0630700c..0cfab1d0 100644 --- a/src/lang/eval.h +++ b/src/lang/eval.h @@ -364,6 +364,9 @@ ray_t* ray_let_fn(ray_t* name_obj, ray_t* val_expr); ray_t* ray_cond_fn(ray_t** args, int64_t n); ray_t* ray_do_fn(ray_t** args, int64_t n); ray_t* ray_while_fn(ray_t** args, int64_t n); +ray_t* ray_times_fn(ray_t** args, int64_t n); +ray_t* ray_times_norm_fn(ray_t* x); +ray_t* ray_times_dec_fn(ray_t* x); ray_t* ray_fn(ray_t** args, int64_t n); ray_t* ray_raise_fn(ray_t* val); ray_t* ray_try_fn(ray_t* expr, ray_t* handler_expr); diff --git a/src/lang/internal.h b/src/lang/internal.h index 103bdd8d..4a513686 100644 --- a/src/lang/internal.h +++ b/src/lang/internal.h @@ -526,6 +526,7 @@ ray_t* ray_binr_fn(ray_t* sorted, ray_t* val); ray_t* ray_map_left_fn(ray_t** args, int64_t n); ray_t* ray_map_right_fn(ray_t** args, int64_t n); ray_t* ray_fold_left_fn(ray_t** args, int64_t n); +ray_t* ray_fold_while_fn(ray_t** args, int64_t n); ray_t* ray_fold_right_fn(ray_t** args, int64_t n); ray_t* ray_scan_left_fn(ray_t** args, int64_t n); ray_t* ray_scan_right_fn(ray_t** args, int64_t n); diff --git a/src/ops/collection.c b/src/ops/collection.c index b3e29db1..b9ac009d 100644 --- a/src/ops/collection.c +++ b/src/ops/collection.c @@ -4366,6 +4366,54 @@ ray_t* ray_fold_left_fn(ray_t** args, int64_t n) { return ray_fold_fn(args, n); } +/* (fold-while pred f init coll) — a fold that stops when the running + * result says stop. + * + * Before each step the accumulator is offered to `pred`; a falsy answer + * ends the fold and yields the accumulator as it stands. The test comes + * BEFORE the first element, so a predicate that is false at the start + * returns `init` untouched and touches nothing. + * + * Deliberately unlike ray_fold_fn in one respect: that routes its + * collection through unbox_vec_arg -> to_boxed_list, boxing every element + * up front. For a primitive whose whole purpose is to stop early, paying + * for the tail it never reaches is exactly the cost being removed here + * (issue 588), so elements are pulled one at a time via collection_elem — + * the same way map_iterate walks its input. Stopping at element three of + * a million costs three boxed atoms, not a million. */ +ray_t* ray_fold_while_fn(ray_t** args, int64_t n) { + if (n != 4) return ray_error("domain", "fold-while: requires exactly 4 args (pred, fn, init, coll), got %lld", (long long)n); + for (int64_t i = 0; i < n; i++) + if (ray_is_lazy(args[i])) args[i] = ray_lazy_materialize(args[i]); + + ray_t* pred = args[0]; + ray_t* fn = args[1]; + ray_t* coll = args[3]; + if (!is_collection(coll)) + return ray_error("type", "fold-while: coll arg must be a collection, got %s", ray_type_name(coll->type)); + + ray_retain(args[2]); + ray_t* acc = args[2]; + int64_t len = ray_len(coll); + for (int64_t i = 0; i < len; i++) { + ray_t* keep = call_fn1(pred, acc); + if (ray_is_lazy(keep)) keep = ray_lazy_materialize(keep); + if (RAY_IS_ERR(keep)) { ray_release(acc); return keep; } + int go = is_truthy(keep); + ray_release(keep); + if (!go) return acc; + + int alloc = 0; + ray_t* elem = collection_elem(coll, i, &alloc); + ray_t* next = call_fn2(fn, acc, elem); + if (alloc) ray_release(elem); + ray_release(acc); + if (RAY_IS_ERR(next)) return next; + acc = next; + } + return acc; +} + /* (fold-right fn init coll) — right fold */ ray_t* ray_fold_right_fn(ray_t** args, int64_t n) { if (n < 2) return ray_error("domain", "fold-right: requires at least 2 args (fn and vec), got %lld", (long long)n); diff --git a/test/rfl/collection/fold_while.rfl b/test/rfl/collection/fold_while.rfl new file mode 100644 index 00000000..8885c3b3 --- /dev/null +++ b/test/rfl/collection/fold_while.rfl @@ -0,0 +1,75 @@ +;; (fold-while pred f init xs) — a fold that stops when the running result +;; says stop. +;; +;; Before each step the accumulator is offered to `pred`; a falsy answer ends +;; the fold and yields the accumulator as it stands. The predicate is tested +;; BEFORE the first element too, so a predicate that is false at the start +;; returns `init` untouched. +;; +;; Unlike `fold`, this does not box the whole collection up front (`fold` +;; routes through to_boxed_list): a primitive whose purpose is to stop early +;; must not pay for the tail it never reaches. The counting assertions below +;; pin that — they fail if the implementation touches more elements than the +;; predicate allows. + +;; ── basic early termination ────────────────────────────────────────── +;; sum until the running total reaches 100 +(fold-while (fn [acc] (< acc 100)) + 0 (til 1000)) -- 105 + +;; double until past 100 — the sequence is only a step counter here +(fold-while (fn [acc] (< acc 100)) (fn [acc _x] (* acc 2)) 1 (til 1000)) -- 128 + +;; ── the predicate is tested before the first step ──────────────────── +(fold-while (fn [acc] false) + 7 (til 10)) -- 7 + +;; a predicate false at the start touches no element at all +(set seen 0) +(fold-while (fn [acc] false) (fn [acc x] (set seen (+ seen 1)) (+ acc x)) 0 (til 1000)) -- 0 +seen -- 0 + +;; ── it touches exactly as many elements as the predicate allows ────── +;; The step runs while acc < 3, so it runs for acc = 0,1,2 — three times — +;; and never looks at the remaining 997 elements. +(set seen 0) +(set r (fold-while (fn [acc] (< acc 3)) (fn [acc x] (set seen (+ seen 1)) (+ acc 1)) 0 (til 1000))) +r -- 3 +seen -- 3 + +;; ── running to completion ──────────────────────────────────────────── +;; a predicate that never fails folds the whole collection, like fold-left +(fold-while (fn [acc] true) + 0 (til 10)) -- 45 +(fold-left + 0 (til 10)) -- 45 + +;; ── empty and single-element collections ───────────────────────────── +(fold-while (fn [acc] true) + 42 (til 0)) -- 42 +(fold-while (fn [acc] false) + 42 (til 0)) -- 42 +(fold-while (fn [acc] true) + 0 (til 1)) -- 0 + +;; ── collection kinds ───────────────────────────────────────────────── +;; a boxed list, not just a typed vector +(fold-while (fn [acc] (< acc 6)) + 0 (list 1 2 3 4 5)) -- 6 + +;; a typed vector of a narrower width +(fold-while (fn [acc] (< acc 6)) + 0 [1 2 3 4 5]) -- 6 + +;; ── arity and type ─────────────────────────────────────────────────── +(fold-while) !- domain +(fold-while (fn [acc] true) + 0) !- domain +(fold-while (fn [acc] true) + 0 (til 5) 9) !- domain +(fold-while (fn [acc] true) + 0 42) !- type + +;; ── errors propagate from both callbacks ───────────────────────────── +;; from the predicate, before any step runs +(set seen 0) +(== (try (fold-while (fn [acc] (raise 'bad-pred)) (fn [acc x] (set seen (+ seen 1)) acc) 0 (til 5)) (fn [e] e)) 'bad-pred) -- true +seen -- 0 + +;; from the step function, on the first element +(== (try (fold-while (fn [acc] true) (fn [acc x] (raise 'bad-step)) 0 (til 5)) (fn [e] e)) 'bad-step) -- true + +;; ── compiled path ──────────────────────────────────────────────────── +;; the whole form inside a lambda, with the bound as a param +((fn [cap] (fold-while (fn [acc] (< acc cap)) + 0 (til 1000))) 100) -- 105 + +;; a builtin as the predicate rather than a lambda +(fold-while nil? (fn [acc _x] 1) null (til 10)) -- 1 diff --git a/test/rfl/lang/times.rfl b/test/rfl/lang/times.rfl new file mode 100644 index 00000000..05fd9c00 --- /dev/null +++ b/test/rfl/lang/times.rfl @@ -0,0 +1,124 @@ +;; (times n body...) — bounded loop: run the body exactly n times. +;; +;; The count is evaluated ONCE, up front, so a body that mutates whatever +;; produced it cannot change how many passes remain. Returns null, pushes +;; no scope, and is asserted on both evaluators — the tree walker +;; (ray_times_fn) and the bytecode compiler, whose sf_times case desugars +;; to a counted loop over a hidden local slot. + +;; ── basic counting ─────────────────────────────────────────────────── +(set n 0) +(times 5 (set n (+ n 1))) +n -- 5 + +;; a statement form: always null, whatever the body evaluated to +(nil? (times 3 'discarded)) -- true + +;; several body expressions, run in order, every pass +(set a 0) +(set b 0) +(times 4 (set a (+ a 1)) (set b (+ b a))) +a -- 4 +b -- 10 + +;; ── zero and negative counts ───────────────────────────────────────── +(set z 99) +(nil? (times 0 (set z 1))) -- true +z -- 99 + +;; a negative count runs zero times rather than trapping +(set z 99) +(nil? (times -3 (set z 1))) -- true +z -- 99 + +;; body-less form is legal and does nothing observable +(nil? (times 5)) -- true + +;; ── the count is evaluated once, up front ──────────────────────────── +;; If the count were re-read each pass this would not terminate at 3. +(set k 3) +(set passes 0) +(times k (set k (+ k 10)) (set passes (+ passes 1))) +passes -- 3 +k -- 33 + +;; the count is an arbitrary expression, evaluated once +(set calls 0) +(set howmany (fn [] (set calls (+ calls 1)) 4)) +(set p 0) +(times (howmany) (set p (+ p 1))) +p -- 4 +calls -- 1 + +;; ── arity and type ─────────────────────────────────────────────────── +(times) !- domain +(times "3" (set n 1)) !- type +(times 1.5 (set n 1)) !- type + +;; ── compiled path ──────────────────────────────────────────────────── +;; count from a lambda param, accumulator in a let-local +((fn [m] (let acc 0) (times m (let acc (+ acc 2))) acc) 5) -- 10 + +;; the count is validated and clamped on the COMPILED path too, where it +;; runs through ray_times_norm_fn rather than the tree walker's check +((fn [] (times "3" 1))) !- type +((fn [] (times 1.5 1))) !- type +((fn [n] (let z 99) (times n (let z 1)) z) -3) -- 99 +((fn [n] (let z 99) (times n (let z 1)) z) 0) -- 99 + +;; the hidden loop counter must not collide with a user local of any name +((fn [] (let n 0) (let c 100) (times 3 (let n (+ n c))) n)) -- 300 + +;; tree-walk path, same arithmetic +(set acc 0) +(times 5 (set acc (+ acc 2))) +acc -- 10 + +;; ── `return` from inside a body exits the enclosing lambda ─────────── +;; Body compiles inline, so return unwinds the lambda rather than +;; degrading to the tree walker's identity return. +(set hits 0) +((fn [] (let i 0) (times 10 (set hits (+ hits 1)) (if (> i 2) (return 42)) (let i (+ i 1))) 7)) -- 42 +hits -- 4 + +;; without the early return the same loop runs all ten passes +(set hits 0) +((fn [] (times 10 (set hits (+ hits 1))) 7)) -- 7 +hits -- 10 + +;; ── errors ─────────────────────────────────────────────────────────── +;; an error in the body propagates and stops the loop on the first pass +(set c 0) +(== (try (times 10 (set c (+ c 1)) (raise 'boom)) (fn [e] e)) 'boom) -- true +c -- 1 + +;; an error in the count propagates before the body ever runs +(set c 0) +(== (try (times (raise 'bad-count) (set c (+ c 1))) (fn [e] e)) 'bad-count) -- true +c -- 0 + +;; ── nesting ────────────────────────────────────────────────────────── +;; nested counters must not clobber one another +(set total 0) +(times 3 (times 2 (set total (+ total 1)))) +total -- 6 + +;; nested, compiled, with let-locals on both levels +((fn [] (let t 0) (times 3 (times 2 (let t (+ t 1)))) t)) -- 6 + +;; nested three deep +(set total 0) +(times 2 (times 3 (times 4 (set total (+ total 1))))) +total -- 24 + +;; ── composes with while ────────────────────────────────────────────── +(set out 0) +(set i 0) +(while (< i 3) (times 2 (set out (+ out 1))) (set i (+ i 1))) +out -- 6 + +;; ── per-pass isolation via do, as with while ───────────────────────── +(set out (list)) +(set i 0) +(times 3 (do (let sq (* i i)) (set out (concat out sq))) (set i (+ i 1))) +out -- (list 0 1 4) From 62e44476b0f585352b7d71c96a04a9485c90b5c3 Mon Sep 17 00:00:00 2001 From: Anton Date: Sun, 20 Sep 2026 11:37:36 +0200 Subject: [PATCH 04/51] perf(join): skip per-cell key null tests on provably null-free columns MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ray_vec_is_null is out-of-line (no LTO) and the join called it once per key column per row in hash_row_keys — on the build side, the probe side, and the prefetch lookahead — plus twice per key column per hash-chain step in join_keys_eq, across both the count and the fill pass. A reported profile put 13.28% of a service's samples there, on a join keyed by two SYM columns that structurally never hold a null. Prove once per join that no key column can hold a null and drop the call. SYM/STR nulls are canonical empty payloads (id 0 / length 0) that HAS_NULLS does not track, so text columns are proven by the chunked zero-scan from #533 rather than by the flag; everything else reads the flag through slices and takes a set bit at face value, so the proof stays O(n) and never degrades into ray_vec_has_nulls' per-element walk. Flag-readable columns are settled first, so a nullable numeric key short-circuits before any text column is scanned. OP_CONST (atom) key slots are refused as unprovable. Also skip the #458 null-run pre-scan under the proof. It gates on ray_vec_may_have_nulls, which is unconditionally true for SYM/STR, so a SYM-keyed join ran a full per-key-per-build-row ray_vec_is_null scan before the join proper on every execution; a null-free key set cannot contain an all-null row, so the scan is dead. Measured with bench/join_nullfree (release, 1:1 book join, 4M probe x 500K build): two SYM keys 275 -> 249 ms (-9.5% median, -7.8% min); single I64 key 106 -> 99 ms (-6.3% median, -7.9% min). A proof that fails on a text key costs a partial scan (~8ms on a 4M-row SYM column) and gains nothing — only nullable SYM/STR key columns pay it. ray_join_force_null_checks forces the null-aware loops so the differential tests and the perf gate can compare both paths in one binary; ray_join_nullfree_keys counts the joins that took the fast path. Closes #597 --- bench/join_nullfree/main.c | 307 +++++++++++++++++++++++++++++++++++++ src/ops/internal.h | 2 + src/ops/join.c | 129 +++++++++++++--- test/test_join_buildside.c | 184 ++++++++++++++++++++++ 4 files changed, 599 insertions(+), 23 deletions(-) create mode 100644 bench/join_nullfree/main.c diff --git a/bench/join_nullfree/main.c b/bench/join_nullfree/main.c new file mode 100644 index 00000000..bdb3252f --- /dev/null +++ b/bench/join_nullfree/main.c @@ -0,0 +1,307 @@ +/* Null-free join key perf gate (#597). + * + * The join tests every key cell for null on every row: once per key column + * in hash_row_keys (build and probe, plus the prefetch lookahead) and twice + * per key column per hash-chain step in join_keys_eq, across both the count + * and the fill pass. ray_vec_is_null is out-of-line (no LTO), so each test + * is a call. A reporter profiled 13.28% of a service's samples there, on a + * join keyed by two SYM columns that structurally never hold a null. + * + * join_keys_nullfree proves once per join that no key column can hold a + * null and the loops drop the call. This measures what that is worth. + * + * Cases (all two-key SYM joins, mirroring the reported venue+instrument + * book shape): + * SYM2 right=500K (unique venue+instrument pairs), left=4M + * drawn from those pairs, both key columns null-free, so + * each left row matches exactly one right row — the shape + * of a book join, not a fan-out. The fast path must fire. + * SYM2-NULL identical, except one left venue cell is the SYM null. + * The fast path must NOT fire; both sides must time alike, + * bounding what the proof scan itself costs. + * I64 right=500K, left=4M, single I64 key, HAS_NULLS clear. + * Fast path fires via the attrs bit rather than a scan. + * + * Mechanism: ray_join_nullfree_keys must advance on SYM2 and I64 and must + * not advance on SYM2-NULL. ray_join_force_null_checks supplies the + * null-aware baseline in the same binary. + * + * Timing: CLOCK_MONOTONIC around ray_execute only. Tables built once + * outside the timed loop; graph rebuilt per rep; sides interleaved per rep + * so drift hits both equally. + */ +#if defined(__APPLE__) +# define _DARWIN_C_SOURCE +#else +# define _POSIX_C_SOURCE 200809L +#endif + +#include +#include "mem/heap.h" +#include "ops/ops.h" +#include "ops/internal.h" +#include "table/sym.h" +#include +#include +#include +#include +#include +#include + +/* ---------- timing ---------- */ +static double now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return (double)ts.tv_sec * 1e3 + (double)ts.tv_nsec * 1e-6; +} + +static int cmp_double(const void* a, const void* b) { + double x = *(const double*)a, y = *(const double*)b; + return (x > y) - (x < y); +} +static double medianN(double arr[], int n) { + double tmp[64]; + memcpy(tmp, arr, (size_t)n * sizeof(double)); + qsort(tmp, (size_t)n, sizeof(double), cmp_double); + return tmp[n / 2]; +} +static double minN(double arr[], int n) { + double m = arr[0]; + for (int i = 1; i < n; i++) if (arr[i] < m) m = arr[i]; + return m; +} + +/* ---------- SYM column over a vocabulary of `vocab` interned symbols ------ + * Cell i takes vocabulary entry (i * stride) % vocab. null_at, when >= 0, + * is written as SYM id 0 — the canonical SYM null, which HAS_NULLS does not + * track, so only a payload scan can see it. */ +static ray_t* make_sym_col(const char* prefix, int64_t n, int64_t vocab, + int64_t stride, int64_t null_at) { + ray_t* col = ray_sym_vec_new(RAY_SYM_W64, n); + if (!col || RAY_IS_ERR(col)) { fprintf(stderr, "make_sym_col: alloc\n"); abort(); } + col->len = n; + + int64_t* ids = (int64_t*)malloc((size_t)vocab * sizeof(int64_t)); + if (!ids) { fprintf(stderr, "make_sym_col: OOM vocab\n"); abort(); } + for (int64_t v = 0; v < vocab; v++) { + char b[32]; + int m = snprintf(b, sizeof(b), "%s%lld", prefix, (long long)v); + ids[v] = ray_sym_intern(b, (size_t)m); + } + for (int64_t i = 0; i < n; i++) + ray_write_sym(ray_data(col), i, (uint64_t)ids[(i * stride) % vocab], + RAY_SYM, col->attrs); + if (null_at >= 0 && null_at < n) + ray_write_sym(ray_data(col), null_at, 0, RAY_SYM, col->attrs); + free(ids); + return col; +} + +static ray_t* make_i64_col(int64_t n, int64_t mod) { + int64_t* v = (int64_t*)malloc((size_t)n * sizeof(int64_t)); + if (!v) { fprintf(stderr, "make_i64_col: OOM\n"); abort(); } + for (int64_t i = 0; i < n; i++) v[i] = i % mod; + ray_t* col = ray_vec_from_raw(RAY_I64, v, n); + free(v); + if (!col || RAY_IS_ERR(col)) { fprintf(stderr, "make_i64_col: from_raw\n"); abort(); } + return col; +} + +static ray_t* table_of(const char* const* names, ray_t* const* cols, int64_t ncols) { + ray_t* tbl = ray_table_new(ncols); + for (int64_t c = 0; c < ncols; c++) + tbl = ray_table_add_col(tbl, ray_sym_intern(names[c], strlen(names[c])), cols[c]); + if (!tbl || RAY_IS_ERR(tbl)) { fprintf(stderr, "table_of: add_col\n"); abort(); } + return tbl; +} + +/* ---------- one inner-join rep ---------- */ +static double run_join_rep(ray_t* lt, const char* const* lkeys, + ray_t* rt, const char* const* rkeys, + uint32_t n_keys, int64_t* rows_out) { + ray_graph_t* g = ray_graph_new(lt); + if (!g) { fprintf(stderr, "run_join_rep: graph alloc\n"); abort(); } + + ray_op_t* lt_node = ray_const_table(g, lt); + ray_op_t* rt_node = ray_const_table(g, rt); + ray_op_t* lk_arr[4]; + ray_op_t* rk_arr[4]; + for (uint32_t k = 0; k < n_keys; k++) { + lk_arr[k] = ray_scan(g, lkeys[k]); + rk_arr[k] = ray_scan(g, rkeys[k]); + if (!lk_arr[k] || !rk_arr[k]) { fprintf(stderr, "run_join_rep: key node\n"); abort(); } + } + if (!lt_node || !rt_node) { fprintf(stderr, "run_join_rep: node alloc\n"); abort(); } + + ray_op_t* jn = ray_join(g, lt_node, lk_arr, rt_node, rk_arr, n_keys, 0); + if (!jn) { fprintf(stderr, "run_join_rep: join node\n"); abort(); } + jn = ray_optimize(g, jn); + + double t0 = now_ms(); + ray_t* result = ray_execute(g, jn); + double t1 = now_ms(); + + if (!result || RAY_IS_ERR(result)) { + fprintf(stderr, "run_join_rep: execute returned error\n"); abort(); + } + if (rows_out) *rows_out = ray_table_nrows(result); + ray_release(result); + ray_graph_free(g); + return t1 - t0; +} + +#define NREPS 11 + +typedef struct { + const char* name; + double fast_ms[NREPS]; /* knob off — null-free proof allowed */ + double base_ms[NREPS]; /* knob on — null-aware loops (pre-#597) */ + int64_t rows_out; +} case_result_t; + +static void run_case(const char* name, + ray_t* lt, const char* const* lkeys, + ray_t* rt, const char* const* rkeys, + uint32_t n_keys, bool expect_fast, + case_result_t* cr) { + cr->name = name; + cr->rows_out = -1; + + printf("Running case %-12s (%d reps)...\n", name, NREPS); + fflush(stdout); + + uint64_t nf_before = ray_join_nullfree_keys; + + for (int rep = 0; rep < NREPS; rep++) { + ray_join_force_null_checks = false; + int64_t rows_f = -1; + cr->fast_ms[rep] = run_join_rep(lt, lkeys, rt, rkeys, n_keys, &rows_f); + + ray_join_force_null_checks = true; + int64_t rows_b = -1; + cr->base_ms[rep] = run_join_rep(lt, lkeys, rt, rkeys, n_keys, &rows_b); + ray_join_force_null_checks = false; + + if (rows_f != rows_b) { + fprintf(stderr, + "CORRECTNESS FAILURE case %s rep %d: fast=%lld rows, baseline=%lld rows\n", + name, rep, (long long)rows_f, (long long)rows_b); + abort(); + } + cr->rows_out = rows_f; + } + + bool fired = ray_join_nullfree_keys > nf_before; + if (expect_fast != fired) { + fprintf(stderr, + "MECHANISM FAILURE case %s: expected null-free path %s " + "(before=%llu after=%llu)\n", + name, expect_fast ? "to fire" : "NOT to fire", + (unsigned long long)nf_before, + (unsigned long long)ray_join_nullfree_keys); + abort(); + } + + printf(" nullfree counter: before=%llu after=%llu fired=%s rows=%lld\n", + (unsigned long long)nf_before, + (unsigned long long)ray_join_nullfree_keys, + fired ? "YES" : "NO", + (long long)cr->rows_out); + fflush(stdout); +} + +static void report(const case_result_t* cr) { + double fmed = medianN((double*)cr->fast_ms, NREPS); + double bmed = medianN((double*)cr->base_ms, NREPS); + double fmin = minN((double*)cr->fast_ms, NREPS); + double bmin = minN((double*)cr->base_ms, NREPS); + printf("%-12s median %8.2f -> %8.2f ms (%+6.1f%%) min %8.2f -> %8.2f ms (%+6.1f%%)\n", + cr->name, bmed, fmed, 100.0 * (fmed - bmed) / bmed, + bmin, fmin, 100.0 * (fmin - bmin) / bmin); +} + +int main(void) { + ray_heap_init(); + (void)ray_sym_init(); + ray_join_force_null_checks = false; + + printf("=== bench-join-nullfree (#597) ===\n"); + printf("NREPS=%d RAY_PARALLEL_THRESHOLD=%d\n\n", + NREPS, (int)RAY_PARALLEL_THRESHOLD); + fflush(stdout); + + const int64_t nl = 4000000L; /* probe side */ + const int64_t nr = 500000L; /* build side */ + + /* ---- SYM2: two null-free SYM key columns ---- */ + printf("Building SYM2 tables (left=%lld, right=%lld)...\n", + (long long)nl, (long long)nr); + fflush(stdout); + { + /* Right: instrument unique per row, venue spread over 64 — 500K + * distinct (venue, instrument) pairs. Left: the same pairs cycled, + * so every left row has exactly one match. */ + ray_t* rv = make_sym_col("venue", nr, nr, 1, -1); + ray_t* ri = make_sym_col("inst", nr, nr, 1, -1); + ray_t* lv = make_sym_col("venue", nl, nr, 1, -1); + ray_t* li = make_sym_col("inst", nl, nr, 1, -1); + + /* SYM2-NULL shares the build side and differs only in one left cell. */ + ray_t* lv_null = make_sym_col("venue", nl, nr, 1, nl / 2); + ray_t* li_null = make_sym_col("inst", nl, nr, 1, -1); + + static const char* const lnames[] = { "lvenue", "linst" }; + static const char* const rnames[] = { "rvenue", "rinst" }; + static const char* const lkeys[] = { "lvenue", "linst" }; + static const char* const rkeys[] = { "rvenue", "rinst" }; + + ray_t* lcols[2] = { lv, li }; + ray_t* rcols[2] = { rv, ri }; + ray_t* lcols_n[2] = { lv_null, li_null }; + ray_t* lt = table_of(lnames, lcols, 2); + ray_t* rt = table_of(rnames, rcols, 2); + ray_t* lt_null = table_of(lnames, lcols_n, 2); + ray_release(lv); ray_release(li); ray_release(rv); ray_release(ri); + ray_release(lv_null); ray_release(li_null); + + case_result_t cr_sym2, cr_sym2n; + run_case("SYM2", lt, lkeys, rt, rkeys, 2, true, &cr_sym2); + run_case("SYM2-NULL", lt_null, lkeys, rt, rkeys, 2, false, &cr_sym2n); + + printf("\n--- results (baseline = forced null-aware loops) ---\n"); + report(&cr_sym2); + report(&cr_sym2n); + + ray_release(lt); ray_release(rt); ray_release(lt_null); + } + + /* ---- I64: single key, proof is the HAS_NULLS bit ---- */ + printf("\nBuilding I64 tables (left=%lld, right=%lld)...\n", + (long long)nl, (long long)nr); + fflush(stdout); + { + ray_t* lc = make_i64_col(nl, nr); + ray_t* rc = make_i64_col(nr, nr); + static const char* const lnames[] = { "lk" }; + static const char* const rnames[] = { "rk" }; + static const char* const lkeys[] = { "lk" }; + static const char* const rkeys[] = { "rk" }; + ray_t* lcols[1] = { lc }; + ray_t* rcols[1] = { rc }; + ray_t* lt = table_of(lnames, lcols, 1); + ray_t* rt = table_of(rnames, rcols, 1); + ray_release(lc); ray_release(rc); + + case_result_t cr_i64; + run_case("I64", lt, lkeys, rt, rkeys, 1, true, &cr_i64); + printf("\n--- results (baseline = forced null-aware loops) ---\n"); + report(&cr_i64); + + ray_release(lt); ray_release(rt); + } + + printf("\n(negative %% = faster with the null-free path)\n"); + ray_sym_destroy(); + ray_heap_destroy(); + return 0; +} diff --git a/src/ops/internal.h b/src/ops/internal.h index 4ff40059..2bb8dd9e 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -804,6 +804,8 @@ extern bool ray_join_force_dup_fallback; extern bool ray_join_no_dup_fallback; extern uint64_t ray_join_dup_fallbacks; extern uint64_t ray_join_null_fallbacks; +extern bool ray_join_force_null_checks; +extern uint64_t ray_join_nullfree_keys; extern bool ray_agg_engine_v2; /* route OP_GROUP through v2 agg engine; default ON (agg_engine.c) */ void ray_expr_stats_init(void); diff --git a/src/ops/join.c b/src/ops/join.c index bc8e744c..7a53f57f 100644 --- a/src/ops/join.c +++ b/src/ops/join.c @@ -44,6 +44,67 @@ uint64_t ray_join_dup_fallbacks = 0; /* Diagnostic: radix joins routed to the chained path upfront because the * build side carries more all-null key rows than RADIX_DUP_RUN_MAX (#458). */ uint64_t ray_join_null_fallbacks = 0; +/* Test knob: suppress the null-free key fast path (#597) so the differential + * harness can compare it against the null-aware loops in one binary. */ +bool ray_join_force_null_checks = false; +/* Diagnostic: joins whose key columns were all proven null-free. */ +uint64_t ray_join_nullfree_keys = 0; + +/* ── #597: null-free key proof ─────────────────────────────────────────── + * ray_vec_is_null is an out-of-line call (no LTO), and the join makes one + * per key column per row in hash_row_keys plus two per key column per + * hash-chain step in join_keys_eq — on both the count and the fill pass. + * A profiled service spent 13% of its samples there. Prove ONCE per join + * that no key column can hold a null and the loops drop the call entirely. + * + * SYM/STR nulls are canonical empty payloads (id 0 / length 0) that + * HAS_NULLS does not track, so text columns are proven by the chunked, + * vectorized zero-scan from #533 rather than by the flag — which is why + * the flag alone would not have helped a SYM-keyed join. Everything else + * is the flag, read through slices by ray_vec_may_have_nulls; a set flag + * is taken at face value (a column that merely may hold a null keeps the + * null-aware path) so the proof stays O(n) and never degrades into the + * per-element walk ray_vec_has_nulls would do. + * + * An absent key column is not a null cell: hash_row_keys skips it and + * join_keys_eq rejects the pair before either reaches the null test, so + * it cannot affect the proof. */ +static bool join_key_col_nullfree(const ray_t* v) { + if (!v) return true; + /* A key slot may hold an OP_CONST literal, i.e. an atom, whose null + * state is RAY_ATOM_IS_NULL and not the vector attrs the proof reads. + * Nothing in the current tree builds such a key, so this is a guard + * rather than a fix: refuse to prove what this function cannot see. */ + if (ray_is_atom(v)) return false; + if (v->type == RAY_SYM || v->type == RAY_STR) return !ray_vec_text_has_nulls(v); + return !ray_vec_may_have_nulls(v); +} + +/* Both sides must be clean: join_keys_eq tests the left and the right cell. + * + * Flag-readable columns are settled first, in one O(1) pass, so a nullable + * numeric key short-circuits the whole proof before any text column is + * scanned. A failed proof on a text column still costs a partial scan + * (~8ms on a 4M-row SYM column, measured) with nothing to show for it — + * that is the price of a payload-encoded null, and only nullable SYM/STR + * keys pay it. */ +static bool join_key_col_scanned(const ray_t* v) { + return v && !ray_is_atom(v) && (v->type == RAY_SYM || v->type == RAY_STR); +} + +static bool join_keys_nullfree(ray_t* const* l_vecs, ray_t* const* r_vecs, + uint32_t n_keys) { + if (ray_join_force_null_checks) return false; + for (uint32_t k = 0; k < n_keys; k++) { + if (!join_key_col_scanned(l_vecs[k]) && !join_key_col_nullfree(l_vecs[k])) return false; + if (!join_key_col_scanned(r_vecs[k]) && !join_key_col_nullfree(r_vecs[k])) return false; + } + for (uint32_t k = 0; k < n_keys; k++) { + if (join_key_col_scanned(l_vecs[k]) && !join_key_col_nullfree(l_vecs[k])) return false; + if (join_key_col_scanned(r_vecs[k]) && !join_key_col_nullfree(r_vecs[k])) return false; + } + return true; +} static int join_store_key_cell(ray_t* dst, int64_t dst_row, ray_t* src, int64_t src_row) { @@ -112,13 +173,14 @@ static inline bool join_str_eq_hashed(const ray_str_t* a, const char* pool_a, * real key's hash is resolved by join_keys_eq, as for any other hash. */ #define JOIN_NULL_KEY_HASH INT64_C(0x5B7A6D3F2E1C0A94) -static uint64_t hash_row_keys(ray_t** key_vecs, uint32_t n_keys, int64_t row) { +static uint64_t hash_row_keys(ray_t** key_vecs, uint32_t n_keys, int64_t row, + bool nullfree) { uint64_t h = 0; for (uint32_t k = 0; k < n_keys; k++) { ray_t* col = key_vecs[k]; if (!col) continue; uint64_t kh; - if (ray_vec_is_null(col, row)) { + if (!nullfree && ray_vec_is_null(col, row)) { /* null == null: hash every null cell alike instead of giving the * row a private hash. A per-row hash made a null key unmatchable * even against itself, so `anti-join [c] X X` came back non-empty @@ -224,6 +286,7 @@ typedef struct { uint32_t* hashes; /* output: hash[row] */ const ray_str_t* str_desc; /* one-key STR specialization, else NULL */ const char* str_pool; + bool nullfree; /* #597: no key column can hold a null */ } join_radix_hash_ctx_t; static void join_radix_hash_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { @@ -241,14 +304,14 @@ static void join_radix_hash_fn(void* raw, uint32_t wid, int64_t start, int64_t e return; } for (int64_t r = start; r < end; r++) - c->hashes[r] = (uint32_t)hash_row_keys(c->key_vecs, c->n_keys, r); + c->hashes[r] = (uint32_t)hash_row_keys(c->key_vecs, c->n_keys, r, c->nullfree); } static join_radix_hash_ctx_t join_radix_hash_ctx(ray_t** keys, uint32_t n_keys, - uint32_t* hashes) { + uint32_t* hashes, bool nullfree) { join_radix_hash_ctx_t c = { .key_vecs = keys, .n_keys = n_keys, .hashes = hashes, - .str_desc = NULL, .str_pool = NULL, + .str_desc = NULL, .str_pool = NULL, .nullfree = nullfree, }; if (n_keys == 1 && keys[0] && keys[0]->type == RAY_STR) { ray_t* col = keys[0]; @@ -584,7 +647,7 @@ static ray_t* join_gather_col_serial(ray_t* src, const int64_t* idx, /* Key equality helper — shared by count + fill phases */ static inline bool join_keys_eq(ray_t* const* l_vecs, ray_t* const* r_vecs, uint32_t n_keys, - int64_t l, int64_t r) { + int64_t l, int64_t r, bool nullfree) { for (uint32_t k = 0; k < n_keys; k++) { ray_t* lc = l_vecs[k]; ray_t* rc = r_vecs[k]; @@ -598,8 +661,8 @@ static inline bool join_keys_eq(ray_t* const* l_vecs, ray_t* const* r_vecs, uint * whenever c held a null and made left-join miss a null-keyed match. * As-of join keeps its own documented NULLs-never-match rule; it * does not route through here. */ - bool l_null = ray_vec_is_null(lc, l); - bool r_null = ray_vec_is_null(rc, r); + bool l_null = !nullfree && ray_vec_is_null(lc, l); + bool r_null = !nullfree && ray_vec_is_null(rc, r); if (l_null || r_null) { if (l_null != r_null) return false; /* No looser than the value arms below: they pair STR only with @@ -667,6 +730,7 @@ typedef struct { const char* l_str_pool; const char* r_str_pool; uint8_t join_type; + bool nullfree; /* #597: no key column can hold a null */ /* Per-partition output: pp_l[p], pp_r[p] are local buffers */ int32_t** pp_l; /* per-partition left indices (int32_t) */ int32_t** pp_r; /* per-partition right indices (int32_t) */ @@ -858,7 +922,7 @@ static void join_radix_build_probe_fn(void* raw, uint32_t wid, int64_t task_star ? join_str_eq_hashed(&c->l_str_desc[lr], c->l_str_pool, &c->r_str_desc[rr], c->r_str_pool) : join_keys_eq(c->l_key_vecs, c->r_key_vecs, c->n_keys, - (int64_t)lr, (int64_t)rr); + (int64_t)lr, (int64_t)rr, c->nullfree); if (keys_equal) { if (!bp_grow_bufs(c, p, &pl, &pr, &cap, cnt)) goto done; @@ -908,6 +972,7 @@ typedef struct { /* ASP-Join: semijoin filter from factorized left side (NULL if N/A) */ uint64_t* asp_bits; int64_t asp_key_max; + bool nullfree; /* #597: no key column can hold a null */ } join_build_ctx_t; static void join_build_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { @@ -930,10 +995,10 @@ static void join_build_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { continue; } if (r + 8 < end) { - uint64_t pf_h = hash_row_keys(c->r_key_vecs, c->n_keys, r + 8); + uint64_t pf_h = hash_row_keys(c->r_key_vecs, c->n_keys, r + 8, c->nullfree); __builtin_prefetch(&heads[(uint32_t)(pf_h & mask)], 1, 1); } - uint64_t h = hash_row_keys(c->r_key_vecs, c->n_keys, r); + uint64_t h = hash_row_keys(c->r_key_vecs, c->n_keys, r, c->nullfree); uint32_t slot = (uint32_t)(h & mask); uint32_t row32 = (uint32_t)r; uint32_t old = atomic_load_explicit(&heads[slot], memory_order_relaxed); @@ -966,6 +1031,7 @@ typedef struct { /* S-Join: semijoin filter bitmap (NULL if not applicable) */ uint64_t* sjoin_bits; int64_t sjoin_key_max; + bool nullfree; /* #597: no key column can hold a null */ } join_probe_ctx_t; /* Pass 2a: count matches per morsel */ @@ -993,14 +1059,14 @@ static void join_count_fn(void* raw, uint32_t wid, int64_t task_start, int64_t t } if (l + 8 < row_end) { - uint64_t pf_h = hash_row_keys(c->l_key_vecs, c->n_keys, l + 8); + uint64_t pf_h = hash_row_keys(c->l_key_vecs, c->n_keys, l + 8, c->nullfree); __builtin_prefetch(&c->ht_heads[(uint32_t)(pf_h & ht_mask)], 0, 1); } - uint64_t h = hash_row_keys(c->l_key_vecs, c->n_keys, l); + uint64_t h = hash_row_keys(c->l_key_vecs, c->n_keys, l, c->nullfree); uint32_t slot = (uint32_t)(h & ht_mask); bool matched = false; for (uint32_t r = c->ht_heads[slot]; r != JHT_EMPTY; r = c->ht_next[r]) { - if (join_keys_eq(c->l_key_vecs, c->r_key_vecs, c->n_keys, l, (int64_t)r)) { + if (join_keys_eq(c->l_key_vecs, c->r_key_vecs, c->n_keys, l, (int64_t)r, c->nullfree)) { count++; matched = true; } @@ -1042,14 +1108,14 @@ static void join_fill_fn(void* raw, uint32_t wid, int64_t task_start, int64_t ta } if (l + 8 < row_end) { - uint64_t pf_h = hash_row_keys(c->l_key_vecs, c->n_keys, l + 8); + uint64_t pf_h = hash_row_keys(c->l_key_vecs, c->n_keys, l + 8, c->nullfree); __builtin_prefetch(&c->ht_heads[(uint32_t)(pf_h & ht_mask)], 0, 1); } - uint64_t h = hash_row_keys(c->l_key_vecs, c->n_keys, l); + uint64_t h = hash_row_keys(c->l_key_vecs, c->n_keys, l, c->nullfree); uint32_t slot = (uint32_t)(h & ht_mask); bool matched = false; for (uint32_t r = c->ht_heads[slot]; r != JHT_EMPTY; r = c->ht_next[r]) { - if (join_keys_eq(c->l_key_vecs, c->r_key_vecs, c->n_keys, l, (int64_t)r)) { + if (join_keys_eq(c->l_key_vecs, c->r_key_vecs, c->n_keys, l, (int64_t)r, c->nullfree)) { li[off] = l; ri[off] = (int64_t)r; off++; @@ -1156,6 +1222,10 @@ static ray_t* exec_join_flat(ray_graph_t* g, ray_op_t* op, ray_t* left_table, ra return ray_error("oom", "join: sym domain runtime-id LUT build failed"); } + /* #597: one proof for the whole join — see join_keys_nullfree. */ + bool keys_nullfree = join_keys_nullfree(l_key_vecs, r_key_vecs, n_keys); + if (keys_nullfree) ray_join_nullfree_keys++; + ray_pool_t* pool = ray_pool_get(); /* Shared output state — used by both radix and chained HT paths */ @@ -1203,8 +1273,14 @@ static ray_t* exec_join_flat(ray_graph_t* g, ray_op_t* op, ray_t* left_table, ra * hash_row_keys (so it counts exactly what would form the run), and * stops at the threshold. */ { + /* #597: ray_vec_may_have_nulls is unconditionally true for + * SYM/STR (their nulls are in the payload, not in attrs), so a + * SYM-keyed join always reached the row scan below — a second + * out-of-line ray_vec_is_null per key per build row, before the + * join proper, every execution. A proven null-free key set + * cannot contain an all-null row, so the whole scan is dead. */ bool any_nullable = false; - for (uint32_t k = 0; k < n_keys && !any_nullable; k++) + for (uint32_t k = 0; k < n_keys && !any_nullable && !keys_nullfree; k++) if (build_keys[k] && ray_vec_may_have_nulls(build_keys[k])) any_nullable = true; if (any_nullable) { @@ -1236,8 +1312,8 @@ static ray_t* exec_join_flat(ray_graph_t* g, ray_op_t* op, ray_t* left_table, ra if (l_hash_hdr) scratch_free(l_hash_hdr); goto chained_ht_fallback; } - join_radix_hash_ctx_t rhctx = join_radix_hash_ctx(build_keys, n_keys, r_hashes); - join_radix_hash_ctx_t lhctx = join_radix_hash_ctx(probe_keys, n_keys, l_hashes); + join_radix_hash_ctx_t rhctx = join_radix_hash_ctx(build_keys, n_keys, r_hashes, keys_nullfree); + join_radix_hash_ctx_t lhctx = join_radix_hash_ctx(probe_keys, n_keys, l_hashes, keys_nullfree); if (pool) { ray_pool_dispatch(pool, join_radix_hash_fn, &rhctx, build_rows); ray_pool_dispatch(pool, join_radix_hash_fn, &lhctx, probe_rows); @@ -1331,7 +1407,7 @@ static ray_t* exec_join_flat(ray_graph_t* g, ray_op_t* op, ray_t* left_table, ra join_radix_bp_ctx_t bp_ctx = { .l_parts = l_parts, .r_parts = r_parts, .l_key_vecs = probe_keys, .r_key_vecs = build_keys, - .n_keys = n_keys, .join_type = join_type, + .n_keys = n_keys, .join_type = join_type, .nullfree = keys_nullfree, .l_str_desc = lhctx.str_desc, .r_str_desc = rhctx.str_desc, .l_str_pool = lhctx.str_pool, .r_str_pool = rhctx.str_pool, .pp_l = pp_l, .pp_r = pp_r, @@ -1508,6 +1584,7 @@ chained_ht_fallback:; .n_keys = n_keys, .asp_bits = asp_bits, .asp_key_max = asp_key_max, + .nullfree = keys_nullfree, }; if (pool && right_rows > RAY_PARALLEL_THRESHOLD) ray_pool_dispatch(pool, join_build_fn, &bctx, right_rows); @@ -1585,6 +1662,7 @@ chained_ht_fallback:; .matched_right = matched_right, .sjoin_bits = sjoin_bits, .sjoin_key_max = sjoin_key_max, + .nullfree = keys_nullfree, }; /* 2a: Count matches per morsel */ @@ -2100,6 +2178,10 @@ static ray_t* exec_antijoin_flat(ray_graph_t* g, ray_op_t* op, return ray_error("oom", "join: sym domain runtime-id LUT build failed"); } + /* #597: one proof for the whole anti-join — see join_keys_nullfree. */ + bool keys_nullfree = join_keys_nullfree(l_key_vecs, r_key_vecs, n_keys); + if (keys_nullfree) ray_join_nullfree_keys++; + /* Build chained hash table from right side */ ray_t* ht_next_hdr = NULL; ray_t* ht_heads_hdr = NULL; @@ -2133,6 +2215,7 @@ static ray_t* exec_antijoin_flat(ray_graph_t* g, ray_op_t* op, .n_keys = n_keys, .asp_bits = NULL, .asp_key_max = 0, + .nullfree = keys_nullfree, }; if (pool && right_rows > RAY_PARALLEL_THRESHOLD) ray_pool_dispatch(pool, join_build_fn, &bctx, right_rows); @@ -2171,11 +2254,11 @@ static ray_t* exec_antijoin_flat(ray_graph_t* g, ray_op_t* op, scratch_free(key_vecs_hdr); return ray_error("cancel", NULL); } - uint64_t h = hash_row_keys(l_key_vecs, n_keys, l); + uint64_t h = hash_row_keys(l_key_vecs, n_keys, l, keys_nullfree); uint32_t slot = (uint32_t)(h & ht_mask); bool matched = false; for (uint32_t r = ht_heads[slot]; r != JHT_EMPTY; r = ht_next[r]) { - if (join_keys_eq(l_key_vecs, r_key_vecs, n_keys, l, (int64_t)r)) { + if (join_keys_eq(l_key_vecs, r_key_vecs, n_keys, l, (int64_t)r, keys_nullfree)) { matched = true; break; /* anti-join: one match is enough to exclude */ } diff --git a/test/test_join_buildside.c b/test/test_join_buildside.c index 30ebccaf..a39b9415 100644 --- a/test/test_join_buildside.c +++ b/test/test_join_buildside.c @@ -1071,6 +1071,186 @@ static test_result_t test_jb_mixed_type_radix(void) { PASS(); } + +/* ── #597: null-free key-column fast path ───────────────────────────────── + * A join proves once, per key column, whether the column can hold a null + * (payload scan for SYM/STR, the HAS_NULLS bit otherwise) and skips the + * per-cell null test in hash_row_keys / join_keys_eq when no key column can. + * + * ray_join_nullfree_keys counts the joins that took that path; + * ray_join_force_null_checks suppresses it, giving the differential tests a + * null-aware oracle inside one binary (same shape as the build-swap knob). + * ──────────────────────────────────────────────────────────────────────── */ + +/* SYM table; a "" entry is the canonical SYM null (id 0). */ +static ray_t* jb_sym_table1(const char* name, const char* const* vals, int64_t n) { + ray_t* col = ray_sym_vec_new(RAY_SYM_W64, n); + if (!col || RAY_IS_ERR(col)) return col; + col->len = n; + for (int64_t i = 0; i < n; i++) { + int64_t id = vals[i][0] ? ray_sym_intern(vals[i], strlen(vals[i])) : 0; + ray_write_sym(ray_data(col), i, (uint64_t)id, RAY_SYM, col->attrs); + } + ray_t* tbl = ray_table_new(1); + int64_t sym = ray_sym_intern(name, strlen(name)); + tbl = ray_table_add_col(tbl, sym, col); + ray_release(col); + return tbl; +} + +/* I64 table with an optional null at `null_at` (-1 for none). */ +static ray_t* jb_table1_null(const char* name, const int64_t* vals, int64_t n, + int64_t null_at) { + ray_t* col = ray_vec_from_raw(RAY_I64, vals, n); + if (!col || RAY_IS_ERR(col)) return col; + if (null_at >= 0) ray_vec_set_null(col, null_at, true); + ray_t* tbl = ray_table_new(1); + int64_t sym = ray_sym_intern(name, strlen(name)); + tbl = ray_table_add_col(tbl, sym, col); + ray_release(col); + return tbl; +} + +/* SYM key columns with no null cell take the null-free path. */ +static test_result_t test_jb_nf_sym_keys_prove(void) { + ray_heap_init(); + (void)ray_sym_init(); + + static const char* const lv[] = { "NYSE", "LSE", "NYSE", "TSE" }; + static const char* const rv[] = { "LSE", "NYSE", "XETR" }; + ray_t* lt = jb_sym_table1("venue", lv, 4); + ray_t* rt = jb_sym_table1("rvenue", rv, 3); + TEST_ASSERT(lt && !RAY_IS_ERR(lt), "left SYM table"); + TEST_ASSERT(rt && !RAY_IS_ERR(rt), "right SYM table"); + + uint64_t before = ray_join_nullfree_keys; + ray_t* got = jb_inner_join(lt, "venue", rt, "rvenue"); + TEST_ASSERT(got && !RAY_IS_ERR(got), "join execution"); + TEST_ASSERT(ray_join_nullfree_keys > before, + "null-free SYM key columns must take the null-free path"); + TEST_ASSERT_EQ_I(ray_table_nrows(got), 3); + + ray_release(got); + ray_release(lt); + ray_release(rt); + ray_sym_destroy(); + ray_heap_destroy(); + PASS(); +} + +/* One null SYM cell on the left blocks the fast path. SYM nulls are id 0 + * in the payload and are NOT reflected in HAS_NULLS, so this case is the + * reason the proof scans the payload instead of reading the flag. */ +static test_result_t test_jb_nf_sym_null_blocks(void) { + ray_heap_init(); + (void)ray_sym_init(); + + static const char* const lv[] = { "NYSE", "", "NYSE", "TSE" }; + static const char* const rv[] = { "LSE", "NYSE", "XETR" }; + ray_t* lt = jb_sym_table1("venue", lv, 4); + ray_t* rt = jb_sym_table1("rvenue", rv, 3); + TEST_ASSERT(lt && !RAY_IS_ERR(lt), "left SYM table"); + TEST_ASSERT(rt && !RAY_IS_ERR(rt), "right SYM table"); + TEST_ASSERT_FALSE(ray_table_get_col_idx(lt, 0)->attrs & RAY_ATTR_HAS_NULLS); + + uint64_t before = ray_join_nullfree_keys; + ray_t* got = jb_inner_join(lt, "venue", rt, "rvenue"); + TEST_ASSERT(got && !RAY_IS_ERR(got), "join execution"); + TEST_ASSERT(ray_join_nullfree_keys == before, + "a null SYM key cell must block the null-free path"); + + ray_release(got); + ray_release(lt); + ray_release(rt); + ray_sym_destroy(); + ray_heap_destroy(); + PASS(); +} + +/* I64 keys: the HAS_NULLS bit decides. */ +static test_result_t test_jb_nf_i64_flag_decides(void) { + ray_heap_init(); + (void)ray_sym_init(); + + static const int64_t lv[] = { 1, 2, 3, 4 }; + static const int64_t rv[] = { 2, 3, 9 }; + + ray_t* lt_clean = jb_table1_null("lk", lv, 4, -1); + ray_t* rt = jb_table1_null("rk", rv, 3, -1); + uint64_t before = ray_join_nullfree_keys; + ray_t* got = jb_inner_join(lt_clean, "lk", rt, "rk"); + TEST_ASSERT(got && !RAY_IS_ERR(got), "clean join execution"); + TEST_ASSERT(ray_join_nullfree_keys > before, + "HAS_NULLS-clear I64 keys must take the null-free path"); + ray_release(got); + ray_release(lt_clean); + + ray_t* lt_null = jb_table1_null("lk", lv, 4, 1); + TEST_ASSERT_TRUE(ray_table_get_col_idx(lt_null, 0)->attrs & RAY_ATTR_HAS_NULLS); + before = ray_join_nullfree_keys; + got = jb_inner_join(lt_null, "lk", rt, "rk"); + TEST_ASSERT(got && !RAY_IS_ERR(got), "nullable join execution"); + TEST_ASSERT(ray_join_nullfree_keys == before, + "a HAS_NULLS I64 key column must block the null-free path"); + + ray_release(got); + ray_release(lt_null); + ray_release(rt); + ray_sym_destroy(); + ray_heap_destroy(); + PASS(); +} + +/* The null-free path must not change any answer: run every join type, at + * both the chained and the radix size, with and without null keys, against + * the forced null-aware oracle. */ +static test_result_t test_jb_nf_differential(void) { + ray_heap_init(); + (void)ray_sym_init(); + + const int64_t sizes[] = { 64, RAY_PARALLEL_THRESHOLD + 5000 }; + const uint8_t types[] = { 0, 1, 2 }; + test_result_t result = (test_result_t){ TEST_PASS, NULL }; + + for (size_t s = 0; s < sizeof sizes / sizeof *sizes && result.status == TEST_PASS; s++) { + int64_t n = sizes[s]; + int64_t* lv = malloc((size_t)n * sizeof(*lv)); + int64_t* rv = malloc((size_t)n * sizeof(*rv)); + if (!lv || !rv) { free(lv); free(rv); ray_sym_destroy(); ray_heap_destroy(); + return (test_result_t){ TEST_FAIL, "malloc" }; } + for (int64_t i = 0; i < n; i++) { lv[i] = i % (n / 4 + 1); rv[i] = i % (n / 3 + 1); } + + for (int nulls = 0; nulls < 2 && result.status == TEST_PASS; nulls++) { + ray_t* lt = jb_table1_null("lk", lv, n, nulls ? n / 2 : -1); + ray_t* rt = jb_table1_null("rk", rv, n, nulls ? n / 3 : -1); + + for (size_t t = 0; t < sizeof types / sizeof *types && result.status == TEST_PASS; t++) { + ray_join_force_null_checks = true; + ray_t* oracle = jb_join(lt, "lk", rt, "rk", types[t]); + ray_join_force_null_checks = false; + ray_t* fast = jb_join(lt, "lk", rt, "rk", types[t]); + + if (!oracle || RAY_IS_ERR(oracle) || !fast || RAY_IS_ERR(fast)) { + result = (test_result_t){ TEST_FAIL, "differential join execution" }; + } else { + result = jb_results_equal(fast, oracle); + } + ray_release(oracle); + ray_release(fast); + } + ray_release(lt); + ray_release(rt); + } + free(lv); + free(rv); + } + + ray_join_force_null_checks = false; + ray_sym_destroy(); + ray_heap_destroy(); + return result; +} + /* ── Entry table ─────────────────────────────────────────────────────────── */ const test_entry_t join_buildside_entries[] = { @@ -1093,5 +1273,9 @@ const test_entry_t join_buildside_entries[] = { { "join_buildside/not_sticky", test_jb_not_sticky, NULL, NULL }, { "join_buildside/mixed_type_radix", test_jb_mixed_type_radix, NULL, NULL }, { "join_buildside/null_run_upfront_fallback", test_jb_null_run_upfront_fallback, NULL, NULL }, + { "join_buildside/nf_sym_keys_prove", test_jb_nf_sym_keys_prove, NULL, NULL }, + { "join_buildside/nf_sym_null_blocks", test_jb_nf_sym_null_blocks, NULL, NULL }, + { "join_buildside/nf_i64_flag_decides", test_jb_nf_i64_flag_decides, NULL, NULL }, + { "join_buildside/nf_differential", test_jb_nf_differential, NULL, NULL }, { NULL, NULL, NULL, NULL }, }; From 06449e6f6e0b5afc0c3523d6d1fdd2230cbd6f82 Mon Sep 17 00:00:00 2001 From: belowzeroff Date: Fri, 18 Sep 2026 12:10:50 -0400 Subject: [PATCH 05/51] fix(system): validate launcher and timeit integer arguments --- src/app/main.c | 9 ++++- src/lang/syscmd.c | 19 +++++++++-- src/ops/system.c | 53 +++++++++++++++++++++++++---- test/rfl/system/syscmd_coverage.rfl | 10 +++--- test/test_runtime.c | 17 ++++++++- 5 files changed, 92 insertions(+), 16 deletions(-) diff --git a/src/app/main.c b/src/app/main.c index f6ff95f3..33277314 100644 --- a/src/app/main.c +++ b/src/app/main.c @@ -214,7 +214,14 @@ int main(int argc, char** argv) { } /* Expose the full command line to Rayfall via (.sys.args). */ - ray_runtime_set_sys_args(ray_build_sys_args(argc, argv)); + ray_t* sys_args = ray_build_sys_args(argc, argv); + if (!sys_args || RAY_IS_ERR(sys_args)) { + fprintf(stderr, "error: invalid command-line arguments\n"); + if (sys_args && RAY_IS_ERR(sys_args)) ray_error_free(sys_args); + ray_runtime_destroy(rt); + return 2; + } + ray_runtime_set_sys_args(sys_args); /* Initialise the worker pool before anything else that might use it * (file load, REPL eval, builtins). If -c wasn't given, leave the diff --git a/src/lang/syscmd.c b/src/lang/syscmd.c index a02c07c2..9fd7b44c 100644 --- a/src/lang/syscmd.c +++ b/src/lang/syscmd.c @@ -38,6 +38,8 @@ void* ray_runtime_get_poll(void); #include +#include +#include #include #include #include @@ -79,9 +81,20 @@ static int64_t arg_as_i64(ray_t* arg, int* err) { int sign = 1; if (i < len && (p[i] == '+' || p[i] == '-')) { if (p[i] == '-') sign = -1; i++; } if (i >= len || p[i] < '0' || p[i] > '9') { *err = 1; return 0; } - int64_t v = 0; - while (i < len && p[i] >= '0' && p[i] <= '9') { v = v * 10 + (p[i] - '0'); i++; } - return sign * v; + uint64_t v = 0; + uint64_t limit = sign < 0 ? (uint64_t)INT64_MAX + 1u : (uint64_t)INT64_MAX; + while (i < len && p[i] >= '0' && p[i] <= '9') { + uint64_t digit = (uint64_t)(p[i] - '0'); + if (v > (limit - digit) / 10u) { *err = 1; return 0; } + v = v * 10u + digit; + i++; + } + if (i != len) { *err = 1; return 0; } + if (sign < 0) { + if (v == (uint64_t)INT64_MAX + 1u) return INT64_MIN; + return -(int64_t)v; + } + return (int64_t)v; } *err = 1; return 0; diff --git a/src/ops/system.c b/src/ops/system.c index dec663f7..17641564 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -53,10 +53,12 @@ void* ray_runtime_get_poll(void); void ray_runtime_set_sys_args(void* dict); void* ray_runtime_get_sys_args(void); #include +#include #include #include #include #include + #if !defined(RAY_OS_WINDOWS) #include #include /* WIFEXITED/WEXITSTATUS — .sys.exec exit codes */ @@ -67,6 +69,25 @@ void* ray_runtime_get_sys_args(void); #define RAY_PCLOSE(f) _pclose(f) #endif +static int parse_sys_int_arg(const char* s, bool nonnegative, bool core_count, + int64_t* out) { + if (!s || !*s) return 0; + const unsigned char* p = (const unsigned char*)s; + if (!nonnegative && (*p == '+' || *p == '-')) p++; + if (!isdigit(*p)) return 0; + for (const unsigned char* q = p; *q; q++) + if (!isdigit(*q)) return 0; + + char* end = NULL; + errno = 0; + long long v = strtoll(s, &end, 10); + if (errno == ERANGE || end == s || *end != '\0' || + (nonnegative && v < 0) || (core_count && v > INT_MAX)) + return 0; + *out = (int64_t)v; + return 1; +} + /* ══════════════════════════════════════════ * Serialization / storage * ══════════════════════════════════════════ */ @@ -1553,13 +1574,33 @@ ray_t* ray_build_sys_args(int argc, char** argv) { interactive = true; else if ((strcmp(argv[i], "-p") == 0 || strcmp(argv[i], "--port") == 0) && i + 1 < argc) port = (int64_t)atoll(argv[++i]); - else if ((strcmp(argv[i], "-c") == 0 || strcmp(argv[i], "--cores") == 0) && i + 1 < argc) { - long long v = atoll(argv[++i]); if (v < 0) v = 0; cores = (int64_t)v; + else if (strcmp(argv[i], "-c") == 0 || strcmp(argv[i], "--cores") == 0) { + if (i + 1 >= argc) + return ray_error("rank", ".sys.args: %s requires an argument", argv[i]); + int64_t v; + if (!parse_sys_int_arg(argv[++i], true, true, &v)) + return ray_error("domain", ".sys.args: invalid %s value \"%s\"", + argv[i - 1], argv[i]); + cores = v; + } + else if (strcmp(argv[i], "-t") == 0 || strcmp(argv[i], "--timeit") == 0) { + if (i + 1 >= argc) + return ray_error("rank", ".sys.args: %s requires an argument", argv[i]); + int64_t v; + if (!parse_sys_int_arg(argv[++i], false, false, &v)) + return ray_error("domain", ".sys.args: invalid %s value \"%s\"", + argv[i - 1], argv[i]); + timeit = (v != 0); + } + else if (strcmp(argv[i], "-Q") == 0 || strcmp(argv[i], "--querylog") == 0) { + if (i + 1 >= argc) + return ray_error("rank", ".sys.args: %s requires an argument", argv[i]); + int64_t v; + if (!parse_sys_int_arg(argv[++i], false, false, &v)) + return ray_error("domain", ".sys.args: invalid %s value \"%s\"", + argv[i - 1], argv[i]); + querylog = (v != 0); } - else if ((strcmp(argv[i], "-t") == 0 || strcmp(argv[i], "--timeit") == 0) && i + 1 < argc) - timeit = (atoll(argv[++i]) != 0); - else if ((strcmp(argv[i], "-Q") == 0 || strcmp(argv[i], "--querylog") == 0) && i + 1 < argc) - querylog = (atoll(argv[++i]) != 0); else if ((strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--file") == 0) && i + 1 < argc) file = argv[++i]; else if (strcmp(argv[i], "-u") == 0 && i + 1 < argc) i++; /* skip secret */ diff --git a/test/rfl/system/syscmd_coverage.rfl b/test/rfl/system/syscmd_coverage.rfl index 8a0a18fe..a69bf0af 100644 --- a/test/rfl/system/syscmd_coverage.rfl +++ b/test/rfl/system/syscmd_coverage.rfl @@ -55,14 +55,14 @@ (.sys.cmd "timeit 100") -- 1 (.sys.cmd "timeit 0") -- 0 -;; Empty-after-sign / non-digit start → type error (line 80). +;; Empty-after-sign / non-digit start / trailing garbage → type error. (.sys.cmd "timeit +") !- type (.sys.cmd "timeit -") !- type (.sys.cmd "timeit abc") !- type -(.sys.cmd "timeit 1.5") -- 1 -;; ^ "1.5" parses up to the dot → 1, profile turns on; the trailing -;; ".5" is not an error (the parser stops at first non-digit and -;; returns what it has). Reset state for any later toggle tests. +(.sys.cmd "timeit 1.5") !- type +(.sys.timeit "1abc") !- type +(.sys.timeit "9223372036854775808") !- type +;; Reset state for any later toggle tests. (.sys.timeit 0) -- 0 ;; ────────────── arg_as_i64: type-fallthrough (lines 85-86) ────────────── diff --git a/test/test_runtime.c b/test/test_runtime.c index 8e6094fa..35160723 100644 --- a/test/test_runtime.c +++ b/test/test_runtime.c @@ -635,6 +635,21 @@ static test_result_t test_build_sys_args_edges(void) { PASS(); } +static test_result_t test_build_sys_args_rejects_malformed_numbers(void) { + char* cases[][3] = { + { "rayforce", "-c", "2abc" }, + { "rayforce", "-c", "999999999999999999999999" }, + { "rayforce", "-t", "1abc" }, + { "rayforce", "-Q", "0abc" }, + }; + for (size_t i = 0; i < sizeof cases / sizeof cases[0]; i++) { + ray_t* d = ray_build_sys_args(3, cases[i]); + TEST_ASSERT_TRUE(RAY_IS_ERR(d)); + ray_error_free(d); + } + PASS(); +} + /* .sys.args builtin: only `source` when unset; reflects stored dict when set */ static test_result_t test_sys_args_builtin(void) { /* unset → a dict holding just `source` (empty here: ray_eval_str is @@ -1288,6 +1303,7 @@ const test_entry_t runtime_entries[] = { { "runtime/build_sys_args_defaults", test_build_sys_args_defaults, sys_setup, sys_teardown }, { "runtime/build_sys_args_flags_user", test_build_sys_args_flags_and_user, sys_setup, sys_teardown }, { "runtime/build_sys_args_edges", test_build_sys_args_edges, sys_setup, sys_teardown }, + { "runtime/build_sys_args_rejects_malformed_numbers", test_build_sys_args_rejects_malformed_numbers, sys_setup, sys_teardown }, { "runtime/sys_args_builtin", test_sys_args_builtin, sys_setup, sys_teardown }, { "runtime/syscov_rc", test_syscov_rc, sys_setup, sys_teardown }, { "runtime/syscov_time_now", test_syscov_time_now, sys_setup, sys_teardown }, @@ -1315,4 +1331,3 @@ const test_entry_t runtime_entries[] = { { NULL, NULL, NULL, NULL }, }; - From 525a51dc41cc6ce927e99931cdd668ef8111cb79 Mon Sep 17 00:00:00 2001 From: Anton Date: Mon, 21 Sep 2026 01:11:48 +0200 Subject: [PATCH 06/51] fix(cli): reject a flag as another flag's value, and unknown options MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every value-taking startup flag was gated on `i + 1 < argc` and then consumed argv[++i] blindly, so one flag silently ate the next. The shape that was reported is the worst one for a service: rayforce -Q -p 5099 svc.rfl `-Q` took `-p` as its value, the port was never bound, and nothing on stdout or stderr said so. Under a supervisor the process starts, the script runs, the unit looks healthy, and every client gets connection refused. `-c` and `-t` had the identical hole. Two neighbouring cases came out of the same code. A trailing flag with no value fell through to the positional-file arm and reported the misleading `cannot open '-Q'`. And that arm took ANY unrecognized token as the script name, which the real positional then overwrote, so a typo'd option was not merely ignored — it was silently swallowed: `rayforce -x svc.rfl` ran the script as if `-x` had never been typed. Every value-taking flag now goes through flag_value(), which refuses a missing value and a value that is EXACTLY one of this program's flag tokens (including the `--` app-args terminator), with a diagnostic and exit 2 before the script is loaded. A value that merely STARTS with '-' stays legal: a password or a negative number is a legitimate value, and refusing those would break working command lines for no gain. An unrecognized `-`-prefixed token is now an error too; a lone `-` is still a positional, and tokens after `--` belong to the app and are never validated. `-Q, --querylog N` was a real flag, referenced twice in the docs, that --help never listed beside its sibling `-t, --timeit N` — added, along with `[-Q 0|1]` in the synopsis. The suggestion to fail when `-p` was requested but nothing bound is already in (#473, listen_fatal.rfl), and cannot catch this: the parser never saw the eaten `-p`, so there was no request to check against. The guard on the value is what closes the class. Also: the two pre-existing `-p` validation bail-outs now release the runtime like every other error path, since these paths are exercised under ASan by the new test. Closes #600 --- src/app/main.c | 126 +++++++++++++++++++++------- test/rfl/system/cli_flag_values.rfl | 47 +++++++++++ 2 files changed, 142 insertions(+), 31 deletions(-) create mode 100644 test/rfl/system/cli_flag_values.rfl diff --git a/src/app/main.c b/src/app/main.c index f6ff95f3..ad2d16bc 100644 --- a/src/app/main.c +++ b/src/app/main.c @@ -67,6 +67,56 @@ static int64_t parse_size_arg(const char* s) { return (int64_t)(v * (double)mult); } +/* -------------------------------------------------------------------------- + * Flag-value guard (#600) + * + * Every value-taking flag used to be gated on `i + 1 < argc` and then + * consume argv[++i] blindly, so one flag silently ate the next: `-Q -p + * 5099` started a server with NO listener, no error and no warning — + * the worst shape for a service, because the unit looks healthy while + * every client gets connection refused. A trailing flag fell through to + * the positional-file arm and reported `cannot open '-Q'`. + * + * A value is refused when it is missing, or when it is EXACTLY one of + * this program's flag tokens. A value that merely STARTS with '-' (a + * password, a negative number) stays legal — refusing those would break + * working command lines for no gain. + * + * Keep this list in step with the flags handled in main's parse loop and + * with --help. (ray_build_sys_args in src/ops/system.c walks argv a + * second time for .sys.args; it only ever sees a command line that + * already passed this guard, so it cannot mis-pair.) + * -------------------------------------------------------------------------- */ +static const char* const k_known_flags[] = { + "-i", "--interactive", "-p", "--port", "-c", "--cores", + "-t", "--timeit", "-Q", "--querylog", "-f", "--file", + "-u", "-U", "-l", "-L", "-m", "--mem", "-h", "--help", + "--", /* the app-args terminator is not a value either */ + NULL +}; + +static bool arg_is_known_flag(const char* tok) { + for (int k = 0; k_known_flags[k]; k++) + if (strcmp(tok, k_known_flags[k]) == 0) return true; + return false; +} + +/* The value for a value-taking flag, or NULL with the diagnostic already + * printed. Advances *i past the value it returns. */ +static const char* flag_value(int argc, char** argv, int* i, const char* name) { + if (*i + 1 >= argc) { + fprintf(stderr, "error: %s expects a value\n", name); + return NULL; + } + const char* v = argv[*i + 1]; + if (arg_is_known_flag(v)) { + fprintf(stderr, "error: %s expects a value, got flag '%s'\n", name, v); + return NULL; + } + ++*i; + return v; +} + int main(int argc, char** argv) { /* Install the fatal-signal handler first: a crash during startup * should still produce a symbolized backtrace. Handles the fault @@ -104,17 +154,19 @@ int main(int argc, char** argv) { for (int i = 1; i < argc; i++) { if (strcmp(argv[i], "-i") == 0 || strcmp(argv[i], "--interactive") == 0) interactive = 1; - else if ((strcmp(argv[i], "-p") == 0 || strcmp(argv[i], "--port") == 0) && i + 1 < argc) { + else if (strcmp(argv[i], "-p") == 0 || strcmp(argv[i], "--port") == 0) { /* PORT binds all interfaces; HOST:PORT confines the listener * to one IPv4 address (#427). Strict digits-only port — a * silently truncated "65536x" listener is worse than exiting. */ - const char* parg = argv[++i]; + const char* parg = flag_value(argc, argv, &i, argv[i]); + if (!parg) { ray_runtime_destroy(rt); return 2; } const char* colon = strchr(parg, ':'); const char* pstr = parg; if (colon) { size_t hlen = (size_t)(colon - parg); if (hlen == 0 || hlen >= sizeof(bind_host)) { fprintf(stderr, "bad -p bind address: %s\n", parg); + ray_runtime_destroy(rt); return 2; } memcpy(bind_host, parg, hlen); @@ -126,44 +178,45 @@ int main(int argc, char** argv) { long pv = strtol(pstr, &endp, 10); if (endp == pstr || *endp != '\0' || errno == ERANGE || pv <= 0 || pv > 65535) { fprintf(stderr, "bad -p port (must be 1..65535): %s\n", parg); + ray_runtime_destroy(rt); return 2; } port = (uint16_t)pv; } - else if ((strcmp(argv[i], "-c") == 0 || strcmp(argv[i], "--cores") == 0) && i + 1 < argc) { - int v = atoi(argv[++i]); - if (v < 0) v = 0; - n_cores = v; - } - else if ((strcmp(argv[i], "-t") == 0 || strcmp(argv[i], "--timeit") == 0) && i + 1 < argc) { - int v = atoi(argv[++i]); - timeit_init = (v != 0); - } - else if ((strcmp(argv[i], "-Q") == 0 || strcmp(argv[i], "--querylog") == 0) && i + 1 < argc) { - int v = atoi(argv[++i]); - qlog_init = (v != 0); + else if (strcmp(argv[i], "-c") == 0 || strcmp(argv[i], "--cores") == 0) { + const char* v = flag_value(argc, argv, &i, argv[i]); + if (!v) { ray_runtime_destroy(rt); return 2; } + n_cores = atoi(v); + if (n_cores < 0) n_cores = 0; } - else if ((strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--file") == 0) && i + 1 < argc) { - file = argv[++i]; + else if (strcmp(argv[i], "-t") == 0 || strcmp(argv[i], "--timeit") == 0) { + const char* v = flag_value(argc, argv, &i, argv[i]); + if (!v) { ray_runtime_destroy(rt); return 2; } + timeit_init = (atoi(v) != 0); } - else if (strcmp(argv[i], "-u") == 0 && i + 1 < argc) { - auth_pw = argv[++i]; - auth_restricted = false; + else if (strcmp(argv[i], "-Q") == 0 || strcmp(argv[i], "--querylog") == 0) { + const char* v = flag_value(argc, argv, &i, argv[i]); + if (!v) { ray_runtime_destroy(rt); return 2; } + qlog_init = (atoi(v) != 0); } - else if (strcmp(argv[i], "-U") == 0 && i + 1 < argc) { - auth_pw = argv[++i]; - auth_restricted = true; + else if (strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--file") == 0) { + file = flag_value(argc, argv, &i, argv[i]); + if (!file) { ray_runtime_destroy(rt); return 2; } } - else if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) { - log_base = argv[++i]; - log_mode = RAY_JOURNAL_ASYNC; + else if (strcmp(argv[i], "-u") == 0 || strcmp(argv[i], "-U") == 0) { + auth_restricted = (argv[i][1] == 'U'); + auth_pw = flag_value(argc, argv, &i, argv[i]); + if (!auth_pw) { ray_runtime_destroy(rt); return 2; } } - else if (strcmp(argv[i], "-L") == 0 && i + 1 < argc) { - log_base = argv[++i]; - log_mode = RAY_JOURNAL_SYNC; + else if (strcmp(argv[i], "-l") == 0 || strcmp(argv[i], "-L") == 0) { + log_mode = (argv[i][1] == 'L') ? RAY_JOURNAL_SYNC : RAY_JOURNAL_ASYNC; + log_base = flag_value(argc, argv, &i, argv[i]); + if (!log_base) { ray_runtime_destroy(rt); return 2; } } - else if ((strcmp(argv[i], "-m") == 0 || strcmp(argv[i], "--mem") == 0) && i + 1 < argc) { - int64_t wm = parse_size_arg(argv[++i]); + else if (strcmp(argv[i], "-m") == 0 || strcmp(argv[i], "--mem") == 0) { + const char* mv = flag_value(argc, argv, &i, argv[i]); + if (!mv) { ray_runtime_destroy(rt); return 2; } + int64_t wm = parse_size_arg(mv); if (wm <= 0) { fprintf(stderr, "error: -m/--mem expects a size like 4G, 512M, or a byte count\n"); ray_runtime_destroy(rt); @@ -173,13 +226,15 @@ int main(int argc, char** argv) { } else if (strcmp(argv[i], "-h") == 0 || strcmp(argv[i], "--help") == 0) { fprintf(stdout, - "usage: %s [-f file] [-p port] [-c cores] [-t 0|1] [-i]" + "usage: %s [-f file] [-p port] [-c cores] [-t 0|1] [-Q 0|1] [-i]" " [-u PW | -U PW] [-l BASE | -L BASE] [-m SIZE] [file.rfl] [-- app args]\n" " -f, --file FILE run script file (or pass as a positional arg)\n" " -p, --port PORT listen for IPC clients on PORT (all interfaces)\n" " --port HOST:PORT bind the listener to HOST only (e.g. 127.0.0.1:7701)\n" " -c, --cores N total execution cores, main included (0 = auto)\n" " -t, --timeit N enable profiler at startup (N != 0)\n" + " -Q, --querylog N enable query logging at startup (N != 0);\n" + " read the ring with (.sys.querylog)\n" " -i, --interactive start the REPL even after running a file\n" " -u PW set plain auth password\n" " -U PW set restricted auth password\n" @@ -209,6 +264,15 @@ int main(int argc, char** argv) { } else if (strcmp(argv[i], "--") == 0) break; /* stop flag parsing; remaining tokens are app args */ + else if (argv[i][0] == '-' && argv[i][1] != '\0') { + /* Not a known flag. This used to fall into the positional arm + * below, so a typo'd option was silently swallowed as the script + * name and then overwritten by the real one — the flag simply + * did nothing (#600). A lone "-" stays a positional. */ + fprintf(stderr, "error: unknown option '%s'\n", argv[i]); + ray_runtime_destroy(rt); + return 2; + } else file = argv[i]; } diff --git a/test/rfl/system/cli_flag_values.rfl b/test/rfl/system/cli_flag_values.rfl new file mode 100644 index 00000000..c014eada --- /dev/null +++ b/test/rfl/system/cli_flag_values.rfl @@ -0,0 +1,47 @@ +;; cli_flag_values.rfl — a value-taking flag must not swallow the next flag, +;; and an unknown option must not be mistaken for the script (#600). +;; +;; Regression: every value-taking flag consumed argv[++i] blindly, so +;; `-Q -p 5099` ate `-p` and started a server with NO listener, no error +;; and no warning — the worst shape for a service, since the unit looks +;; healthy and every client gets connection refused. `-c` and `-t` had +;; the same hole. A trailing `-Q` fell through to the positional-file +;; arm and failed with the misleading `cannot open '-Q'`, and an unknown +;; option was silently swallowed by that same arm. +(.sys.exec "printf '(println \"script ran\")\n' > /tmp/rfl_cfv.rfl") + +;; ---- a flag as another flag's value is rejected, script never runs ---- +;; (.sys.exec returns the raw wait status, so each case folds to 0/1) +(.sys.exec "./rayforce -Q -p 19961 /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q 'script ran' /tmp/rfl_cfv.out && exit 99; grep -q \"\\-Q expects a value, got flag '-p'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; the same hole in -c and -t +(.sys.exec "./rayforce -c -p 19962 /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q 'script ran' /tmp/rfl_cfv.out && exit 99; grep -q \"\\-c expects a value, got flag '-p'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 +(.sys.exec "./rayforce -t -p 19963 /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q 'script ran' /tmp/rfl_cfv.out && exit 99; grep -q \"\\-t expects a value, got flag '-p'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; long form too +(.sys.exec "./rayforce --querylog --port 19964 /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q 'script ran' /tmp/rfl_cfv.out && exit 99; grep -q \"expects a value, got flag '--port'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; ---- a missing value says so, instead of 'cannot open' ---- +(.sys.exec "./rayforce /tmp/rfl_cfv.rfl -Q /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q \"cannot open\" /tmp/rfl_cfv.out && exit 99; grep -q \"\\-Q expects a value\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; ---- an unknown option is rejected, not taken as the script ---- +(.sys.exec "./rayforce -x /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q 'script ran' /tmp/rfl_cfv.out && exit 99; grep -q \"unknown option '-x'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; ---- values that merely START with '-' are still accepted ---- +;; A password or a negative number is a legitimate value; only an exact +;; match against a known flag token is refused. +(.sys.exec "./rayforce -u -s3cret -t -1 /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; grep -q 'script ran' /tmp/rfl_cfv.out") -- 0 + +;; ---- the -Q that #600 was reached through still works, and still binds ---- +(.sys.exec "./rayforce -Q 1 -p 19965 /tmp/rfl_cfv.rfl /tmp/rfl_cfv2.out 2>&1 & p=$!; for i in $(seq 50); do grep -q 'script ran' /tmp/rfl_cfv2.out && break; sleep 0.1; done; bash -c '(echo > /dev/tcp/127.0.0.1/19965) 2>/dev/null'; rc=$?; kill $p 2>/dev/null; [ $rc -eq 0 ]") -- 0 + +;; the app-args terminator is not a value either +(.sys.exec "./rayforce -Q -- /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out 2>&1; rc=$?; grep -q \"\\-Q expects a value, got flag '--'\" /tmp/rfl_cfv.out || exit 98; [ $rc -eq 2 ]") -- 0 + +;; ---- tokens after `--` belong to the app and are never validated ---- +(.sys.exec "./rayforce /tmp/rfl_cfv.rfl -- -Q -p /tmp/rfl_cfv.out 2>&1; grep -q 'script ran' /tmp/rfl_cfv.out") -- 0 + +;; ---- --help still lists -Q, beside its sibling -t ---- +(.sys.exec "./rayforce --help 2>&1 | grep -q -- '-Q, --querylog'") -- 0 + +(.sys.exec "rm -f /tmp/rfl_cfv.rfl /tmp/rfl_cfv.out /tmp/rfl_cfv2.out; true") From 63d91b1ef5cee7b488184c7dfe9c4ba8fba69329 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Mon, 21 Sep 2026 10:56:19 +0300 Subject: [PATCH 07/51] fix(exec): release the input table of window, sort and limit unconditionally MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every node the executor evaluates returns an owned reference — the constant table node included, it retains its literal. The OP_WINDOW, OP_SORT and OP_HEAD cases released their input only when it differed from the graph's table, taking an equal pointer for a borrowed one. A query's root is a constant node over that very table, so the reference was never released: one input table per windowed, sorted or limited query. Invisible while a global kept the table alive; a whole table per call when the input was built for that call, as a service that windowed a freshly concatenated buffer on a timer found (#602). The four cases now release the input on every path, as the join and the plain head/tail cases already did. Test: window, sorted and limited selects over a table built per call, measured with the two-window bytes-allocated method of the other memory probes, plus a shared input reused across fifty calls. Co-Authored-By: Claude Fable 5.1 --- src/ops/exec.c | 27 ++++++++++-------- test/rfl/mem/query_input_release.rfl | 42 ++++++++++++++++++++++++++++ 2 files changed, 58 insertions(+), 11 deletions(-) create mode 100644 test/rfl/mem/query_input_release.rfl diff --git a/src/ops/exec.c b/src/ops/exec.c index 93b54944..d762388d 100644 --- a/src/ops/exec.c +++ b/src/ops/exec.c @@ -2610,20 +2610,26 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { } case OP_SORT: { + /* exec_node hands back an OWNED reference for every node, + * the constant table node included (it retains its literal). + * A `!= g->table` guard here treated a child that evaluates + * to the query table as borrowed and leaked one table per + * call — the ordinary shape, since a select's root is a + * constant node over that very table. */ ray_t* input = exec_node(g, op_child(g, op, 0)); if (!input || RAY_IS_ERR(input)) return input; ray_t* tbl = (input->type == RAY_TABLE) ? input : g->table; /* Compact lazy selection before sort (needs dense data) */ if (g->selection && tbl && !RAY_IS_ERR(tbl) && tbl->type == RAY_TABLE) { ray_t* compacted = sel_compact(g, tbl, g->selection, NULL, 0); - if (input != g->table) ray_release(input); + ray_release(input); ray_release(g->selection); g->selection = NULL; input = compacted; tbl = compacted; } ray_t* result = exec_sort(g, op, tbl, 0); - if (input != g->table) ray_release(input); + ray_release(input); return result; } @@ -2835,20 +2841,21 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { } case OP_WINDOW: { + /* Owned input, released unconditionally: see OP_SORT. */ ray_t* input = exec_node(g, op_child(g, op, 0)); if (!input || RAY_IS_ERR(input)) return input; ray_t* wdf = (input->type == RAY_TABLE) ? input : g->table; /* Compact lazy selection before window (needs dense data) */ if (g->selection && wdf && !RAY_IS_ERR(wdf) && wdf->type == RAY_TABLE) { ray_t* compacted = sel_compact(g, wdf, g->selection, NULL, 0); - if (input != g->table) ray_release(input); + ray_release(input); ray_release(g->selection); g->selection = NULL; input = compacted; wdf = compacted; } ray_t* result = exec_window(g, op, wdf); - if (input != g->table) ray_release(input); + ray_release(input); return result; } @@ -2865,14 +2872,14 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { /* Compact lazy selection before sort */ if (g->selection && tbl && !RAY_IS_ERR(tbl) && tbl->type == RAY_TABLE) { ray_t* compacted = sel_compact(g, tbl, g->selection, NULL, 0); - if (sort_input != g->table) ray_release(sort_input); + ray_release(sort_input); /* owned: see OP_SORT */ ray_release(g->selection); g->selection = NULL; sort_input = compacted; tbl = compacted; } ray_t* result = exec_sort(g, child_op, tbl, n); - if (sort_input != g->table) ray_release(sort_input); + ray_release(sort_input); /* Top-level statement GC catches intermediates. */ return result; } @@ -2935,7 +2942,7 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { ? filter_input : g->table; if (g->selection && ftbl && ftbl->type == RAY_TABLE) { ray_t* compacted = sel_compact(g, ftbl, g->selection, NULL, 0); - if (filter_input != g->table) ray_release(filter_input); + ray_release(filter_input); /* owned: see OP_SORT */ ray_release(g->selection); g->selection = NULL; filter_input = compacted; @@ -2949,15 +2956,13 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { g->table = saved_table; if (!pred || RAY_IS_ERR(pred)) { - if (filter_input != saved_table) - ray_release(filter_input); + ray_release(filter_input); return pred; } ray_t* result = exec_filter_head(ftbl, pred, n); ray_release(pred); - if (filter_input != saved_table) - ray_release(filter_input); + ray_release(filter_input); /* Top-level statement GC catches intermediates. */ return result; } else { diff --git a/test/rfl/mem/query_input_release.rfl b/test/rfl/mem/query_input_release.rfl new file mode 100644 index 00000000..80d4c0c0 --- /dev/null +++ b/test/rfl/mem/query_input_release.rfl @@ -0,0 +1,42 @@ +;; query_input_release.rfl — a query over a table built for that one call +;; must give the table back. +;; +;; Every node the executor evaluates hands back an OWNED reference, the +;; constant table node included. The window, sort and limit cases once +;; skipped the release when the input equalled the graph's table, taking it +;; for a borrowed reference — and a query's root is a constant node over that +;; very table, so one input table leaked per call. Invisible while a global +;; kept the table alive; a whole table per call on a fresh one. +;; +;; Same method as the other probes in this directory: warm up, then two equal +;; windows of calls; the first may settle a slab or an arena, the second must +;; not move. `bytes-allocated` is exact, so the bound is a few KB; the table +;; below is ~19 KB, so a single leaked call already fails it. +(set syms (as 'SYMBOL (map (fn [i] (format "inst%" i)) (til 700)))) +(set mk (fn [p] (table [instrument rid v] (list (take syms 786) (as 'I64 (+ p (til 786))) (as 'I64 (til 786)))))) +(set heapb (fn [] (at (.sys.mem) 'bytes-allocated))) +(set probe (fn [f warm reps] (do (map f (til warm)) (map f (til reps)) (set h0 (heapb)) (map f (til reps)) (< (- (heapb) h0) 4096)))) + +;; partitioned + ordered, trailing frame (the shape of the report) +(probe (fn [p] (count (window {from: (mk p) part: [instrument] order: [rid] frame: 2 funcs: {prev: (first v) n: (count v)}}))) 5 60) -- true +;; no part/order: the whole table is one partition +(probe (fn [p] (count (window {from: (mk p) funcs: {n: (count v)}}))) 5 60) -- true +;; running frame +(probe (fn [p] (count (window {from: (mk p) part: [instrument] order: [rid] frame: 'running funcs: {s: (sum v)}}))) 5 60) -- true +;; a where: on the window input (from: is materialised before the window) +(probe (fn [p] (count (window {from: (select {from: (mk p) where: (> rid 3)}) part: [instrument] order: [rid] funcs: {n: (count v)}}))) 5 60) -- true +;; the same guard sat under sort and limit: a sorted or limited select over +;; a table built for the call leaked it the same way +(probe (fn [p] (count (select {from: (mk p) asc: rid}))) 5 60) -- true +(probe (fn [p] (count (select {from: (mk p) desc: v take: 5}))) 5 60) -- true +(probe (fn [p] (count (select {from: (mk p) where: (> rid 3) take: 5}))) 5 60) -- true +(probe (fn [p] (count (select {from: (mk p) where: (> rid 3) asc: v}))) 5 60) -- true +(probe (fn [p] (count (select {from: (mk p) by: instrument s: (sum v) asc: s}))) 5 60) -- true +(probe (fn [p] (count (select {from: (mk p) by: instrument s: (sum v) desc: s take: 3}))) 5 60) -- true +;; the same table reused across calls must not be released twice: the +;; result is still right after many calls +(set T (mk 1)) +(map (fn [p] (window {from: T part: [instrument] order: [rid] funcs: {n: (count v)}})) (til 50)) +(count T) -- 786 +(count (window {from: T part: [instrument] order: [rid] funcs: {n: (count v)}})) -- 786 +(== (sum (at (window {from: T part: [instrument] order: [rid] frame: 'running funcs: {s: (sum v)}}) 's)) (sum (at (window {from: T part: [instrument] order: [rid] frame: 'running funcs: {s: (sum v)}}) 's))) -- true From a2fe4960d46a529d1717df161a2a9d6217eeae1b Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Mon, 21 Sep 2026 11:16:20 +0300 Subject: [PATCH 08/51] test(lang): pin the refcount of a table across window, sort and limit queries Twenty calls of each shape over one global table; the table's refcount must be exactly what it was before, and the table still whole after. Fails on the unfixed executor at the first window shape. Co-Authored-By: Claude Fable 5.1 --- test/test_lang.c | 53 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/test/test_lang.c b/test/test_lang.c index 1aaa33c6..afbdf69d 100644 --- a/test/test_lang.c +++ b/test/test_lang.c @@ -8703,6 +8703,58 @@ static test_result_t test_builtin_group_guid_rfl(void) { * the expression over the vocabulary bytes a chunk at a time. With the * chunk shrunk to 100 values, a 200-value vocabulary crosses chunk * boundaries; the groups and aggregates must equal the in-memory answer. */ +/* ---- Test: a query releases the table it was given ------------------- + * Every node the executor evaluates returns an owned reference, the + * constant table node a query is rooted on included. The window, sort + * and limit cases used to skip that release when the input equalled the + * graph's table, so each such query left one reference behind: the + * table's refcount climbed by one per call and, for a table built for + * the call, the whole table stayed allocated. Pin the refcount: after + * twenty calls of each shape it must be exactly what it was before. */ +static test_result_t test_select_releases_input_table(void) { + ray_t* setup = ray_eval_str( + "(do (set __rl_i (til 786)) " + " (set __rl_syms (as 'SYMBOL (map (fn [i] (format \"inst%\" i)) (til 700)))) " + " (set __rl_T (table [instrument rid v] (list (take __rl_syms 786) __rl_i (% (* 7 __rl_i) 101)))) 0)"); + TEST_ASSERT_NOT_NULL(setup); TEST_ASSERT_FALSE(RAY_IS_ERR(setup)); ray_release(setup); + ray_t* T = ray_eval_str("__rl_T"); /* one more ref: ours */ + TEST_ASSERT_NOT_NULL(T); TEST_ASSERT_FALSE(RAY_IS_ERR(T)); + TEST_ASSERT_EQ_I(T->type, RAY_TABLE); + uint32_t rc0 = ray_atomic_load(&T->rc); + static const char* const shapes[] = { + "(window {from: __rl_T part: [instrument] order: [rid] frame: 2 funcs: {n: (count v)}})", + "(window {from: __rl_T funcs: {n: (count v)}})", + "(window {from: __rl_T part: [instrument] order: [rid] frame: 'running funcs: {s: (sum v)}})", + "(select {from: __rl_T asc: rid})", + "(select {from: __rl_T desc: v take: 5})", + "(select {from: __rl_T where: (> rid 3) take: 5})", + "(select {from: __rl_T where: (> rid 3) asc: v})", + "(select {from: __rl_T by: instrument s: (sum v) asc: s})", + "(select {from: __rl_T by: instrument s: (sum v) desc: s take: 3})", + "(select {from: __rl_T take: 5})", + "(select {from: __rl_T where: (> rid 3)})", + }; + for (size_t k = 0; k < sizeof(shapes) / sizeof(shapes[0]); k++) { + for (int i = 0; i < 20; i++) { + ray_t* r = ray_eval_str(shapes[k]); + TEST_ASSERT_NOT_NULL(r); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + TEST_ASSERT_EQ_I(r->type, RAY_TABLE); + ray_release(r); + } + if (ray_atomic_load(&T->rc) != rc0) + FAIL(shapes[k]); + } + /* the table is still whole after all of that */ + ray_t* n = ray_eval_str("(count __rl_T)"); + TEST_ASSERT_NOT_NULL(n); TEST_ASSERT_FALSE(RAY_IS_ERR(n)); + TEST_ASSERT_EQ_I(n->i64, 786); + ray_release(n); + ray_release(T); + ray_release(ray_eval_str("(set __rl_T 0) (set __rl_i 0) (set __rl_syms 0)")); + PASS(); +} + static test_result_t test_select_derived_key_file_chunks(void) { ray_t* r = ray_eval_str( "(do (set __dk_i (til 20000)) " @@ -9600,6 +9652,7 @@ const test_entry_t lang_entries[] = { { "lang/builtin/group_guid_rfl", test_builtin_group_guid_rfl, lang_setup, lang_teardown }, { "lang/builtin/group_empty_list", test_builtin_group_empty_and_list, lang_setup, lang_teardown }, { "lang/select/derived_key_file_chunks", test_select_derived_key_file_chunks, lang_setup, lang_teardown }, + { "lang/select/releases_input_table", test_select_releases_input_table, lang_setup, lang_teardown }, { "lang/temporal/extract_builtins_fn", test_temporal_extract_builtins_fn, lang_setup, lang_teardown }, { "lang/temporal/extract_time_atom", test_temporal_extract_time_atom, lang_setup, lang_teardown }, { "lang/temporal/extract_time_vector", test_temporal_extract_time_vector, lang_setup, lang_teardown }, From 2ccebde24b0c9d3614b7b649b2b631b34689d5d0 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Mon, 21 Sep 2026 11:26:16 +0300 Subject: [PATCH 09/51] fix(exec): the reduction case owns its input too The same borrowed-query-table guard sat under the reductions; no child evaluates to the query table there today, so nothing leaked, but the premise is the one the sort, window and limit cases just dropped. Co-Authored-By: Claude Fable 5.1 --- src/ops/exec.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ops/exec.c b/src/ops/exec.c index d762388d..de07dbbe 100644 --- a/src/ops/exec.c +++ b/src/ops/exec.c @@ -2025,7 +2025,9 @@ static ray_t* exec_node_inner(ray_graph_t* g, ray_op_t* op) { if (!input || RAY_IS_ERR(input)) return input; /* Compact lazy selection before reducing — filters may have * set g->selection without materializing a compacted table. */ - bool own_input = (input != g->table); + /* Owned, like every exec_node result (see OP_SORT); no child + * evaluates to a borrowed query table. */ + bool own_input = true; if (g->selection && input->type == RAY_TABLE) { ray_t* compacted = sel_compact(g, input, g->selection, NULL, 0); if (own_input) ray_release(input); From a7b9b3991bcee6f4a24c2cf46e52b6dc5ef83176 Mon Sep 17 00:00:00 2001 From: Anton Date: Mon, 21 Sep 2026 15:45:51 +0200 Subject: [PATCH 10/51] perf(pool): wake only the workers a dispatch can keep busy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ray_pool_dispatch and ray_pool_dispatch_n signalled the whole pool on every dispatch, however narrow the window. The main thread participates as worker 0, so at most n_tasks-1 helpers can ever claim anything; the surplus threads woke, raced to an already-drained window, and went straight back to the semaphore. On a dispatch narrower than the machine that is pure overhead — a partition-parallel step over three partitions woke every core on the box to run two tasks. Signal min(n_tasks-1, n_workers) instead. Signalling FEWER is safe: completion is governed by the `pending` spin-wait, never by the signal count, so an unsignalled worker simply stays asleep, and under-signalling cannot produce the surplus-signal problem between consecutive dispatches that the spin-wait exists to avoid. Measured (10M-row ClickBench, 43 queries x 3, splayed): sum of per-query hot times 5690.7 ms -> 5667.3 ms (-0.4%, within noise), user CPU 55.1 s -> 54.8 s. So this is waste removal rather than a speedup on workloads whose dispatches are already wider than the pool. Not a fix for #599. That report's shape — a 130k-row join probe, which splits into 16 tasks against 8 workers — is unaffected, because the clamp lands on the same worker count it already used (measured: 5.19 s -> 5.18 s of CPU, unchanged). The investigation there points at the pool defaulting to logical rather than physical cores; that is a separate change with its own benchmark trade and is filed on its own. test_pool.c gains pool/dispatch_narrow, which is the case the existing coverage missed: dispatch_n_small uses 4 tasks on a 2-worker pool, so the clamp never binds. The new test drives windows of 1..4 tasks on a 4-worker pool, then alternates narrow and wide windows 25 times over the same pool, asserting exact task and element counts throughout — the two risks the change introduces are a task nobody claims and signal accounting that drifts across dispatches and starves a later wide window. --- src/core/pool.c | 24 ++++++++++++++++++---- test/test_pool.c | 52 ++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 72 insertions(+), 4 deletions(-) diff --git a/src/core/pool.c b/src/core/pool.c index 0b81070a..c3cac800 100644 --- a/src/core/pool.c +++ b/src/core/pool.c @@ -340,8 +340,16 @@ void ray_pool_dispatch(ray_pool_t* pool, ray_pool_fn fn, void* ctx, const bool prog = ray_qstats_mode() & RAY_QS_PROGRESS; if (prog) ray_progress_dispatch_begin((uint64_t)total_elems); - /* Wake worker threads */ - for (uint32_t i = 0; i < pool->n_workers; i++) { + /* Wake worker threads — only as many as have something to claim. Main + * participates as worker 0, so at most n_tasks-1 helpers can be useful; + * signalling the whole pool woke threads that raced to an already-drained + * window and went straight back to sleep, which is pure overhead on any + * dispatch narrower than the machine (#599). Signalling FEWER is safe: + * completion is governed by the `pending` spin-wait below, never by the + * signal count, so an unsignalled worker simply stays asleep. */ + uint32_t wake = n_tasks > 0 ? n_tasks - 1 : 0; + if (wake > pool->n_workers) wake = pool->n_workers; + for (uint32_t i = 0; i < wake; i++) { ray_sem_signal(&pool->work_ready); } @@ -413,8 +421,16 @@ static void dispatch_n_round(ray_pool_t* pool, ray_pool_fn fn, void* ctx, const bool prog = ray_qstats_mode() & RAY_QS_PROGRESS; if (prog) ray_progress_dispatch_begin((uint64_t)n_tasks); - /* Wake worker threads */ - for (uint32_t i = 0; i < pool->n_workers; i++) { + /* Wake worker threads — only as many as have something to claim. Main + * participates as worker 0, so at most n_tasks-1 helpers can be useful; + * signalling the whole pool woke threads that raced to an already-drained + * window and went straight back to sleep, which is pure overhead on any + * dispatch narrower than the machine (#599). Signalling FEWER is safe: + * completion is governed by the `pending` spin-wait below, never by the + * signal count, so an unsignalled worker simply stays asleep. */ + uint32_t wake = n_tasks > 0 ? n_tasks - 1 : 0; + if (wake > pool->n_workers) wake = pool->n_workers; + for (uint32_t i = 0; i < wake; i++) { ray_sem_signal(&pool->work_ready); } diff --git a/test/test_pool.c b/test/test_pool.c index 7e2f177d..8eb499a1 100644 --- a/test/test_pool.c +++ b/test/test_pool.c @@ -462,6 +462,57 @@ static test_result_t test_dispatch_small(void) { * spin-wait for completion). * -------------------------------------------------------------------------- */ +/* -------------------------------------------------------------------------- + * Test: a dispatch NARROWER than the pool still runs every task, and leaves + * the pool fully usable for a later wide one (#599). + * + * ray_pool_dispatch/_n now signal only min(n_tasks-1, n_workers) workers + * instead of the whole pool, so the under-signalled workers stay asleep. The + * risks that buys are (a) a task nobody claims and (b) signal accounting that + * drifts across dispatches, starving a later wide window. Alternating narrow + * and wide dispatches over one pool catches both: every window is verified for + * exact task and element counts. + * -------------------------------------------------------------------------- */ +static test_result_t test_dispatch_narrower_than_pool(void) { + ray_heap_init(); + + ray_pool_t pool; + TEST_ASSERT_EQ_I(ray_pool_create(&pool, 4), RAY_OK); + + /* n_tasks below the worker count: 1 wakes nobody, 2 wakes one, etc. */ + for (uint32_t n = 1; n <= 4; n++) { + pool_count_ctx_t ctx = {0}; + ray_pool_dispatch_n(&pool, pool_count_fn, &ctx, n); + TEST_ASSERT_EQ_I(atomic_load(&ctx.calls), n); + TEST_ASSERT_EQ_I(atomic_load(&ctx.elem_sum), n); + } + + /* The element form, sized to a single task. */ + { + pool_count_ctx_t ctx = {0}; + ray_pool_dispatch(&pool, pool_count_fn, &ctx, 1); + TEST_ASSERT_EQ_I(atomic_load(&ctx.calls), 1); + TEST_ASSERT_EQ_I(atomic_load(&ctx.elem_sum), 1); + } + + /* Alternating narrow/wide over the same pool: a wide window must still be + * fully served after windows that signalled fewer workers than exist. */ + for (int rep = 0; rep < 25; rep++) { + pool_count_ctx_t narrow = {0}; + ray_pool_dispatch_n(&pool, pool_count_fn, &narrow, 1); + TEST_ASSERT_EQ_I(atomic_load(&narrow.calls), 1); + + pool_count_ctx_t wide = {0}; + ray_pool_dispatch_n(&pool, pool_count_fn, &wide, 32); + TEST_ASSERT_EQ_I(atomic_load(&wide.calls), 32); + TEST_ASSERT_EQ_I(atomic_load(&wide.elem_sum), 32); + } + + ray_pool_free(&pool); + ray_heap_destroy(); + PASS(); +} + static test_result_t test_dispatch_n_small(void) { ray_heap_init(); @@ -1357,6 +1408,7 @@ const test_entry_t pool_entries[] = { { "pool/dispatch_zero_elems", test_dispatch_zero_elems, NULL, NULL }, { "pool/dispatch_small", test_dispatch_small, NULL, NULL }, { "pool/dispatch_n_small", test_dispatch_n_small, NULL, NULL }, + { "pool/dispatch_narrow", test_dispatch_narrower_than_pool, NULL, NULL }, { "pool/dispatch_n_ring_grow", test_dispatch_n_ring_growth, NULL, NULL }, { "pool/dispatch_ring_grow", test_dispatch_ring_growth, NULL, NULL }, { "pool/dispatch_n_cancelled", test_dispatch_n_cancelled, NULL, NULL }, From 08c5918cd7f81c1ef3c8890e18d484d2d99aaef7 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:56 +0300 Subject: [PATCH 11/51] build: Windows (MSYS2 CLANG64/MINGW64) toolchain in the Makefile Detect Windows via $(OS) or uname (MSYS2's make hides $(OS)), link Winsock and a statically linked winpthreads, build with 64-bit off_t (_FILE_OFFSET_BITS=64) and an 8 MiB stack like a Linux main thread. Co-Authored-By: Claude Opus 5 (1M context) --- Makefile | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/Makefile b/Makefile index aa6b666e..f6c170f3 100644 --- a/Makefile +++ b/Makefile @@ -81,7 +81,25 @@ COVERAGE_CFLAGS = -fPIC $(WARNS) -std=$(STD) -g -O0 -march=$(RAY_MARCH) -DDEBUG -fno-omit-frame-pointer -fprofile-instr-generate -fcoverage-mapping COVERAGE_LDFLAGS = -fprofile-instr-generate -fcoverage-mapping -ifeq ($(UNAME_S),Linux) +# Windows: MSYS2 CLANG64/MINGW64 toolchain (x86_64-w64-windows-gnu). MSYS2's +# own make hides $(OS), so also match `uname -s` (MINGW64_NT-*, CLANG64_NT-*, +# MSYS_NT-*); a native make started from PowerShell/cmd sees OS=Windows_NT. +RAY_WINDOWS := $(if $(filter Windows_NT,$(OS))$(findstring _NT-,$(UNAME_S)),1,) + +ifeq ($(RAY_WINDOWS),1) + # 64-bit off_t / struct stat.st_size: MinGW defaults both to 32 bits, which + # silently truncates sizes of files over 2 GiB (stat, lseek, ftruncate). + DEFS += -D_FILE_OFFSET_BITS=64 + # --stack: 8 MiB like a Linux main thread (the Windows default is 1 MiB, + # too little for deep DAG/eval recursion). CreateThread(size 0) inherits + # it too, so pool workers get the same. winpthreads (sched_yield, + # clock_gettime) is linked statically so the binary needs no MSYS2 DLL; the + # UCRT it also uses ships with Windows 10+. + LIBS = -lws2_32 -lmswsock -lkernel32 -ladvapi32 \ + -Wl,-Bstatic -lpthread -Wl,-Bdynamic \ + -Wl,--stack,8388608 + RELEASE_LDFLAGS = -Wl,--gc-sections +else ifeq ($(UNAME_S),Linux) LIBS = -lm -lpthread RELEASE_LDFLAGS = -Wl,--gc-sections -Wl,--as-needed else From 5ee60ea89b0702c3d708256f5d4eaf4b6b7bcd5e Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:56 +0300 Subject: [PATCH 12/51] feat(core): WSAPoll event loop for Windows Replace the IOCP stub with a readiness-based loop that mirrors the epoll backend's dispatch order, so the selector state machine is identical on every platform. stdin (console or pipe) is not a socket, so RAY_SEL_STDIN selectors are probed directly and the socket wait is sliced while one is registered. Co-Authored-By: Claude Opus 5 (1M context) --- src/core/iocp.c | 431 ++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 416 insertions(+), 15 deletions(-) diff --git a/src/core/iocp.c b/src/core/iocp.c index 636c3fed..dd75fd8f 100644 --- a/src/core/iocp.c +++ b/src/core/iocp.c @@ -21,36 +21,181 @@ * SOFTWARE. */ +#include "core/platform.h" + #if defined(RAY_OS_WINDOWS) +#define WIN32_LEAN_AND_MEAN +#include +#include +#include +#include +#include +#include + #include "core/poll.h" -#include +#include "core/mcast.h" +#include "core/timer.h" +#include "mem/sys.h" +#include "mem/heap.h" /* idle decay: bound the wait, sweep after wakeup */ + +/* Windows event loop. + * + * Readiness-based, like the epoll and kqueue backends, so the selector + * state machine (rx fill -> read_fn -> data_fn, tx flush) is the same on + * every platform. Sockets are waited on with WSAPoll. Standard input is + * not a socket and WSAPoll cannot watch it, so RAY_SEL_STDIN selectors are + * probed directly (console input queue, bytes in a pipe) and the socket + * wait is cut into short slices while one is registered. + * + * WSAPoll keeps no kernel-side registration, so the wait set is rebuilt + * from poll->sels on every pass: register/deregister and tx request/cancel + * need no OS call, and a selector's write interest simply follows whether + * it has a pending tx buffer. */ + +#define RAY_POLL_INITIAL_CAP 16 +#define RAY_POLL_STDIN_SLICE_MS 10 + +enum { EV_IN = 1, EV_OUT = 2, EV_HUP = 4 }; + +/* ===== stdin readiness ===== */ + +/* Would a read of `fd` return now (data or EOF) instead of blocking? */ +static int stdin_events(int64_t fd) +{ + HANDLE h = (HANDLE)_get_osfhandle((int)fd); + if (h == INVALID_HANDLE_VALUE) return EV_HUP; + + switch (GetFileType(h)) { + case FILE_TYPE_CHAR: { + DWORD mode; + if (!GetConsoleMode(h, &mode)) return EV_IN; /* NUL device: never blocks */ + for (;;) { + INPUT_RECORD rec; + DWORD n = 0; + if (!PeekConsoleInputW(h, &rec, 1, &n)) return EV_HUP; + if (n == 0) return 0; + if (rec.EventType == KEY_EVENT && rec.Event.KeyEvent.bKeyDown && + rec.Event.KeyEvent.uChar.UnicodeChar != 0) + return EV_IN; + /* Focus, mouse, resize and key-up records never yield a byte to + * ReadFile: drop them, or a "ready" console would make the + * reader block in ReadFile until the next real keystroke. */ + if (!ReadConsoleInputW(h, &rec, 1, &n)) return EV_HUP; + } + } + case FILE_TYPE_PIPE: { + DWORD avail = 0; + if (!PeekNamedPipe(h, NULL, 0, NULL, &avail, NULL)) + return EV_IN | EV_HUP; /* writer gone: the read returns EOF */ + return avail ? EV_IN : 0; + } + default: + return EV_IN; /* disk file: reads never block */ + } +} -/* Windows IOCP implementation — stub for now. - * Full IOCP support is deferred to a future release. */ +/* ===== Lifecycle ===== */ ray_poll_t* ray_poll_create(void) { - fprintf(stderr, "ray_poll_create: IOCP not yet implemented\n"); - return NULL; + WSADATA wsa; + if (WSAStartup(MAKEWORD(2, 2), &wsa) != 0) return NULL; + + ray_poll_t* poll = (ray_poll_t*)ray_sys_alloc(sizeof(ray_poll_t)); + if (!poll) return NULL; + + memset(poll, 0, sizeof(*poll)); + poll->fd = -1; /* no kernel object behind a WSAPoll loop */ + poll->code = -1; + poll->sel_cap = RAY_POLL_INITIAL_CAP; + poll->sels = (ray_selector_t**)ray_sys_alloc( + poll->sel_cap * sizeof(ray_selector_t*)); + if (!poll->sels) { + ray_sys_free(poll); + return NULL; + } + memset(poll->sels, 0, poll->sel_cap * sizeof(ray_selector_t*)); + return poll; } void ray_poll_destroy(ray_poll_t* poll) { - (void)poll; + if (!poll) return; + + for (uint32_t i = 0; i < poll->n_sels; i++) { + ray_selector_t* sel = poll->sels[i]; + if (!sel) continue; + if (sel->close_fn) sel->close_fn(poll, sel); + if (sel->rx.buf) ray_poll_buf_free(sel->rx.buf); + ray_poll_buf_free(sel->tx.buf); + ray_sys_free(sel); + poll->sels[i] = NULL; + } + + if (poll->sels) ray_sys_free(poll->sels); + if (poll->timers) { + ray_timers_destroy((ray_timers_t*)poll->timers); + poll->timers = NULL; + } + if (poll->mcast) { + ray_mcast_destroy((ray_mcast_t*)poll->mcast); + poll->mcast = NULL; + } + ray_sys_free(poll); } +/* ===== Registration ===== */ + int64_t ray_poll_register(ray_poll_t* poll, ray_poll_reg_t* reg) { - (void)poll; (void)reg; - return -1; -} + if (!poll || !reg) return -1; -void ray_poll_deregister(ray_poll_t* poll, int64_t id) -{ - (void)poll; (void)id; + /* Find free slot or grow */ + int64_t id = -1; + for (uint32_t i = 0; i < poll->n_sels; i++) { + if (!poll->sels[i]) { id = (int64_t)i; break; } + } + if (id < 0) { + if (poll->n_sels >= poll->sel_cap) { + uint32_t new_cap = poll->sel_cap * 2; + ray_selector_t** ns = (ray_selector_t**)ray_sys_alloc( + new_cap * sizeof(ray_selector_t*)); + if (!ns) return -1; + memcpy(ns, poll->sels, poll->n_sels * sizeof(ray_selector_t*)); + memset(ns + poll->n_sels, 0, + (new_cap - poll->n_sels) * sizeof(ray_selector_t*)); + ray_sys_free(poll->sels); + poll->sels = ns; + poll->sel_cap = new_cap; + } + id = (int64_t)poll->n_sels; + poll->n_sels++; + } + + ray_selector_t* sel = (ray_selector_t*)ray_sys_alloc(sizeof(ray_selector_t)); + if (!sel) return -1; + memset(sel, 0, sizeof(*sel)); + + sel->fd = reg->fd; + sel->id = id; + sel->type = reg->type; + sel->data = reg->data; + sel->open_fn = reg->open_fn; + sel->close_fn = reg->close_fn; + sel->error_fn = reg->error_fn; + sel->data_fn = reg->data_fn; + sel->rx.recv_fn = reg->recv_fn; + sel->rx.read_fn = reg->read_fn; + sel->tx.send_fn = reg->send_fn; + + poll->sels[id] = sel; + poll->n_live++; + if (sel->open_fn) sel->open_fn(poll, sel); + return id; } +/* Write interest is derived from sel->tx.buf when the wait set is built. */ void ray_poll_tx_request(ray_poll_t* poll, ray_selector_t* sel) { (void)poll; (void)sel; @@ -61,10 +206,266 @@ void ray_poll_tx_cancel(ray_poll_t* poll, ray_selector_t* sel) (void)poll; (void)sel; } +void ray_poll_deregister(ray_poll_t* poll, int64_t id) +{ + if (!poll || id < 0 || (uint32_t)id >= poll->n_sels) return; + ray_selector_t* sel = poll->sels[id]; + if (!sel) return; + + if (sel->close_fn) sel->close_fn(poll, sel); + if (sel->rx.buf) ray_poll_buf_free(sel->rx.buf); + ray_poll_buf_free(sel->tx.buf); + ray_sys_free(sel); + poll->sels[id] = NULL; + if (poll->n_live > 0) poll->n_live--; +} + +/* ===== Event dispatch (same order and rules as the epoll backend) ===== */ + +static void poll_dispatch(ray_poll_t* poll, uint64_t eid, int events) +{ + ray_selector_t* sel = NULL; + if (eid < poll->n_sels) + sel = poll->sels[eid]; + if (!sel) return; + + /* Process readable data first — even if hangup is also set. A client + * may send a message and close; both arrive in the same wakeup. */ + if (events & EV_IN) { + /* Loop: read data -> call read_fn -> if state advanced, read more. + * Handles multi-phase protocols (handshake -> header -> payload) + * arriving in a single wakeup. */ + for (;;) { + if (sel->rx.recv_fn && sel->rx.buf) { + while (sel->rx.buf->offset < sel->rx.buf->size) { + int64_t nr = sel->rx.recv_fn( + sel->fd, + sel->rx.buf->data + sel->rx.buf->offset, + sel->rx.buf->size - sel->rx.buf->offset); + if (nr <= 0) { + if (nr < 0 && errno == EINTR) continue; + if (nr < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) + break; + /* Error or peer closed mid-read */ + if (sel->error_fn) + sel->error_fn(poll, sel); + else + ray_poll_deregister(poll, sel->id); + return; + } + sel->rx.buf->offset += nr; + } + } + + /* Not enough data for current phase */ + if (sel->rx.buf && sel->rx.buf->offset < sel->rx.buf->size) + break; + + /* Call read_fn — may advance state and request new buffer */ + if (!sel->rx.read_fn) break; + ray_t* obj = sel->rx.read_fn(poll, sel); + + /* Re-validate: read_fn may have deregistered this selector */ + if (eid >= poll->n_sels || !poll->sels[eid]) return; + sel = poll->sels[eid]; + + if (obj && sel->data_fn) + sel->data_fn(poll, sel, obj); + + if (eid >= poll->n_sels || !poll->sels[eid]) return; + sel = poll->sels[eid]; + + /* If no rx buffer (state machine done or not set), stop */ + if (!sel->rx.buf) break; + /* If buffer already has enough data for next phase, loop */ + if (sel->rx.buf->offset >= sel->rx.buf->size) continue; + /* Otherwise try reading more (may EAGAIN -> break) */ + } + } + + if (events & EV_OUT) { + if (eid >= poll->n_sels || !poll->sels[eid]) return; + sel = poll->sels[eid]; + if (sel->tx.buf && ray_poll_tx_flush(poll, sel) < 0) { + if (eid < poll->n_sels && poll->sels[eid]) { + sel = poll->sels[eid]; + if (sel->error_fn) + sel->error_fn(poll, sel); + else + ray_poll_deregister(poll, sel->id); + } + return; + } + } + + /* Error / hangup — after data is drained */ + if (events & EV_HUP) { + if (eid < poll->n_sels && poll->sels[eid]) { + sel = poll->sels[eid]; + if (sel->error_fn) + sel->error_fn(poll, sel); + else + ray_poll_deregister(poll, sel->id); + } + } +} + +/* ===== Run loop ===== */ + int64_t ray_poll_run_for(ray_poll_t* poll, int timeout_ms) { - (void)poll; (void)timeout_ms; - return -1; + if (!poll) return -1; + + bool bounded = timeout_ms >= 0; + int64_t end_ms = bounded ? ray_time_now_ms() + timeout_ms : INT64_MAX; + + while (poll->code < 0) { + int wait_ms = -1; + if (bounded) { + int64_t remaining = end_ms - ray_time_now_ms(); + if (remaining < 0) remaining = 0; + if (remaining > INT_MAX) remaining = INT_MAX; + wait_ms = (int)remaining; + } + + /* Nothing registered and nothing scheduled: an unbounded loop + * would block here forever. Return instead, so a process that + * stayed only for its timers can end once they are spent. */ + if (!bounded && ray_poll_idle(poll)) return 0; + + if (poll->timers) { + int64_t deadline = ray_timers_next_deadline_ms( + (ray_timers_t*)poll->timers); + if (deadline != INT64_MAX) { + int64_t delta = deadline - ray_time_now_ms(); + if (delta < 0) delta = 0; + if (delta > INT_MAX) delta = INT_MAX; + if (wait_ms < 0 || delta < wait_ms) + wait_ms = (int)delta; + } + } + + /* Idle allocator decay bounds an unbounded wait, exactly as in the + * epoll backend (see the comment there). */ + { + int64_t decay = bounded ? -1 : ray_heap_decay_due_ms(); + if (decay >= 0) { + if (decay > INT_MAX) decay = INT_MAX; + if (wait_ms < 0 || decay < wait_ms) wait_ms = (int)decay; + } + } + + /* Build this pass's wait set: sockets for WSAPoll, stdin probed. */ + uint32_t cap = poll->n_sels ? poll->n_sels : 1; + WSAPOLLFD* pfds = (WSAPOLLFD*)ray_sys_alloc(cap * sizeof(WSAPOLLFD)); + uint32_t* pids = (uint32_t*)ray_sys_alloc(cap * sizeof(uint32_t)); + uint32_t* sids = (uint32_t*)ray_sys_alloc(cap * sizeof(uint32_t)); + int* sevs = (int*)ray_sys_alloc(cap * sizeof(int)); + if (!pfds || !pids || !sids || !sevs) { + if (pfds) ray_sys_free(pfds); + if (pids) ray_sys_free(pids); + if (sids) ray_sys_free(sids); + if (sevs) ray_sys_free(sevs); + return -1; + } + + uint32_t nfds = 0, nstd = 0; + for (uint32_t i = 0; i < poll->n_sels; i++) { + ray_selector_t* sel = poll->sels[i]; + if (!sel) continue; + if (sel->type == RAY_SEL_STDIN) { + sids[nstd] = i; + sevs[nstd] = 0; + nstd++; + continue; + } + pfds[nfds].fd = (SOCKET)sel->fd; + pfds[nfds].events = POLLRDNORM | (sel->tx.buf ? POLLWRNORM : 0); + pfds[nfds].revents = 0; + pids[nfds] = i; + nfds++; + } + + /* Wait. With stdin registered the socket wait is sliced so the + * console/pipe is re-probed every RAY_POLL_STDIN_SLICE_MS. */ + int64_t wait_end = wait_ms < 0 ? INT64_MAX : ray_time_now_ms() + wait_ms; + int n = 0; + bool failed = false; + for (;;) { + bool std_ready = false; + for (uint32_t k = 0; k < nstd; k++) { + sevs[k] = stdin_events(poll->sels[sids[k]]->fd); + if (sevs[k]) std_ready = true; + } + + int remain = -1; + if (wait_end != INT64_MAX) { + int64_t r = wait_end - ray_time_now_ms(); + remain = r < 0 ? 0 : (r > INT_MAX ? INT_MAX : (int)r); + } + int slice = std_ready ? 0 : remain; + if (nstd && !std_ready && + (slice < 0 || slice > RAY_POLL_STDIN_SLICE_MS)) + slice = RAY_POLL_STDIN_SLICE_MS; + + if (nfds) { + n = WSAPoll(pfds, nfds, slice); + if (n == SOCKET_ERROR) { + if (WSAGetLastError() == WSAEINTR) continue; + failed = true; + break; + } + } else if (slice > 0) { + Sleep((DWORD)slice); + } else if (slice < 0) { + /* Only reachable with a live selector that is neither a + * socket nor stdin; nothing can wake us, so just yield. */ + Sleep(RAY_POLL_STDIN_SLICE_MS); + } + + if (n > 0 || std_ready) break; + if (remain == 0) break; /* deadline reached */ + if (!nstd && slice == remain) break; /* whole wait elapsed */ + } + + if (!failed) { + for (uint32_t j = 0; j < nfds && n > 0; j++) { + short re = pfds[j].revents; + if (!re) continue; + int ev = 0; + if (re & (POLLRDNORM | POLLRDBAND)) ev |= EV_IN; + if (re & POLLWRNORM) ev |= EV_OUT; + if (re & (POLLERR | POLLHUP | POLLNVAL)) ev |= EV_HUP; + /* The slot may have been reused by an earlier dispatch in + * this pass (deregister + accept): only act on the socket + * the event was reported for. */ + ray_selector_t* sel = pids[j] < poll->n_sels ? poll->sels[pids[j]] : NULL; + if (!sel || (SOCKET)sel->fd != pfds[j].fd) continue; + poll_dispatch(poll, pids[j], ev); + } + for (uint32_t k = 0; k < nstd; k++) { + if (!sevs[k]) continue; + ray_selector_t* sel = sids[k] < poll->n_sels ? poll->sels[sids[k]] : NULL; + if (!sel || sel->type != RAY_SEL_STDIN) continue; + poll_dispatch(poll, sids[k], sevs[k]); + } + } + + ray_sys_free(pfds); + ray_sys_free(pids); + ray_sys_free(sids); + ray_sys_free(sevs); + if (failed) return -1; + + if (poll->timers) { + if (ray_timers_fire_expired((ray_timers_t*)poll->timers) > 0) + ray_heap_note_activity(); + } + ray_heap_decay(); + if (bounded) break; + } + + return poll->code >= 0 ? poll->code : 0; } int64_t ray_poll_run(ray_poll_t* poll) @@ -72,4 +473,4 @@ int64_t ray_poll_run(ray_poll_t* poll) return ray_poll_run_for(poll, -1); } -#endif /* _WIN32 */ +#endif /* RAY_OS_WINDOWS */ From 7cb12aff93839c848f89eb30fc72a281f017304b Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:56 +0300 Subject: [PATCH 13/51] fix(net): Winsock errno mapping, exclusive bind and IPC on Windows - mirror WSAGetLastError()/SO_ERROR into errno for send/recv/connect/ accept/bind/listen: callers branch on EAGAIN/ECONNREFUSED; - ipc_send_fn always overwrites errno, so a stale EAGAIN can no longer park a frame on a dead socket; - SO_EXCLUSIVEADDRUSE instead of SO_REUSEADDR (which lets a bind steal a port that is in use); - WSAStartup at load time; sock.h pulls in platform.h; - verbose IPC capture uses a temp file that works without admin rights; - the SIGURG out-of-band cancel stays POSIX-only. Co-Authored-By: Claude Opus 5 (1M context) --- src/core/ipc.c | 27 +++++++++++++++- src/core/sock.c | 84 ++++++++++++++++++++++++++++++++++++++++++++++--- src/core/sock.h | 2 +- 3 files changed, 107 insertions(+), 6 deletions(-) diff --git a/src/core/ipc.c b/src/core/ipc.c index 5143b5ce..c2593510 100644 --- a/src/core/ipc.c +++ b/src/core/ipc.c @@ -39,6 +39,7 @@ #ifdef RAY_OS_WINDOWS #define WIN32_LEAN_AND_MEAN + #include /* dup, dup2, close */ #include #include #else @@ -719,13 +720,31 @@ static ray_t* eval_payload_core(uint8_t* payload, size_t payload_len, * to fail the whole request because /tmp is full. The captured * string is then empty, and the response shape is still the 2-elem * list — clients can rely on that invariant. */ +#ifdef RAY_OS_WINDOWS +/* MSVCRT's tmpfile() creates its file in the drive root (admin-only on a + * stock system); use the user's temp directory, deleted on close ("D"). + * With no TMP/TEMP/USERPROFILE in the environment GetTempPath falls back + * to the (unwritable) Windows directory, so retry in the working dir. */ +static FILE* ipc_tmpfile(void) +{ + char dir[MAX_PATH + 1], path[MAX_PATH + 1]; + DWORD n = GetTempPathA(sizeof(dir), dir); + if (n == 0 || n > sizeof(dir) || !GetTempFileNameA(dir, "ray", 0, path)) { + if (!GetTempFileNameA(".", "ray", 0, path)) return NULL; + } + return fopen(path, "w+bD"); +} +#else +#define ipc_tmpfile() tmpfile() +#endif + static ray_t* eval_payload(uint8_t* payload, size_t payload_len, ray_ipc_header_t* hdr) { if (!(hdr->flags & RAY_IPC_FLAG_VERBOSE)) return eval_payload_core(payload, payload_len, hdr); - FILE* cap = tmpfile(); + FILE* cap = ipc_tmpfile(); int saved_out = -1, saved_err = -1; bool capturing = false; @@ -831,9 +850,13 @@ static int64_t ipc_send_fn(int64_t fd, uint8_t* buf, int64_t len) #ifdef RAY_OS_WINDOWS int n = send((ray_sock_t)fd, (const char*)buf, (int)len, 0); if (n < 0) { + /* Always overwrite errno: callers treat EAGAIN as "queue and retry", + * so a stale EAGAIN left over from an earlier call would turn a dead + * socket into a silently parked frame. */ int e = WSAGetLastError(); if (e == WSAEWOULDBLOCK) errno = EAGAIN; else if (e == WSAEINTR) errno = EINTR; + else errno = EIO; } return n; #else @@ -855,6 +878,7 @@ static int64_t ipc_send_fn(int64_t fd, uint8_t* buf, int64_t len) * ray_request_interrupt(). */ static volatile sig_atomic_t g_ipc_active_fd = -1; +#ifndef RAY_OS_WINDOWS /* no SIGURG on Windows: OOB cancel is POSIX-only */ static void ipc_sigurg_handler(int sig) { (void)sig; @@ -867,6 +891,7 @@ static void ipc_sigurg_handler(int sig) if (ray_sock_take_oob((ray_sock_t)fd)) ray_request_interrupt(); } +#endif static void ipc_install_oob_cancel(void) { diff --git a/src/core/sock.c b/src/core/sock.c index 750af93a..18f0c502 100644 --- a/src/core/sock.c +++ b/src/core/sock.c @@ -21,7 +21,7 @@ * SOFTWARE. */ -#ifndef RAY_OS_WINDOWS +#ifndef _WIN32 #define _GNU_SOURCE #endif @@ -48,6 +48,44 @@ /* ===== Socket Implementation ===== */ +#ifdef RAY_OS_WINDOWS +/* Winsock must be initialised before the process's first socket call; do it + * at load time so no entry point (listen, connect, a bare client) can miss + * it. Never paired with WSACleanup: process exit releases it. */ +__attribute__((constructor)) static void sock_win_startup(void) +{ + WSADATA wsa; + (void)WSAStartup(MAKEWORD(2, 2), &wsa); +} + +/* Winsock reports failures through WSAGetLastError(), never errno. The + * callers (poll loop, IPC) branch on errno == EINTR / EAGAIN, so mirror the + * Winsock code into errno after every failed socket call (or a code taken + * from SO_ERROR). */ +static void sock_win_errno_code(int code) +{ + switch (code) { + case WSAEWOULDBLOCK: errno = EAGAIN; break; + case WSAEINTR: errno = EINTR; break; + case WSAEINPROGRESS: errno = EINPROGRESS; break; + case WSAETIMEDOUT: errno = ETIMEDOUT; break; + case WSAECONNRESET: errno = ECONNRESET; break; + case WSAECONNABORTED: errno = ECONNABORTED; break; + case WSAECONNREFUSED: errno = ECONNREFUSED; break; + case WSAENOTCONN: errno = ENOTCONN; break; + case WSAENOTSOCK: errno = ENOTSOCK; break; + case WSAEADDRINUSE: errno = EADDRINUSE; break; + case WSAEINVAL: errno = EINVAL; break; + default: errno = EIO; break; + } +} + +static void sock_win_errno(void) +{ + sock_win_errno_code(WSAGetLastError()); +} +#endif + ray_sock_t ray_sock_listen_at(const char* host, uint16_t port) { /* NULL/empty host keeps the historical INADDR_ANY bind. A host that @@ -68,7 +106,15 @@ ray_sock_t ray_sock_listen_at(const char* host, uint16_t port) if (fd == RAY_INVALID_SOCK) return RAY_INVALID_SOCK; int yes = 1; +#ifdef RAY_OS_WINDOWS + /* Windows SO_REUSEADDR lets a second socket bind a port another one is + * actively listening on (the bind "steals" it). The POSIX meaning — + * rebind right after a restart, but fail while the port is in use — is + * SO_EXCLUSIVEADDRUSE here (TIME_WAIT never blocks a Windows bind). */ + setsockopt(fd, SOL_SOCKET, SO_EXCLUSIVEADDRUSE, (const char*)&yes, sizeof(yes)); +#else setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, (const char*)&yes, sizeof(yes)); +#endif struct sockaddr_in addr; memset(&addr, 0, sizeof(addr)); @@ -77,10 +123,16 @@ ray_sock_t ray_sock_listen_at(const char* host, uint16_t port) addr.sin_port = htons(port); if (bind(fd, (struct sockaddr*)&addr, sizeof(addr)) < 0) { +#ifdef RAY_OS_WINDOWS + sock_win_errno(); /* e.g. EADDRINUSE for a taken port */ +#endif ray_sock_close(fd); return RAY_INVALID_SOCK; } if (listen(fd, 128) < 0) { +#ifdef RAY_OS_WINDOWS + sock_win_errno(); +#endif ray_sock_close(fd); return RAY_INVALID_SOCK; } @@ -97,6 +149,9 @@ ray_sock_t ray_sock_accept(ray_sock_t srv) ray_sock_t fd; do { fd = (ray_sock_t)accept(srv, NULL, NULL); +#ifdef RAY_OS_WINDOWS + if (fd == RAY_INVALID_SOCK) sock_win_errno(); +#endif } while (fd == RAY_INVALID_SOCK && errno == EINTR); if (fd == RAY_INVALID_SOCK) return RAY_INVALID_SOCK; @@ -115,8 +170,13 @@ ray_sock_t ray_sock_accept(ray_sock_t srv) static int sock_connect_one(ray_sock_t fd, const struct sockaddr* addr, socklen_t addrlen, int timeout_ms) { - if (timeout_ms <= 0) - return connect(fd, addr, addrlen) < 0 ? -1 : 0; + if (timeout_ms <= 0) { + if (connect(fd, addr, addrlen) == 0) return 0; +#ifdef RAY_OS_WINDOWS + sock_win_errno(); /* callers tell refusal from other failures by errno */ +#endif + return -1; + } ray_sock_set_nonblocking(fd); int rc = connect(fd, addr, addrlen); @@ -124,6 +184,7 @@ static int sock_connect_one(ray_sock_t fd, const struct sockaddr* addr, #ifdef RAY_OS_WINDOWS int werr = WSAGetLastError(); int in_progress = (werr == WSAEWOULDBLOCK || werr == WSAEINPROGRESS); + if (!in_progress) sock_win_errno_code(werr); #else int in_progress = (errno == EINPROGRESS); #endif @@ -136,13 +197,22 @@ static int sock_connect_one(ray_sock_t fd, const struct sockaddr* addr, do { pr = poll(&pfd, 1, timeout_ms); } while (pr < 0 && errno == EINTR); #endif if (pr == 0) { errno = ETIMEDOUT; return -1; } - if (pr < 0) return -1; + if (pr < 0) { +#ifdef RAY_OS_WINDOWS + sock_win_errno(); +#endif + return -1; + } /* Writable: harvest the pending connect result via SO_ERROR. */ int soerr = 0; socklen_t soerr_len = sizeof(soerr); if (getsockopt(fd, SOL_SOCKET, SO_ERROR, (char*)&soerr, &soerr_len) < 0 || soerr != 0) { +#ifdef RAY_OS_WINDOWS + if (soerr != 0) sock_win_errno_code(soerr); /* a WSAE* code */ +#else if (soerr != 0) errno = soerr; +#endif return -1; } } @@ -213,6 +283,7 @@ int64_t ray_sock_send(ray_sock_t s, const void* buf, size_t len) while (rem > 0) { #ifdef RAY_OS_WINDOWS int n = send(s, (const char*)p, (int)rem, 0); + if (n < 0) sock_win_errno(); #else ssize_t n = send(s, p, rem, MSG_NOSIGNAL); #endif @@ -221,7 +292,11 @@ int64_t ray_sock_send(ray_sock_t s, const void* buf, size_t len) if (errno == EAGAIN || errno == EWOULDBLOCK) { /* Wait for write-readiness before retry */ struct pollfd pfd = { .fd = s, .events = POLLOUT }; +#ifdef RAY_OS_WINDOWS + WSAPoll(&pfd, 1, -1); +#else poll(&pfd, 1, -1); +#endif continue; } return -1; @@ -237,6 +312,7 @@ int64_t ray_sock_recv(ray_sock_t s, void* buf, size_t len) for (;;) { #ifdef RAY_OS_WINDOWS int n = recv(s, (char*)buf, (int)len, 0); + if (n < 0) sock_win_errno(); #else ssize_t n = recv(s, buf, len, 0); #endif diff --git a/src/core/sock.h b/src/core/sock.h index e35e566c..d400dbef 100644 --- a/src/core/sock.h +++ b/src/core/sock.h @@ -24,7 +24,7 @@ #ifndef RAY_SOCK_H #define RAY_SOCK_H -#include +#include "core/platform.h" /* ===== Socket Abstraction ===== */ From 4a392b761cba159a3f4507adf46a07ece85eee47 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:56 +0300 Subject: [PATCH 14/51] fix: Windows platform layer (file mapping, pools, crash report, paths) - ray_vm_unmap_file only unmaps at a view's own base: UnmapViewOfFile drops the whole view for an interior pointer, which freed columns that carry a passenger index (munmap is a no-op there); - ray_vm_alloc_aligned returns its own allocation base, so pools are really released by ray_vm_free; - ray_vm_map_fd_ro maps for real (CSV reads always failed with io); - crash report via SetUnhandledExceptionFilter; - heap: file-backed spill stays POSIX-only (docs/architecture/memory.md); - domain: realpath substitute; symfile paths compare case/separator- insensitively so one file never gets two domains; - aof/csr/hnsw: platform handle for fsync, portable ray_mkdir. Co-Authored-By: Claude Opus 5 (1M context) --- src/core/crash.c | 107 +++++++++++++++++++++++++++++++++----------- src/core/platform.c | 59 ++++++++++++++++++------ src/io/csv.c | 4 +- src/mem/heap.c | 36 +++++++++++++-- src/store/aof.c | 13 +++++- src/store/csr.c | 3 +- src/store/fileio.h | 5 ++- src/store/hnsw.c | 3 +- src/table/domain.c | 42 +++++++++++++++-- 9 files changed, 218 insertions(+), 54 deletions(-) diff --git a/src/core/crash.c b/src/core/crash.c index 17d644ed..76f894b2 100644 --- a/src/core/crash.c +++ b/src/core/crash.c @@ -19,10 +19,13 @@ #include "core/crash.h" #include +#include #include #include -#if !defined(_WIN32) +#if defined(_WIN32) +#include +#else #include #endif @@ -42,7 +45,7 @@ static void cw(const char* s) { /* Write an unsigned value as 0x-prefixed hex. No libc formatting. * Routes through cw() so the write return value is handled (gcc's * warn_unused_result on write() is not silenced by a (void) cast). */ -static void cw_hex(unsigned long v) { +static void cw_hex(uint64_t v) { char buf[2 + 16 + 1]; static const char hexd[] = "0123456789abcdef"; buf[0] = '0'; buf[1] = 'x'; @@ -52,7 +55,8 @@ static void cw_hex(unsigned long v) { cw(buf); } -/* Write a small non-negative integer as decimal. */ +#if !defined(_WIN32) +/* Write a small non-negative integer as decimal (backtrace frame count). */ static void cw_int(int v) { if (v < 0) { cw("-"); v = -v; } char buf[16]; @@ -62,6 +66,75 @@ static void cw_int(int v) { while (v > 0 && i > 0) { buf[--i] = (char)('0' + v % 10); v /= 10; } cw(&buf[i]); } +#endif + +/* Banner precomputed at install time so the handler doesn't format it. */ +static char g_banner[128]; + +static void crash_banner_init(void) { + const char* v = +#ifdef RAYFORCE_VERSION + "rayforce " RAYFORCE_VERSION +#else + "rayforce" +#endif +#ifdef RAYFORCE_GIT_COMMIT + " (" RAYFORCE_GIT_COMMIT ")" +#endif + "\n"; + size_t vl = strlen(v); + if (vl >= sizeof(g_banner)) vl = sizeof(g_banner) - 1; + memcpy(g_banner, v, vl); + g_banner[vl] = '\0'; +} + +#if defined(_WIN32) + +/* ── Windows: structured exceptions ───────────────────────────────── */ + +static const char* exc_name(DWORD code) { + switch (code) { + case EXCEPTION_ACCESS_VIOLATION: return "EXCEPTION_ACCESS_VIOLATION"; + case EXCEPTION_STACK_OVERFLOW: return "EXCEPTION_STACK_OVERFLOW"; + case EXCEPTION_ILLEGAL_INSTRUCTION: return "EXCEPTION_ILLEGAL_INSTRUCTION"; + case EXCEPTION_INT_DIVIDE_BY_ZERO: return "EXCEPTION_INT_DIVIDE_BY_ZERO"; + case EXCEPTION_ARRAY_BOUNDS_EXCEEDED: return "EXCEPTION_ARRAY_BOUNDS_EXCEEDED"; + case EXCEPTION_IN_PAGE_ERROR: return "EXCEPTION_IN_PAGE_ERROR"; + default: return "exception"; + } +} + +/* Top-level filter: runs only for exceptions nobody else handled. Report + * and let the default handling (WER / exit with the exception code) run, + * which keeps the process exit status meaningful to the orchestrator. */ +static LONG WINAPI crash_filter(EXCEPTION_POINTERS* ep) { + const EXCEPTION_RECORD* er = ep ? ep->ExceptionRecord : NULL; + cw("\n=== rayforce fatal "); + cw(er ? exc_name(er->ExceptionCode) : "exception"); + if (er) { + cw(" code "); cw_hex((uint64_t)er->ExceptionCode); + cw(" at "); cw_hex((uint64_t)(uintptr_t)er->ExceptionAddress); + if (er->ExceptionCode == EXCEPTION_ACCESS_VIOLATION && + er->NumberParameters >= 2) { + cw(" fault addr "); + cw_hex((uint64_t)er->ExceptionInformation[1]); + } + } + cw(" ===\n"); + cw(g_banner); + return EXCEPTION_CONTINUE_SEARCH; +} + +void ray_crash_install(void) { + crash_banner_init(); + /* Reserve stack for the filter itself, so a stack-overflow exception + * can still be reported (the Windows analogue of sigaltstack). */ + ULONG guarantee = 64 * 1024; + (void)SetThreadStackGuarantee(&guarantee); + SetUnhandledExceptionFilter(crash_filter); +} + +#else /* POSIX */ static const char* sig_name(int sig) { switch (sig) { @@ -74,9 +147,6 @@ static const char* sig_name(int sig) { } } -/* Banner precomputed at install time so the handler doesn't format it. */ -static char g_banner[128]; - /* ── the handler ──────────────────────────────────────────────────── */ static void crash_handler(int sig, siginfo_t* info, void* ucontext) { @@ -84,11 +154,10 @@ static void crash_handler(int sig, siginfo_t* info, void* ucontext) { cw("\n=== rayforce fatal "); cw(sig_name(sig)); - if (info) { cw(" at fault addr "); cw_hex((unsigned long)info->si_addr); } + if (info) { cw(" at fault addr "); cw_hex((uint64_t)(uintptr_t)info->si_addr); } cw(" ===\n"); cw(g_banner); -#if !defined(_WIN32) void* frames[64]; int n = backtrace(frames, 64); /* backtrace_symbols_fd writes directly to the fd without allocating. */ @@ -96,7 +165,6 @@ static void crash_handler(int sig, siginfo_t* info, void* ucontext) { cw("=== end backtrace ("); cw_int(n); cw(" frames) ===\n"); -#endif /* Restore the default disposition and re-raise, so the process dies * from the original signal: this preserves the core dump and reports @@ -116,24 +184,8 @@ static void crash_handler(int sig, siginfo_t* info, void* ucontext) { static char g_altstack[RAY_CRASH_ALTSTACK_SZ]; void ray_crash_install(void) { -#if !defined(_WIN32) /* Precompute the version banner once (async-signal-safe reuse). */ - { - const char* v = -#ifdef RAYFORCE_VERSION - "rayforce " RAYFORCE_VERSION -#else - "rayforce" -#endif -#ifdef RAYFORCE_GIT_COMMIT - " (" RAYFORCE_GIT_COMMIT ")" -#endif - "\n"; - size_t vl = strlen(v); - if (vl >= sizeof(g_banner)) vl = sizeof(g_banner) - 1; - memcpy(g_banner, v, vl); - g_banner[vl] = '\0'; - } + crash_banner_init(); /* Warm up the unwinder: the first backtrace() may dlopen libgcc and * allocate, which must not happen inside the handler. */ @@ -157,5 +209,6 @@ void ray_crash_install(void) { static const int sigs[] = { SIGSEGV, SIGBUS, SIGILL, SIGFPE, SIGABRT }; for (size_t i = 0; i < sizeof(sigs) / sizeof(sigs[0]); i++) (void)sigaction(sigs[i], &sa, NULL); -#endif } + +#endif /* _WIN32 */ diff --git a/src/core/platform.c b/src/core/platform.c index 89852918..bad99c91 100644 --- a/src/core/platform.c +++ b/src/core/platform.c @@ -469,6 +469,7 @@ void ray_sem_signal(ray_sem_t* s) { #define WIN32_LEAN_AND_MEAN #endif #include +#include /* _get_osfhandle */ #include "mem/sys.h" /* -------------------------------------------------------------------------- @@ -518,13 +519,34 @@ void* ray_vm_map_file(const char* path, size_t* out_size) { void ray_vm_unmap_file(void* ptr, size_t size) { if (!ptr) return; - UnmapViewOfFile(ptr); + /* munmap releases exactly [ptr, ptr+size) and is a no-op (EINVAL) for an + * unaligned ptr — the heap relies on that when it frees a block living + * inside a column's mapping (a passenger index). UnmapViewOfFile instead + * drops the WHOLE view containing ptr, which would pull the column out + * from under its live references. So only unmap at the view's own base; + * an interior block goes away with the view. The byte accounting still + * follows the call, exactly as on POSIX. */ + MEMORY_BASIC_INFORMATION mbi; + if (VirtualQuery(ptr, &mbi, sizeof(mbi)) && mbi.AllocationBase == ptr) + UnmapViewOfFile(ptr); ray_sys_track_file_sub((int64_t)size); } -/* Windows never reaches the fd/mmap CSV path (#ifndef RAY_OS_WINDOWS), so this - * is an unused stub kept only for API completeness. */ -void* ray_vm_map_fd_ro(int fd, size_t size) { (void)fd; (void)size; return NULL; } +/* Read-only view of an open CRT descriptor (the CSV reader maps the file this + * way, then closes the fd). The view keeps the file alive on its own, so both + * the mapping handle and the caller's fd may be closed afterwards. */ +void* ray_vm_map_fd_ro(int fd, size_t size) { + if (size == 0) return NULL; + HANDLE hFile = (HANDLE)_get_osfhandle(fd); + if (hFile == INVALID_HANDLE_VALUE) return NULL; + HANDLE hMap = CreateFileMappingA(hFile, NULL, PAGE_READONLY, 0, 0, NULL); + if (!hMap) return NULL; + void* p = MapViewOfFile(hMap, FILE_MAP_READ, 0, 0, size); + CloseHandle(hMap); + if (!p) return NULL; + ray_sys_track_file_add((int64_t)size); + return p; +} void ray_vm_advise_seq(void* ptr, size_t size) { /* PrefetchVirtualMemory is Win8.1+. Best-effort; ignore failure. */ @@ -546,16 +568,25 @@ void ray_vm_release_block(void* blk, size_t bsize, bool hugepage) { } void* ray_vm_alloc_aligned(size_t size, size_t alignment) { - /* Over-allocate, find aligned offset. Can't trim on Windows, so the - * pool header's vm_base field stores the original base for VirtualFree. */ - void* mem = VirtualAlloc(NULL, size + alignment, - MEM_RESERVE | MEM_COMMIT, PAGE_READWRITE); - if (!mem) return NULL; - uintptr_t aligned = ((uintptr_t)mem + alignment - 1) & ~(alignment - 1); - /* Count the kept `size` to balance ray_vm_free(ptr, size); the alignment - * slack Windows cannot trim is left uncounted (parity with POSIX). */ - ray_sys_track_add((int64_t)size); - return (void*)aligned; + /* VirtualFree(MEM_RELEASE) only accepts the exact base VirtualAlloc + * returned, and a reservation cannot be trimmed. So find an aligned + * hole by reserving size+alignment, release it, and allocate exactly + * `size` at the aligned address inside it. Another thread may take the + * hole in between; retry a few times. The result is its own allocation + * base, so ray_vm_free(ptr, size) releases it like any other block. */ + for (int attempt = 0; attempt < 16; attempt++) { + void* probe = VirtualAlloc(NULL, size + alignment, MEM_RESERVE, PAGE_NOACCESS); + if (!probe) return NULL; + uintptr_t aligned = ((uintptr_t)probe + alignment - 1) & ~(alignment - 1); + VirtualFree(probe, 0, MEM_RELEASE); + void* p = VirtualAlloc((void*)aligned, size, + MEM_RESERVE | MEM_COMMIT, PAGE_READWRITE); + if (p) { + ray_sys_track_add((int64_t)size); + return p; + } + } + return NULL; } bool ray_vm_hugepage(void* ptr, size_t size) { (void)ptr; (void)size; return false; } diff --git a/src/io/csv.c b/src/io/csv.c index 269fb3da..25302c1f 100644 --- a/src/io/csv.c +++ b/src/io/csv.c @@ -65,8 +65,8 @@ #include #ifndef RAY_OS_WINDOWS #include -#endif #include +#endif /* -------------------------------------------------------------------------- * Constants @@ -104,7 +104,9 @@ static inline uint64_t csv_prog_at(size_t file_size, unsigned pct) { * mmap flags * -------------------------------------------------------------------------- */ +#ifndef RAY_OS_WINDOWS #define MMAP_FLAGS MAP_PRIVATE +#endif /* -------------------------------------------------------------------------- * Scratch memory helpers (same pattern as exec.c). diff --git a/src/mem/heap.c b/src/mem/heap.c index 1f82a7e6..906ef225 100644 --- a/src/mem/heap.c +++ b/src/mem/heap.c @@ -44,11 +44,21 @@ #include /* getpid, close, ftruncate, unlink */ #include /* open, fcntl, F_PREALLOCATE on macOS */ #include -#include /* mmap, munmap */ #include /* O_* modes */ #include #include +/* File-backed spill (a pool or direct block mapped over a preallocated temp + * file when the anonymous mapping is refused) is POSIX-only. Windows pools + * take only the anonymous VirtualAlloc path (docs/architecture/memory.md): + * the spill helpers are compiled out and swap_fd stays -1 there. */ +#if defined(RAY_OS_WINDOWS) +# define RAY_HEAP_FILE_SPILL 0 +#else +# define RAY_HEAP_FILE_SPILL 1 +# include /* mmap, munmap */ +#endif + #ifdef DEBUG /* ===================================================================== * Debug-only stale-pointer detector (issue #240 investigation). @@ -64,7 +74,9 @@ * choice for chasing double releases (found while debugging #240). * RAY_DFD_NO_ABORT=1 reports without aborting. * ===================================================================== */ +#if !defined(RAY_OS_WINDOWS) #include +#endif #define DFD_CAP_BITS 20 #define DFD_CAP (1u << DFD_CAP_BITS) @@ -149,10 +161,12 @@ static void dfd_purge_range(uintptr_t lo, uintptr_t hi) { } static void dfd_report(const char* who, const void* p) { + fprintf(stderr, "\n=== DFD: %s on FREED block %p ===\n", who, p); +#if !defined(RAY_OS_WINDOWS) void* frames[64]; int n = backtrace(frames, 64); - fprintf(stderr, "\n=== DFD: %s on FREED block %p ===\n", who, p); backtrace_symbols_fd(frames, n, 2); +#endif fflush(stderr); if (!getenv("RAY_DFD_NO_ABORT")) abort(); } @@ -179,6 +193,7 @@ static void dfd_validate_freelists(void); * contiguous first, fall back to non-contiguous, then ftruncate to * extend the file size if needed (F_PREALLOCATE doesn't grow the file * beyond its current size). */ +#if RAY_HEAP_FILE_SPILL static int heap_preallocate(int fd, off_t offset, off_t len) { #if defined(__APPLE__) fstore_t fs = { @@ -202,6 +217,7 @@ static int heap_preallocate(int fd, off_t offset, off_t len) { return posix_fallocate(fd, offset, len); #endif } +#endif /* RAY_HEAP_FILE_SPILL */ /* -------------------------------------------------------------------------- * Static asserts @@ -574,6 +590,9 @@ static bool heap_add_pool(ray_heap_t* h, uint8_t order) { if (!heap_anon_would_exceed(pool_size)) mem = ray_vm_alloc_aligned(pool_size, pool_size); +#if !RAY_HEAP_FILE_SPILL + if (!mem) return false; +#else if (!mem) { /* Anonymous mmap refused — usually means RAM+swap can't satisfy * pool_size right now. Fall back to file-backed mmap: create a @@ -668,6 +687,7 @@ static bool heap_add_pool(ray_heap_t* h, uint8_t order) { ray_sys_free(swap_path); swap_path = NULL; } +#endif /* RAY_HEAP_FILE_SPILL */ /* Enable transparent huge pages on anon pools (Linux). Self-aligned * 32MB pools are 2MB-aligned, hence THP-eligible. Never on file-backed @@ -1184,6 +1204,10 @@ static void ray_detach_owned_refs(ray_t* v) { * failure. */ static void* heap_direct_map_file(ray_heap_t* h, size_t map_size, int* out_fd, char** out_path) { +#if !RAY_HEAP_FILE_SPILL + (void)h; (void)map_size; (void)out_fd; (void)out_path; + return NULL; +#else static _Atomic uint64_t direct_swap_counter = 0; uint64_t cnt = atomic_fetch_add_explicit(&direct_swap_counter, 1, memory_order_relaxed); @@ -1213,6 +1237,7 @@ static void* heap_direct_map_file(ray_heap_t* h, size_t map_size, *out_fd = fd; *out_path = path; return mapped; +#endif /* RAY_HEAP_FILE_SPILL */ } /* -------------------------------------------------------------------------- @@ -1713,6 +1738,7 @@ void ray_free(ray_t* v) { if (h) RAY_STAT(h->stats.free_count++); atomic_fetch_sub_explicit(&g_direct_bytes, (int64_t)map_size, memory_order_relaxed); atomic_fetch_sub_explicit(&g_direct_count, 1, memory_order_relaxed); +#if RAY_HEAP_FILE_SPILL if (swap_fd >= 0) { /* File-backed spill: mapped directly (not via ray_vm_alloc), so * unmap + uncount by hand, then close and unlink the spill file. */ @@ -1720,7 +1746,11 @@ void ray_free(ray_t* v) { ray_sys_track_sub((int64_t)map_size); close(swap_fd); if (swap_path) { unlink(swap_path); ray_sys_free(swap_path); } - } else if (direct_cache_put(base, map_size)) { + } else +#else + (void)swap_fd; (void)swap_path; +#endif + if (direct_cache_put(base, map_size)) { /* Stashed for reuse: pages stay resident, so the block keeps * its committed-RAM and watermark accounting; only the live * stats above dropped. Eviction (direct_cache_drain) performs diff --git a/src/store/aof.c b/src/store/aof.c index fe01b480..e85831ae 100644 --- a/src/store/aof.c +++ b/src/store/aof.c @@ -49,6 +49,15 @@ #include #include +/* ray_file_sync takes the platform handle: the descriptor itself on POSIX, + * the underlying HANDLE on Windows. */ +#ifdef RAY_OS_WINDOWS +#include +#define AOF_FP_HANDLE(fp) ((ray_fd_t)_get_osfhandle(fileno(fp))) +#else +#define AOF_FP_HANDLE(fp) ((ray_fd_t)fileno(fp)) +#endif + #define AOF_PATH_MAX 1024 /* Segment-path buffers are sized past the worst case (dir + '/' + 24-char * segment name) so gcc's -Wformat-truncation can prove snprintf fits even @@ -390,7 +399,7 @@ static ray_err_t aof_rotate(ray_aof_t* log) { if (err != RAY_OK) return err; } if (fflush(log->fp) != 0) return RAY_ERR_IO; - if (ray_file_sync((ray_fd_t)fileno(log->fp)) != RAY_OK) return RAY_ERR_IO; + if (ray_file_sync(AOF_FP_HANDLE(log->fp)) != RAY_OK) return RAY_ERR_IO; if (fclose(log->fp) != 0) { log->fp = NULL; return RAY_ERR_IO; } char path[AOF_SEGPATH_MAX]; @@ -442,7 +451,7 @@ ray_err_t ray_aof_commit(ray_aof_t* log) { ray_err_t err = aof_write_frame(log); if (err != RAY_OK) return err; if (fflush(log->fp) != 0) return RAY_ERR_IO; - return ray_file_sync((ray_fd_t)fileno(log->fp)); + return ray_file_sync(AOF_FP_HANDLE(log->fp)); } int64_t ray_aof_next_lsn(const ray_aof_t* log) { diff --git a/src/store/csr.c b/src/store/csr.c index 59f43b5a..b042fb51 100644 --- a/src/store/csr.c +++ b/src/store/csr.c @@ -24,6 +24,7 @@ #include "csr.h" #include "store/col.h" #include "mem/sys.h" +#include "store/fileio.h" /* ray_mkdir */ #include #include #include @@ -464,7 +465,7 @@ ray_err_t ray_rel_save(ray_rel_t* rel, const char* dir) { if (!rel || !dir) return RAY_ERR_IO; /* Create directory */ - if (mkdir(dir, 0755) != 0 && errno != EEXIST) return RAY_ERR_IO; + if (ray_mkdir(dir) != RAY_OK) return RAY_ERR_IO; ray_err_t err = csr_save(&rel->fwd, dir, "fwd"); if (err != RAY_OK) return err; diff --git a/src/store/fileio.h b/src/store/fileio.h index 658e5606..c6f954c8 100644 --- a/src/store/fileio.h +++ b/src/store/fileio.h @@ -24,10 +24,13 @@ #ifndef RAY_FILEIO_H #define RAY_FILEIO_H -#include +#include "core/platform.h" /* Cross-platform file I/O (locking, sync, atomic rename) */ #ifdef RAY_OS_WINDOWS + #ifndef WIN32_LEAN_AND_MEAN + #define WIN32_LEAN_AND_MEAN /* keep / macros out */ + #endif #include typedef HANDLE ray_fd_t; #define RAY_FD_INVALID INVALID_HANDLE_VALUE diff --git a/src/store/hnsw.c b/src/store/hnsw.c index 339119a1..a350c433 100644 --- a/src/store/hnsw.c +++ b/src/store/hnsw.c @@ -23,6 +23,7 @@ #include "hnsw.h" #include "mem/sys.h" +#include "store/fileio.h" /* ray_mkdir */ #include #include #include @@ -818,7 +819,7 @@ typedef struct { ray_err_t ray_hnsw_save(const ray_hnsw_t* idx, const char* dir) { if (!idx || !dir) return RAY_ERR_IO; - if (mkdir(dir, 0755) != 0 && errno != EEXIST) return RAY_ERR_IO; + if (ray_mkdir(dir) != RAY_OK) return RAY_ERR_IO; char path[1024]; FILE* f; diff --git a/src/table/domain.c b/src/table/domain.c index ba87c492..0e9f5869 100644 --- a/src/table/domain.c +++ b/src/table/domain.c @@ -210,6 +210,40 @@ static inline void dom_unlock(void) { /* ---- FILE domain construction / destruction ------------------------------- */ +/* realpath(3) contract: absolute path of an EXISTING file, else NULL. + * Windows has no realpath; _fullpath only normalizes (it succeeds for + * missing files too), so existence is checked separately. */ +static char* dom_realpath(const char* path, char resolved[PATH_MAX]) { +#if defined(RAY_OS_WINDOWS) + if (!_fullpath(resolved, path, PATH_MAX)) return NULL; + if (GetFileAttributesA(resolved) == INVALID_FILE_ATTRIBUTES) return NULL; + return resolved; +#else + return realpath(path, resolved); +#endif +} + +/* Do two resolved paths name the same file? On Windows one file is + * reachable as C:\db\sym, C:\db/sym or C:\DB\Sym (NTFS is case-insensitive + * and _fullpath keeps the caller's separators and case), so compare with + * '\\' == '/' and ASCII case folded — a plain strcmp would key one symfile + * twice and open two diverging domains for it. The paths themselves are + * left as given: they are also used to create the file. */ +static bool dom_path_eq(const char* a, const char* b) { +#if defined(RAY_OS_WINDOWS) + for (;; a++, b++) { + char ca = *a == '\\' ? '/' : *a; + char cb = *b == '\\' ? '/' : *b; + if (ca >= 'A' && ca <= 'Z') ca = (char)(ca - 'A' + 'a'); + if (cb >= 'A' && cb <= 'Z') cb = (char)(cb - 'A' + 'a'); + if (ca != cb) return false; + if (!ca) return true; + } +#else + return strcmp(a, b) == 0; +#endif +} + /* Resolved cache key for `path`. realpath of the file when it exists; * for to-be-created symfiles, realpath of the parent + "/" + basename * (the parent must exist). malloc'd. */ @@ -218,7 +252,7 @@ static char* dom_resolve_path(const char* path) { * realpath(NULL)/strdup's libc-malloc'd buffers, so the returned key is * uniformly buddy-allocated and the caller releases it with ray_free_raw. */ char resolved[PATH_MAX]; - if (realpath(path, resolved)) { + if (dom_realpath(path, resolved)) { size_t n = strlen(resolved); char* out = (char*)ray_sys_alloc(n + 1); if (out) memcpy(out, resolved, n + 1); @@ -238,7 +272,7 @@ static char* dom_resolve_path(const char* path) { memcpy(tmp, path, plen + 1); char* dir = dirname(tmp); char rdir[PATH_MAX]; - if (!realpath(dir, rdir)) return NULL; + if (!dom_realpath(dir, rdir)) return NULL; size_t dlen = strlen(rdir); char* out = (char*)ray_sys_alloc(dlen + 1 + blen + 1); if (!out) return NULL; @@ -519,7 +553,7 @@ static ray_sym_domain_t* dom_open_impl(const char* path, bool create) { dom_lock(); for (ray_sym_domain_t* d = g_domains; d; d = d->next) { - if (strcmp(d->path, rpath) == 0) { + if (dom_path_eq(d->path, rpath)) { /* Revalidate: external append-only growth extends in place; * any other divergence is loud (NULL). */ size_t cur_size = exists ? (size_t)st.st_size : 0; @@ -591,7 +625,7 @@ static ray_sym_domain_t* dom_open_impl(const char* path, bool create) { * the winner (pointer equality must hold for one resolved path). */ dom_lock(); for (ray_sym_domain_t* e = g_domains; e; e = e->next) { - if (strcmp(e->path, d->path) == 0) { + if (dom_path_eq(e->path, d->path)) { e->rc++; dom_unlock(); dom_destroy(d); From 9f5bbaf9995cbf9f771e3a652d0a7ec0d8a61bf3 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:56 +0300 Subject: [PATCH 15/51] fix: REPL, profiler and system builtins build and work on Windows - term.h/profile.h include platform.h and a lean ; - term_write for the Windows console, errno.h and core count in the REPL; - .sys.info reports page-size and total-mem on Windows too; - KEY_READ no longer collides with . Co-Authored-By: Claude Opus 5 (1M context) --- src/app/repl.c | 6 +++++- src/app/term.c | 7 ++++++- src/app/term.h | 5 ++++- src/core/profile.h | 3 +++ src/ops/query.c | 4 +++- src/ops/system.c | 16 +++++++++++++++- 6 files changed, 36 insertions(+), 5 deletions(-) diff --git a/src/app/repl.c b/src/app/repl.c index 6775c890..7036c63c 100644 --- a/src/app/repl.c +++ b/src/app/repl.c @@ -51,6 +51,7 @@ #include #include #include +#include #if defined(RAY_OS_WINDOWS) #include @@ -59,7 +60,6 @@ #define STDIN_FD 0 #else #include -#include #include #define STDIN_FD STDIN_FILENO #endif @@ -385,7 +385,11 @@ static void print_banner(void) { char cpu[256]; get_cpu_name(cpu, sizeof(cpu)); int64_t mem_mb = get_total_mem_mb(); +#if defined(RAY_OS_WINDOWS) + int ncores = (int)ray_thread_count(); +#else int ncores = (int)sysconf(_SC_NPROCESSORS_ONLN); +#endif /* "Using" count reflects the actual worker-pool size, not ncores. * ray_pool_get() is a lazy initializer — callers might not have diff --git a/src/app/term.c b/src/app/term.c index ffae9548..79509f18 100644 --- a/src/app/term.c +++ b/src/app/term.c @@ -67,7 +67,12 @@ typedef struct stat hist_stat_t; #define RAY_BLOCK_FROM_DATA(ptr) ((ray_t*)((char*)(ptr) - sizeof(ray_t))) /* Suppress -Wunused-result for terminal I/O writes to stdout. */ -#if !defined(RAY_OS_WINDOWS) +#if defined(RAY_OS_WINDOWS) +static inline void term_write(const void* buf, size_t len) { + int r = _write(1, buf, (unsigned)len); + (void)r; +} +#else static inline void term_write(const void* buf, size_t len) { ssize_t r = write(STDOUT_FILENO, buf, len); (void)r; diff --git a/src/app/term.h b/src/app/term.h index cf89457b..8c7f23fa 100644 --- a/src/app/term.h +++ b/src/app/term.h @@ -24,9 +24,12 @@ #ifndef RAY_TERM_H #define RAY_TERM_H -#include +#include "core/platform.h" #if defined(RAY_OS_WINDOWS) +#ifndef WIN32_LEAN_AND_MEAN +#define WIN32_LEAN_AND_MEAN /* keep / macros out */ +#endif #include #define KEYCODE_RETURN '\r' #else diff --git a/src/core/profile.h b/src/core/profile.h index 02a71ab6..cc5ab6ad 100644 --- a/src/core/profile.h +++ b/src/core/profile.h @@ -28,6 +28,9 @@ #include #if defined(RAY_OS_WINDOWS) +#ifndef WIN32_LEAN_AND_MEAN +#define WIN32_LEAN_AND_MEAN /* keep / macros out */ +#endif #include #else #include diff --git a/src/ops/query.c b/src/ops/query.c index 2008af99..056d2388 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -11017,7 +11017,9 @@ ray_t* ray_select(ray_t** args, int64_t n) { /* Type-aware key element reader. Normalizes any * comparable scalar key into an int64_t so linear * scans can use equality. For floats we bitcast so - * NaN and -0/+0 match the DAG's hash-equality. */ + * NaN and -0/+0 match the DAG's hash-equality. + * (via ) owns the name on Windows. */ + #undef KEY_READ #define KEY_READ(dst, vec, base_type, idx) do { \ const void* _d = ray_data(vec); \ switch (base_type) { \ diff --git a/src/ops/system.c b/src/ops/system.c index 17641564..6d5a55cc 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -65,6 +65,7 @@ void* ray_runtime_get_sys_args(void); #define RAY_POPEN(c, m) popen((c), (m)) #define RAY_PCLOSE(f) pclose(f) #else +#include /* access, F_OK */ #define RAY_POPEN(c, m) _popen((c), (m)) #define RAY_PCLOSE(f) _pclose(f) #endif @@ -1532,10 +1533,23 @@ ray_t* ray_sysinfo_fn(ray_t** args, int64_t n) { ray_t* v3 = make_i64(ray_sys_total_ram()); vals = ray_list_append(vals, v3); ray_release(v3); #else + SYSTEM_INFO si; + GetSystemInfo(&si); + int64_t s1 = ray_sym_intern("cores", 5); keys = ray_vec_append(keys, &s1); - ray_t* v1 = make_i64(1); + ray_t* v1 = make_i64((int64_t)si.dwNumberOfProcessors); vals = ray_list_append(vals, v1); ray_release(v1); + + int64_t s2 = ray_sym_intern("page-size", 9); + keys = ray_vec_append(keys, &s2); + ray_t* v2 = make_i64((int64_t)si.dwPageSize); + vals = ray_list_append(vals, v2); ray_release(v2); + + int64_t s3 = ray_sym_intern("total-mem", 9); + keys = ray_vec_append(keys, &s3); + ray_t* v3 = make_i64(ray_sys_total_ram()); + vals = ray_list_append(vals, v3); ray_release(v3); #endif /* Process and host identity (#573). An embedded process previously had From 0807e1cb7b2f4fb4a7681a961c1904f2ab2cf4dd Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:57 +0300 Subject: [PATCH 16/51] test: run the suite on Windows - runner: ';; @requires: posix' marks .rfl files whose fixtures or checks need a POSIX shell/filesystem; on Windows they are reported as SKIP; - shell-free ray_test_rm_rf / ray_test_mkdir_p replace system("rm -rf") and "mkdir -p" in C tests; test.h maps the few POSIX helpers tests use; - tests of POSIX-only behaviour (setrlimit, ENOTDIR, read-only dirs, AF_UNIX, file spill, Winsock send buffering) skip with the reason; - fixture fixes that were latent on any platform: binary-mode CSV fixtures, a per-test AOF dir, journal closed before the crash rename. Co-Authored-By: Claude Opus 5 (1M context) --- test/ipc_harness.h | 5 + test/main.c | 62 ++++++++++++ test/rfl/collection/atomic_map_coverage.rfl | 1 + test/rfl/collection/collection_branch_cov.rfl | 1 + test/rfl/collection/cov3.rfl | 1 + test/rfl/collection/cov4.rfl | 1 + test/rfl/collection/cov5.rfl | 1 + test/rfl/collection/cov6.rfl | 1 + test/rfl/group/group_key_types.rfl | 1 + test/rfl/io/csv_branch_cov.rfl | 1 + test/rfl/io/csv_parallel_scan.rfl | 1 + test/rfl/io/csv_rayfall_temporal.rfl | 1 + test/rfl/io/csv_types.rfl | 1 + test/rfl/io/read_until_eof.rfl | 1 + test/rfl/journal/ops_journal.rfl | 1 + test/rfl/journal/ops_journal_purge.rfl | 1 + test/rfl/lang/guid_entropy.rfl | 1 + test/rfl/lang/parse_branch_cov.rfl | 1 + test/rfl/linkop/coverage.rfl | 1 + test/rfl/null/slice_has_nulls.rfl | 1 + test/rfl/null/sort_null_placement.rfl | 1 + test/rfl/null/sym_str_null_compare.rfl | 1 + test/rfl/ops/internal_coverage.rfl | 1 + test/rfl/query/update_by_sym_width.rfl | 1 + test/rfl/regress/load_select_isnull_mask.rfl | 1 + test/rfl/sort/sort_coverage2.rfl | 1 + test/rfl/storage/shared_sym_domain.rfl | 1 + test/rfl/storage/splay_coverage.rfl | 1 + test/rfl/store/col_format_generation.rfl | 1 + test/rfl/store/indexed_col_aux_refs.rfl | 1 + test/rfl/strop/like_seen_proj.rfl | 1 + test/rfl/strop/string_branch_cov.rfl | 1 + test/rfl/symbol/sym_coverage.rfl | 1 + test/rfl/system/cli_flag_values.rfl | 1 + test/rfl/system/csv_auto_int_width.rfl | 1 + .../rfl/system/csv_explicit_numeric_types.rfl | 1 + test/rfl/system/db_get.rfl | 1 + test/rfl/system/db_parted_fill.rfl | 1 + test/rfl/system/db_sym_resolution.rfl | 1 + test/rfl/system/ipc_diff.rfl | 1 + test/rfl/system/ipc_first_last.rfl | 1 + test/rfl/system/ipc_open_errors.rfl | 1 + test/rfl/system/ipc_open_timeout.rfl | 1 + test/rfl/system/listen_fatal.rfl | 1 + test/rfl/system/load_errors.rfl | 1 + test/rfl/system/load_home.rfl | 1 + test/rfl/system/log_journal.rfl | 1 + test/rfl/system/log_journal_advanced.rfl | 1 + test/rfl/system/os_fs.rfl | 1 + test/rfl/system/part.rfl | 1 + test/rfl/system/part_branch_cov.rfl | 1 + test/rfl/system/piped_timers.rfl | 1 + test/rfl/system/process_identity.rfl | 1 + test/rfl/system/querylog_ipc.rfl | 1 + test/rfl/system/read_csv.rfl | 1 + test/rfl/system/startup_script_fatal.rfl | 1 + test/rfl/system/system_branch_cov.rfl | 1 + test/rfl/system/timer_overrun.rfl | 1 + test/stress_store.c | 36 +++++-- test/test.h | 54 +++++++++++ test/test_aof.c | 7 +- test/test_csv.c | 70 ++++++------- test/test_heap.c | 31 ++++++ test/test_ipc.c | 97 +++++++++++++------ test/test_journal.c | 64 +++++++++++- test/test_link.c | 24 +++-- test/test_mcast.c | 30 +++--- test/test_repl.c | 6 +- test/test_runtime.c | 14 ++- test/test_splay.c | 43 ++++---- test/test_store.c | 53 +++++----- test/test_traverse.c | 8 +- 72 files changed, 496 insertions(+), 164 deletions(-) diff --git a/test/ipc_harness.h b/test/ipc_harness.h index 5f9066dd..8dad45ae 100644 --- a/test/ipc_harness.h +++ b/test/ipc_harness.h @@ -48,8 +48,13 @@ #include "core/runtime.h" #include "mem/sys.h" #include "lang/internal.h" +#ifdef RAY_OS_WINDOWS +#include +#include +#else #include #include +#endif #include #include diff --git a/test/main.c b/test/main.c index 6cf59894..42426e6a 100644 --- a/test/main.c +++ b/test/main.c @@ -50,6 +50,7 @@ #include "lang/format.h" #include "ops/internal.h" #include "ops/idxop.h" +#include "store/fileio.h" /* ray_mkdir_p — ray_test_mkdir_p */ /* __RUNTIME is internal test plumbing; runtime API declarations come from * . */ @@ -376,6 +377,16 @@ static test_result_t run_rfl_file(const char* path) { src[r] = '\0'; fclose(f); + /* ";; @requires: posix" anywhere in a file marks it as depending on a + * POSIX shell / filesystem (.sys.exec pipelines, /proc). Where that + * does not exist the file is reported as SKIP, never silently dropped. */ +#if defined(_WIN32) + if (strstr(src, ";; @requires: posix")) { + free(src); + SKIP("requires POSIX shell/filesystem"); + } +#endif + int line_no = 0; int assert_count = 0; /* tallies LHS -- RHS and EXPR !- SUBSTR lines */ char* p = src; @@ -848,7 +859,58 @@ static int name_matches_filter(const char* name, const char* filter) { return strstr(name, filter) != NULL; } +/* ---- Shell-free filesystem helpers (declared in test.h) ---- */ + +int ray_test_rm_rf(const char* path) { + struct stat st; + if (lstat(path, &st) != 0) return 0; /* already gone */ + if (S_ISDIR(st.st_mode)) { + DIR* d = opendir(path); + if (d) { + struct dirent* ent; + char child[4096]; + while ((ent = readdir(d)) != NULL) { + if (strcmp(ent->d_name, ".") == 0 || strcmp(ent->d_name, "..") == 0) + continue; + snprintf(child, sizeof(child), "%s/%s", path, ent->d_name); + ray_test_rm_rf(child); + } + closedir(d); + } + return rmdir(path); + } + return unlink(path); +} + +int ray_test_mkdir_p(const char* path) { + return ray_mkdir_p(path) == RAY_OK ? 0 : -1; +} + +#if defined(_WIN32) +#define WIN32_LEAN_AND_MEAN +#define NOMINMAX +#include +long ray_test_sysconf(int name) { + SYSTEM_INFO si; + GetSystemInfo(&si); + switch (name) { + case _SC_PAGESIZE: return (long)si.dwPageSize; + case _SC_NPROCESSORS_ONLN: return (long)si.dwNumberOfProcessors; + case _SC_PHYS_PAGES: { + MEMORYSTATUSEX ms; + ms.dwLength = sizeof(ms); + if (!GlobalMemoryStatusEx(&ms)) return -1; + return (long)(ms.ullTotalPhys / si.dwPageSize); + } + default: errno = EINVAL; return -1; + } +} +#endif + int main(int argc, char** argv) { +#if defined(_WIN32) + (void)_mkdir("/tmp"); /* tests use "/tmp/..." paths (see test.h) */ +#endif ray_expr_stats_init(); ray_idx_stats_init(); g_color = isatty(fileno(stdout)); diff --git a/test/rfl/collection/atomic_map_coverage.rfl b/test/rfl/collection/atomic_map_coverage.rfl index 7eba0de7..28c7e7b2 100644 --- a/test/rfl/collection/atomic_map_coverage.rfl +++ b/test/rfl/collection/atomic_map_coverage.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; atomic_map_binary_op / atomic_map_unary coverage. ;; Exercises boxed-list paths, empty collections, nested auto-map, ;; recursive map, and error propagation. diff --git a/test/rfl/collection/collection_branch_cov.rfl b/test/rfl/collection/collection_branch_cov.rfl index b187b9e0..dc0e28bb 100644 --- a/test/rfl/collection/collection_branch_cov.rfl +++ b/test/rfl/collection/collection_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; collection_branch_cov.rfl — branch coverage for src/ops/collection.c ;; ;; Targets uncovered branches identified at 64.84% baseline. Each section diff --git a/test/rfl/collection/cov3.rfl b/test/rfl/collection/cov3.rfl index 0c2e08f5..dc23bea6 100644 --- a/test/rfl/collection/cov3.rfl +++ b/test/rfl/collection/cov3.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; cov3.rfl — additional targeted coverage for collection.c remaining gaps ;; Focuses on: atom_eq LIST path, propagate_sym_dict, list_to_typed_vec empty SYM/STR, ;; take STR range out-of-bounds, take dict with typed vals, diff --git a/test/rfl/collection/cov4.rfl b/test/rfl/collection/cov4.rfl index 61b0a9bb..07923cbf 100644 --- a/test/rfl/collection/cov4.rfl +++ b/test/rfl/collection/cov4.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; cov4 — targeted coverage for collection.c remaining gaps ;; Focuses on: atom_eq different-length vecs, range-take type errors, ;; STR typed vec from CSV, STR range-take out-of-bounds, diff --git a/test/rfl/collection/cov5.rfl b/test/rfl/collection/cov5.rfl index 0dd92350..2be5c80b 100644 --- a/test/rfl/collection/cov5.rfl +++ b/test/rfl/collection/cov5.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; cov5 — targeted coverage: distinct_sort_cmp default branch (lines 282-291) ;; ;; F32 (type=6) is not in hs_hash_row switch → hashes by index (all "distinct"). diff --git a/test/rfl/collection/cov6.rfl b/test/rfl/collection/cov6.rfl index d6517d32..efd9d21b 100644 --- a/test/rfl/collection/cov6.rfl +++ b/test/rfl/collection/cov6.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; cov6 — targeted coverage: parted_to_flat_vec STR path (lines 778-790) ;; ;; parted_to_flat_vec has two branches: diff --git a/test/rfl/group/group_key_types.rfl b/test/rfl/group/group_key_types.rfl index 8c868fe2..aca43f3f 100644 --- a/test/rfl/group/group_key_types.rfl +++ b/test/rfl/group/group_key_types.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage for group.c — key type diversity paths ;; ;; Targets: diff --git a/test/rfl/io/csv_branch_cov.rfl b/test/rfl/io/csv_branch_cov.rfl index 562e0e21..93eead2d 100644 --- a/test/rfl/io/csv_branch_cov.rfl +++ b/test/rfl/io/csv_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Branch coverage for src/io/csv.c — targets reachable functional branches ;; left uncovered by csv_types.rfl, csv_round2.rfl, csv_splayed.rfl, ;; system/{read,write}_csv.rfl, and test/test_csv.c. diff --git a/test/rfl/io/csv_parallel_scan.rfl b/test/rfl/io/csv_parallel_scan.rfl index d8b0e737..470235d4 100644 --- a/test/rfl/io/csv_parallel_scan.rfl +++ b/test/rfl/io/csv_parallel_scan.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Parallel row-offset scan (csv.c build_row_offsets_par) — exactness. ;; ;; The scan splits the file into chunks and reconciles quote parity across diff --git a/test/rfl/io/csv_rayfall_temporal.rfl b/test/rfl/io/csv_rayfall_temporal.rfl index 5163c192..6f18420c 100644 --- a/test/rfl/io/csv_rayfall_temporal.rfl +++ b/test/rfl/io/csv_rayfall_temporal.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; CSV type inference must recognize Rayfall's display temporal forms. ;; These are the forms users see when DATE/TIMESTAMP values are printed. (.sys.exec "printf 'd,ts\n2024.01.02,2024.01.02D01:02:03.004005006\n' > rf_test_csv_rayfall_temporal.csv") -- 0 diff --git a/test/rfl/io/csv_types.rfl b/test/rfl/io/csv_types.rfl index 671f7366..f2a63358 100644 --- a/test/rfl/io/csv_types.rfl +++ b/test/rfl/io/csv_types.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage for src/io/csv.c — type inference, edge cases, parted writer. ;; ;; Targets (by approximate line number): diff --git a/test/rfl/io/read_until_eof.rfl b/test/rfl/io/read_until_eof.rfl index a4ea55ab..2fe117c3 100644 --- a/test/rfl/io/read_until_eof.rfl +++ b/test/rfl/io/read_until_eof.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; `read` / `read-bytes` must read until EOF, not to the size the file ;; reports — issue #572. ;; diff --git a/test/rfl/journal/ops_journal.rfl b/test/rfl/journal/ops_journal.rfl index b962325c..0c04bd19 100644 --- a/test/rfl/journal/ops_journal.rfl +++ b/test/rfl/journal/ops_journal.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage extension for src/ops/journal.c. ;; ;; The bulk of src/ops/journal.c is exercised by diff --git a/test/rfl/journal/ops_journal_purge.rfl b/test/rfl/journal/ops_journal_purge.rfl index e24369c1..f21da713 100644 --- a/test/rfl/journal/ops_journal_purge.rfl +++ b/test/rfl/journal/ops_journal_purge.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage for .log.purge (src/ops/journal.c ray_log_purge_fn -> ;; src/store/journal.c ray_journal_purge) — issue #279. ;; diff --git a/test/rfl/lang/guid_entropy.rfl b/test/rfl/lang/guid_entropy.rfl index b77ceadb..6d4e5511 100644 --- a/test/rfl/lang/guid_entropy.rfl +++ b/test/rfl/lang/guid_entropy.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; `guid` must be unique across processes — issue #571. ;; ;; The generator seeded its per-thread xorshift state from rand(), and diff --git a/test/rfl/lang/parse_branch_cov.rfl b/test/rfl/lang/parse_branch_cov.rfl index 046871d5..44ca9bf0 100644 --- a/test/rfl/lang/parse_branch_cov.rfl +++ b/test/rfl/lang/parse_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Branch-coverage extension for src/lang/parse.c — the Rayfall parser ;; (tokenizer, atom-literal parsing, list/dict/vector syntax, comments, ;; escape sequences, error recovery). diff --git a/test/rfl/linkop/coverage.rfl b/test/rfl/linkop/coverage.rfl index 6aebc2bb..6bba5bea 100644 --- a/test/rfl/linkop/coverage.rfl +++ b/test/rfl/linkop/coverage.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage workout for src/ops/linkop.c ;; Targets the regions NOT exercised by test/test_link.c: ;; - ray_col_link_fn error paths (lines 291, 293) diff --git a/test/rfl/null/slice_has_nulls.rfl b/test/rfl/null/slice_has_nulls.rfl index c30a383e..60e0e980 100644 --- a/test/rfl/null/slice_has_nulls.rfl +++ b/test/rfl/null/slice_has_nulls.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; slice_has_nulls.rfl — a slice sees its parent's nulls, in memory and on ;; disk (#495). ;; diff --git a/test/rfl/null/sort_null_placement.rfl b/test/rfl/null/sort_null_placement.rfl index 7f11b4a4..6a71ba30 100644 --- a/test/rfl/null/sort_null_placement.rfl +++ b/test/rfl/null/sort_null_placement.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Where a null lands in a sort, pinned for every path that can run. ;; ;; A null is the SMALLEST value: ascending puts nulls first, descending last. diff --git a/test/rfl/null/sym_str_null_compare.rfl b/test/rfl/null/sym_str_null_compare.rfl index 40a13b4f..bde1fa4e 100644 --- a/test/rfl/null/sym_str_null_compare.rfl +++ b/test/rfl/null/sym_str_null_compare.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Equality comparison on NULLABLE SYM / STR columns — fused path vs unfused. ;; ;; A SYM null is sym id 0 and a STR null is a zero-length descriptor: both are diff --git a/test/rfl/ops/internal_coverage.rfl b/test/rfl/ops/internal_coverage.rfl index 15633da0..a033b33a 100644 --- a/test/rfl/ops/internal_coverage.rfl +++ b/test/rfl/ops/internal_coverage.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage for src/ops/internal.h static-inline helpers that are ;; instantiated in production TUs (exec.c, filter.c, expr.c, etc.) ;; but have never been exercised through the test suite. diff --git a/test/rfl/query/update_by_sym_width.rfl b/test/rfl/query/update_by_sym_width.rfl index dc52d022..c9b3d38a 100644 --- a/test/rfl/query/update_by_sym_width.rfl +++ b/test/rfl/query/update_by_sym_width.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Regression: `update ... by: ` on a narrow-width SYM key column. ;; ;; SYM columns use an adaptive dictionary-index width (W8/W16/W32/W64 in attrs, diff --git a/test/rfl/regress/load_select_isnull_mask.rfl b/test/rfl/regress/load_select_isnull_mask.rfl index d0fffb14..db46bbd4 100644 --- a/test/rfl/regress/load_select_isnull_mask.rfl +++ b/test/rfl/regress/load_select_isnull_mask.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Regression: load → select ISNULL WHERE-mask corruption. ;; ;; exec_elementwise_unary once had dedicated ISNULL kernels for only some diff --git a/test/rfl/sort/sort_coverage2.rfl b/test/rfl/sort/sort_coverage2.rfl index 3d1dec16..e0b34c28 100644 --- a/test/rfl/sort/sort_coverage2.rfl +++ b/test/rfl/sort/sort_coverage2.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Pass-7 additional sort.c coverage. ;; ;; Targets uncovered regions NOT hit by sort_coverage.rfl: diff --git a/test/rfl/storage/shared_sym_domain.rfl b/test/rfl/storage/shared_sym_domain.rfl index 119d4901..780677f7 100644 --- a/test/rfl/storage/shared_sym_domain.rfl +++ b/test/rfl/storage/shared_sym_domain.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; The client layout, end-to-end through the language surface (sym-domain ;; architecture, Task 7b): a parted `hist` plus a splayed `live` sharing ;; ONE symfile (root/.sym) — write, read, query, join (same-domain fast diff --git a/test/rfl/storage/splay_coverage.rfl b/test/rfl/storage/splay_coverage.rfl index ffc09d7f..51b1be34 100644 --- a/test/rfl/storage/splay_coverage.rfl +++ b/test/rfl/storage/splay_coverage.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage extension for src/store/splay.c. ;; ;; src/store/splay.c is exercised by: diff --git a/test/rfl/store/col_format_generation.rfl b/test/rfl/store/col_format_generation.rfl index a5d3cccc..d9b2de41 100644 --- a/test/rfl/store/col_format_generation.rfl +++ b/test/rfl/store/col_format_generation.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; On-disk column format generation (the header `order` byte, offset 17). ;; ;; Regression: an engine build briefly stamped the generation to 1 and demanded diff --git a/test/rfl/store/indexed_col_aux_refs.rfl b/test/rfl/store/indexed_col_aux_refs.rfl index 962539ea..0d2731be 100644 --- a/test/rfl/store/indexed_col_aux_refs.rfl +++ b/test/rfl/store/indexed_col_aux_refs.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; An indexed column still owns what sits at aux bytes 8-15. ;; ;; A SYM column keeps its resolution domain there, a STR column its pool. diff --git a/test/rfl/strop/like_seen_proj.rfl b/test/rfl/strop/like_seen_proj.rfl index 84aa9ec2..5e78351f 100644 --- a/test/rfl/strop/like_seen_proj.rfl +++ b/test/rfl/strop/like_seen_proj.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Targeted parallel SYM-LIKE coverage for like_seen_fn / like_proj_fn. ;; ;; Both kernels are the worker bodies dispatched by ray_pool_dispatch diff --git a/test/rfl/strop/string_branch_cov.rfl b/test/rfl/strop/string_branch_cov.rfl index 24838bfb..27b26bb8 100644 --- a/test/rfl/strop/string_branch_cov.rfl +++ b/test/rfl/strop/string_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; string_branch_cov.rfl -- targeted branch coverage for src/ops/string.c ;; ;; Baseline: 61.41% (345 uncovered branches). This file targets the diff --git a/test/rfl/symbol/sym_coverage.rfl b/test/rfl/symbol/sym_coverage.rfl index 1b9e2b4a..01a23e7d 100644 --- a/test/rfl/symbol/sym_coverage.rfl +++ b/test/rfl/symbol/sym_coverage.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Coverage extension for src/table/sym.c. ;; ;; src/table/sym.c is exercised by many existing tests via CSV/splayed I/O. diff --git a/test/rfl/system/cli_flag_values.rfl b/test/rfl/system/cli_flag_values.rfl index c014eada..2eb4e703 100644 --- a/test/rfl/system/cli_flag_values.rfl +++ b/test/rfl/system/cli_flag_values.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; cli_flag_values.rfl — a value-taking flag must not swallow the next flag, ;; and an unknown option must not be mistaken for the script (#600). ;; diff --git a/test/rfl/system/csv_auto_int_width.rfl b/test/rfl/system/csv_auto_int_width.rfl index fd8ff119..91c37ea7 100644 --- a/test/rfl/system/csv_auto_int_width.rfl +++ b/test/rfl/system/csv_auto_int_width.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; `INT` schema token → auto narrowest integer width on .csv.splayed. ;; ;; A column declared `INT` in a .csv.splayed schema is parsed as int64, diff --git a/test/rfl/system/csv_explicit_numeric_types.rfl b/test/rfl/system/csv_explicit_numeric_types.rfl index d053c520..facf8928 100644 --- a/test/rfl/system/csv_explicit_numeric_types.rfl +++ b/test/rfl/system/csv_explicit_numeric_types.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Explicit numeric schemas must survive the streaming .csv.splayed writer ;; and the mmap-backed .db.splayed.get reload without widening. diff --git a/test/rfl/system/db_get.rfl b/test/rfl/system/db_get.rfl index cb03cd4a..66f6cf33 100644 --- a/test/rfl/system/db_get.rfl +++ b/test/rfl/system/db_get.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Multi-table roots through the explicit .db trio — the surface left ;; after the .db.*.mount removal: every table is opened by name with ;; .db.splayed.get / .db.parted.get; nothing is discovered or bound diff --git a/test/rfl/system/db_parted_fill.rfl b/test/rfl/system/db_parted_fill.rfl index 2574b271..b0e3a947 100644 --- a/test/rfl/system/db_parted_fill.rfl +++ b/test/rfl/system/db_parted_fill.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; .db.parted.fill — fill missing tables across a parted db's partitions. ;; ;; For every table present in ANY partition, write an empty copy (schema diff --git a/test/rfl/system/db_sym_resolution.rfl b/test/rfl/system/db_sym_resolution.rfl index daed3ae8..56fce717 100644 --- a/test/rfl/system/db_sym_resolution.rfl +++ b/test/rfl/system/db_sym_resolution.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; SYM-domain symfile resolution precedence, end-to-end (sym-domain ;; architecture spec, "Surface"): ;; explicit argument | dir/.sym | partition-shaped parent -> root/.sym diff --git a/test/rfl/system/ipc_diff.rfl b/test/rfl/system/ipc_diff.rfl index e02f1f08..31f627bd 100644 --- a/test/rfl/system/ipc_diff.rfl +++ b/test/rfl/system/ipc_diff.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Differential IPC oracle. ;; ;; Spawns a fresh `./rayforce -p PORT` process as an IPC server, then diff --git a/test/rfl/system/ipc_first_last.rfl b/test/rfl/system/ipc_first_last.rfl index fcd36642..b468e0bd 100644 --- a/test/rfl/system/ipc_first_last.rfl +++ b/test/rfl/system/ipc_first_last.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Regression: `first`/`last` returned over IPC used to hang the client ;; forever (issue #285). ;; diff --git a/test/rfl/system/ipc_open_errors.rfl b/test/rfl/system/ipc_open_errors.rfl index 525da04f..7760319a 100644 --- a/test/rfl/system/ipc_open_errors.rfl +++ b/test/rfl/system/ipc_open_errors.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; ipc_open_errors.rfl — `.ipc.open` names the failure it actually hit. ;; ;; Regression (#472): the connect and the wire handshake share one budget, diff --git a/test/rfl/system/ipc_open_timeout.rfl b/test/rfl/system/ipc_open_timeout.rfl index 6a525b1b..896254d6 100644 --- a/test/rfl/system/ipc_open_timeout.rfl +++ b/test/rfl/system/ipc_open_timeout.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Connect timeout argument for `.ipc.open` (issue #286). ;; ;; `.ipc.open` is now a variadic builtin accepting an optional second diff --git a/test/rfl/system/listen_fatal.rfl b/test/rfl/system/listen_fatal.rfl index 7417ae32..d4f93aa4 100644 --- a/test/rfl/system/listen_fatal.rfl +++ b/test/rfl/system/listen_fatal.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; listen_fatal.rfl — a `-p` that cannot bind is fatal (#473). ;; ;; Regression: when ray_ipc_listen_at failed, main printed diff --git a/test/rfl/system/load_errors.rfl b/test/rfl/system/load_errors.rfl index f2fdcf82..99959bf6 100644 --- a/test/rfl/system/load_errors.rfl +++ b/test/rfl/system/load_errors.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; load_errors.rfl — a failing `load` names the file and the OS cause (#505). ;; ;; Regression: ray_load_file_fn returned a bare `io` for every failure, so diff --git a/test/rfl/system/load_home.rfl b/test/rfl/system/load_home.rfl index 4ce8575e..febcad8e 100644 --- a/test/rfl/system/load_home.rfl +++ b/test/rfl/system/load_home.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; load_home.rfl — relative `load` paths fall back to RAYFORCE_HOME, and ;; (.sys.args) reports the file being evaluated as `source` (#506). ;; diff --git a/test/rfl/system/log_journal.rfl b/test/rfl/system/log_journal.rfl index bca8730f..a6d1e753 100644 --- a/test/rfl/system/log_journal.rfl +++ b/test/rfl/system/log_journal.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Journal end-to-end — transaction-log journaling (-l/-L). ;; ;; Two named processes are spawned in sequence under -l ; this diff --git a/test/rfl/system/log_journal_advanced.rfl b/test/rfl/system/log_journal_advanced.rfl index 7d1a7e22..e0916179 100644 --- a/test/rfl/system/log_journal_advanced.rfl +++ b/test/rfl/system/log_journal_advanced.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Journal feature — invariants the basic test (log_journal.rfl) does ;; not cover. Each phase uses a distinct base + port so a phase ;; failing mid-run doesn't pollute the next one's state. diff --git a/test/rfl/system/os_fs.rfl b/test/rfl/system/os_fs.rfl index 9bccd27e..76630f81 100644 --- a/test/rfl/system/os_fs.rfl +++ b/test/rfl/system/os_fs.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; .fs.size and .fs.list — filesystem metadata primitives, issue #36. ;; ;; Two functions on purpose: every other predicate (exists, is-file, diff --git a/test/rfl/system/part.rfl b/test/rfl/system/part.rfl index 9084f24f..0af05319 100644 --- a/test/rfl/system/part.rfl +++ b/test/rfl/system/part.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; src/store/part.c — exercise every reachable branch of ray_read_parted ;; (the function backing .db.parted.get) plus the ;; static helpers infer_mc_type, parse_date_dir, parse_int_dir, diff --git a/test/rfl/system/part_branch_cov.rfl b/test/rfl/system/part_branch_cov.rfl index 9553fc94..5b71277d 100644 --- a/test/rfl/system/part_branch_cov.rfl +++ b/test/rfl/system/part_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; src/store/part.c — branch-coverage top-up for ray_read_parted and its ;; static helpers (is_date_dir, is_integer_str, infer_mc_type, ;; parse_date_dir, parse_int_dir, collect_part_dirs). diff --git a/test/rfl/system/piped_timers.rfl b/test/rfl/system/piped_timers.rfl index 83cb6a64..689987ce 100644 --- a/test/rfl/system/piped_timers.rfl +++ b/test/rfl/system/piped_timers.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; piped_timers.rfl — timers keep firing while a piped (non-TTY) stdin is ;; held open, and a process stays for its pending timers after input ends ;; (#493, the narrowed report). diff --git a/test/rfl/system/process_identity.rfl b/test/rfl/system/process_identity.rfl index d0baa387..af3e5183 100644 --- a/test/rfl/system/process_identity.rfl +++ b/test/rfl/system/process_identity.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Process and host identity, plus .sys.exec output capture — issue #573. ;; ;; An embedded process could not learn anything about itself: no PID, no diff --git a/test/rfl/system/querylog_ipc.rfl b/test/rfl/system/querylog_ipc.rfl index f65b18c2..08dd04fb 100644 --- a/test/rfl/system/querylog_ipc.rfl +++ b/test/rfl/system/querylog_ipc.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Query-statistics ring end-to-end over IPC (the server / cloud path). ;; ;; The capture hook lives in eval_payload_core (src/core/ipc.c), so it only diff --git a/test/rfl/system/read_csv.rfl b/test/rfl/system/read_csv.rfl index 5ecd1080..fcbe301f 100644 --- a/test/rfl/system/read_csv.rfl +++ b/test/rfl/system/read_csv.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; Ported from test_lang_rf.inc::test_rf_read_csv. ;; Rayfall's str-pool hits "error: limit" when raze/fold accumulates ;; ~1000 strings, so we shell out via .sys.exec to write the 20k-row diff --git a/test/rfl/system/startup_script_fatal.rfl b/test/rfl/system/startup_script_fatal.rfl index 34373856..d935cc27 100644 --- a/test/rfl/system/startup_script_fatal.rfl +++ b/test/rfl/system/startup_script_fatal.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; startup_script_fatal.rfl — a failing startup script under -p is fatal ;; non-interactively (#507). ;; diff --git a/test/rfl/system/system_branch_cov.rfl b/test/rfl/system/system_branch_cov.rfl index 32adf1dd..bb67d9ed 100644 --- a/test/rfl/system/system_branch_cov.rfl +++ b/test/rfl/system/system_branch_cov.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; system_branch_cov.rfl — branch coverage for src/ops/system.c (part 1). ;; ;; Covers: ser/de, splayed set/get, parted get, os.size, os.list, diff --git a/test/rfl/system/timer_overrun.rfl b/test/rfl/system/timer_overrun.rfl index 9bd67729..125b2254 100644 --- a/test/rfl/system/timer_overrun.rfl +++ b/test/rfl/system/timer_overrun.rfl @@ -1,3 +1,4 @@ +;; @requires: posix (fixtures/checks run through a POSIX shell via .sys.exec) ;; timer_overrun.rfl — a periodic timer does not replay missed intervals, ;; and a failing callback's line carries the message (#474). ;; diff --git a/test/stress_store.c b/test/stress_store.c index 71ac326d..c92453e4 100644 --- a/test/stress_store.c +++ b/test/stress_store.c @@ -4,6 +4,11 @@ * Rayforce heap under test. */ +#if !defined(_WIN32) && !defined(_GNU_SOURCE) +#define _GNU_SOURCE /* lstat under strict -std=c17 */ +#endif + +#include "test.h" /* POSIX shims on Windows (lstat) */ #include "stress_store.h" #include "store/splay.h" #include "store/part.h" @@ -14,6 +19,9 @@ #include #include #include /* getpid — per-process scratch paths */ +#include +#include +#include "store/fileio.h" /* ray_mkdir_p */ const char* stress_db_path(const char* name) { static char buf[256]; @@ -148,10 +156,28 @@ void stress_part_dir(const stress_ctx_t* c, int i, char* buf, size_t n) { snprintf(buf, n, "%s/%s/hist", c->db_root, c->part_dates[i]); } +/* Recursive delete without a shell (portable to Windows, where system() + * runs cmd.exe and has no `rm -rf`). */ static void rm_rf(const char* path) { - char cmd[600]; - snprintf(cmd, sizeof(cmd), "rm -rf '%s'", path); - (void)!system(cmd); + struct stat st; + if (lstat(path, &st) != 0) return; /* never follow a symlink out */ + if (S_ISDIR(st.st_mode)) { + DIR* d = opendir(path); + if (d) { + struct dirent* ent; + char child[1024]; + while ((ent = readdir(d)) != NULL) { + if (strcmp(ent->d_name, ".") == 0 || strcmp(ent->d_name, "..") == 0) + continue; + snprintf(child, sizeof(child), "%s/%s", path, ent->d_name); + rm_rf(child); + } + closedir(d); + } + (void)rmdir(path); + } else { + (void)unlink(path); + } } /* ---- ray table <-> shadow rows ------------------------------------------ */ @@ -276,9 +302,7 @@ bool stress_init(stress_ctx_t* c, const char* db_root, uint64_t seed) { c->oplog = (char(*)[128])malloc((size_t)STRESS_OPLOG_CAP * 128); if (!c->oplog) return false; rm_rf(c->db_root); - char cmd[600]; - snprintf(cmd, sizeof(cmd), "mkdir -p '%s'", c->db_root); - if (system(cmd) != 0) { + if (ray_mkdir_p(c->db_root) != RAY_OK) { free(c->oplog); c->oplog = NULL; return false; diff --git a/test/test.h b/test/test.h index 8a54c42e..8f67db58 100644 --- a/test/test.h +++ b/test/test.h @@ -41,6 +41,60 @@ #include #include +/* Shell-free filesystem helpers (test/main.c). Tests must not depend on + * /bin/sh: on Windows system() runs cmd.exe. Both return 0 on success. */ +int ray_test_rm_rf(const char* path); /* rm -rf path */ +int ray_test_mkdir_p(const char* path); /* mkdir -p path */ + +#if defined(_WIN32) +/* POSIX helpers the tests rely on, mapped onto their MSVCRT equivalents. + * Test paths use "/tmp/...", which Windows resolves to :\tmp; the + * runner creates that directory at startup (see test/main.c). */ +#include +#include +#include +#include +static inline int setenv(const char* k, const char* v, int overwrite) { + if (!overwrite && getenv(k)) return 0; + return _putenv_s(k, v) == 0 ? 0 : -1; +} +static inline int unsetenv(const char* k) { + return _putenv_s(k, "") == 0 ? 0 : -1; /* "" removes the variable */ +} +static inline char* mkdtemp(char* tmpl) { + if (!_mktemp(tmpl)) return NULL; + return _mkdir(tmpl) == 0 ? tmpl : NULL; +} +static inline unsigned geteuid(void) { return 1; } /* never "root" */ +#define lstat stat /* no symlinks to skip */ +#include +#include +static inline int symlink(const char* target, const char* path) { + (void)target; (void)path; + errno = ENOSYS; /* callers skip when symlink fails */ + return -1; +} +#define mkdir(p, mode) _mkdir(p) +#define pipe(fds) _pipe((fds), 65536, _O_BINARY) +/* sysconf subset (page size, physical pages, CPUs); see test/main.c. */ +#define _SC_PAGESIZE 1 +#define _SC_PAGE_SIZE _SC_PAGESIZE +#define _SC_PHYS_PAGES 2 +#define _SC_NPROCESSORS_ONLN 3 +long ray_test_sysconf(int name); +#define sysconf ray_test_sysconf +/* MSVCRT's tmpfile() creates its file in the drive root, which needs admin + * rights. Use the temp directory instead; "D" deletes the file on close. */ +static inline FILE* ray_test_tmpfile(void) { + char* name = _tempnam("/tmp", "rayt"); + if (!name) return NULL; + FILE* f = fopen(name, "w+bD"); + free(name); + return f; +} +#define tmpfile ray_test_tmpfile +#endif + typedef enum { TEST_PASS = 0, TEST_FAIL, TEST_SKIP } test_status_t; typedef struct { diff --git a/test/test_aof.c b/test/test_aof.c index 4705436f..469f5491 100644 --- a/test/test_aof.c +++ b/test/test_aof.c @@ -58,7 +58,12 @@ static void aof_rm_rf(const char* dir) { } static void aof_setup(void) { - snprintf(g_aof_dir, sizeof g_aof_dir, "/tmp/ray_test_aof_%d", (int)getpid()); + /* A fresh dir per test: the crash test deliberately leaks an open + * writer, and on Windows an open file cannot be deleted, so a shared + * dir would hand its stale segment to every later test. */ + static int seq = 0; + snprintf(g_aof_dir, sizeof g_aof_dir, "/tmp/ray_test_aof_%d_%d", + (int)getpid(), seq++); aof_rm_rf(g_aof_dir); } diff --git a/test/test_csv.c b/test/test_csv.c index 9a964fc2..3f015624 100644 --- a/test/test_csv.c +++ b/test/test_csv.c @@ -213,7 +213,7 @@ static test_result_t test_csv_null_i64(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\n\n30\n"); fclose(f); @@ -245,7 +245,7 @@ static test_result_t test_csv_null_i64_unparseable(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\nN/A\n30\n"); fclose(f); @@ -274,7 +274,7 @@ static test_result_t test_csv_null_f64(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n1.5\n\n3.5\n"); fclose(f); @@ -305,7 +305,7 @@ static test_result_t test_csv_null_i16(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\n\n30\n"); fclose(f); @@ -336,7 +336,7 @@ static test_result_t test_csv_null_i32(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\n\n30\n"); fclose(f); @@ -367,7 +367,7 @@ static test_result_t test_csv_null_date(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "d\n2025-01-02\n\n2026-12-31\n"); fclose(f); @@ -396,7 +396,7 @@ static test_result_t test_csv_null_time(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "t\n12:34:56\n\n23:59:59\n"); fclose(f); @@ -425,7 +425,7 @@ static test_result_t test_csv_null_timestamp(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "ts\n2025-01-02T03:04:05\n\n2026-12-31T23:59:59\n"); fclose(f); @@ -455,7 +455,7 @@ static test_result_t test_csv_null_bool(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "flag\ntrue\n\nfalse\n"); fclose(f); @@ -484,7 +484,7 @@ static test_result_t test_csv_null_sym(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "name\nalice\n\nbob\n"); fclose(f); @@ -515,7 +515,7 @@ static test_result_t test_csv_no_nulls_no_null_bitmap(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\n20\n30\n"); fclose(f); @@ -537,7 +537,7 @@ static test_result_t test_csv_null_mixed_columns(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "id,val,name\n1,1.5,alice\n,2.5,\n3,,bob\n"); fclose(f); @@ -579,7 +579,7 @@ static test_result_t test_csv_explicit_str_schema(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); /* Mix inline (<=12B), pooled (>12B), empty/null, and a short */ fprintf(f, "id,note\n" "1,hi\n" @@ -628,7 +628,7 @@ static test_result_t test_csv_escaped_str_roundtrip(void) { (void)ray_sym_init(); /* Write a CSV with fields that require quoting/escaping */ - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "s\n" "\"he,llo\"\n" "\"wo\"\"rld\"\n" @@ -748,7 +748,7 @@ static test_result_t test_csv_infer_date(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "d\n2025-01-02\n2026-12-31\n2000-03-15\n"); fclose(f); @@ -769,7 +769,7 @@ static test_result_t test_csv_infer_time(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "t\n12:34:56\n00:00:00\n23:59:59.123\n"); fclose(f); @@ -792,7 +792,7 @@ static test_result_t test_csv_infer_timestamp_promotion(void) { /* Mix of full timestamps with both 'T' and ' ' separators, plus a * date-only sentinel that should be promoted to TIMESTAMP. */ - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "ts\n2025-01-02T03:04:05\n2025-06-07 08:09:10.123456\n2024-12-31\n"); fclose(f); @@ -813,7 +813,7 @@ static test_result_t test_csv_infer_bool(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "b\ntrue\nfalse\nTRUE\nFALSE\n"); fclose(f); @@ -834,7 +834,7 @@ static test_result_t test_csv_infer_f64_specials(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "v\n1.0\n2e10\n-3.5E-2\nnan\nInf\n+inf\n-INF\n"); fclose(f); @@ -856,7 +856,7 @@ static test_result_t test_csv_infer_null_sentinels(void) { (void)ray_sym_init(); /* Sentinel rows alternating with i64 values; column should infer I64. */ - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n10\nN/A\nNA\nnull\nNULL\nNone\nnone\nn/a\nna\n.\n42\n"); fclose(f); @@ -882,7 +882,7 @@ static test_result_t test_csv_infer_promotions(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "n,b\n1,true\n2,0\n3.5,1\n"); fclose(f); @@ -905,7 +905,7 @@ static test_result_t test_csv_tab_delimiter(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "a\tb\tc\n1\t2\t3\n4\t5\t6\n"); fclose(f); @@ -926,7 +926,7 @@ static test_result_t test_csv_no_header(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "10,20\n30,40\n50,60\n"); fclose(f); @@ -988,7 +988,7 @@ static test_result_t test_csv_invalid_schema_type(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "x\n1\n2\n"); fclose(f); @@ -1002,7 +1002,7 @@ static test_result_t test_csv_invalid_schema_type(void) { /* Schema too short for ncols also errors out. */ int8_t one_only[1] = { RAY_I64 }; - FILE* g = fopen(TMP_CSV, "w"); + FILE* g = fopen(TMP_CSV, "wb"); fprintf(g, "a,b\n1,2\n"); fclose(g); ray_t* loaded3 = ray_read_csv_opts(TMP_CSV, ',', true, one_only, 1); @@ -1044,7 +1044,7 @@ static test_result_t test_csv_truncated_row(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "a,b,c\n1,2,3\n4\n7,8,9\n"); fclose(f); @@ -1277,7 +1277,7 @@ static test_result_t test_csv_parallel_parse(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "i,s\n"); /* 9000 rows so n_rows > 8192. */ for (int i = 0; i < 9000; i++) @@ -1306,7 +1306,7 @@ static test_result_t test_csv_sym_narrowing(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "k\n"); /* Only three distinct values across many rows. */ for (int i = 0; i < 200; i++) @@ -1342,7 +1342,7 @@ static test_result_t test_csv_explicit_u8_schema(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "v\n"); /* 10 000 rows ⇒ parallel parse path; values 0..255 cycling so the * truncated bytes fully exercise the U8 range. */ @@ -1384,7 +1384,7 @@ static test_result_t test_csv_explicit_i16_schema_with_nulls(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "v\n"); const int N = 1500; for (int i = 0; i < N; i++) { @@ -1427,7 +1427,7 @@ static test_result_t test_csv_explicit_i32_schema(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "v\n"); const int N = 500; for (int i = 0; i < N; i++) fprintf(f, "%d\n", -100000 + i * 137); @@ -1460,7 +1460,7 @@ static test_result_t test_csv_explicit_u8_schema_serial(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "a,b\n"); /* 200 rows; second column missing on every 50th row → triggers * past-row-boundary fill in the parser. */ @@ -1501,7 +1501,7 @@ static test_result_t test_csv_infer_high_cardinality_str(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); fprintf(f, "payload\n"); for (int i = 0; i < 100; i++) fprintf(f, "unique_payload_%03d\n", i); @@ -1570,7 +1570,7 @@ static test_result_t test_csv_interrupt_mid_parse(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); TEST_ASSERT_NOT_NULL(f); fputs("id,payload\n", f); for (int i = 0; i < 200000; i++) { @@ -1610,7 +1610,7 @@ static test_result_t test_csv_progress_never_goes_backwards(void) { ray_heap_init(); (void)ray_sym_init(); - FILE* f = fopen(TMP_CSV, "w"); + FILE* f = fopen(TMP_CSV, "wb"); TEST_ASSERT_NOT_NULL(f); fputs("payload,symbol\n", f); for (int i = 0; i < 100000; i++) diff --git a/test/test_heap.c b/test/test_heap.c index 2054f7b1..0e14473f 100644 --- a/test/test_heap.c +++ b/test/test_heap.c @@ -50,7 +50,33 @@ #include #include #include +#if defined(_WIN32) +#define WIN32_LEAN_AND_MEAN +#include +/* Anonymous mmap for the fake file-mapped (mmod==1) blocks below: a + * pagefile-backed view, so the library's ray_vm_unmap_file + * (UnmapViewOfFile) releases it just as it would a real file mapping. */ +#define PROT_READ 1 +#define PROT_WRITE 2 +#define MAP_PRIVATE 2 +#define MAP_ANONYMOUS 0x20 +#define MAP_FAILED ((void*)-1) +static void* mmap(void* addr, size_t len, int prot, int flags, int fd, long off) { + (void)addr; (void)prot; (void)flags; (void)fd; (void)off; + HANDLE m = CreateFileMappingA(INVALID_HANDLE_VALUE, NULL, PAGE_READWRITE, + 0, (DWORD)len, NULL); + if (!m) return MAP_FAILED; + void* p = MapViewOfFile(m, FILE_MAP_WRITE, 0, 0, len); + CloseHandle(m); + return p ? p : MAP_FAILED; +} +static int munmap(void* p, size_t len) { + (void)len; + return UnmapViewOfFile(p) ? 0 : -1; +} +#else #include +#endif #include #include @@ -2565,6 +2591,11 @@ static test_result_t test_direct_cache_concurrent_replacement(void) { /* Drive the anon-to-file crossing with a low watermark, independent of RAM. */ static test_result_t test_anon_watermark_spill(void) { +#if defined(_WIN32) + /* Windows takes only the anonymous path; the file-backed spill is + * POSIX-only (docs/architecture/memory.md, RAY_HEAP_FILE_SPILL). */ + SKIP("file-backed spill is POSIX-only"); +#endif size_t sz = 40 * 1024 * 1024 - 128; /* order 26 → direct path */ /* Start from an empty reuse cache: leftover cached blocks from earlier * tests would (a) inflate the baseline and (b) be drained by the diff --git a/test/test_ipc.c b/test/test_ipc.c index 21b3dfa1..aa1752f0 100644 --- a/test/test_ipc.c +++ b/test/test_ipc.c @@ -54,11 +54,16 @@ #include "test.h" #include "ipc_harness.h" +#ifdef RAY_OS_WINDOWS +#include +#include +#else #include #include #include #include #include +#endif #include #include "core/ipc.h" #include "core/sock.h" @@ -1047,12 +1052,17 @@ static test_result_t test_ipc_send_large_compressible(void) { * Open a journal, then connect an IPC server on top; each SYNC message * should flow through ray_journal_write_bytes. */ +/* Drop a journal's .log / .qdb pair (no shell). */ +static void ipc_rm_journal(const char* jbase) { + char path[256]; + snprintf(path, sizeof(path), "%s.log", jbase); (void)remove(path); + snprintf(path, sizeof(path), "%s.qdb", jbase); (void)remove(path); +} + static test_result_t test_ipc_journal_path(void) { const char* jbase = "/tmp/rayforce_test_ipc_journal"; /* Remove stale files */ - char cmd[256]; - snprintf(cmd, sizeof(cmd), "rm -f %s.log %s.qdb", jbase, jbase); - system(cmd); + ipc_rm_journal(jbase); /* Open journal */ ray_err_t jerr = ray_journal_open(jbase, RAY_JOURNAL_ASYNC); @@ -1081,7 +1091,7 @@ static test_result_t test_ipc_journal_path(void) { ray_test_server_stop(&srv); ray_journal_close(); - system(cmd); /* cleanup */ + ipc_rm_journal(jbase); /* cleanup */ PASS(); } @@ -1453,9 +1463,7 @@ static test_result_t test_ipc_send_verbose_large_result(void) { */ static test_result_t test_ipc_journal_restricted(void) { const char* jbase = "/tmp/rayforce_test_ipc_jrestr"; - char cmd[256]; - snprintf(cmd, sizeof(cmd), "rm -f %s.log %s.qdb", jbase, jbase); - system(cmd); + ipc_rm_journal(jbase); ray_err_t jerr = ray_journal_open(jbase, RAY_JOURNAL_ASYNC); if (jerr != RAY_OK) { @@ -1486,7 +1494,7 @@ static test_result_t test_ipc_journal_restricted(void) { ray_test_server_stop(&srv); ray_journal_close(); - system(cmd); + ipc_rm_journal(jbase); PASS(); } @@ -1979,6 +1987,7 @@ static test_result_t test_ipc_addr_local_ipv6_remote(void) { PASS(); } +#ifndef RAY_OS_WINDOWS /* AF_UNIX locality is POSIX-only (see sock.c) */ static test_result_t test_ipc_addr_local_af_unix(void) { struct sockaddr_un sa; memset(&sa, 0, sizeof(sa)); @@ -1986,6 +1995,7 @@ static test_result_t test_ipc_addr_local_af_unix(void) { TEST_ASSERT_TRUE(ray_sock_addr_is_local(&sa, sizeof(sa))); PASS(); } +#endif static test_result_t test_ipc_addr_local_rejects_garbage(void) { struct sockaddr_in sa; @@ -1998,24 +2008,53 @@ static test_result_t test_ipc_addr_local_rejects_garbage(void) { PASS(); } -/* A real AF_UNIX pair resolves as local through getpeername. */ +/* A connected pair of sockets on this machine: an AF_UNIX socketpair on + * POSIX; Windows has no socketpair(2), so a loopback TCP pair there. */ +static int test_local_sock_pair(ray_sock_t sv[2]) { +#ifdef RAY_OS_WINDOWS + ray_sock_t srv = ray_sock_listen_at("127.0.0.1", 0); + if (srv == RAY_INVALID_SOCK) return -1; + struct sockaddr_in addr; + int len = sizeof(addr); + if (getsockname((SOCKET)srv, (struct sockaddr*)&addr, &len) != 0) { + ray_sock_close(srv); + return -1; + } + sv[0] = ray_sock_connect("127.0.0.1", ntohs(addr.sin_port), 0); + sv[1] = sv[0] == RAY_INVALID_SOCK ? RAY_INVALID_SOCK : ray_sock_accept(srv); + ray_sock_close(srv); + if (sv[1] == RAY_INVALID_SOCK) { + if (sv[0] != RAY_INVALID_SOCK) ray_sock_close(sv[0]); + return -1; + } + return 0; +#else + int fds[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, fds) != 0) return -1; + sv[0] = fds[0]; + sv[1] = fds[1]; + return 0; +#endif +} + +/* A real local pair resolves as local through getpeername. */ static test_result_t test_ipc_peer_is_local_socketpair(void) { - int sv[2]; - TEST_ASSERT_EQ_I(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); - TEST_ASSERT_TRUE(ray_sock_peer_is_local((ray_sock_t)sv[0])); - TEST_ASSERT_TRUE(ray_sock_peer_is_local((ray_sock_t)sv[1])); - close(sv[0]); - close(sv[1]); + ray_sock_t sv[2]; + TEST_ASSERT_EQ_I(test_local_sock_pair(sv), 0); + TEST_ASSERT_TRUE(ray_sock_peer_is_local(sv[0])); + TEST_ASSERT_TRUE(ray_sock_peer_is_local(sv[1])); + ray_sock_close(sv[0]); + ray_sock_close(sv[1]); PASS(); } /* An unconnected socket has no peer: getpeername fails, and an unknown * peer must fall back to the compressing default, not to "local". */ static test_result_t test_ipc_peer_is_local_unconnected(void) { - int fd = socket(AF_INET, SOCK_STREAM, 0); - TEST_ASSERT_TRUE(fd >= 0); - TEST_ASSERT_FALSE(ray_sock_peer_is_local((ray_sock_t)fd)); - close(fd); + ray_sock_t fd = (ray_sock_t)socket(AF_INET, SOCK_STREAM, 0); + TEST_ASSERT_TRUE(fd != RAY_INVALID_SOCK); + TEST_ASSERT_FALSE(ray_sock_peer_is_local(fd)); + ray_sock_close(fd); PASS(); } @@ -2027,18 +2066,18 @@ static test_result_t test_ipc_peer_is_local_invalid_fd(void) { /* The policy the send paths consult: local links never compress, others * keep the compiled-in default. */ static test_result_t test_ipc_link_threshold_local_vs_remote(void) { - int sv[2]; - TEST_ASSERT_EQ_I(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); - TEST_ASSERT_EQ_U(ray_ipc_link_threshold((ray_sock_t)sv[0]), + ray_sock_t sv[2]; + TEST_ASSERT_EQ_I(test_local_sock_pair(sv), 0); + TEST_ASSERT_EQ_U(ray_ipc_link_threshold(sv[0]), RAY_IPC_COMPRESS_NEVER); - close(sv[0]); - close(sv[1]); + ray_sock_close(sv[0]); + ray_sock_close(sv[1]); - int fd = socket(AF_INET, SOCK_STREAM, 0); - TEST_ASSERT_TRUE(fd >= 0); - TEST_ASSERT_EQ_U(ray_ipc_link_threshold((ray_sock_t)fd), + ray_sock_t fd = (ray_sock_t)socket(AF_INET, SOCK_STREAM, 0); + TEST_ASSERT_TRUE(fd != RAY_INVALID_SOCK); + TEST_ASSERT_EQ_U(ray_ipc_link_threshold(fd), (size_t)RAY_IPC_COMPRESS_THRESHOLD); - close(fd); + ray_sock_close(fd); PASS(); } @@ -2747,7 +2786,9 @@ const test_entry_t ipc_entries[] = { { "ipc/addr_local/ipv6_loopback", test_ipc_addr_local_ipv6_loopback, ipc_setup, ipc_teardown }, { "ipc/addr_local/ipv6_mapped_loopback",test_ipc_addr_local_ipv6_mapped_loopback, ipc_setup, ipc_teardown }, { "ipc/addr_local/ipv6_remote", test_ipc_addr_local_ipv6_remote, ipc_setup, ipc_teardown }, +#ifndef RAY_OS_WINDOWS { "ipc/addr_local/af_unix", test_ipc_addr_local_af_unix, ipc_setup, ipc_teardown }, +#endif { "ipc/addr_local/rejects_garbage", test_ipc_addr_local_rejects_garbage, ipc_setup, ipc_teardown }, { "ipc/peer_is_local/socketpair", test_ipc_peer_is_local_socketpair, ipc_setup, ipc_teardown }, { "ipc/peer_is_local/unconnected", test_ipc_peer_is_local_unconnected, ipc_setup, ipc_teardown }, diff --git a/test/test_journal.c b/test/test_journal.c index 8277a729..9904893f 100644 --- a/test/test_journal.c +++ b/test/test_journal.c @@ -40,7 +40,44 @@ #include #include #include +#if defined(_WIN32) +#define WIN32_LEAN_AND_MEAN +#include +/* Minimal glob(3) for the archive checks below: wildcards only in the + * last path component, which is all ".*.log" needs. */ +typedef struct { size_t gl_pathc; char** gl_pathv; } glob_t; +static void globfree(glob_t* g) { + for (size_t i = 0; i < g->gl_pathc; i++) free(g->gl_pathv[i]); + free(g->gl_pathv); + g->gl_pathc = 0; g->gl_pathv = NULL; +} +static int glob(const char* pat, int flags, void* errfunc, glob_t* g) { + (void)flags; (void)errfunc; + g->gl_pathc = 0; g->gl_pathv = NULL; + const char* slash = strrchr(pat, '/'); + const char* bs = strrchr(pat, '\\'); + if (bs && (!slash || bs > slash)) slash = bs; + size_t dlen = slash ? (size_t)(slash - pat) + 1 : 0; + WIN32_FIND_DATAA fd; + HANDLE h = FindFirstFileA(pat, &fd); + if (h == INVALID_HANDLE_VALUE) return 3; /* GLOB_NOMATCH */ + do { + if (fd.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) continue; + size_t nlen = strlen(fd.cFileName); + char* path = (char*)malloc(dlen + nlen + 1); + char** v = (char**)realloc(g->gl_pathv, (g->gl_pathc + 1) * sizeof(char*)); + if (!path || !v) { free(path); if (v) g->gl_pathv = v; FindClose(h); globfree(g); return 1; } + memcpy(path, pat, dlen); + memcpy(path + dlen, fd.cFileName, nlen + 1); + g->gl_pathv = v; + g->gl_pathv[g->gl_pathc++] = path; + } while (FindNextFileA(h, &fd)); + FindClose(h); + return g->gl_pathc ? 0 : 3; +} +#else #include +#endif /* ── Runtime fixture (same pattern as test_link.c) ─────────────────── */ @@ -98,8 +135,17 @@ static void cleanup_base(const char* base) { snprintf(path, sizeof(path), "%s.qdb", base); unlink(path); snprintf(path, sizeof(path), "%s.qdb.tmp", base); unlink(path); /* Archived rolls have the form base..log — remove with glob via shell. */ +#if defined(_WIN32) + snprintf(path, sizeof(path), "%s.*.log", base); /* no POSIX shell here */ + glob_t g; + if (glob(path, 0, NULL, &g) == 0) { + for (size_t i = 0; i < g.gl_pathc; i++) unlink(g.gl_pathv[i]); + globfree(&g); + } +#else snprintf(path, sizeof(path), "rm -f '%s'.*.log 2>/dev/null", base); (void)system(path); +#endif } /* ═══════════════════════════════════════════════════════════════════════ @@ -1041,15 +1087,19 @@ static test_result_t test_journal_crash_window_no_double_apply(void) { /* Simulate the crash IN the window: put the just-archived log back * under its live name, exactly the on-disk state a crash between - * the two renames leaves behind. */ + * the two renames leaves behind. The journal is closed first (the + * "process" is gone), and the live file removed before the rename: + * Windows can neither replace a file that is still open nor rename + * onto an existing one. */ + TEST_ASSERT_EQ_I(ray_journal_close(), RAY_OK); char pattern[300]; snprintf(pattern, sizeof(pattern), "%s.*.log", base); glob_t g; TEST_ASSERT_EQ_I(glob(pattern, 0, NULL, &g), 0); TEST_ASSERT_EQ_I((int64_t)g.gl_pathc, 1); + (void)remove(lpath); TEST_ASSERT_EQ_I(rename(g.gl_pathv[0], lpath), 0); globfree(&g); - TEST_ASSERT_EQ_I(ray_journal_close(), RAY_OK); /* Restart: clobber the binding, recover. The covered log must be * skipped — jw_x comes back as the snapshot value, not value+1. */ @@ -2294,9 +2344,13 @@ static bool purge_write_one(int64_t x) { /* True iff at least one rolled archive (base..log) exists. */ static bool archive_exists(const char* base) { - char cmd[1200]; - snprintf(cmd, sizeof(cmd), "test -n \"$(ls '%s'.*.log 2>/dev/null)\"", base); - return system(cmd) == 0; + char pattern[1200]; + snprintf(pattern, sizeof(pattern), "%s.*.log", base); + glob_t g; + if (glob(pattern, 0, NULL, &g) != 0) return false; /* no match */ + bool found = g.gl_pathc > 0; + globfree(&g); + return found; } /* P1. Full purge while the journal is OPEN: closes it, unlinks the active diff --git a/test/test_link.c b/test/test_link.c index cd430538..7f1c8d56 100644 --- a/test/test_link.c +++ b/test/test_link.c @@ -918,9 +918,7 @@ static ray_err_t write_link_partition(const char* part_dir, int64_t custs_sym) { char dir[1024]; snprintf(dir, sizeof(dir), TMP_LINK_PART_DB "/%s/" TMP_LINK_PART_TBL, part_dir); - char cmd[1100]; - snprintf(cmd, sizeof(cmd), "mkdir -p %s", dir); - if (system(cmd) != 0) return RAY_ERR_IO; + if (ray_test_mkdir_p(dir) != 0) return RAY_ERR_IO; ray_t* ridcol = ray_vec_from_raw(RAY_I64, (void*)rids, n_rid); if (!ridcol || RAY_IS_ERR(ridcol)) return RAY_ERR_OOM; @@ -954,7 +952,7 @@ static ray_err_t write_link_partition(const char* part_dir, static test_result_t test_link_parted_load_propagates(void) { int64_t custs_sym = setup_custs_dim(); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); int64_t r1[] = { 0, 1, 2 }; int64_t q1[] = { 10, 20, 30 }; @@ -1020,7 +1018,7 @@ static test_result_t test_link_parted_load_propagates(void) { ray_release(ages1); ray_release(parted); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); PASS(); } @@ -1032,7 +1030,7 @@ static test_result_t test_link_attach_rejects_parted_target(void) { /* Build a parted table on disk and load via ray_read_parted so we have a * real RAY_TABLE-with-RAY_PARTED-cols handle to point at. */ int64_t custs_sym = setup_custs_dim(); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); int64_t r1[] = { 0, 1, 2 }; int64_t q1[] = { 10, 20, 30 }; @@ -1065,7 +1063,7 @@ static test_result_t test_link_attach_rejects_parted_target(void) { ray_release(w); ray_release(v); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); PASS(); } @@ -1095,7 +1093,7 @@ static test_result_t test_link_deref_rejects_parted_after_rebind(void) { ray_release(good); /* Build a parted table on disk and rebind `custs` to it. */ - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); int64_t r1[] = { 0, 1 }; int64_t q1[] = { 10, 20 }; TEST_ASSERT_EQ_I(write_link_partition("2024.01.01", r1, 2, q1, 2, custs_sym), RAY_OK); @@ -1118,7 +1116,7 @@ static test_result_t test_link_deref_rejects_parted_after_rebind(void) { ray_release(w); ray_release(v); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); PASS(); } @@ -1155,7 +1153,7 @@ static test_result_t test_link_dotted_resolve_propagates_parted_error(void) { ray_release(good); /* Rebind custs to a parted table on disk. */ - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); int64_t r1[] = { 0, 1 }; int64_t q1[] = { 10, 20 }; TEST_ASSERT_EQ_I(write_link_partition("2024.01.01", r1, 2, q1, 2, custs_sym), RAY_OK); @@ -1180,7 +1178,7 @@ static test_result_t test_link_dotted_resolve_propagates_parted_error(void) { ray_release(w); ray_release(v); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); PASS(); } @@ -1212,7 +1210,7 @@ static test_result_t test_link_vm_eval_propagates_parted_error(void) { ray_release(good); /* Rebind custs to a parted table. */ - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); int64_t r1[] = { 0, 1 }; int64_t q1[] = { 10, 20 }; TEST_ASSERT_EQ_I(write_link_partition("2024.01.01", r1, 2, q1, 2, custs_sym), RAY_OK); @@ -1233,7 +1231,7 @@ static test_result_t test_link_vm_eval_propagates_parted_error(void) { ray_release(w); ray_release(v); - (void)!system("rm -rf " TMP_LINK_PART_DB); + (void)ray_test_rm_rf(TMP_LINK_PART_DB); PASS(); } diff --git a/test/test_mcast.c b/test/test_mcast.c index 8fa81053..e71e174b 100644 --- a/test/test_mcast.c +++ b/test/test_mcast.c @@ -26,6 +26,9 @@ #include #include #include +#else + #include + #include #endif extern ray_runtime_t* __RUNTIME; @@ -631,7 +634,6 @@ static test_result_t test_mcast_large_payload_queues_until_writable(void) { TEST_ASSERT_FALSE(RAY_IS_ERR(sub)); ray_release(sub); -#ifndef RAY_OS_WINDOWS ray_t* server_h = ray_env_get(ray_sym_intern("_mc_sub_handle", 14)); TEST_ASSERT_NOT_NULL(server_h); TEST_ASSERT_EQ_I(server_h->type, -RAY_I64); @@ -639,8 +641,7 @@ static test_result_t test_mcast_large_payload_queues_until_writable(void) { TEST_ASSERT_NOT_NULL(server_sel); int sndbuf = 4096; setsockopt((ray_sock_t)server_sel->fd, SOL_SOCKET, SO_SNDBUF, - &sndbuf, sizeof(sndbuf)); -#endif + (const char*)&sndbuf, sizeof(sndbuf)); const char* pub_src = "(.mc.pub \"big\" (+ (* (til 100000) 1103515245) 12345))"; @@ -728,7 +729,6 @@ static test_result_t test_mcast_shared_frame_across_subscribers(void) { int64_t hp = ray_ipc_connect("127.0.0.1", port, NULL, NULL, 0); TEST_ASSERT((hp) >= (0), "publisher connected"); -#ifndef RAY_OS_WINDOWS ray_t* server_hs = ray_env_get(ray_sym_intern("_mc_handles", 11)); TEST_ASSERT_NOT_NULL(server_hs); TEST_ASSERT((ray_len(server_hs)) >= (3), "three server-side subscriber handles"); @@ -737,9 +737,8 @@ static test_result_t test_mcast_shared_frame_across_subscribers(void) { ray_selector_t* ssel = ray_poll_get(poll, sh); TEST_ASSERT_NOT_NULL(ssel); int sndbuf = 4096; - setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_SNDBUF, &sndbuf, sizeof(sndbuf)); + setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_SNDBUF, (const char*)&sndbuf, sizeof(sndbuf)); } -#endif const char* pub_src = "(.mc.pub \"big\" (+ (* (til 100000) 1103515245) 12345))"; ray_t* msg = ray_str(pub_src, strlen(pub_src)); @@ -802,6 +801,13 @@ static test_result_t test_mcast_shared_frame_across_subscribers(void) { * frame keeps receiving. The active limits and the high-water mark are * readable from .mc.stats and per handle from (.ipc.handle h). */ static test_result_t test_mcast_txlimit_overflow_disconnects(void) { +#if defined(_WIN32) + /* The backlog this test needs never forms on Windows: Winsock accepts a + * single non-blocking send() larger than SO_SNDBUF whole (it pins the + * caller's buffer), so the 800 KiB frame leaves at once and no + * subscriber crosses its tx limit. */ + SKIP("Winsock accepts oversized sends whole; no tx backlog forms"); +#endif ray_t* r = ray_eval_str( "(set _mc_count 0)" "(set _mc_close_count 0)" @@ -840,14 +846,12 @@ static test_result_t test_mcast_txlimit_overflow_disconnects(void) { TEST_ASSERT((ray_len(server_hs)) >= (2), "two server-side subscriber handles"); int64_t s1 = ((int64_t*)ray_data(server_hs))[0]; int64_t s2 = ((int64_t*)ray_data(server_hs))[1]; -#ifndef RAY_OS_WINDOWS for (int i = 0; i < 2; i++) { ray_selector_t* ssel = ray_poll_get(poll, i == 0 ? s1 : s2); TEST_ASSERT_NOT_NULL(ssel); int sndbuf = 4096; - setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_SNDBUF, &sndbuf, sizeof(sndbuf)); + setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_SNDBUF, (const char*)&sndbuf, sizeof(sndbuf)); } -#endif /* Process default: 64 KiB. h2 alone may hold 4 MiB. */ ray_t* msg = ray_str("(.ipc.txlimit 65536 0)", strlen("(.ipc.txlimit 65536 0)")); @@ -1040,14 +1044,12 @@ static test_result_t test_ipc_outbound_close_hook(void) { TEST_ASSERT_TRUE(pump_until_env_i64_at_least("_oc_out", 7, 1, 1000)); /* B: the peer resets h2 (linger zero, then close → RST) */ -#ifndef RAY_OS_WINDOWS { ray_selector_t* ssel = ray_poll_get(poll, s2); TEST_ASSERT_NOT_NULL(ssel); struct linger lg = { 1, 0 }; - setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_LINGER, &lg, sizeof(lg)); + setsockopt((ray_sock_t)ssel->fd, SOL_SOCKET, SO_LINGER, (const char*)&lg, sizeof(lg)); } -#endif n = snprintf(src, sizeof(src), "(.ipc.close %lld)", (long long)s2); msg = ray_str(src, (size_t)n); cr = ray_ipc_send(hc, msg); @@ -1224,7 +1226,6 @@ static test_result_t test_mcast_sync_reply_after_queued_frame(void) { TEST_ASSERT_FALSE(RAY_IS_ERR(sub)); ray_release(sub); -#ifndef RAY_OS_WINDOWS /* Shrink the server->h1 send buffer so a big frame parks on sel->tx.buf * instead of leaving in a single write. */ ray_t* server_h = ray_env_get(ray_sym_intern("_mc_sub_handle", 14)); @@ -1234,8 +1235,7 @@ static test_result_t test_mcast_sync_reply_after_queued_frame(void) { TEST_ASSERT_NOT_NULL(server_sel); int sndbuf = 4096; setsockopt((ray_sock_t)server_sel->fd, SOL_SOCKET, SO_SNDBUF, - &sndbuf, sizeof(sndbuf)); -#endif + (const char*)&sndbuf, sizeof(sndbuf)); const char* pub_src = "(.mc.pub \"big\" (+ (* (til 100000) 1103515245) 12345))"; diff --git a/test/test_repl.c b/test/test_repl.c index a341cc4f..5397c207 100644 --- a/test/test_repl.c +++ b/test/test_repl.c @@ -88,10 +88,8 @@ extern void* ray_runtime_get_poll(void); * poll first (closes any leftover conns), runtime second. */ static void repl_setup(void) { ray_runtime_create(0, NULL); -#ifndef RAY_OS_WINDOWS ray_poll_t* p = ray_poll_create(); if (p) ray_runtime_set_poll(p); -#endif } static void repl_teardown(void) { @@ -103,7 +101,6 @@ static void repl_teardown(void) { ray_t* args = NULL; ray_release(ray_repl_disconnect_fn(&args, 0)); } -#ifndef RAY_OS_WINDOWS { ray_poll_t* p = (ray_poll_t*)ray_runtime_get_poll(); if (p) { @@ -111,7 +108,6 @@ static void repl_teardown(void) { ray_poll_destroy(p); } } -#endif ray_runtime_destroy(__RUNTIME); } @@ -670,7 +666,9 @@ static test_result_t test_repl_pty_ctrl_d(void) { * code proves it ran while the prompt was idle, and the helper's 5 s * timeout (-2) is what a starved timer would produce. Nothing after * the timer line is ever written to the pty. */ +#ifndef RAY_OS_WINDOWS static int run_pty_listen_with_poll(const char* input); +#endif static test_result_t test_repl_pty_timer_fires_while_idle(void) { #ifndef RAY_OS_WINDOWS int rc = run_pty_listen_with_poll("(.time.timer.set 150 1 (fn [t] (exit 7)))\n"); diff --git a/test/test_runtime.c b/test/test_runtime.c index 35160723..e6bb4f06 100644 --- a/test/test_runtime.c +++ b/test/test_runtime.c @@ -37,8 +37,13 @@ #include #include #include +#ifdef RAY_OS_WINDOWS +#include +#include +#else #include #include +#endif static char* make_tmpdir(void) { char tmpl[] = "/tmp/rayforce-rt-test-XXXXXX"; @@ -70,6 +75,11 @@ static test_result_t test_create_with_sym_absent_is_ok(void) { * passing a path whose parent exists but isn't a directory (ENOTDIR) — * portable across Linux/macOS without needing root or chmod games. */ static test_result_t test_create_with_sym_io_error_surfaces(void) { +#if defined(_WIN32) + /* Win32 reports a path through a regular file as "path not found" + * (ENOENT, the missing-file case), never ENOTDIR. */ + SKIP("no ENOTDIR on Windows"); +#endif char* dir = make_tmpdir(); TEST_ASSERT_NOT_NULL(dir); @@ -1265,9 +1275,7 @@ static test_result_t test_syscov_splayed_set_with_sym_path(void) { } /* cleanup — no free(dir), it's a stack pointer */ - char cmd[512]; - snprintf(cmd, sizeof(cmd), "rm -rf %s", dir); - system(cmd); + (void)ray_test_rm_rf(dir); PASS(); } diff --git a/test/test_splay.c b/test/test_splay.c index 1137cbbd..34188b46 100644 --- a/test/test_splay.c +++ b/test/test_splay.c @@ -65,9 +65,7 @@ static void splay_teardown(void) { /* Remove temp dir tree */ static void rm_rf(const char* path) { - char cmd[512]; - snprintf(cmd, sizeof(cmd), "rm -rf %s", path); - (void)!system(cmd); + (void)ray_test_rm_rf(path); } /* ========================================================================= @@ -181,9 +179,7 @@ static test_result_t test_load_missing_schema(void) { /* Directory exists but contains no .d file */ const char* dir = TMP_SPLAY_BASE "/no_schema"; rm_rf(dir); - char cmd[512]; - snprintf(cmd, sizeof(cmd), "mkdir -p %s", dir); - (void)!system(cmd); + (void)ray_test_mkdir_p(dir); ray_t* r = ray_splay_load(dir, NULL); /* ray_col_load of missing file returns an error object */ @@ -499,9 +495,7 @@ static test_result_t test_save_sym_error(void) { char sym_as_dir[512]; snprintf(sym_as_dir, sizeof(sym_as_dir), "%s/sym_dir", dir); /* Ensure parent dir exists first */ - char mk[600]; - snprintf(mk, sizeof(mk), "mkdir -p %s", sym_as_dir); - (void)!system(mk); + (void)ray_test_mkdir_p(sym_as_dir); ray_err_t err = ray_splay_save(tbl, dir, sym_as_dir); /* Either succeeds (some impls tolerate it) or returns an error — either @@ -732,8 +726,8 @@ static test_result_t test_save_bulk_with_sym_path(void) { * 19. splay_save_impl: snprintf overflow for the column / ".d" paths. * Requires strlen(dir) >= 1021 so that strlen(dir)+3 >= 1024. * Build a deeply nested path using short components (≤ 50 chars each) - * so the filesystem NAME_MAX (255) is not exceeded, then call mkdir_p - * via system(), then ray_splay_save → snprintf("%s/.d") fires range. + * so the filesystem NAME_MAX (255) is not exceeded, then create it + * with ray_test_mkdir_p, then ray_splay_save → snprintf("%s/.d") fires range. * * Path layout (each component 50 chars): * /tmp/rft_deep_save/ (18 chars) @@ -748,6 +742,10 @@ static test_result_t test_save_dir_path_too_long(void) { * fires under the same condition on Linux PATH_MAX = 4096. Skip * on Darwin — the Linux runner covers the regression. */ SKIP("PATH_MAX=1024 on macOS — deep-mkdir fixture not portable"); +#elif defined(_WIN32) + /* Win32 directory paths stop at MAX_PATH (~260) without long-path + * opt-in, far short of the 1021-char tree. */ + SKIP("MAX_PATH=260 on Windows — deep-mkdir fixture not portable"); #endif /* Construct the nested path in a buffer */ char long_dir[2048]; @@ -771,10 +769,8 @@ static test_result_t test_save_dir_path_too_long(void) { TEST_ASSERT_TRUE((size_t)off >= 1021); /* Create the directory tree so ray_mkdir_p inside save succeeds. - * system("mkdir -p ...") handles arbitrarily deep paths. */ - char mk[4096]; - snprintf(mk, sizeof(mk), "mkdir -p \"%s\"", long_dir); - (void)!system(mk); + * ray_test_mkdir_p handles arbitrarily deep paths. */ + (void)ray_test_mkdir_p(long_dir); int64_t id_v2 = ray_sym_intern("v2long", 6); int64_t raw[] = {1}; @@ -794,9 +790,7 @@ static test_result_t test_save_dir_path_too_long(void) { ray_release(col); ray_release(tbl); /* Cleanup entire nested tree from the base */ - char rm_cmd[256]; - snprintf(rm_cmd, sizeof(rm_cmd), "rm -rf /tmp/rft_deep_save"); - (void)!system(rm_cmd); + (void)ray_test_rm_rf("/tmp/rft_deep_save"); PASS(); } @@ -884,9 +878,7 @@ static test_result_t test_trace_missing_schema(void) { const char* dir = TMP_SPLAY_BASE "/trace_noschema"; rm_rf(dir); /* Create dir without .d file */ - char mk[512]; - snprintf(mk, sizeof(mk), "mkdir -p %s", dir); - (void)!system(mk); + (void)ray_test_mkdir_p(dir); setenv("RAY_CSV_TRACE", "1", 1); ray_t* r = ray_splay_load(dir, NULL); @@ -985,11 +977,14 @@ static test_result_t test_trace_fresh_load(void) { * (with .d last, the column save is the first write to hit the dir). * ========================================================================= */ static test_result_t test_save_schema_write_fails(void) { +#if defined(_WIN32) + /* chmod cannot make a Windows directory refuse new files: the + * read-only attribute is ignored for directories. */ + SKIP("read-only directories are not enforced on Windows"); +#endif const char* dir = TMP_SPLAY_BASE "/no_write_schema"; rm_rf(dir); - char mk[512]; - snprintf(mk, sizeof(mk), "mkdir -p %s", dir); - (void)!system(mk); + (void)ray_test_mkdir_p(dir); /* Make dir read-only so .d cannot be written */ chmod(dir, 0555); diff --git a/test/test_store.c b/test/test_store.c index f4ca1056..1e7d849a 100644 --- a/test/test_store.c +++ b/test/test_store.c @@ -326,7 +326,7 @@ static test_result_t test_col_mmap_nofile(void) { static test_result_t test_splay_open_roundtrip(void) { /* Clean up any leftover splay dir */ - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); /* Build a 3-column table: I64, F64, I32 */ ray_t* tbl = ray_table_new(4); @@ -398,14 +398,14 @@ static test_result_t test_splay_open_roundtrip(void) { ray_release(tbl); /* Cleanup */ - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } /* ---- test_splay_str_column_roundtrip ----------------------------------- */ static test_result_t test_splay_str_column_roundtrip(void) { - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); ray_t* tbl = ray_table_new(2); TEST_ASSERT_NOT_NULL(tbl); @@ -471,7 +471,7 @@ static test_result_t test_splay_str_column_roundtrip(void) { ray_release(names); ray_release(tbl); - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } @@ -485,7 +485,7 @@ static test_result_t test_splay_str_column_roundtrip(void) { * ---------------------------------------------------------------------- */ static test_result_t test_splay_short_strv_roundtrip(void) { - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); int64_t id_short = ray_sym_intern("short", 5); int64_t id_empty = ray_sym_intern("empty", 5); @@ -547,13 +547,13 @@ static test_result_t test_splay_short_strv_roundtrip(void) { ray_release(tbl); ray_release(tbl2); - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } /* ---- test_splay_dict_column_roundtrip --------------------------------- */ static test_result_t test_splay_dict_column_roundtrip(void) { - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); int64_t ids_raw[] = {1, 2}; ray_t* ids = ray_vec_from_raw(RAY_I64, ids_raw, 2); @@ -608,13 +608,13 @@ static test_result_t test_splay_dict_column_roundtrip(void) { ray_release(tbl); ray_release(ids); ray_release(sched); - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } /* ---- test_splay_empty_list_column_roundtrip --------------------------- */ static test_result_t test_splay_empty_list_column_roundtrip(void) { - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); ray_t* ids = ray_vec_new(RAY_I64, 0); ray_t* who = ray_vec_new(RAY_SYM, 0); @@ -657,14 +657,14 @@ static test_result_t test_splay_empty_list_column_roundtrip(void) { ray_release(ids); ray_release(who); ray_release(sched); - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } /* A deterministic unsupported column must be rejected before an earlier * column can replace the committed generation. */ static test_result_t test_splay_save_preflight_preserves_generation(void) { - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); int64_t k_id = ray_sym_intern("k", 1); int64_t v_id = ray_sym_intern("v", 1); @@ -705,7 +705,7 @@ static test_result_t test_splay_save_preflight_preserves_generation(void) { ray_release(good); ray_release(old_v); ray_release(old_k); - (void)!system("rm -rf " TMP_SPLAY_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_DIR); PASS(); } @@ -838,9 +838,9 @@ static test_result_t test_parted_release(void) { static test_result_t test_part_open(void) { /* Setup: create a 2-partition db with 2 columns each */ - (void)!system("rm -rf " TMP_PART_DB); - (void)!system("mkdir -p " TMP_PART_DB "/2024.01.01/" TMP_TABLE_NAME); - (void)!system("mkdir -p " TMP_PART_DB "/2024.01.02/" TMP_TABLE_NAME); + (void)ray_test_rm_rf(TMP_PART_DB); + (void)ray_test_mkdir_p(TMP_PART_DB "/2024.01.01/" TMP_TABLE_NAME); + (void)ray_test_mkdir_p(TMP_PART_DB "/2024.01.02/" TMP_TABLE_NAME); /* Partition 1: 3 rows */ int64_t raw_a1[] = {10, 20, 30}; @@ -941,7 +941,7 @@ static test_result_t test_part_open(void) { /* Release — should unmap all segments */ ray_release(parted); - (void)!system("rm -rf " TMP_PART_DB); + (void)ray_test_rm_rf(TMP_PART_DB); PASS(); } @@ -949,7 +949,7 @@ static test_result_t test_part_open(void) { /* ray_parted_tables lists the splayed-table subdirectories of the first * partition as a sorted SYM vector usable with ray_read_parted. */ static test_result_t test_parted_tables(void) { - (void)!system("rm -rf " TMP_PART_DB); + (void)ray_test_rm_rf(TMP_PART_DB); /* Two tables (trades, quotes) across two partitions. */ const char* dirs[] = { TMP_PART_DB "/2024.01.01/trades", TMP_PART_DB "/2024.01.01/quotes", @@ -986,16 +986,17 @@ static test_result_t test_parted_tables(void) { /* An existing-but-empty root (no partition dirs) lists no tables — * an empty SYM vector, not an error. */ - (void)!system("rm -rf " TMP_PART_DB "_np && mkdir -p " TMP_PART_DB "_np"); + (void)ray_test_rm_rf(TMP_PART_DB "_np"); + (void)ray_test_mkdir_p(TMP_PART_DB "_np"); ray_t* empty = ray_parted_tables(TMP_PART_DB "_np"); TEST_ASSERT_NOT_NULL(empty); TEST_ASSERT_FALSE(RAY_IS_ERR(empty)); TEST_ASSERT_EQ_I(empty->type, RAY_SYM); TEST_ASSERT_EQ_I(empty->len, 0); ray_release(empty); - (void)!system("rm -rf " TMP_PART_DB "_np"); + (void)ray_test_rm_rf(TMP_PART_DB "_np"); - (void)!system("rm -rf " TMP_PART_DB); + (void)ray_test_rm_rf(TMP_PART_DB); PASS(); } @@ -1662,7 +1663,7 @@ static test_result_t test_sym_col_valid_roundtrip(void) { #define TMP_SYM_PATH "/tmp/rayforce_test_splay_sym_file" static test_result_t test_splay_load_with_sym(void) { - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); unlink(TMP_SYM_PATH); /* Intern symbols and build a table with a RAY_SYM column */ @@ -1711,7 +1712,7 @@ static test_result_t test_splay_load_with_sym(void) { ray_release(col_name); ray_release(col_age); ray_release(tbl); - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); unlink(TMP_SYM_PATH); unlink(TMP_SYM_PATH ".lk"); PASS(); @@ -1720,7 +1721,7 @@ static test_result_t test_splay_load_with_sym(void) { /* ---- test_splay_load_sym_missing_corrupt ------------------------------- */ static test_result_t test_splay_load_sym_missing_corrupt(void) { - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); unlink(TMP_SYM_PATH); /* Intern symbols and build a table with a RAY_SYM column */ @@ -1757,7 +1758,7 @@ static test_result_t test_splay_load_sym_missing_corrupt(void) { ray_release(col); ray_release(tbl); - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); unlink(TMP_SYM_PATH); unlink(TMP_SYM_PATH ".lk"); PASS(); @@ -1766,7 +1767,7 @@ static test_result_t test_splay_load_sym_missing_corrupt(void) { /* ---- test_read_splayed_bad_sym_fatal ----------------------------------- */ static test_result_t test_read_splayed_bad_sym_fatal(void) { - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); /* Build a simple table (no RAY_SYM columns needed) */ int64_t id_x = ray_sym_intern("x", 1); @@ -1793,7 +1794,7 @@ static test_result_t test_read_splayed_bad_sym_fatal(void) { ray_release(col_x); ray_release(tbl); - (void)!system("rm -rf " TMP_SPLAY_SYM_DIR); + (void)ray_test_rm_rf(TMP_SPLAY_SYM_DIR); PASS(); } diff --git a/test/test_traverse.c b/test/test_traverse.c index db72da8d..d976d0d4 100644 --- a/test/test_traverse.c +++ b/test/test_traverse.c @@ -33,7 +33,7 @@ #include #include #include -#ifndef __SANITIZE_ADDRESS__ +#if !defined(__SANITIZE_ADDRESS__) && !defined(_WIN32) /* setrlimit: POSIX only */ #include #endif @@ -3904,7 +3904,7 @@ static test_result_t test_k_shortest_found_path_dup(void) { /* -------------------------------------------------------------------------- * Helper: read VmSize from /proc/self/status; returns 0 on failure. * -------------------------------------------------------------------------- */ -#ifndef __SANITIZE_ADDRESS__ +#if !defined(__SANITIZE_ADDRESS__) && !defined(_WIN32) #include static size_t get_vmsize_bytes(void) { FILE* f = fopen("/proc/self/status", "r"); @@ -4241,7 +4241,7 @@ static test_result_t test_traverse_oom_paths(void) { ray_heap_destroy(); PASS(); } -#endif /* __SANITIZE_ADDRESS__ */ +#endif /* !__SANITIZE_ADDRESS__ && !_WIN32 */ /* -------------------------------------------------------------------------- * Test: exec_expand with SIP bitmap build where rev.n_nodes > fwd.n_nodes. @@ -5725,7 +5725,7 @@ const test_entry_t traverse_entries[] = { { "traverse/k_shortest_large_k", test_k_shortest_large_k, NULL, NULL }, { "traverse/betweenness_with_rev_edges", test_betweenness_with_rev_edges, NULL, NULL }, { "traverse/closeness_sampled_norm", test_closeness_sampled_norm, NULL, NULL }, -#ifndef __SANITIZE_ADDRESS__ +#if !defined(__SANITIZE_ADDRESS__) && !defined(_WIN32) { "traverse/traverse_oom_paths", test_traverse_oom_paths, NULL, NULL }, #endif { "traverse/shortest_path_exceeds_254", test_shortest_path_exceeds_254, NULL, NULL }, From fbf55cf8baaceea02c68eec4591fb7535de5e407 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 22 Sep 2026 14:58:57 +0300 Subject: [PATCH 17/51] docs: Windows build instructions and platform status Co-Authored-By: Claude Opus 5 (1M context) --- CONTRIBUTING.md | 22 ++++++++++++++++++++++ RELEASE.md | 8 ++++---- docs/docs/guides/memory.md | 4 ---- docs/docs/index.md | 2 +- docs/docs/namespaces/sys.md | 2 +- 5 files changed, 28 insertions(+), 10 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index aea56e49..cbbaf68d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -69,6 +69,28 @@ a focused subset with: ./rayforce.test -f ``` +A `.rfl` file that needs a POSIX shell or filesystem (fixtures built or +checked through `.sys.exec`, `/proc`, `/dev/tcp`, …) carries the line +`;; @requires: posix`; on Windows the runner reports it as `SKIP` rather +than running it. C tests use `#ifndef RAY_OS_WINDOWS` / `SKIP(...)` for the +same purpose. Prefer the shell-free helpers `ray_test_rm_rf` / +`ray_test_mkdir_p` (`test/test.h`) over `system("rm -rf …")` in C tests. + +### Windows + +Build with the MSYS2 CLANG64 (or MINGW64) toolchain — `pacman -S +mingw-w64-clang-x86_64-clang make` — from an MSYS2 shell or with +`C:\msys64\clang64\bin` and `C:\msys64\usr\bin` on `PATH`: + +```sh +make # debug build (ASan + UBSan) +make test +make release +``` + +The debug binaries load the ASan runtime DLL from `clang64\bin`, so keep it on +`PATH` when running them. + ## Stability tooling Beyond the ASan/UBSan test run, the repo carries a stability toolset. These diff --git a/RELEASE.md b/RELEASE.md index fb141e6e..76421ebe 100644 --- a/RELEASE.md +++ b/RELEASE.md @@ -136,7 +136,7 @@ Each release publishes, in addition to the source: ## Platform support -Linux and macOS binaries are published today. Windows is not build-ready yet -(IOCP backend is a stub, `main.c`/`heap.c` have unguarded POSIX calls, and the -Makefile has no Windows toolchain path); once ported, add a `windows-latest` row -to the `build` matrix in `.github/workflows/release.yml`. +Linux and macOS binaries are published today. Windows builds and passes the +test suite from source with the MSYS2 CLANG64 toolchain (see CONTRIBUTING.md), +but no Windows binary is published yet; to ship one, add a `windows-latest` +row (MSYS2 CLANG64) to the `build` matrix in `.github/workflows/release.yml`. diff --git a/docs/docs/guides/memory.md b/docs/docs/guides/memory.md index 998c1d2d..a0050195 100644 --- a/docs/docs/guides/memory.md +++ b/docs/docs/guides/memory.md @@ -104,10 +104,6 @@ total-mem | 16777216000 | `page-size` | OS page size in bytes | | `total-mem` | Total physical RAM in bytes | -!!! note "Note" - - On Windows, only `cores` is currently reported. - ## 5. Progress Monitoring Long-running queries display a progress bar automatically in the REPL. The bar appears after approximately 2 seconds of execution and shows real-time feedback. diff --git a/docs/docs/index.md b/docs/docs/index.md index 6550f46a..162253e7 100644 --- a/docs/docs/index.md +++ b/docs/docs/index.md @@ -2,7 +2,7 @@ Embeddable columnar analytics and graph traversal engine in pure C. -Rayforce combines morsel-driven vectorized execution, a multi-pass query optimizer, and a native CSR graph engine in a single pipeline. It is queried through the **Rayfall** language, exposes a C API for embedding, and runs on Linux and macOS (Windows is planned — the IOCP backend is still a stub). +Rayforce combines morsel-driven vectorized execution, a multi-pass query optimizer, and a native CSR graph engine in a single pipeline. It is queried through the **Rayfall** language, exposes a C API for embedding, and runs on Linux and macOS, and builds on Windows from source (MSYS2 CLANG64). [Quick Start](getting-started/quick-start.md){ .md-button .md-button--primary } [Functions Reference](language/functions.md){ .md-button } diff --git a/docs/docs/namespaces/sys.md b/docs/docs/namespaces/sys.md index 5e13d189..f6eefbff 100644 --- a/docs/docs/namespaces/sys.md +++ b/docs/docs/namespaces/sys.md @@ -75,7 +75,7 @@ Signature: `(.sys.build)`. Returns a dict with `version` (string) and `build-dat ## `.sys.info` { #sys-info } -Signature: `(.sys.info)`. Returns `{cores: i64, page-size: i64, total-mem: i64, pid: i64, hostname: str}` on POSIX. On Windows the machine facts fall back to `{cores: 1}` (the sysconf-backed values aren't wired), but `pid` and `hostname` are answered on both platforms. +Signature: `(.sys.info)`. Returns `{cores: i64, page-size: i64, total-mem: i64, pid: i64, hostname: str}` on every platform (on Windows from `GetSystemInfo` / `GlobalMemoryStatusEx`). ```lisp (.sys.info) From cfdbb34ed95a25789d173ef7687ad5aad2f021e3 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 13:23:54 +0300 Subject: [PATCH 18/51] fix: readable REPL prompt and CPU name in the banner on Windows No console font Windows ships has U+2023 (checked Consolas, Cascadia Mono, Lucida Console, Courier New) and the classic console does no font fallback, so the prompt rendered as '?'. Use the nearest filled triangle they all do have, U+25BA, which is also three UTF-8 bytes. The banner's CPU line said 'unknown': read ProcessorNameString instead. Co-Authored-By: Claude Opus 5 (1M context) --- docs/docs/getting-started/quick-start.md | 2 +- src/app/repl.c | 13 ++++++++++++- src/app/term.c | 12 +++++++++++- test/test_term.c | 7 ++++++- 4 files changed, 30 insertions(+), 4 deletions(-) diff --git a/docs/docs/getting-started/quick-start.md b/docs/docs/getting-started/quick-start.md index 73a2dc04..143da205 100644 --- a/docs/docs/getting-started/quick-start.md +++ b/docs/docs/getting-started/quick-start.md @@ -58,7 +58,7 @@ The Rayfall REPL provides an interactive environment with syntax highlighting, b ./rayforce ``` -You will see the `‣` prompt (a green triangle bullet): +You will see the `‣` prompt (a green triangle bullet; `►` on Windows, whose console fonts have no `‣`): ```text ‣ diff --git a/src/app/repl.c b/src/app/repl.c index 7036c63c..b5cd32d8 100644 --- a/src/app/repl.c +++ b/src/app/repl.c @@ -353,7 +353,18 @@ static void get_cpu_name(char* buf, size_t sz) { if (sysctlbyname("machdep.cpu.brand_string", buf, &len, NULL, 0) != 0) snprintf(buf, sz, "unknown"); #elif defined(RAY_OS_WINDOWS) - snprintf(buf, sz, "unknown"); + /* The brand string the firmware reported, same text as /proc/cpuinfo's + * "model name"; it is padded with trailing spaces, so trim them. */ + DWORD n = (DWORD)sz; + if (RegGetValueA(HKEY_LOCAL_MACHINE, + "HARDWARE\\DESCRIPTION\\System\\CentralProcessor\\0", + "ProcessorNameString", RRF_RT_REG_SZ, NULL, + buf, &n) == ERROR_SUCCESS) { + size_t len = strlen(buf); + while (len > 0 && buf[len - 1] == ' ') buf[--len] = '\0'; + } else { + snprintf(buf, sz, "unknown"); + } #else snprintf(buf, sz, "unknown"); #endif diff --git a/src/app/term.c b/src/app/term.c index 79509f18..3b1560a6 100644 --- a/src/app/term.c +++ b/src/app/term.c @@ -1517,8 +1517,18 @@ int32_t ray_term_count_unmatched(ray_term_t* term) { /* ===== Prompt ===== */ -/* Green ‣ (U+2023) prompt, matching Rayforce style */ +/* Green ‣ (U+2023) prompt, matching Rayforce style. + * + * No console font Windows ships has U+2023 — not Consolas, Cascadia Mono, + * Lucida Console or Courier New — and the classic console does no font + * fallback, so the prompt renders as "?" there. Windows uses ► (U+25BA), + * the nearest filled triangle all four of them do have; it is also three + * UTF-8 bytes, so the byte and visual widths below are unchanged. */ +#if defined(RAY_OS_WINDOWS) +#define PROMPT_STR "\033[32m\xe2\x96\xba\033[0m " +#else #define PROMPT_STR "\033[32m\xe2\x80\xa3\033[0m " +#endif #define PROMPT_LEN 13 /* ESC[32m (5) + ‣ (3) + ESC[0m (4) + space (1) = 13 bytes */ #define PROMPT_VIS 2 /* visual: ‣ + space */ #define CONT_PROMPT_STR "\033[90m\xe2\x80\xa6\033[0m " /* gray … (U+2026) */ diff --git a/test/test_term.c b/test/test_term.c index f29ddc08..319c2f98 100644 --- a/test/test_term.c +++ b/test/test_term.c @@ -1185,7 +1185,12 @@ static test_result_t test_term_prompt_emits_bytes(void) { ray_term_prompt(t); fflush(stdout); int32_t n = capture_end(saved, path, cap, sizeof cap); - int saw_arrow = strstr(cap, "\xe2\x80\xa3") != NULL; /* ‣ */ + /* ‣ (U+2023), or ► (U+25BA) where no console font has ‣ — see term.c */ +#if defined(RAY_OS_WINDOWS) + int saw_arrow = strstr(cap, "\xe2\x96\xba") != NULL; +#else + int saw_arrow = strstr(cap, "\xe2\x80\xa3") != NULL; +#endif int saw_green = strstr(cap, "\033[32m") != NULL; TEST_ASSERT_FMT(n > 0, "no prompt output"); TEST_ASSERT_FMT(saw_arrow, "missing ‣ arrow in prompt: bytes=%d", n); From bb6b5d8875ad6eb4ed3b8b57d15e3b0e8771019b Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 13:32:14 +0300 Subject: [PATCH 19/51] bench: build the micro-benchmarks on Windows, record a Windows/Linux check alloc and agg_v2 read peak RSS through getrusage; use GetProcessMemoryInfo there. windows_vs_linux.md records the numbers the port was checked against. Co-Authored-By: Claude Opus 5 (1M context) --- bench/agg_v2/main.c | 14 ++++++- bench/alloc/main.c | 14 ++++++- bench/bottleneck/windows_vs_linux.md | 56 ++++++++++++++++++++++++++++ 3 files changed, 82 insertions(+), 2 deletions(-) create mode 100644 bench/bottleneck/windows_vs_linux.md diff --git a/bench/agg_v2/main.c b/bench/agg_v2/main.c index 401365ce..908fc3f4 100644 --- a/bench/agg_v2/main.c +++ b/bench/agg_v2/main.c @@ -46,7 +46,13 @@ #include #include #include -#include +#if defined(_WIN32) +# define WIN32_LEAN_AND_MEAN +# include +# include +#else +# include +#endif /* ---------- timing ---------- */ static double now_ms(void) { @@ -69,12 +75,18 @@ static double vmin(const double* arr, int n) { return m; } static long max_rss_kb(void) { +#if defined(_WIN32) + PROCESS_MEMORY_COUNTERS pmc; + if (!GetProcessMemoryInfo(GetCurrentProcess(), &pmc, sizeof(pmc))) return 0; + return (long)(pmc.PeakWorkingSetSize / 1024); +#else struct rusage ru; getrusage(RUSAGE_SELF, &ru); #if defined(__APPLE__) return ru.ru_maxrss / 1024; #else return ru.ru_maxrss; #endif +#endif } /* ---------- deterministic PRNG (splitmix64) ---------- */ diff --git a/bench/alloc/main.c b/bench/alloc/main.c index 9bbefc12..3294e918 100644 --- a/bench/alloc/main.c +++ b/bench/alloc/main.c @@ -11,7 +11,13 @@ #include #include #include -#include +#if defined(_WIN32) +# define WIN32_LEAN_AND_MEAN +# include +# include +#else +# include +#endif static double now_s(void) { struct timespec ts; @@ -66,6 +72,11 @@ static void* consumer(void* _) { } static long max_rss_kb(void) { +#if defined(_WIN32) + PROCESS_MEMORY_COUNTERS pmc; + if (!GetProcessMemoryInfo(GetCurrentProcess(), &pmc, sizeof(pmc))) return 0; + return (long)(pmc.PeakWorkingSetSize / 1024); +#else struct rusage ru; getrusage(RUSAGE_SELF, &ru); /* Linux: ru_maxrss is KB; macOS: bytes. Normalize to KB. */ #if defined(__APPLE__) @@ -73,6 +84,7 @@ static long max_rss_kb(void) { #else return ru.ru_maxrss; #endif +#endif } int main(void) { diff --git a/bench/bottleneck/windows_vs_linux.md b/bench/bottleneck/windows_vs_linux.md new file mode 100644 index 00000000..92874cf8 --- /dev/null +++ b/bench/bottleneck/windows_vs_linux.md @@ -0,0 +1,56 @@ +# Windows vs Linux sanity check + +Not a performance study: a coarse check that the Windows port lands in the +same ballpark as Linux, run while porting (`serhii/windows-port`). The two +sides do not share a compiler, an allocator-visible kernel, or a filesystem, +so only order-of-magnitude gaps are meaningful here. + +## Environment + +**CPU**: 11th Gen Intel Core i7 (8 logical cores) — one laptop, both runs +**Windows**: Windows 11, clang 20.1.8 (MSYS2 CLANG64), 32 GiB +**Linux**: WSL2 (kernel 6.6.87.2-microsoft-standard-WSL2, Ubuntu 24.04), gcc 13.3.0, 15 GiB to the VM +**Build**: `make release` both sides (`-O3 -march=native`, no sanitizers — `nm bench-alloc | grep -ci asan` → 0) + +WSL2 is a virtual machine with its own memory budget and a virtualised +filesystem; that alone moves I/O and page-fault numbers. Treat the file-backed +rows as indicative only. + +## Allocator micro-benchmark (`bench/alloc`) + +| case | Windows | Linux | +|------|---------|-------| +| atom-64B | 85.7 Mops/s | 80.9 Mops/s | +| vec-256B | 85.5 Mops/s | 78.6 Mops/s | +| morsel-8K | 88.2 Mops/s | 80.3 Mops/s | +| morsel-16K | 88.4 Mops/s | 78.2 Mops/s | +| large-1M | 62.7 Mops/s | 56.4 Mops/s | +| producer-consumer | 9.2 Mops/s, peak RSS 28 MB | 6.0 Mops/s, peak RSS 68 MB | + +## Engine operations (5M rows, `timeit`, median of 3) + +| operation | Windows (ms) | Linux (ms) | Win/Lin | +|-----------|-------------:|-----------:|--------:| +| arith-f64 (`sum (* f 1.5)`) | 7.94 | 5.10 | 1.56 | +| sort-i64 | 9.89 | 8.78 | 1.13 | +| distinct-i64 | 2.96 | 4.69 | 0.63 | +| group-by sym | 3.44 | 3.93 | 0.88 | +| select where | 7.13 | 14.14 | 0.50 | +| inner-join | 108.6 | 108.3 | 1.00 | +| csv write (5M rows) | 2201 | 1654 | 1.33 | +| csv read | 155 | 186 | 0.83 | +| splayed set | 509 | 580 | 0.88 | +| splayed get + count | 5.9 | 0.28 | 21 | + +`sum`/`avg` are omitted: both platforms report ~0.005 ms, the DAG elides them. + +## Reading + +Compute paths agree within a factor of ~1.6 either way, which is compiler and +noise territory, and the allocator is slightly ahead on Windows. + +The one real gap is `splayed get + count` (5.9 ms vs 0.28 ms). It opens and +maps one file per column, so it measures file-open cost, not the engine: +`CreateFileA` + `CreateFileMapping` + `MapViewOfFile` per column, plus +whatever on-access scanning is installed. The absolute cost is small and it is +paid per table open, but a wide table opened in a loop would feel it. From fb392d92cb104f58904bbce23b5594f5f5612599 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 14:11:18 +0300 Subject: [PATCH 20/51] bench: record the recent perf benches on Windows vs Linux Co-Authored-By: Claude Opus 5 (1M context) --- bench/bottleneck/windows_vs_linux.md | 36 ++++++++++++++++++++++++++++ 1 file changed, 36 insertions(+) diff --git a/bench/bottleneck/windows_vs_linux.md b/bench/bottleneck/windows_vs_linux.md index 92874cf8..c7b0e30a 100644 --- a/bench/bottleneck/windows_vs_linux.md +++ b/bench/bottleneck/windows_vs_linux.md @@ -54,3 +54,39 @@ maps one file per column, so it measures file-open cost, not the engine: `CreateFileA` + `CreateFileMapping` + `MapViewOfFile` per column, plus whatever on-access scanning is installed. The absolute cost is small and it is paid per table open, but a wide table opened in a loop would feel it. + +## The benches recent PRs shipped + +Same binaries, built from the release objects on each side. + +**`bench/join_nullfree` (#598, null-free key fast path).** The optimisation +engages on Windows — the `nullfree` counter advances on the null-free cases +and stays put on the nullable one, as on Linux. + +| case | Windows median (baseline → fast) | Linux median | +|------|---------------------------------|--------------| +| SYM2 | 355.3 → 336.0 ms (-5.4%) | 413.6 → 408.4 ms (-1.3%) | +| SYM2-NULL (must not fire) | 348.2 → 349.8 ms (+0.5%) | 407.1 → 412.9 ms (+1.4%) | +| I64 | 102.7 → 89.2 ms (-13.2%) | 100.9 → 96.3 ms (-4.6%) | + +**`bench/join_dup` (duplicate-key fallback).** The pathological case is fixed +on both: CATASTROPHIC-INNER post-fix ~170 ms on Windows and ~230 ms on Linux, +against ~2.6 s pre-fix (Windows) — the same order-of-magnitude win. + +**`bench/join_buildside` (build-side swap).** The swap fires on Windows and +pays off by the same factor: MANY-TO-MANY 207 ms swapped vs 494 ms legacy +(2.4x); Linux 188 vs 431 (2.3x). HEAVY-DUP-WIN: 1.7 s vs 6.2 s (Windows), +1.8 s vs 8.2 s (Linux). + +**`bench/idx_route` Q3** (1000 lookups/rep): indexed 0.014 ms/batch on +Windows, 0.009 on Linux; the unindexed control is 0.004 on both. + +Not runnable as-is: + +- `bench/group_pushdown` and `bench/agg_v2` no longer compile **on either + platform** — they use `ray_op.inputs` and `ray_group2/3`, which the engine + no longer has. Pre-existing, unrelated to the port. +- `bench/groupby_shapes/*.py` needs python3, which a stock Windows lacks; the + `.rfl` cases in that directory run directly under `rayforce` on both. +- `scripts/soak.sh` and `scripts/fuzz-seed-*.sh` are bash and stay POSIX-only + (the fuzzing runtime is Linux-only anyway, see the Makefile). From 55430068b15aab977ae0b74b1703c7c136be4676 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 14:43:41 +0300 Subject: [PATCH 21/51] fix(format): render i64 at full width, not through 32-bit long MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Interpolation (format/println), print and the pivot / column-name helpers formatted an i64 with "%ld" and a (long) cast. long is 64-bit on LP64 and 32-bit on Windows, so there 10^18 printed as -1486618624 and a pivot keyed on values above 2^31 produced truncated column NAMES — a data defect, not only a display one. Use PRId64 throughout; regression test included. Co-Authored-By: Claude Opus 5 (1M context) --- src/ops/builtins.c | 8 ++++---- src/ops/pivot.c | 5 +++-- src/ops/tblop.c | 4 ++-- test/rfl/regress/i64_text_width.rfl | 18 ++++++++++++++++++ 4 files changed, 27 insertions(+), 8 deletions(-) create mode 100644 test/rfl/regress/i64_text_width.rfl diff --git a/src/ops/builtins.c b/src/ops/builtins.c index 18dd8d35..9b2fb31b 100644 --- a/src/ops/builtins.c +++ b/src/ops/builtins.c @@ -110,7 +110,7 @@ void ray_lang_print(FILE* fp, ray_t* val) { return; } switch (val->type) { - case -RAY_I64: fprintf(fp, "%ld", (long)val->i64); break; + case -RAY_I64: fprintf(fp, "%" PRId64, val->i64); break; case -RAY_F64: { double fv = val->f64; fv = clear_neg_zero(fv); @@ -142,8 +142,8 @@ void ray_lang_print(FILE* fp, ray_t* val) { break; } case RAY_TABLE: - fprintf(fp, "", - (long)ray_table_nrows(val), (long)ray_table_ncols(val)); + fprintf(fp, "
", + ray_table_nrows(val), ray_table_ncols(val)); break; case RAY_UNARY: case RAY_BINARY: case RAY_VARY: { const char* name = ray_fn_name(val); @@ -204,7 +204,7 @@ static char* fmt_interpolate(const char* fmt, size_t flen, ray_t** args, int64_t RAY_ATOM_IS_NULL(a)) { tlen = snprintf(tmp, sizeof(tmp), "%s", null_literal_str(a->type)); } else if (a->type == -RAY_I64) { - tlen = snprintf(tmp, sizeof(tmp), "%ld", (long)a->i64); + tlen = snprintf(tmp, sizeof(tmp), "%" PRId64, a->i64); } else if (a->type == -RAY_F64) { double fv = a->f64; fv = clear_neg_zero(fv); diff --git a/src/ops/pivot.c b/src/ops/pivot.c index f3b6eb52..0be4a557 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -21,6 +21,7 @@ * SOFTWARE. */ +#include #include "ops/internal.h" #include "ops/hash.h" #include "ops/idxop.h" @@ -1910,9 +1911,9 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { len = snprintf(buf, sizeof(buf), "%s", pval ? "true" : "false"); } else if (pt == RAY_I64 || pt == RAY_I32 || pt == RAY_I16 || pt == RAY_DATE || pt == RAY_TIME || pt == RAY_TIMESTAMP) { - len = snprintf(buf, sizeof(buf), "%ld", (long)pval); + len = snprintf(buf, sizeof(buf), "%" PRId64, pval); } else { - len = snprintf(buf, sizeof(buf), "col%ld", (long)pval); + len = snprintf(buf, sizeof(buf), "col%" PRId64, pval); } col_sym = ray_sym_intern(buf, (size_t)len); } diff --git a/src/ops/tblop.c b/src/ops/tblop.c index 7da33443..375d4ae3 100644 --- a/src/ops/tblop.c +++ b/src/ops/tblop.c @@ -627,7 +627,7 @@ static ray_t* pivot_fn_impl(ray_t* tbl, ray_t* index_arg, ray_t* pivot_col_name, if (pval->type == -RAY_SYM) { col_sym = pval->i64; } else if (pval->type == -RAY_I64) { - char buf[64]; int len = snprintf(buf, sizeof(buf), "%ld", (long)pval->i64); + char buf[64]; int len = snprintf(buf, sizeof(buf), "%" PRId64, pval->i64); col_sym = ray_sym_intern(buf, (size_t)len); } else if (pval->type == -RAY_F64) { double fv = clear_neg_zero(pval->f64); @@ -636,7 +636,7 @@ static ray_t* pivot_fn_impl(ray_t* tbl, ray_t* index_arg, ray_t* pivot_col_name, } else if (pval->type == -RAY_BOOL) { col_sym = ray_sym_intern(pval->b8 ? "true" : "false", pval->b8 ? 4 : 5); } else { - char buf[64]; int len = snprintf(buf, sizeof(buf), "col%ld", (long)pval->i64); + char buf[64]; int len = snprintf(buf, sizeof(buf), "col%" PRId64, pval->i64); col_sym = ray_sym_intern(buf, (size_t)len); } if (a1) ray_release(pval); diff --git a/test/rfl/regress/i64_text_width.rfl b/test/rfl/regress/i64_text_width.rfl new file mode 100644 index 00000000..8ae3affc --- /dev/null +++ b/test/rfl/regress/i64_text_width.rfl @@ -0,0 +1,18 @@ +;; An i64 must survive every text path in full 64-bit width. The +;; interpolation, print and pivot/column-name helpers used to render it with +;; "%ld" + (long), which is 32-bit on Windows (LLP64): 10^18 came back as +;; -1486618624, and a pivot keyed on large integers produced wrong column +;; names — a data defect, not just a display one. + +(format "%" 1000000000000000000) -- "1000000000000000000" +(format "%" 9223372036854775807) -- "9223372036854775807" +(format "%" -9223372036854775806) -- "-9223372036854775806" +(format "%" (count (til 3000000000))) -- "3000000000" +(format "%,%" 4294967296 4294967297) -- "4294967296,4294967297" + +;; as 'STR shares none of that code — pin it as the independent oracle. +(== (format "%" 1099511627776) (as 'STR 1099511627776)) -- true + +;; pivot builds column names from the key values +(set _t64 (table [g k v] (list ['a 'a 'b] [4294967296 4294967297 4294967296] [1.0 2.0 3.0]))) +(cols (pivot _t64 'g 'k 'v sum)) -- ['g '4294967296 '4294967297] From 7d370d481d72d67faf2a1f7ba405c737c9d43440 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 16:17:25 +0300 Subject: [PATCH 22/51] fix(crash): format the banner with snprintf instead of macro string splicing The banner was spliced from string literals and the RAYFORCE_VERSION / RAYFORCE_GIT_COMMIT macros inside #ifdef arms; without the -D values a static analyser reads the literal followed by the bare macro name as two adjacent tokens and reports a syntax error. Format it with one snprintf at install time (the handler itself still never formats), with the macros defaulting to empty strings. Output unchanged. Co-Authored-By: Claude Fable 5.1 --- src/core/crash.c | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/src/core/crash.c b/src/core/crash.c index 76f894b2..41da2890 100644 --- a/src/core/crash.c +++ b/src/core/crash.c @@ -20,6 +20,7 @@ #include #include +#include #include #include @@ -71,21 +72,20 @@ static void cw_int(int v) { /* Banner precomputed at install time so the handler doesn't format it. */ static char g_banner[128]; -static void crash_banner_init(void) { - const char* v = -#ifdef RAYFORCE_VERSION - "rayforce " RAYFORCE_VERSION -#else - "rayforce" +#ifndef RAYFORCE_VERSION +#define RAYFORCE_VERSION "" #endif -#ifdef RAYFORCE_GIT_COMMIT - " (" RAYFORCE_GIT_COMMIT ")" +#ifndef RAYFORCE_GIT_COMMIT +#define RAYFORCE_GIT_COMMIT "" #endif - "\n"; - size_t vl = strlen(v); - if (vl >= sizeof(g_banner)) vl = sizeof(g_banner) - 1; - memcpy(g_banner, v, vl); - g_banner[vl] = '\0'; + +static void crash_banner_init(void) { + const char* ver = RAYFORCE_VERSION; + const char* rev = RAYFORCE_GIT_COMMIT; + int n = snprintf(g_banner, sizeof(g_banner), "rayforce%s%s%s%s%s\n", + ver[0] ? " " : "", ver, + rev[0] ? " (" : "", rev, rev[0] ? ")" : ""); + if (n < 0) g_banner[0] = '\0'; } #if defined(_WIN32) From 6b4de91312000ab4b322b5344b3efd9b28338510 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Wed, 23 Sep 2026 16:28:19 +0300 Subject: [PATCH 23/51] test(regress): drop the 24 GB til from the i64 width probe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `til` is eager, so counting a three-billion-element range built a 24 GB vector for no extra coverage — the neighbouring literals already cross 2^31, 2^32 and 2^63. Co-Authored-By: Claude Fable 5.1 --- test/rfl/regress/i64_text_width.rfl | 1 - 1 file changed, 1 deletion(-) diff --git a/test/rfl/regress/i64_text_width.rfl b/test/rfl/regress/i64_text_width.rfl index 8ae3affc..cde583f7 100644 --- a/test/rfl/regress/i64_text_width.rfl +++ b/test/rfl/regress/i64_text_width.rfl @@ -7,7 +7,6 @@ (format "%" 1000000000000000000) -- "1000000000000000000" (format "%" 9223372036854775807) -- "9223372036854775807" (format "%" -9223372036854775806) -- "-9223372036854775806" -(format "%" (count (til 3000000000))) -- "3000000000" (format "%,%" 4294967296 4294967297) -- "4294967296,4294967297" ;; as 'STR shares none of that code — pin it as the independent oracle. From cf9aa2188f8f5836e4e146abee75ad5dd3917441 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Thu, 24 Sep 2026 21:00:26 +0300 Subject: [PATCH 24/51] fix(mem): hand workers their freed blocks back at the end of every dispatch (#621) * fix(mem): hand workers their freed blocks back at the end of every dispatch A parallel operator's workers allocate per-task buffers from their own heaps and the main thread frees them once the dispatch has completed, so each block lands on the owning worker's foreign list. The owner drained that list only when an allocation found its freelists empty, which a warm worker with a partly cut pool rarely does: it kept splitting fresh pool space instead, touching new pages every round while its own freed blocks waited. Under a steady stream of such operators (a join against a large table per incoming batch) the process grew by a full 32 MB pool per worker before any block was reused, and only the idle decay, which needs the process to sit quiet, drained it earlier. The dispatcher now drains every registered heap's foreign list at the end of each parallel region (ray_heap_reclaim_workers), under the same conditions the idle decay already relies on: the flag is clear, so every worker has finished its last task and neither allocates nor frees until the next dispatch. The blocks go back to freelists only, no pages are released, so the next round reuses them without faulting. Tests: heap/reclaim_workers_drains_owner (a block owned by one heap and freed from another leaves the owner's list on reclaim and is handed out again at the same address, no new pool) and pool/dispatch_reclaims_worker_blocks (blocks allocated by real workers and freed by the main thread are back with their owners by the end of the next dispatch). Co-Authored-By: Claude Fable 5.1 * fix(mem): reclaim only the pool's own worker heaps after a dispatch The heap registry holds every thread that ever allocated, not only pool workers: a server's poll thread or an embedding's own threads are there too, and they may be allocating while a dispatch on another thread ends. Draining such a heap's foreign list from the dispatcher coalesces into freelists their owner is using at that moment. Only a pool worker parked on its semaphore is known to touch nothing of its own until the next dispatch, so each worker now publishes its heap in the pool (worker_heaps, set after ray_heap_init, cleared before it exits) and the dispatcher reclaims exactly those, plus its own list on its own thread. ray_heap_reclaim_worker takes one heap; the registry walk is gone, which also removes its per-dispatch scan of every registry slot. The heap test now checks the block's bytes leave the owner's books only on reclaim, instead of expecting the next allocation at the same address; the pool test reads the worker heaps the pool publishes. Co-Authored-By: Claude Fable 5.1 * test(pool,heap): make the reclaim tests independent of worker start-up and adopted heaps The pool test read a worker's heap slot right after ray_pool_create, which returns before the workers have started, and passed vacuously when the main thread took every task. It now waits for each worker to publish its heap, records which worker allocated each block and repeats the round until a worker really allocated something, then checks that the next dispatch left no worker with a foreign list. The heap test drains whatever an adopted heap already held before taking its baseline. Co-Authored-By: Claude Fable 5.1 * fix(pool): allocate the worker heap slots without an _Atomic cast cppcheck cannot build an AST for a cast to `_Atomic(void*)*` and fails the static-analysis run on it. The cast is not needed in C: ray_sys_alloc returns void*, which converts to the slot pointer type on its own, and sizeof(*pool->worker_heaps) names the element size without spelling the type. Co-Authored-By: Claude Fable 5.1 --------- Co-authored-by: Claude Fable 5.1 --- src/core/pool.c | 43 ++++++++++++++++++++++++++++--- src/core/pool.h | 10 ++++++++ src/mem/heap.c | 38 +++++++++++++++++++++++++++ src/mem/heap.h | 7 +++++ test/test_heap.c | 63 +++++++++++++++++++++++++++++++++++++++++++++ test/test_pool.c | 67 ++++++++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 225 insertions(+), 3 deletions(-) diff --git a/src/core/pool.c b/src/core/pool.c index c3cac800..904600fd 100644 --- a/src/core/pool.c +++ b/src/core/pool.c @@ -82,6 +82,10 @@ static void worker_loop(void* arg) { * by an earlier worker (pools and slab caches intact). */ ray_heap_init(); ray_rc_sync = true; /* workers always use atomic refcounting */ + /* Publish the heap for the dispatcher's post-dispatch reclaim (release + * pairs with its acquire: the heap is fully initialised when seen). */ + atomic_store_explicit(&pool->worker_heaps[wctx.worker_id - 1], ray_tl_heap, + memory_order_release); for (;;) { ray_sem_wait(&pool->work_ready); @@ -109,9 +113,10 @@ static void worker_loop(void* arg) { memory_order_acq_rel); } - /* No ray_heap_gc() here — removing worker GC between dispatch rounds - * ensures main can safely modify worker heaps in ray_parallel_end(). - * Eager madvise in heap_coalesce already releases pages on free. */ + /* No ray_heap_gc() here — a worker that neither allocates nor frees + * between its last pending-- and sem_wait is what lets the + * dispatcher drain worker foreign lists (ray_heap_reclaim_workers) + * and the idle decay walk worker heaps once the flag is clear. */ } /* Abandon, do not destroy. Another thread may still hold — and later @@ -120,9 +125,26 @@ static void worker_loop(void* arg) { * a heap that could be unregistered and munmapped underneath that lookup * would be a use-after-free. The heap stays registered with its pools * and is adopted by the next worker thread that starts. */ + atomic_store_explicit(&pool->worker_heaps[wctx.worker_id - 1], NULL, + memory_order_release); ray_heap_abandon(); } +/* End of a parallel region: hand every worker the blocks other threads + * freed to it since the last dispatch, so the next round reuses them + * instead of cutting fresh pool space (issue #619). Workers only — they + * are parked on the semaphore and touch nothing of their own until the + * next dispatch; any other registered heap may belong to a live thread. + * The dispatcher's own list is drained too, on its own thread. */ +static void pool_reclaim_worker_heaps(ray_pool_t* pool) { + for (uint32_t i = 0; i < pool->n_workers; i++) { + ray_heap_t* h = (ray_heap_t*)atomic_load_explicit(&pool->worker_heaps[i], + memory_order_acquire); + if (h) ray_heap_reclaim_worker(h); + } + ray_heap_flush_foreign(); +} + /* -------------------------------------------------------------------------- * ray_pool_create * -------------------------------------------------------------------------- */ @@ -192,6 +214,15 @@ static ray_err_t ray_pool_create_impl(ray_pool_t* pool, uint32_t n_workers, ray_sys_free(pool->tasks); return RAY_ERR_OOM; } + pool->worker_heaps = ray_sys_alloc(n_workers * sizeof(*pool->worker_heaps)); + if (!pool->worker_heaps) { + ray_sys_free(pool->threads); + ray_sem_destroy(&pool->work_ready); + ray_sys_free(pool->tasks); + return RAY_ERR_OOM; + } + for (uint32_t i = 0; i < n_workers; i++) + atomic_store_explicit(&pool->worker_heaps[i], NULL, memory_order_relaxed); for (uint32_t i = 0; i < n_workers; i++) { worker_ctx_t* wctx = (worker_ctx_t*)ray_sys_alloc(sizeof(worker_ctx_t)); @@ -204,6 +235,7 @@ static ray_err_t ray_pool_create_impl(ray_pool_t* pool, uint32_t n_workers, for (uint32_t j = 0; j < i; j++) { ray_thread_join(pool->threads[j]); } + ray_sys_free(pool->worker_heaps); ray_sys_free(pool->threads); ray_sem_destroy(&pool->work_ready); ray_sys_free(pool->tasks); @@ -222,6 +254,7 @@ static ray_err_t ray_pool_create_impl(ray_pool_t* pool, uint32_t n_workers, for (uint32_t j = 0; j < i; j++) { ray_thread_join(pool->threads[j]); } + ray_sys_free(pool->worker_heaps); ray_sys_free(pool->threads); ray_sem_destroy(&pool->work_ready); ray_sys_free(pool->tasks); @@ -255,6 +288,7 @@ void ray_pool_free(ray_pool_t* pool) { ray_thread_join(pool->threads[i]); } + ray_sys_free(pool->worker_heaps); ray_sys_free(pool->threads); ray_sem_destroy(&pool->work_ready); ray_sys_free(pool->tasks); @@ -392,6 +426,8 @@ void ray_pool_dispatch(ray_pool_t* pool, ray_pool_fn fn, void* ctx, * be between pending-- and sem_wait. */ atomic_thread_fence(memory_order_seq_cst); ray_rc_sync = false; + + pool_reclaim_worker_heaps(pool); } /* One round of ray_pool_dispatch_n: tasks [first, first+n_tasks), each handed @@ -466,6 +502,7 @@ static void dispatch_n_round(ray_pool_t* pool, ray_pool_fn fn, void* ctx, atomic_store_explicit(&ray_parallel_flag, 0, memory_order_release); atomic_thread_fence(memory_order_seq_cst); ray_rc_sync = false; + pool_reclaim_worker_heaps(pool); } /* -------------------------------------------------------------------------- diff --git a/src/core/pool.h b/src/core/pool.h index db0ea49f..f9914df0 100644 --- a/src/core/pool.h +++ b/src/core/pool.h @@ -52,6 +52,16 @@ struct ray_pool { uint32_t n_workers; /* number of background threads (nproc - 1) */ _Atomic(uint32_t) shutdown; + /* Heap of each worker thread [n_workers] (a ray_heap_t*), published by + * the worker once ray_heap_init has run and cleared before it exits. + * The dispatcher reads them at the end of every parallel region to hand + * each worker the blocks other threads freed to it (ray_heap_reclaim_ + * worker). Only these heaps: a worker parked on the semaphore is the + * one thread known to touch nothing of its own until the next dispatch, + * which is what makes draining its list from here sound. NULL until + * the worker is up. */ + _Atomic(void*)* worker_heaps; + /* SPMC task ring (single producer = main, multi consumer = workers + main). * * Claiming uses two MONOTONIC 64-bit cursors that are never reset, which diff --git a/src/mem/heap.c b/src/mem/heap.c index 906ef225..45244056 100644 --- a/src/mem/heap.c +++ b/src/mem/heap.c @@ -2793,6 +2793,44 @@ void ray_heap_flush_foreign(void) { heap_drain_foreign(h); } +/* -------------------------------------------------------------------------- + * Post-dispatch reclaim (dispatcher side) + * + * A parallel operator's workers allocate their per-task buffers from their + * own heaps and the main thread frees them once the dispatch has completed, + * so every such block ends up on the owning worker's foreign list. The + * owner takes that list back only when an allocation finds its freelists + * empty at the order it needs — and a warm worker with a partly cut pool + * rarely does: it keeps splitting fresh pool space instead, touching new + * pages every round while its own freed blocks wait on the list. Under a + * steady stream of such operators the process grows by one full pool per + * worker before any block is reused, and nothing short of the idle decay + * (which needs the process to sit quiet) drains it earlier. + * + * So the dispatcher drains each worker's list at the end of each parallel + * region. The conditions are the ones the decay sweep relies on: the flag + * is clear, so every worker has done its last pending-- and is claiming + * nothing, allocating nothing and freeing nothing until the next dispatch; + * a concurrent push onto a foreign list from some other thread is safe + * because the drain takes the whole list in one exchange and leaves later + * arrivals for the next round. Only pool workers qualify: the registry + * also holds the heaps of other live threads (a server's poll thread, an + * embedding's own threads), whose freelists only their owner may touch, so + * the pool names the heaps rather than this walking the registry. No + * pages are released here — the blocks only go back to freelists, so the + * next round reuses them without faulting. Cost: one load per worker and + * the coalescing of whatever was freed cross-thread since the last + * dispatch, work the owner would otherwise do on its next dry allocation. + * -------------------------------------------------------------------------- */ + +void ray_heap_reclaim_worker(ray_heap_t* h) { + if (!h) return; + if (atomic_load_explicit(&ray_parallel_flag, memory_order_acquire) != 0) + return; + if (!atomic_load_explicit(&h->foreign, memory_order_relaxed)) return; + heap_drain_foreign(h); +} + /* -------------------------------------------------------------------------- * Pending-merge queue (lock-free LIFO) * diff --git a/src/mem/heap.h b/src/mem/heap.h index 9b8b6129..84a25828 100644 --- a/src/mem/heap.h +++ b/src/mem/heap.h @@ -226,6 +226,13 @@ void ray_heap_destroy(void); void ray_heap_abandon(void); void ray_heap_merge(ray_heap_t* src); void ray_heap_flush_foreign(void); +/* Give a parked worker's heap the blocks other threads freed to it. For the + * dispatcher, once a parallel region has ended: a worker only drains its own + * list when its freelists run dry, so between dispatches the blocks the main + * thread freed sit unused while the worker keeps cutting fresh pool space. + * The caller vouches that the owning thread is idle (a pool worker on its + * semaphore); must run with ray_parallel_flag clear, does nothing otherwise. */ +void ray_heap_reclaim_worker(ray_heap_t* h); void ray_heap_push_pending(ray_heap_t* heap); void ray_heap_drain_pending(void); uint8_t ray_order_for_size(size_t data_size); diff --git a/test/test_heap.c b/test/test_heap.c index 0e14473f..975eb7f6 100644 --- a/test/test_heap.c +++ b/test/test_heap.c @@ -41,6 +41,7 @@ #include "test.h" #include #include "mem/heap.h" +#include "core/platform.h" #include "core/pool.h" /* Restore the shipped policy after a test drives it. */ @@ -640,6 +641,67 @@ static test_result_t test_free_routes_to_owner_list(void) { PASS(); } +/* ---- The dispatcher hands a foreign block back to its owner -------------- + * + * Issue #619: a parallel operator's workers allocate per-task buffers and the + * main thread frees them after the dispatch, so they land on the owning + * worker's foreign list, where they sit until that worker's freelists run + * dry — which a warm worker's rarely do. ray_heap_reclaim_worker, run by + * the dispatcher on each worker's heap at the end of every parallel region, + * drains that list into the worker's freelists. Model it with two heaps: a + * block owned by heap_b and freed from heap_a stays parked (still on heap_b's + * books) until heap_a reclaims heap_b — and is left alone while the parallel + * flag is set. */ + +static test_result_t test_reclaim_worker_drains_owner_list(void) { + ray_heap_t* heap_a = ray_tl_heap; + + ray_tl_heap = NULL; + ray_heap_init(); + ray_heap_t* heap_b = ray_tl_heap; + TEST_ASSERT_NOT_NULL(heap_b); + /* heap_b may be a heap abandoned by an earlier test's worker, with blocks + * freed to it since; take them back first so the books below move only + * for the block this test frees. */ + ray_heap_flush_foreign(); +#if RAY_MEM_STATS + size_t booked0 = heap_b->stats.bytes_allocated; +#endif + ray_t* blk = ray_alloc(256u << 10); /* above the slab orders */ + TEST_ASSERT_NOT_NULL(blk); +#if RAY_MEM_STATS + size_t booked1 = heap_b->stats.bytes_allocated; + TEST_ASSERT(booked1 > booked0, "allocation charged to heap_b"); +#endif + + ray_tl_heap = heap_a; + ray_free(blk); + TEST_ASSERT_EQ_U((uintptr_t)atomic_load(&heap_b->foreign), (uintptr_t)blk); +#if RAY_MEM_STATS + /* Parked, not yet returned: heap_b still carries the bytes. */ + TEST_ASSERT_EQ_U(heap_b->stats.bytes_allocated, booked1); +#endif + + /* Inside a parallel region the reclaim is a no-op. */ + ray_parallel_begin(); + ray_heap_reclaim_worker(heap_b); + TEST_ASSERT_EQ_U((uintptr_t)atomic_load(&heap_b->foreign), (uintptr_t)blk); + atomic_store(&ray_parallel_flag, 0); + + /* Once it is over, another thread reclaims heap_b: the list is empty and + * the block is back on heap_b's freelists, off its books. */ + ray_heap_reclaim_worker(heap_b); + TEST_ASSERT_NULL(atomic_load(&heap_b->foreign)); +#if RAY_MEM_STATS + TEST_ASSERT_EQ_U(heap_b->stats.bytes_allocated, booked0); +#endif + + ray_tl_heap = heap_b; + ray_heap_destroy(); + ray_tl_heap = heap_a; + PASS(); +} + /* ---- A cross-thread free cycle must not grow the owner's pools ---------- * * The regression for issue #439. Workers allocate per-group temporaries from @@ -2961,6 +3023,7 @@ const test_entry_t heap_entries[] = { { "heap/gc_serial", test_heap_gc_serial, heap_setup, heap_teardown }, { "heap/gc_parallel", test_heap_gc_parallel, heap_setup, heap_teardown }, { "heap/free_routes_to_owner", test_free_routes_to_owner_list, heap_setup, heap_teardown }, + { "heap/reclaim_worker_drains_owner", test_reclaim_worker_drains_owner_list, heap_setup, heap_teardown }, { "heap/cross_heap_cycle_bounds", test_cross_heap_cycle_bounds_pools, heap_setup, heap_teardown }, { "heap/abandon_is_adopted", test_heap_abandon_is_adopted, heap_setup, heap_teardown }, { "heap/alloc_copy_list", test_alloc_copy_list_retains, heap_setup, heap_teardown }, diff --git a/test/test_pool.c b/test/test_pool.c index 8eb499a1..6a3069f1 100644 --- a/test/test_pool.c +++ b/test/test_pool.c @@ -397,6 +397,72 @@ static void pool_count_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t } } +/* -------------------------------------------------------------------------- + * Test: blocks a worker allocated inside a dispatch and the main thread freed + * after it are back with their owners by the end of the next dispatch — no + * registered heap carries a foreign list across a parallel region (issue + * #619: a stream of parallel joins grew the process by a pool per worker + * while the freed per-task buffers waited on those lists). + * -------------------------------------------------------------------------- */ + +typedef struct { + ray_t* blocks[64]; + uint32_t owner[64]; /* worker_id that allocated blocks[i]; 0 = main */ +} pool_alloc_ctx_t; + +static void pool_alloc_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { + pool_alloc_ctx_t* c = (pool_alloc_ctx_t*)ctx; + for (int64_t i = start; i < end && i < 64; i++) { + c->blocks[i] = ray_alloc(256u << 10); /* 256 KB from the caller's heap */ + c->owner[i] = worker_id; + } +} + +static test_result_t test_dispatch_reclaims_worker_blocks(void) { + ray_heap_init(); + + ray_pool_t pool; + ray_err_t err = ray_pool_create(&pool, 3); + TEST_ASSERT_EQ_I(err, RAY_OK); + + /* ray_pool_create returns before the workers have started; a worker + * publishes its heap once it has run ray_heap_init. Wait for all three + * on that published state, so every slot below is a real heap. */ + for (uint32_t w = 0; w < pool.n_workers; w++) + while (!atomic_load(&pool.worker_heaps[w])) RAY_CPU_RELAX(); + + /* Rounds of "workers allocate, main frees" until a round in which at + * least one block really came from a worker (main is worker 0 and can + * take every task when the others are slow to wake). Rounds are cheap; + * 64 of them without a worker allocation would mean the pool is not + * running its workers at all. */ + pool_alloc_ctx_t ctx = {0}; + int worker_blocks = 0; + for (int round = 0; round < 64 && worker_blocks == 0; round++) { + ray_pool_dispatch_n(&pool, pool_alloc_fn, &ctx, 64); + for (int i = 0; i < 64; i++) { + TEST_ASSERT_NOT_NULL(ctx.blocks[i]); + if (ctx.owner[i] != 0) worker_blocks++; + ray_free(ctx.blocks[i]); /* cross-thread for worker blocks */ + ctx.blocks[i] = NULL; + } + } + TEST_ASSERT(worker_blocks > 0, "some blocks were allocated by workers"); + + /* Those frees happened after the last dispatch ended, so the blocks sit + * on their owners' foreign lists now; the next dispatch hands them back. */ + pool_count_ctx_t cctx = {0}; + ray_pool_dispatch_n(&pool, pool_count_fn, &cctx, 4); + for (uint32_t w = 0; w < pool.n_workers; w++) { + ray_heap_t* wh = (ray_heap_t*)atomic_load(&pool.worker_heaps[w]); + TEST_ASSERT_NOT_NULL(wh); + TEST_ASSERT_NULL(atomic_load(&wh->foreign)); + } + + ray_pool_free(&pool); + PASS(); +} + /* -------------------------------------------------------------------------- * Test: dispatch with total_elems <= 0 returns immediately, no calls fire * -------------------------------------------------------------------------- */ @@ -1400,6 +1466,7 @@ const test_entry_t pool_entries[] = { { "pool/auto_all_logical_cpus", test_auto_all_logical_cpus, NULL, NULL }, #endif { "pool/parallel_sum", test_parallel_sum, NULL, NULL }, + { "pool/dispatch_reclaims_worker_blocks", test_dispatch_reclaims_worker_blocks, NULL, NULL }, { "pool/parallel_add", test_parallel_add, NULL, NULL }, { "pool/parallel_group_sum", test_parallel_group_sum, NULL, NULL }, { "pool/parallel_min_max", test_parallel_min_max, NULL, NULL }, From a3ba5553e6eed059b1e1d457ff77019716b82f54 Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Thu, 24 Sep 2026 21:01:07 +0300 Subject: [PATCH 25/51] fix(serde): reject trailing payload bytes (#622) --- src/store/serde.c | 8 +++++++- test/test_store.c | 19 +++++++++++++++++++ 2 files changed, 26 insertions(+), 1 deletion(-) diff --git a/src/store/serde.c b/src/store/serde.c index 6bed0ad4..ecfa2c10 100644 --- a/src/store/serde.c +++ b/src/store/serde.c @@ -1250,7 +1250,13 @@ ray_t* ray_de(ray_t* bytes) { return ray_error("domain", "deserialize: ipc header size %lld + header != buffer length %lld", (long long)hdr->size, (long long)total); int64_t len = hdr->size; - return ray_de_raw(buf + sizeof(ray_ipc_header_t), &len); + ray_t* result = ray_de_raw(buf + sizeof(ray_ipc_header_t), &len); + if (!result || RAY_IS_ERR(result)) return result; + if (len != 0) { + ray_release(result); + return ray_error("domain", "deserialize: %lld trailing payload bytes", (long long)len); + } + return result; } /* -------------------------------------------------------------------------- diff --git a/test/test_store.c b/test/test_store.c index 1e7d849a..190a3eae 100644 --- a/test/test_store.c +++ b/test/test_store.c @@ -3061,6 +3061,25 @@ static test_result_t test_serde_de_raw_default_and_errors(void) { TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_TRUE(RAY_IS_ERR(r)); ray_release(r); ray_release(w); } + /* A top-level frame must contain exactly one serialized object. Extra + * payload bytes are not valid framing and must not be silently ignored. */ + { + ray_t* w = ray_ser(ray_i64(42)); + TEST_ASSERT_NOT_NULL(w); TEST_ASSERT_FALSE(RAY_IS_ERR(w)); + int64_t total = w->len; + ray_t* trailing = ray_vec_new(RAY_U8, total + 1); + TEST_ASSERT_NOT_NULL(trailing); TEST_ASSERT_FALSE(RAY_IS_ERR(trailing)); + trailing->len = total + 1; + memcpy(ray_data(trailing), ray_data(w), (size_t)total); + ((uint8_t*)ray_data(trailing))[total] = 0xa5; + ray_ipc_header_t* hdr = (ray_ipc_header_t*)ray_data(trailing); + hdr->size += 1; + + ray_t* r = ray_de(trailing); + TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_TRUE(RAY_IS_ERR(r)); + TEST_ASSERT_MEM_EQ(7, r->sdata, "domain"); + ray_release(r); ray_release(trailing); ray_release(w); + } /* SYM vector where an element has no null terminator within bounds: * craft a payload: type=RAY_SYM(12), attrs=0, len=1, then 4 non-null * bytes and nothing else → safe_strlen returns 4 = *len, domain error */ From 31f45a72b5e086257a535041404a0f6a23cc33d2 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Fri, 25 Sep 2026 10:11:51 +0300 Subject: [PATCH 26/51] fix(select): a projection sees the projections before it (#620) --- docs/docs/queries/select.md | 56 ++ src/lang/env.c | 7 + src/lang/env.h | 3 + src/lang/eval.c | 9 +- src/ops/agg_engine.c | 12 +- src/ops/graph.c | 2 + src/ops/ops.h | 26 + src/ops/query.c | 905 +++++++++++++++++++++++++++++++- test/rfl/agg/parted_f64_agg.rfl | 8 +- test/rfl/regress/issue_617.rfl | 158 ++++++ 10 files changed, 1146 insertions(+), 40 deletions(-) create mode 100644 test/rfl/regress/issue_617.rfl diff --git a/docs/docs/queries/select.md b/docs/docs/queries/select.md index aa2f4d36..9009fb7c 100644 --- a/docs/docs/queries/select.md +++ b/docs/docs/queries/select.md @@ -35,6 +35,62 @@ Select specific columns with computed expressions: ; MSFT 378000 ``` +### Projections that build on earlier projections + +A projection sees every projection defined before it in the same `select`, +by its alias — as a bare name or as a literal symbol: + +```lisp +(select {from: t sym: sym notional: (* price volume) nn: (+ notional 1)}) +; sym notional nn +; --- -------- ------ +; AAPL 75000 75001 +; GOOG 112000 112001 +; MSFT 378000 378001 +``` + +The alias is bound after its own expression, so an alias that shadows a +source column reads the source in its own definition and the new column in +every projection after it: + +```lisp +(select {from: t price: (* price 2) p2: (+ price 1)}) +; price p2 +; ----- --- +; 300 301 +; 560 561 +; 840 841 +``` + +Only projections see aliases. `where:` and `by:` are evaluated against the +source table, so `where: (> notional 100000)` raises `schema` — filter on +the expression itself, or on a nested select. + +In a grouped select a later output may build on an earlier aggregate. An +alias that is one value per group — an aggregate, or an output built on one +— is read from the group result, so `nn` below is computed from the +per-group sum, and a chain of such outputs costs one column each. An alias +of a per-row expression (`vals: price`) is substituted where it is named, +so `m: (max vals)` is `(max price)`. + +```lisp +(select {from: t by: sym notional: (sum (* price volume)) nn: (+ notional 1)}) +``` + +Three rules follow from an alias being one value per group. Inside an +aggregate's argument a name that is a source column is always the source +column, so `s: (sum s) mx: (max s)` takes the maximum of the rows, not of a +sum. An aggregate cannot be aggregated again, through an alias or written +out: `s: (sum price) mx: (max s)` and `mx: (max (sum price))` both raise +`domain`. And an output may not put a source column beside a per-group +value outside an aggregate: `f: (* price n)` with `n: (count price)` raises +`domain`; write `(* (sum price) n)` or whichever aggregate is meant. + +A literal symbol inside a select resolves to an earlier alias or a source +column of that name, in that order; one naming neither stays a constant +symbol. Arithmetic on a symbol — a literal or a symbol column — is a `type` +error inside a select, as it is outside. + ### Whole-column projections A projection may be a whole-column verb — `distinct`, `asc`, `desc`, or diff --git a/src/lang/env.c b/src/lang/env.c index 65ba600a..9fe76eee 100644 --- a/src/lang/env.c +++ b/src/lang/env.c @@ -112,6 +112,13 @@ static struct { * threads with no bound VM, so plain __VM derefs are safe below. */ int32_t ray_env_scope_depth(void) { return __VM ? __VM->scope_depth : 0; } + +bool ray_env_query_scope_above(int32_t depth) { + if (!__VM) return false; + for (int32_t d = depth < 0 ? 0 : depth; d < __VM->scope_depth; d++) + if (__VM->scope_stack[d].kind == RAY_SCOPE_QUERY) return true; + return false; +} int32_t ray_env_global_count(void) { return g_env.count; } /* Reverse-map a builtin function object to the symbol it is bound under in the diff --git a/src/lang/env.h b/src/lang/env.h index 95ccad7d..5054a155 100644 --- a/src/lang/env.h +++ b/src/lang/env.h @@ -143,6 +143,9 @@ ray_err_t ray_env_push_scope(void); ray_err_t ray_env_push_query_scope(void); void ray_env_pop_scope(void); int32_t ray_env_scope_depth(void); +/* True when a query scope frame sits at index >= depth: a nested query has + * bound its own columns since the frame that was at depth-1 was pushed. */ +bool ray_env_query_scope_above(int32_t depth); ray_err_t ray_env_set_local(int64_t sym_id, ray_t* val); ray_t* ray_env_get_lexical_local(int64_t sym_id); bool ray_env_has_lexical_local(int64_t sym_id); diff --git a/src/lang/eval.c b/src/lang/eval.c index ced4f8ad..4e1b5342 100644 --- a/src/lang/eval.c +++ b/src/lang/eval.c @@ -3824,11 +3824,10 @@ ray_t* ray_eval(ray_t* obj) { * the rule fires only while a query is active (ray_active_query_table * is NULL otherwise). A literal naming no column returns itself. */ if (obj->type == -RAY_SYM) { - ray_t* qt = ray_active_query_table(); - if (qt && qt->type == RAY_TABLE) { - ray_t* col = ray_table_get_col(qt, obj->i64); - if (col) { ray_retain(col); ret = col; goto out; } - } + /* The column — or, during a per-row evaluation, its cell in the + * current row (ray_active_query_literal). */ + ray_t* v = ray_active_query_literal(obj->i64); + if (v) { ret = v; goto out; } } ray_retain(obj); ret = obj; goto out; diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index 8f414c61..9f19eb1a 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -928,8 +928,8 @@ static bool agg_desc_init(agg_desc_t* d, ray_graph_t* g, ray_op_ext_t* ext, for (uint32_t k = 0; k < nk; k++) d->key_data[k] = ray_data(key_cols[k]); for (uint32_t a = 0; a < na; a++) { ray_op_ext_t* ie = find_ext(g, ext->agg_ins[a]); - d->agg_syms[a] = ie->sym; - ray_t* vc = (ext->agg_ops[a] != OP_COUNT) ? ray_table_get_col(tbl, ie->sym) : NULL; + d->agg_syms[a] = ie ? ie->sym : 0; + ray_t* vc = (ext->agg_ops[a] != OP_COUNT && ie) ? ray_table_get_col(tbl, ie->sym) : NULL; d->val_data[a] = vc ? ray_data(vc) : NULL; d->val_types[a] = vc ? vc->type : RAY_I64; d->val_hasnull[a] = vc ? ray_vec_may_have_nulls(vc) : false; @@ -966,7 +966,7 @@ static bool agg_vo_init(agg_vo_t* vo, ray_graph_t* g, ray_op_ext_t* ext, ray_t* vo->block = 0; for (uint32_t a = 0; a < na; a++) { ray_op_ext_t* ie = find_ext(g, ext->agg_ins[a]); - ray_t* vc = (ext->agg_ops[a] != OP_COUNT) ? ray_table_get_col(tbl, ie->sym) : NULL; + ray_t* vc = (ext->agg_ops[a] != OP_COUNT && ie) ? ray_table_get_col(tbl, ie->sym) : NULL; int8_t in_type = vc ? vc->type : RAY_I64; vo->vts[a] = agg_resolve(ext->agg_ops[a], in_type); vo->off[a] = vo->block; @@ -5285,13 +5285,15 @@ static ray_t* exec_group_v2_run_inner(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const agg_vtable_t* vt = agg_resolve(ext->agg_ops[a], x_col->type); col = agg_run_one_bin(vt, x_col, y_col, groups.gids, nrows, groups.ngroups, kparam); } else { - ray_t* val_col = (ext->agg_ops[a] != OP_COUNT) ? ray_table_get_col(tbl, ie->sym) : NULL; + ray_t* val_col = (ext->agg_ops[a] != OP_COUNT && ie) ? ray_table_get_col(tbl, ie->sym) : NULL; int8_t in_type = val_col ? val_col->type : RAY_I64; const agg_vtable_t* vt = agg_resolve(ext->agg_ops[a], in_type); col = agg_run_one(vt, val_col, groups.gids, nrows, groups.ngroups, kparam); } if (!col || RAY_IS_ERR(col)) { agg_groups_free(&groups); ray_release(result); return col ? col : ray_error("oom", NULL); } - int64_t agg_name = agg_result_col_name(ie->sym, ext->agg_ops[a]); + /* A COUNT reads no input column, so its input node need not be a + * scan (`(count (* price 2))`) and has no ext to name it by. */ + int64_t agg_name = agg_result_col_name(ie ? ie->sym : 0, ext->agg_ops[a]); result = ray_table_add_col(result, agg_name, col); ray_release(col); } diff --git a/src/ops/graph.c b/src/ops/graph.c index 5608332d..7efba810 100644 --- a/src/ops/graph.c +++ b/src/ops/graph.c @@ -140,6 +140,8 @@ ray_graph_t* ray_graph_new(ray_t* tbl) { void ray_graph_free(ray_graph_t* g) { if (!g) return; + if (g->compile_err) { ray_release(g->compile_err); g->compile_err = NULL; } + /* Unconsumed slice-group hint (error paths / shapes that never * reached exec_group). */ if (g->sg_col) { ray_release(g->sg_col); g->sg_col = NULL; } diff --git a/src/ops/ops.h b/src/ops/ops.h index 33a50f5d..656b4fd5 100644 --- a/src/ops/ops.h +++ b/src/ops/ops.h @@ -547,6 +547,31 @@ typedef struct ray_graph { uint32_t node_id; } cexpr_env[32]; int cexpr_env_top; + + /* Output aliases of the select being compiled (src/ops/query.c): + * the projections compiled so far, in order. A name reference or a + * literal symbol consults them after the lambda/let env and before + * the source table's columns. Borrowed views into the projection + * loop's scratch arrays: set and cleared by that loop, never freed + * here. Kept apart from cexpr_env so a wide select does not eat the + * slots lambda inlining needs. */ + const int64_t* sel_alias_syms; + const uint32_t* sel_alias_ids; + int sel_alias_n; + + /* Set by compile_expr_dag when it declines an expression that can + * only fail (arithmetic on a symbol column). A caller with an + * evaluation fallback ignores it — the evaluator raises the same + * error; one without reports it instead of a generic compile + * failure. Owned; released by ray_graph_free. */ + ray_t* compile_err; + + /* > 0 while compile_expr_dag is inside a branch of `if`/`cond`. The + * DAG evaluates both arms element-wise and the condition picks one, so + * an arm's value is observable only where it is selected; the checks + * that reject an expression outright (arithmetic on a symbol) stay + * quiet inside an arm and let the arm compile as it always did. */ + int if_arm_depth; } ray_graph_t; /* ===== Morsel Iterator ===== */ @@ -918,6 +943,7 @@ ray_t* ray_lazy_append(ray_t* lazy, uint16_t opcode); * so a literal never captures a lambda/let local and resolution fires only * inside a query. Returns NULL when no query is active. */ ray_t* ray_active_query_table(void); +ray_t* ray_active_query_literal(int64_t sym); #ifdef __cplusplus } diff --git a/src/ops/query.c b/src/ops/query.c index 056d2388..f5bc3071 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -314,6 +314,36 @@ static bool dag_type_is_temporal(int8_t t) { return t == RAY_DATE || t == RAY_TIME || t == RAY_TIMESTAMP; } +/* Arithmetic on a symbol has no meaning — the eval path rejects it + * (`cannot add sym and i64`, arith.c). The DAG used to admit it because + * `promote` folds RAY_SYM into I64 and the constant folder reads a symbol + * atom as its interned id, so `(+ 'name 1)` and `(+ sym_col 1)` returned + * ids plus one without a word. Declining here sends the projection to the + * per-row eval fallback, which raises the same error as outside a query. + * Comparisons and membership keep SYM: they are not arithmetic. */ +static const char* dag_arith_verb(const char* fname, size_t fname_len) { + if (fname_len == 1) { + switch (fname[0]) { + case '+': return "add"; + case '-': return "subtract"; + case '*': return "multiply"; + case '/': return "divide"; + case '%': return "mod"; + } + } + return fname_len == 3 && memcmp(fname, "pow", 3) == 0 ? "pow" : "div"; +} + +static bool dag_arith_rejects_sym(const char* fname, size_t fname_len, + int8_t lt, int8_t rt) { + bool arith = (fname_len == 1 && (fname[0] == '+' || fname[0] == '-' || + fname[0] == '*' || fname[0] == '/' || + fname[0] == '%')) || + (fname_len == 3 && (memcmp(fname, "div", 3) == 0 || + memcmp(fname, "pow", 3) == 0)); + return arith && (lt == RAY_SYM || rt == RAY_SYM); +} + static bool dag_temporal_arith_needs_eval(const char* name, size_t len, int8_t left_type, int8_t right_type) { if (!dag_type_is_temporal(left_type) && !dag_type_is_temporal(right_type)) @@ -977,6 +1007,26 @@ static void cexpr_env_pop(ray_graph_t* g, int n) { if (g->cexpr_env_top < 0) g->cexpr_env_top = 0; /* defensive */ } +/* The projections of the select being compiled that precede the one + * being compiled now (g->sel_alias_*, published by the projection loop). + * Latest binding wins, so an alias that shadows an earlier alias — or a + * source column — is the one a later projection sees. */ +static ray_op_t* sel_alias_lookup(ray_graph_t* g, int64_t sym) { + for (int i = g->sel_alias_n - 1; i >= 0; i--) + if (g->sel_alias_syms[i] == sym) + return &g->nodes[g->sel_alias_ids[i]]; + return NULL; +} + +/* Takes the error compile_expr_dag left on the graph (see ops.h), or + * NULL. Callers that report a compile failure use it so the message + * names the actual problem when there is one. */ +static ray_t* graph_take_compile_err(ray_graph_t* g) { + ray_t* e = g->compile_err; + g->compile_err = NULL; + return e; +} + static int const_str_expr_len(ray_t* expr, size_t* out_len) { if (!expr || !out_len) return 0; if (expr->type == -RAY_STR && !(expr->attrs & ATTR_QUOTED)) { @@ -1062,6 +1112,8 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { if (expr->type == -RAY_SYM && !(expr->attrs & ATTR_QUOTED)) { ray_op_t* bound = cexpr_env_lookup(g, expr->i64); if (bound) return bound; + ray_op_t* alias = sel_alias_lookup(g, expr->i64); + if (alias) return alias; ray_t* local = ray_env_get_lexical_local(expr->i64); if (local) { if (ray_is_atom(local)) return ray_const_atom(g, local); @@ -1157,6 +1209,13 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { * the rule fires only when a query table is bound. A literal naming no * column stays a const atom node. */ if (expr->type == -RAY_SYM) { + /* A literal naming an EARLIER PROJECTION of the same select resolves + * to it, exactly as the bare name does — and ahead of a source + * column of the same name, as the bare name does. The alias store + * holds projections only, never lambda formals or let bindings, so + * a literal still never captures those. */ + ray_op_t* alias = sel_alias_lookup(g, expr->i64); + if (alias) return alias; if (g->table && g->table->type == RAY_TABLE && ray_table_get_col(g->table, expr->i64)) { ray_t* s = ray_sym_str(expr->i64); @@ -1276,7 +1335,16 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { g->cexpr_env_top++; pushed++; } + /* The body was written outside this select: a free + * name in it is a global (or a column), never an + * output alias of the select that happens to call + * the lambda. Hide the alias store while the body + * compiles; the actuals above were compiled in the + * projection's own scope and keep seeing aliases. */ + int saved_aliases = g->sel_alias_n; + g->sel_alias_n = 0; ray_op_t* result = compile_expr_dag(g, body); + g->sel_alias_n = saved_aliases; cexpr_env_pop(g, pushed); return result; } @@ -1342,11 +1410,12 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { ray_op_t* c = compile_expr_dag(g, elems[1]); if (!c) return NULL; uint32_t c_id = c->id; + g->if_arm_depth++; ray_op_t* t = compile_expr_dag(g, elems[2]); - if (!t) return NULL; - uint32_t t_id = t->id; - ray_op_t* e = compile_expr_dag(g, elems[3]); - if (!e) return NULL; + uint32_t t_id = t ? t->id : 0; + ray_op_t* e = t ? compile_expr_dag(g, elems[3]) : NULL; + g->if_arm_depth--; + if (!t || !e) return NULL; c = &g->nodes[c_id]; t = &g->nodes[t_id]; return ray_if(g, c, t, e); @@ -1575,7 +1644,9 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { } if (is_else) { if (i != n - 1) return NULL; + g->if_arm_depth++; ray_op_t* c = compile_expr_dag(g, cpair[1]); + g->if_arm_depth--; if (!c) return NULL; chain_id = c->id; } else { @@ -1583,7 +1654,9 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { ray_op_t* pred = compile_expr_dag(g, cpair[0]); if (!pred) return NULL; uint32_t pred_id = pred->id; + g->if_arm_depth++; ray_op_t* body = compile_expr_dag(g, cpair[1]); + g->if_arm_depth--; if (!body) return NULL; pred = &g->nodes[pred_id]; ray_op_t* chain = &g->nodes[chain_id]; @@ -1689,6 +1762,16 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { left->out_type, right->out_type)) return NULL; + if (g->if_arm_depth == 0 && + dag_arith_rejects_sym(fname, fname_len, + left->out_type, right->out_type)) { + if (!g->compile_err) + g->compile_err = ray_error("type", "cannot %s %s and %s", + dag_arith_verb(fname, fname_len), + ray_type_name((int8_t)-left->out_type), + ray_type_name((int8_t)-right->out_type)); + return NULL; + } if (fname_len == 3 && memcmp(fname, "pow", 3) == 0 && (!dag_pow_type_admitted(left->out_type) || !dag_pow_type_admitted(right->out_type))) @@ -1761,6 +1844,15 @@ ray_op_t* compile_expr_dag(ray_graph_t* g, ray_t* expr) { * Column-membership scoped and query-only by construction — see * ray_active_query_table. */ static _Thread_local ray_t* g_active_query_table = NULL; +/* The per-row evaluation in progress, if any: the row, the table it indexes + * and the scope depth just above its own query frame. A literal column + * name stands for the row's cell only while the active table is that table + * and no nested query has opened a scope since — a nested select, a + * per-group evaluation or a where-mask over any table (the same one + * included) binds whole columns and must see whole columns. */ +static _Thread_local int64_t g_active_query_row = -1; +static _Thread_local ray_t* g_active_query_row_tbl = NULL; +static _Thread_local int32_t g_active_query_row_depth = 0; ray_t* ray_active_query_table(void) { return g_active_query_table; } @@ -2026,6 +2118,18 @@ static int hidden_agg_shape_ok(ray_t* expr) { return op == OP_QUANTILE; } +static bool agg_arith_head_is_control(int64_t sym) { + static const char* const forms[] = { "if", "while", "times", + "do", "let", "set", "try" }; + ray_t* s = ray_sym_str(sym); + if (!s) return false; + const char* p = ray_str_ptr(s); + size_t l = ray_str_len(s); + for (size_t i = 0; i < sizeof forms / sizeof forms[0]; i++) + if (l == strlen(forms[i]) && memcmp(p, forms[i], l) == 0) return true; + return false; +} + static ray_t* agg_arith_rewrite(ray_t* expr, ray_t* tbl, ray_t** hexprs, int64_t* hnames, int* n_hidden, int cap, int* ok) { @@ -2053,6 +2157,15 @@ static ray_t* agg_arith_rewrite(ray_t* expr, ray_t* tbl, * Reject the whole output; it keeps the per-group scatter path. */ if (n > 0 && el[0] && el[0]->type == -RAY_SYM && el[0]->i64 == ray_sym_intern("fn", 2)) { *ok = 0; return NULL; } + /* `if` (and the loops) take one condition, not a vector of them: + * evaluated once over the n_groups-row result they would pick a + * single branch for every group. The scope-making forms (do, let, + * set, try) evaluate their body outside the scope the hidden slots + * are bound in. Reject such outputs so they keep the per-group + * evaluation, where the condition is a scalar and names are plain. */ + if (n > 0 && el[0] && el[0]->type == -RAY_SYM && + !(el[0]->attrs & ATTR_QUOTED) && agg_arith_head_is_control(el[0]->i64)) + { *ok = 0; return NULL; } ray_t* out = ray_list_new(0); if (!out || RAY_IS_ERR(out)) { *ok = 0; return out && RAY_IS_ERR(out) ? NULL : NULL; } for (int64_t i = 0; i < n && *ok; i++) { @@ -3448,6 +3561,29 @@ static ray_t* nonagg_eval_per_group_buf(ray_t* expr, ray_t* tbl, return res; } +/* The value a literal column-name symbol stands for while a query is + * active (see the literal rule in eval.c): the whole column, or — while a + * per-row evaluation is running — that column's cell in the current row, + * so `'price` reads what `price` reads wherever the expression is + * evaluated. Owned ref; NULL when no query is active or the name is not a + * column of it. */ +ray_t* ray_active_query_literal(int64_t sym) { + ray_t* qt = g_active_query_table; + if (!qt || qt->type != RAY_TABLE) return NULL; + ray_t* col = ray_table_get_col(qt, sym); + if (!col) return NULL; + if (g_active_query_row < 0 || qt != g_active_query_row_tbl || + g_active_query_row >= ray_len(col) || + ray_env_query_scope_above(g_active_query_row_depth)) { + ray_retain(col); return col; + } + int allocated = 0; + ray_t* cell = collection_elem(col, g_active_query_row, &allocated); + if (!cell || RAY_IS_ERR(cell)) return cell; + if (!allocated) ray_retain(cell); + return cell; +} + static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { /* Exact-size carve, mirrors nonagg_eval_per_group_core: ncols(tbl) is * a hard upper bound for collect_col_refs's deduplicated table-column @@ -3471,7 +3607,12 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { * the previous value and restore it before EVERY exit from this scope * (including error returns), exactly as the bind_all_columns sites do. */ ray_t* _aqt = g_active_query_table; - g_active_query_table = tbl; + int64_t _aqr = g_active_query_row; + ray_t* _aqrt = g_active_query_row_tbl; + int32_t _aqrd = g_active_query_row_depth; + g_active_query_table = tbl; + g_active_query_row_tbl = tbl; + g_active_query_row_depth = ray_env_scope_depth(); /* just above our own frame */ ray_t* result = NULL; int direct_typed = 0; @@ -3482,7 +3623,7 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { int allocated = 0; ray_t* arg = collection_elem(cols[i], row, &allocated); if (!arg || RAY_IS_ERR(arg)) { - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); if (result) ray_release(result); scratch_free(refs_hdr); @@ -3492,9 +3633,10 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { if (allocated) ray_release(arg); } + g_active_query_row = row; ray_t* cell = ray_eval(expr); if (!cell || RAY_IS_ERR(cell)) { - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); if (result) ray_release(result); scratch_free(refs_hdr); @@ -3507,7 +3649,7 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { if (collapsable) { result = ray_vec_new((int8_t)-t, nrows); if (!result || RAY_IS_ERR(result)) { - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); ray_release(cell); scratch_free(refs_hdr); @@ -3527,7 +3669,7 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { if (!collapsable) { result = ray_alloc(nrows * sizeof(ray_t*)); if (!result) { - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); ray_release(cell); scratch_free(refs_hdr); @@ -3547,7 +3689,7 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { ray_t* list_col = typed_vec_to_list(result, row, nrows); ray_release(result); if (RAY_IS_ERR(list_col)) { - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); ray_release(cell); scratch_free(refs_hdr); @@ -3564,7 +3706,7 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { } } - g_active_query_table = _aqt; + g_active_query_table = _aqt; g_active_query_row = _aqr; g_active_query_row_tbl = _aqrt; g_active_query_row_depth = _aqrd; ray_env_pop_scope(); scratch_free(refs_hdr); if (!result) { @@ -3587,6 +3729,10 @@ static ray_t* eval_expr_per_row(ray_t* expr, ray_t* tbl, int64_t nrows) { static ray_t* eval_expr_whole_column(ray_t* expr, ray_t* tbl) { if (ray_env_push_query_scope() != RAY_OK) return ray_error("oom", NULL); ray_t* _aqt = bind_all_columns(tbl); + /* Whole columns here, even when a per-row evaluation of an outer select + * is in progress (a nested select inside one of its outputs). */ + int64_t _aqr = g_active_query_row; + g_active_query_row = -1; ray_t* result = ray_eval(expr); /* distinct/asc/desc/reverse return a lazy DAG chain (RAY_LAZY) — force it * to a concrete vector while the source columns it captured are still @@ -3594,6 +3740,7 @@ static ray_t* eval_expr_whole_column(ray_t* expr, ray_t* tbl) { if (result && !RAY_IS_ERR(result)) result = ray_lazy_materialize(result); g_active_query_table = _aqt; + g_active_query_row = _aqr; ray_env_pop_scope(); if (!result) return ray_error("domain", "select: whole-column expression evaluation failed"); @@ -5135,6 +5282,7 @@ static ray_t* eval_scalar_agg_outputs(ray_t** dict_elems, int64_t dict_n, /* (select {from: t [where: pred] [by: key] [col: expr ...]}) * Special form — receives unevaluated dict arg. */ ray_t* ray_select(ray_t** args, int64_t n); +static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved); ray_t* ray_update(ray_t** args, int64_t n); ray_t* ray_insert(ray_t** args, int64_t n); ray_t* ray_upsert(ray_t** args, int64_t n); @@ -6196,7 +6344,7 @@ static ray_t* try_temporal_group_materialize(ray_t* dict, ray_t* tbl) { rewritten = ray_dict_upsert(rewritten, from, extended); ray_release(from); ray_release(extended); extended = NULL; MAT_CHECK(rewritten); - ray_t* result = ray_select(&rewritten, 1); + ray_t* result = ray_select_impl(&rewritten, 1, true); ray_release(rewritten); return result; oom: @@ -6207,7 +6355,595 @@ static ray_t* try_temporal_group_materialize(ray_t* dict, ray_t* tbl) { #undef MAT_CHECK } +/* -------------------------------------------------------------------------- + * Grouped select: a projection may refer to an earlier projection. + * + * In a grouped select every output is an aggregate (or a per-group value) + * over the SOURCE rows, so a later output cannot read an earlier one as a + * column the way the ungrouped path does through the compile-time env. + * The reference is resolved at the expression level instead: each output + * value is rewritten with every earlier alias replaced by that alias's + * (already rewritten) expression, and the rewritten dict goes through the + * ordinary classification — `nn: (+ notional 1)` after + * `notional: (sum (* price volume))` becomes `(+ (sum (* price volume)) 1)`, + * which the arith-of-aggs decomposition evaluates over the group result. + * + * Two places keep the source column instead of the alias. An alias is + * visible only to the outputs after it, so its own definition reads the + * source column of the same name. And inside an AGGREGATE's argument a + * name that is a source column stays the source column — an aggregate + * consumes rows, an alias is one value per group, so `s: (sum s) mx: (max s)` + * asks for the max of the rows, not of a sum. Outside aggregates the alias + * wins, as it does in an ungrouped select. where:, by: and the sort keys + * are not outputs and are never rewritten. A literal symbol naming an + * earlier alias resolves like the bare name; any other literal is left + * alone. Nothing is rewritten inside a lambda, a quote or a let, whose own + * bindings could shadow the alias. + * -------------------------------------------------------------------------- */ +/* Binds output `kid` = `col` in `tbl` for the projections still to be + * evaluated: replaces the column of that name when there is one, appends + * otherwise. Consumes one ref of `tbl`, returns an owned (possibly + * copied) table; `col` is retained by the table, the caller keeps its ref. */ +static ray_t* select_fallback_bind_alias(ray_t* tbl, int64_t kid, ray_t* col) { + int64_t ncols = ray_table_ncols(tbl); + for (int64_t c = 0; c < ncols; c++) { + if (ray_table_col_name(tbl, c) != kid) continue; + ray_t* t2 = ray_cow(tbl); /* a failed copy leaves `tbl` untouched */ + if (!t2 || RAY_IS_ERR(t2)) { ray_release(tbl); return t2; } + ray_table_set_col_idx(t2, c, col); + if (ray_table_get_col_idx(t2, c) != col) { ray_release(t2); return ray_error("oom", NULL); } + return t2; + } + return ray_table_add_col(tbl, kid, col); +} + +static bool select_alias_skip_form(ray_t* head) { + if (!head || head->type != -RAY_SYM || (head->attrs & ATTR_QUOTED)) return false; + ray_t* s = ray_sym_str(head->i64); + if (!s) return false; + const char* p = ray_str_ptr(s); + size_t l = ray_str_len(s); + return (l == 2 && memcmp(p, "fn", 2) == 0) || + (l == 3 && memcmp(p, "let", 3) == 0) || + (l == 5 && memcmp(p, "quote", 5) == 0); +} + +static bool select_alias_head_is_agg(ray_t* head) { + if (!head || head->type != -RAY_SYM || (head->attrs & ATTR_QUOTED)) return false; + if (resolve_agg_opcode(head->i64) != 0) return true; + ray_t* s = ray_sym_str(head->i64); + return s && ray_str_len(s) == 8 && memcmp(ray_str_ptr(s), "distinct", 8) == 0; +} + +/* The first aggregate call in `expr` that sits inside another aggregate's + * argument, or NULL. In a grouped select the inner one is already one + * value per group, so the outer one has no rows left to fold; evaluated + * anyway it collapses over the whole table and every group gets the same + * number. A lambda or let body is its own scope and is not descended. */ +static ray_t* grouped_output_nested_agg(ray_t* expr, bool inside_agg) { + if (!expr || expr->type != RAY_LIST || expr->len < 1) return NULL; + ray_t** el = (ray_t**)ray_data(expr); + if (select_alias_skip_form(el[0])) return NULL; + bool agg = is_agg_expr(expr) != 0; + if (agg && inside_agg) return expr; + for (int64_t i = 1; i < expr->len; i++) { + ray_t* hit = grouped_output_nested_agg(el[i], inside_agg || agg); + if (hit) return hit; + } + return NULL; +} + +/* ── Grouped alias planning ─────────────────────────────────────────────── + * A grouped select's outputs may name earlier outputs. Two kinds of alias: + * + * ROW — the expression aggregates nothing (`vals: price`, `s: sym`, + * `p2: (* price 2)`): a value per source row. Substituted by + * value where referenced, so `m: (max vals)` is `(max price)`. + * GROUP — the expression contains an aggregate, or refers to a GROUP + * alias: one value per group. Never substituted. An output + * that refers to one outside an aggregate becomes a DERIVED + * output: it leaves the engine's dict and is evaluated after + * grouping, over the group result, where the alias is a column + * (by reference — copying the expression made a chain like + * `a3: (+ a2 a1)` double in size at every step). Aggregates + * inside a derived output are computed by the engine under + * hidden names and read back from the result. Inside an + * aggregate's argument a GROUP alias has no rows left to fold + * and is rejected. + * + * A lambda's formals shadow aliases in its body; `let` and `quote` forms + * are not descended. */ + +enum { SEL_ALIAS_ROW = 0, SEL_ALIAS_GROUP = 1 }; +#define SEL_ALIAS_MAX_NODES 65536 + +typedef struct { + int64_t n_derived; + int64_t* derived_names; /* [n_derived] */ + ray_t** derived_exprs; /* [n_derived], owned */ + int64_t n_hidden; + int64_t* hidden_names; /* [n_hidden]: engine outputs to drop from the result */ + bool sort_after; /* asc:/desc:/take: taken out of the engine's dict */ + ray_t* hdr; /* scratch block behind the arrays */ +} select_alias_plan_t; + +static void select_alias_plan_free(select_alias_plan_t* p) { + for (int64_t i = 0; i < p->n_derived; i++) + if (p->derived_exprs[i]) ray_release(p->derived_exprs[i]); + if (p->hdr) scratch_free(p->hdr); + memset(p, 0, sizeof(*p)); +} + +static bool select_sym_is(ray_t* s, const char* name, size_t len) { + if (!s || s->type != -RAY_SYM || (s->attrs & ATTR_QUOTED)) return false; + ray_t* str = ray_sym_str(s->i64); + return str && ray_str_len(str) == len && memcmp(ray_str_ptr(str), name, len) == 0; +} + +static bool select_lambda_form(ray_t* expr) { + if (!expr || expr->type != RAY_LIST || expr->len < 3) return false; + ray_t** el = (ray_t**)ray_data(expr); + return select_sym_is(el[0], "fn", 2) && ray_is_vec(el[1]) && el[1]->type == RAY_SYM; +} + +/* Aggregate calls anywhere in `expr` (is_agg_expr, the predicate + * select_extract_aggs extracts on — wider than the DAG-shaped count that + * sizes the engine's own hidden slots). */ +static int64_t count_agg_calls(ray_t* expr) { + if (!expr || expr->type != RAY_LIST) return 0; + int64_t c = is_agg_expr(expr) ? 1 : 0; + ray_t** e = (ray_t**)ray_data(expr); + for (int64_t i = 0; i < expr->len; i++) c += count_agg_calls(e[i]); + return c; +} + +/* True when `expr` reads a column of `tbl` (bare or literal name; a + * lambda's formals shadow). Used on a ROW alias's expression when it is + * substituted outside an aggregate: the source column it carries is then + * read per row, whatever name it came in under. */ +static bool select_expr_reads_col(ray_t* expr, ray_t* tbl, const int64_t* shadow, int n_shadow) { + if (!expr) return false; + if (expr->type == -RAY_SYM) { + for (int i = 0; i < n_shadow; i++) if (shadow[i] == expr->i64) return false; + return ray_table_get_col(tbl, expr->i64) != NULL; + } + if (expr->type != RAY_LIST || expr->len < 1) return false; + ray_t** el = (ray_t**)ray_data(expr); + if (select_sym_is(el[0], "quote", 5)) return false; + if (select_lambda_form(expr)) { + int64_t nf = ray_len(el[1]); + ray_t* hdr = NULL; + int64_t* merged = (int64_t*)scratch_alloc(&hdr, (size_t)(n_shadow + nf + 1) * sizeof(int64_t)); + if (!merged) return true; /* cannot tell: report the safe answer */ + for (int i = 0; i < n_shadow; i++) merged[i] = shadow[i]; + for (int64_t i = 0; i < nf; i++) merged[n_shadow + i] = sym_cell_runtime_id(el[1], i); + bool r = false; + for (int64_t i = 2; i < expr->len && !r; i++) + r = select_expr_reads_col(el[i], tbl, merged, n_shadow + (int)nf); + scratch_free(hdr); + return r; + } + for (int64_t i = 0; i < expr->len; i++) + if (select_expr_reads_col(el[i], tbl, shadow, n_shadow)) return true; + return false; +} + +static int64_t expr_node_count(ray_t* expr, int64_t cap) { + if (!expr) return 0; + if (expr->type != RAY_LIST) return 1; + int64_t n = 1; + ray_t** el = (ray_t**)ray_data(expr); + for (int64_t i = 0; i < expr->len && n < cap; i++) + n += expr_node_count(el[i], cap - n); + return n; +} + +/* Rebuilds `expr` with element i replaced by f(el[i]) for i >= from; the + * list is copied only once an element changes. `sub` returns an owned + * expression. Shared by the three walkers below. */ +typedef ray_t* (*select_alias_map_fn)(ray_t* e, void* ctx); + +static ray_t* select_alias_map_list(ray_t* expr, int64_t from, select_alias_map_fn fn, void* ctx) { + ray_t** el = (ray_t**)ray_data(expr); + ray_t* out = NULL; + for (int64_t i = from; i < expr->len; i++) { + ray_t* e2 = fn(el[i], ctx); + if (!e2 || RAY_IS_ERR(e2)) { if (out) ray_release(out); return e2 ? e2 : ray_error("oom", NULL); } + if (e2 != el[i] && !out) { + out = ray_list_new(expr->len); + if (!out || RAY_IS_ERR(out)) { ray_release(e2); return out ? out : ray_error("oom", NULL); } + for (int64_t j = 0; j < i; j++) { + out = ray_list_append(out, el[j]); + if (!out || RAY_IS_ERR(out)) { ray_release(e2); return out ? out : ray_error("oom", NULL); } + } + } + if (out) { + out = ray_list_append(out, e2); + ray_release(e2); + if (!out || RAY_IS_ERR(out)) return out ? out : ray_error("oom", NULL); + } else { + ray_release(e2); + } + } + if (!out) { ray_retain(expr); return expr; } + return out; +} + +typedef struct { + const int64_t* names; + ray_t** exprs; + const uint8_t* kinds; + int n_alias; + ray_t* tbl; + bool in_agg; + const int64_t* shadow; + int n_shadow; + bool changed; + bool refs_group; + bool refs_row; /* a source column read outside an aggregate */ + const int64_t* keys; /* by: names — columns of the group result, not per-row reads */ + int n_keys; +} select_alias_subst_ctx_t; + +static bool select_sym_in(int64_t sym, const int64_t* syms, int n) { + for (int i = 0; i < n; i++) if (syms[i] == sym) return true; + return false; +} + +static ray_t* select_alias_subst(ray_t* expr, void* vctx) { + select_alias_subst_ctx_t* c = (select_alias_subst_ctx_t*)vctx; + if (!expr) return NULL; + if (expr->type == -RAY_SYM) { + for (int i = 0; i < c->n_shadow; i++) + if (c->shadow[i] == expr->i64) { ray_retain(expr); return expr; } + /* inside an aggregate's argument a source column is the source column */ + if (c->in_agg && ray_table_get_col(c->tbl, expr->i64)) { ray_retain(expr); return expr; } + for (int i = c->n_alias - 1; i >= 0; i--) { + if (c->names[i] != expr->i64) continue; + if (c->kinds[i] == SEL_ALIAS_GROUP) { + if (c->in_agg) { + ray_t* s = ray_sym_str(expr->i64); + return ray_error("domain", + "select by: `%.*s` is an aggregate of the group and cannot be aggregated again", + s ? (int)ray_str_len(s) : 1, s ? ray_str_ptr(s) : "?"); + } + c->refs_group = true; + ray_retain(expr); return expr; + } + c->changed = true; + if (!c->in_agg && select_expr_reads_col(c->exprs[i], c->tbl, c->keys, c->n_keys)) c->refs_row = true; + ray_retain(c->exprs[i]); + return c->exprs[i]; + } + if (!c->in_agg && !select_sym_in(expr->i64, c->keys, c->n_keys) && + ray_table_get_col(c->tbl, expr->i64)) c->refs_row = true; + ray_retain(expr); return expr; + } + if (expr->type != RAY_LIST || expr->len < 1) { ray_retain(expr); return expr; } + ray_t** el = (ray_t**)ray_data(expr); + if (select_sym_is(el[0], "let", 3) || select_sym_is(el[0], "quote", 5)) { ray_retain(expr); return expr; } + if (select_lambda_form(expr)) { + int64_t nf = ray_len(el[1]); + ray_t* sh_hdr = NULL; + int64_t* merged = (int64_t*)scratch_alloc(&sh_hdr, (size_t)(c->n_shadow + nf + 1) * sizeof(int64_t)); + if (!merged) return ray_error("oom", NULL); + for (int i = 0; i < c->n_shadow; i++) merged[i] = c->shadow[i]; + for (int64_t i = 0; i < nf; i++) merged[c->n_shadow + i] = sym_cell_runtime_id(el[1], i); + select_alias_subst_ctx_t inner = *c; + inner.shadow = merged; inner.n_shadow = c->n_shadow + (int)nf; + ray_t* out = select_alias_map_list(expr, 2, select_alias_subst, &inner); + scratch_free(sh_hdr); + c->changed |= inner.changed; c->refs_group |= inner.refs_group; c->refs_row |= inner.refs_row; + return out; + } + select_alias_subst_ctx_t inner = *c; + inner.in_agg = c->in_agg || select_alias_head_is_agg(el[0]); + /* a call whose head is itself a form — `((fn [x] …) 1)` — is walked + * from the head, which is where such a lambda's body lives */ + ray_t* out = select_alias_map_list(expr, el[0]->type == RAY_LIST ? 0 : 1, + select_alias_subst, &inner); + c->changed |= inner.changed; c->refs_group |= inner.refs_group; c->refs_row |= inner.refs_row; + return out; +} + +/* Replaces every aggregate call in a derived output by a hidden engine + * output `__ad` (added to `*engine`) and returns the rewritten + * expression, owned. Lambda, let and quote forms are left alone. */ +typedef struct { + ray_t** engine; + select_alias_plan_t* plan; + const int64_t* shadow; /* lambda formals in scope: an aggregate over one stays put */ + int n_shadow; +} select_extract_ctx_t; + +static bool select_expr_mentions(ray_t* expr, const int64_t* syms, int n) { + if (!expr) return false; + if (expr->type == -RAY_SYM) { + for (int i = 0; i < n; i++) if (syms[i] == expr->i64) return true; + return false; + } + if (expr->type != RAY_LIST) return false; + ray_t** el = (ray_t**)ray_data(expr); + for (int64_t i = 0; i < expr->len; i++) + if (select_expr_mentions(el[i], syms, n)) return true; + return false; +} + +static ray_t* select_extract_aggs(ray_t* expr, void* vctx) { + select_extract_ctx_t* c = (select_extract_ctx_t*)vctx; + if (!expr || expr->type != RAY_LIST || expr->len < 1) { ray_retain(expr); return expr; } + ray_t** el = (ray_t**)ray_data(expr); + if (select_sym_is(el[0], "let", 3) || select_sym_is(el[0], "quote", 5)) { ray_retain(expr); return expr; } + if (select_lambda_form(expr)) { + int64_t nf = ray_len(el[1]); + ray_t* hdr = NULL; + int64_t* merged = (int64_t*)scratch_alloc(&hdr, (size_t)(c->n_shadow + nf + 1) * sizeof(int64_t)); + if (!merged) return ray_error("oom", NULL); + for (int i = 0; i < c->n_shadow; i++) merged[i] = c->shadow[i]; + for (int64_t i = 0; i < nf; i++) merged[c->n_shadow + i] = sym_cell_runtime_id(el[1], i); + select_extract_ctx_t inner = *c; + inner.shadow = merged; inner.n_shadow = c->n_shadow + (int)nf; + ray_t* out = select_alias_map_list(expr, 2, select_extract_aggs, &inner); + scratch_free(hdr); + return out; + } + if (is_agg_expr(expr) && !select_expr_mentions(expr, c->shadow, c->n_shadow)) { + char buf[24]; + int bn = snprintf(buf, sizeof buf, "__ad%lld", (long long)c->plan->n_hidden); + int64_t hn = ray_sym_intern(buf, (size_t)bn); + ray_t* key = ray_sym(hn); + if (!key || RAY_IS_ERR(key)) return key ? key : ray_error("oom", NULL); + *c->engine = ray_dict_upsert(*c->engine, key, expr); /* retains expr */ + ray_release(key); + if (!*c->engine || RAY_IS_ERR(*c->engine)) { ray_t* e = *c->engine; *c->engine = NULL; return e ? e : ray_error("oom", NULL); } + c->plan->hidden_names[c->plan->n_hidden++] = hn; + return ray_sym(hn); + } + return select_alias_map_list(expr, el[0]->type == RAY_LIST ? 0 : 1, select_extract_aggs, c); +} + +/* Plans the grouped select: NULL when the dict has no by: or nothing refers + * to an earlier alias (nothing to do); an error; or an owned dict for the + * engine (ROW aliases substituted, derived outputs and — when there are + * any — asc:/desc:/take: removed, hidden aggregates added, from: replaced by + * the evaluated table) with `plan` filled in for select_apply_derived. */ +/* Does `expr` mention (bare or literal) any of the first `n` names? */ +static bool select_expr_mentions_key(ray_t* expr, ray_t* keys, int64_t n) { + if (!expr) return false; + if (expr->type == -RAY_SYM) { + for (int64_t i = 0; i < n; i++) if (sym_cell_runtime_id(keys, i) == expr->i64) return true; + return false; + } + if (expr->type != RAY_LIST) return false; + ray_t** el = (ray_t**)ray_data(expr); + for (int64_t i = 0; i < expr->len; i++) + if (select_expr_mentions_key(el[i], keys, n)) return true; + return false; +} + +static ray_t* select_plan_grouped_aliases(ray_t* dict, ray_t* tbl, select_alias_plan_t* plan) { + memset(plan, 0, sizeof(*plan)); + if (!dict || dict->type != RAY_DICT || !dict_get(dict, "by")) return NULL; + ray_t* keys = ray_dict_keys(dict); + ray_t* vals = ray_dict_vals(dict); + if (!keys || keys->type != RAY_SYM || !vals || vals->type != RAY_LIST) return NULL; + int64_t nd = ray_dict_len(dict); + /* Most grouped selects name no earlier output: leave before any + * allocation unless some value mentions a key that precedes it. */ + { + bool any = false; + for (int64_t i = 1; i < nd && !any; i++) + any = select_expr_mentions_key(((ray_t**)ray_data(vals))[i], keys, i); + if (!any) return NULL; + } + static const char* const reserved[] = { "from", "where", "by", "take", "asc", "desc", "nearest" }; + int64_t rid[7]; + for (int i = 0; i < 7; i++) rid[i] = ray_sym_intern(reserved[i], strlen(reserved[i])); + int64_t hidden_max = 1; + for (int64_t i = 0; i < nd; i++) + hidden_max += count_agg_calls(((ray_t**)ray_data(vals))[i]); + ray_t* work_hdr = NULL; + int64_t* names = (int64_t*)scratch_alloc(&work_hdr, + (size_t)(nd > 0 ? nd : 1) * (sizeof(int64_t) + sizeof(ray_t*) + sizeof(uint8_t))); + if (!names) return ray_error("oom", NULL); + ray_t** exprs = (ray_t**)(names + nd); + uint8_t* kinds = (uint8_t*)(exprs + nd); + plan->derived_names = (int64_t*)scratch_alloc(&plan->hdr, + (size_t)(nd > 0 ? nd : 1) * (sizeof(int64_t) + sizeof(ray_t*)) + (size_t)hidden_max * sizeof(int64_t)); + if (!plan->derived_names) { scratch_free(work_hdr); return ray_error("oom", NULL); } + plan->derived_exprs = (ray_t**)(plan->derived_names + nd); + plan->hidden_names = (int64_t*)(plan->derived_exprs + nd); + /* The by: names: a group key is a column of the group result (one value + * per group), so reading it beside a GROUP alias is not a per-row read. */ + ray_t* by_expr = dict_get(dict, "by"); + ray_t* keys_hdr = NULL; + int64_t* key_syms = NULL; + int n_key_syms = 0; + { + int64_t nk = 0; + if (by_expr && by_expr->type == -RAY_SYM) nk = 1; + else if (by_expr && ray_is_vec(by_expr) && by_expr->type == RAY_SYM) nk = ray_len(by_expr); + else if (by_expr && by_expr->type == RAY_DICT) nk = ray_dict_len(by_expr); + key_syms = (int64_t*)scratch_alloc(&keys_hdr, (size_t)(nk > 0 ? nk : 1) * sizeof(int64_t)); + if (!key_syms) { scratch_free(work_hdr); select_alias_plan_free(plan); return ray_error("oom", NULL); } + if (by_expr && by_expr->type == -RAY_SYM) key_syms[n_key_syms++] = by_expr->i64; + else if (by_expr && ray_is_vec(by_expr) && by_expr->type == RAY_SYM) + for (int64_t k = 0; k < nk; k++) key_syms[n_key_syms++] = sym_cell_runtime_id(by_expr, k); + else if (by_expr && by_expr->type == RAY_DICT) { + ray_t* bk = ray_dict_keys(by_expr); + if (bk && bk->type == RAY_SYM) + for (int64_t k = 0; k < nk; k++) key_syms[n_key_syms++] = sym_cell_runtime_id(bk, k); + } + } + int n_alias = 0; + ray_t* engine = NULL; /* owned once anything changes */ + ray_t* err = NULL; + for (int64_t i = 0; i < nd && !err; i++) { + int64_t kid = sym_cell_runtime_id(keys, i); + bool is_reserved = false; + for (int r = 0; r < 7; r++) if (rid[r] == kid) { is_reserved = true; break; } + if (is_reserved) continue; + ray_t* v = ((ray_t**)ray_data(vals))[i]; + select_alias_subst_ctx_t sc = { names, exprs, kinds, n_alias, tbl, false, NULL, 0, false, false, false, key_syms, n_key_syms }; + ray_t* v1 = select_alias_subst(v, &sc); + if (!v1 || RAY_IS_ERR(v1)) { err = v1 ? v1 : ray_error("oom", NULL); break; } + if (sc.refs_group && sc.refs_row) { + /* one value per group beside a value per row: the output has + * no single shape; the column needs an aggregate around it */ + ray_t* s = ray_sym_str(kid); + ray_release(v1); + err = ray_error("domain", "select by: output `%.*s` combines a source column with a per-group value; aggregate the column", + s ? (int)ray_str_len(s) : 1, s ? ray_str_ptr(s) : "?"); + break; + } + if (sc.changed && expr_node_count(v1, SEL_ALIAS_MAX_NODES) >= SEL_ALIAS_MAX_NODES) { + ray_t* s = ray_sym_str(kid); + ray_release(v1); + err = ray_error("limit", "select by: output `%.*s` grows too large once its aliases are expanded", + s ? (int)ray_str_len(s) : 1, s ? ray_str_ptr(s) : "?"); + break; + } + ray_t* key = ray_sym(kid); + if (!key || RAY_IS_ERR(key)) { ray_release(v1); err = key ? key : ray_error("oom", NULL); break; } + if (sc.refs_group) { + /* derived: out of the engine's dict, aggregates inside it in */ + if (!engine) { engine = dict; ray_retain(engine); } + engine = ray_dict_remove(engine, key); + if (!engine || RAY_IS_ERR(engine)) { ray_release(key); ray_release(v1); err = engine ? engine : ray_error("oom", NULL); engine = NULL; break; } + select_extract_ctx_t xc = { &engine, plan, NULL, 0 }; + ray_t* v2 = select_extract_aggs(v1, &xc); + ray_release(v1); + if (!v2 || RAY_IS_ERR(v2)) { ray_release(key); err = v2 ? v2 : ray_error("oom", NULL); break; } + plan->derived_names[plan->n_derived] = kid; + plan->derived_exprs[plan->n_derived] = v2; /* owned by the plan */ + plan->n_derived++; + names[n_alias] = kid; exprs[n_alias] = v2; kinds[n_alias] = SEL_ALIAS_GROUP; + n_alias++; + ray_release(key); + continue; + } + if (sc.changed) { + if (!engine) { engine = dict; ray_retain(engine); } + engine = ray_dict_upsert(engine, key, v1); /* retains v1 */ + if (!engine || RAY_IS_ERR(engine)) { ray_release(key); ray_release(v1); err = engine ? engine : ray_error("oom", NULL); engine = NULL; break; } + } + ray_release(key); + /* bind AFTER the value: visible to later outputs only. exprs[] + * borrows v1 from the dict it now lives in. */ + names[n_alias] = kid; + exprs[n_alias] = v1; + kinds[n_alias] = expr_contains_agg(v1) ? SEL_ALIAS_GROUP : SEL_ALIAS_ROW; + n_alias++; + ray_release(v1); + } + scratch_free(work_hdr); + scratch_free(keys_hdr); + if (err) { if (engine) ray_release(engine); select_alias_plan_free(plan); return err; } + if (!engine) { select_alias_plan_free(plan); return NULL; } + /* Sort keys and take: may name a derived output: apply them after. */ + if (plan->n_derived > 0) { + for (int r = 3; r <= 5; r++) { + ray_t* key = ray_sym(rid[r]); + if (!key || RAY_IS_ERR(key)) { ray_release(engine); select_alias_plan_free(plan); return key ? key : ray_error("oom", NULL); } + if (dict_get(dict, reserved[r])) plan->sort_after = true; + engine = ray_dict_remove(engine, key); + ray_release(key); + if (!engine || RAY_IS_ERR(engine)) { select_alias_plan_free(plan); return engine ? engine : ray_error("oom", NULL); } + } + } + ray_t* from_key = ray_sym(rid[0]); + if (!from_key || RAY_IS_ERR(from_key)) { ray_release(engine); select_alias_plan_free(plan); return from_key ? from_key : ray_error("oom", NULL); } + engine = ray_dict_upsert(engine, from_key, tbl); + ray_release(from_key); + if (!engine || RAY_IS_ERR(engine)) { select_alias_plan_free(plan); return engine ? engine : ray_error("oom", NULL); } + return engine; +} + +/* A derived output is evaluated once over the group result with every + * column bound (vector semantics) unless it holds a form that takes one + * condition or one value — `if` and the other control forms, a lambda + * body — in which case it is evaluated row by row. */ +static bool select_derived_needs_rows(ray_t* expr) { + if (!expr || expr->type != RAY_LIST || expr->len < 1) return false; + ray_t** el = (ray_t**)ray_data(expr); + if (el[0]->type != -RAY_SYM || (el[0]->attrs & ATTR_QUOTED)) return true; /* ((fn …) …) and friends */ + if (agg_arith_head_is_control(el[0]->i64) || select_sym_is(el[0], "cond", 4)) return true; + ray_t* gv = ray_env_get(el[0]->i64); + if (gv && gv->type == RAY_LAMBDA) return true; + for (int64_t i = 1; i < expr->len; i++) + if (select_derived_needs_rows(el[i])) return true; + return false; +} + +/* Evaluates the derived outputs over the engine's result (in order, each + * one visible to the next), drops the hidden aggregate columns, restores + * the dict's output order and applies the sort keys and take: that were + * held back. Consumes `result`. */ +static ray_t* select_apply_derived(ray_t* result, ray_t* dict, select_alias_plan_t* plan) { + if (!result || RAY_IS_ERR(result)) return result; + if (ray_is_lazy(result)) result = ray_lazy_materialize(result); + if (!result || RAY_IS_ERR(result)) return result; + if (result->type != RAY_TABLE) return result; + int64_t n_groups = ray_table_nrows(result); + for (int64_t d = 0; d < plan->n_derived; d++) { + ray_t* expr = plan->derived_exprs[d]; + ray_t* col = select_derived_needs_rows(expr) + ? eval_expr_per_row(expr, result, n_groups) + : eval_expr_whole_column(expr, result); + if (!col || RAY_IS_ERR(col)) { ray_release(result); return col ? col : ray_error("oom", NULL); } + if (ray_is_lazy(col)) col = ray_lazy_materialize(col); + if (!col || RAY_IS_ERR(col)) { ray_release(result); return col ? col : ray_error("oom", NULL); } + if (!(ray_is_vec(col) || col->type == RAY_LIST) || ray_len(col) != n_groups) { + ray_t* nm = ray_sym_str(plan->derived_names[d]); + ray_release(col); ray_release(result); + return ray_error("domain", "select by: output `%.*s` did not evaluate to one value per group", + nm ? (int)ray_str_len(nm) : 1, nm ? ray_str_ptr(nm) : "?"); + } + result = select_fallback_bind_alias(result, plan->derived_names[d], col); + ray_release(col); + if (!result || RAY_IS_ERR(result)) return result ? result : ray_error("oom", NULL); + } + /* Key columns first (whatever the engine emitted that is not an output), + * then the outputs in the dict's order, hidden aggregates dropped. */ + DICT_VIEW_DECL(dv); + DICT_VIEW_OPEN(dict, dv); + if (DICT_VIEW_OVERFLOW(dv)) { DICT_VIEW_CLOSE(dv); ray_release(result); return ray_error("oom", NULL); } + int64_t from_id = ray_sym_intern("from", 4), where_id = ray_sym_intern("where", 5), by_id = ray_sym_intern("by", 2); + int64_t take_id = ray_sym_intern("take", 4), asc_id = ray_sym_intern("asc", 3), desc_id = ray_sym_intern("desc", 4); + int64_t nearest_id = ray_sym_intern("nearest", 7); + int64_t ncols = ray_table_ncols(result); + ray_t* out = ray_table_new(ncols); + if (!out || RAY_IS_ERR(out)) { DICT_VIEW_CLOSE(dv); ray_release(result); return out ? out : ray_error("oom", NULL); } + for (int64_t c = 0; c < ncols && out && !RAY_IS_ERR(out); c++) { + int64_t cn = ray_table_col_name(result, c); + bool is_output = false, is_hidden = false; + for (int64_t i = 0; i + 1 < dv_n; i += 2) + if (dv[i]->i64 == cn && cn != from_id && cn != where_id && cn != by_id && cn != take_id && cn != asc_id && cn != desc_id && cn != nearest_id) { is_output = true; break; } + for (int64_t h = 0; h < plan->n_hidden; h++) if (plan->hidden_names[h] == cn) { is_hidden = true; break; } + if (is_output || is_hidden) continue; + out = ray_table_add_col(out, cn, ray_table_get_col_idx(result, c)); + } + for (int64_t i = 0; i + 1 < dv_n && out && !RAY_IS_ERR(out); i += 2) { + int64_t cn = dv[i]->i64; + if (cn == from_id || cn == where_id || cn == by_id || cn == take_id || cn == asc_id || cn == desc_id || cn == nearest_id) continue; + ray_t* col = ray_table_get_col(result, cn); + if (!col) continue; /* an output the engine folded away (e.g. a key projection) */ + out = ray_table_add_col(out, cn, col); + } + if (!out || RAY_IS_ERR(out)) { DICT_VIEW_CLOSE(dv); ray_release(result); return out ? out : ray_error("oom", NULL); } + ray_release(result); + result = out; + if (plan->sort_after) + result = apply_sort_take(result, dv, dv_n, asc_id, desc_id, take_id, NULL); + DICT_VIEW_CLOSE(dv); + return result; +} + ray_t* ray_select(ray_t** args, int64_t n) { + return ray_select_impl(args, n, false); +} + +static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (n < 1) return ray_error("arity", "select: expects a query dict, got %lld args", (long long)n); ray_t* dict = args[0]; if (!dict || dict->type != RAY_DICT) @@ -6261,6 +6997,20 @@ ray_t* ray_select(ray_t** args, int64_t n) { if (RAY_IS_ERR(tbl)) return tbl; if (tbl->type != RAY_TABLE) { int8_t tbl_t = tbl->type; ray_release(tbl); return ray_error("type", "select: `from:` must evaluate to a table, got %s", ray_type_name(tbl_t)); } + if (!aliases_resolved) { + select_alias_plan_t plan; + ray_t* engine = select_plan_grouped_aliases(dict, tbl, &plan); + if (engine && RAY_IS_ERR(engine)) { ray_release(tbl); return engine; } + if (engine) { + ray_t* r = ray_select_impl(&engine, 1, true); + if (plan.n_derived > 0) r = select_apply_derived(r, dict, &plan); + select_alias_plan_free(&plan); + ray_release(engine); + ray_release(tbl); + return r; + } + } + ray_t* temporal_result = try_temporal_group_materialize(dict, tbl); if (temporal_result) { ray_release(tbl); return temporal_result; } @@ -7128,7 +7878,8 @@ ray_t* ray_select(ray_t** args, int64_t n) { (size_t)hidden_max * (sizeof(ray_t*) + sizeof(int64_t)) + (size_t)nk_max * sizeof(ray_op_t*) + (size_t)aggs_max * (2 * sizeof(ray_op_t*) + 2 * sizeof(int64_t) + sizeof(uint16_t)) + - (size_t)2 * (size_t)n_dep_keys * sizeof(int64_t)); + (size_t)2 * (size_t)n_dep_keys * sizeof(int64_t) + + (size_t)(nk_max + 2 * aggs_max) * sizeof(uint32_t)); if (!nonagg_names) { scratch_free(dep_src_hdr); if (by_sym_vec_owned) ray_release(by_sym_vec_owned); @@ -7147,7 +7898,14 @@ ray_t* ray_select(ray_t** args, int64_t n) { int64_t* agg_names = agg_k + aggs_max; dep_key_names = agg_names + aggs_max; /* [n_dep_keys] */ dep_key_biases = dep_key_names + n_dep_keys; /* [n_dep_keys] */ - uint16_t* agg_ops = (uint16_t*)(dep_key_biases + n_dep_keys); + /* Node ids of key_ops / agg_ins / agg_ins2. g->nodes grows by realloc + * as expressions compile, so a pointer taken from one compile is stale + * after the next; the ids are captured with each compile and the + * pointers re-resolved from them right before the group node is built. */ + uint32_t* key_ids = (uint32_t*)(dep_key_biases + n_dep_keys); /* [nk_max] */ + uint32_t* agg_in_ids = key_ids + nk_max; /* [aggs_max] */ + uint32_t* agg_in2_ids = agg_in_ids + aggs_max; /* [aggs_max] */ + uint16_t* agg_ops = (uint16_t*)(agg_in2_ids + aggs_max); /* Copy the bridged dependent-key names/biases into this block, then release * the transient collector. n_dep_keys == 0 (non-dep path) → no-op, and * dep_src_hdr stays NULL. */ @@ -9183,6 +9941,16 @@ ray_t* ray_select(ray_t** args, int64_t n) { ray_t* expr = dict_elems[i + 1]; if (is_single_group_key_projection(by_expr, expr)) continue; + if (grouped_output_nested_agg(expr, false)) { + ray_t* nm = ray_sym_str(kid); + for (int ci = 0; ci < n_compound; ci++) + ray_release(compound_rw[ci]); + ray_graph_free(g); ray_release(tbl); + scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); + return ray_error("domain", + "select by: output `%.*s` aggregates an aggregate of the group", + nm ? (int)ray_str_len(nm) : 1, nm ? ray_str_ptr(nm) : "?"); + } if (is_group_dag_agg_expr_dag_safe(expr, tbl)) { /* dag-aggs claim output slots in order. Not a flat * forcer. */ @@ -9385,7 +10153,7 @@ ray_t* ray_select(ray_t** args, int64_t n) { } for (int64_t i = 0; i < deferred_nk && n_keys < nk_max; i++) { key_ops[n_keys] = compile_expr_dag(g, dfv[i * 2 + 1]); - if (!key_ops[n_keys]) { DICT_VIEW_CLOSE(dfv); ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("domain", "select by: failed to compile group key expression"); } + if (!key_ops[n_keys]) { ray_t* cerr = graph_take_compile_err(g); DICT_VIEW_CLOSE(dfv); ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return cerr ? cerr : ray_error("domain", "select by: failed to compile group key expression"); } n_keys++; } DICT_VIEW_CLOSE(dfv); @@ -9414,10 +10182,13 @@ ray_t* ray_select(ray_t** args, int64_t n) { /* Only a real expression is renamed: a bare column symbol lands * here too and keeps its own name. */ computed_single_key = (key_ops[0] != NULL && by_expr->type == RAY_LIST); - if (!key_ops[0]) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("domain", "select by: failed to compile group key expression"); } + if (!key_ops[0]) { ray_t* cerr = graph_take_compile_err(g); ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return cerr ? cerr : ray_error("domain", "select by: failed to compile group key expression"); } n_keys = 1; } + for (int64_t ki = 0; ki < n_keys; ki++) + key_ids[ki] = key_ops[ki]->id; + /* Collect aggregation expressions from output columns. * Non-agg expressions are tracked separately for post-DAG scatter. * agg_ins2[] is parallel to agg_ins[] — NULL for unary aggs, @@ -9454,7 +10225,12 @@ ray_t* ray_select(ray_t** args, int64_t n) { agg_ops[n_aggs] = op; /* Compile the aggregation input (the column reference) */ agg_ins[n_aggs] = compile_expr_dag(g, agg_arg); - if (!agg_ins[n_aggs]) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("domain", "select by: failed to compile aggregation argument"); } + agg_in_ids[n_aggs] = agg_ins[n_aggs] ? agg_ins[n_aggs]->id : RAY_OP_NONE; + if (!agg_ins[n_aggs]) { + ray_t* cerr = graph_take_compile_err(g); + ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); + return cerr ? cerr : ray_error("domain", "select by: failed to compile aggregation argument"); + } agg_names[n_aggs] = kid; /* Canonical aggregand type-admission (matches the scalar * builtins): reject non-numeric / absolute-temporal inputs so a @@ -9465,10 +10241,12 @@ ray_t* ray_select(ray_t** args, int64_t n) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("type", "select by: aggregation does not admit input type %s", ray_type_name(in_t)); } agg_ins2[n_aggs] = NULL; + agg_in2_ids[n_aggs] = RAY_OP_NONE; agg_k[n_aggs] = 0; if (agg_is_binary_agg(op)) { if (ray_len(val_expr) < 3) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("arity", "select by: binary aggregation requires two column arguments"); } agg_ins2[n_aggs] = compile_expr_dag(g, agg_elems[2]); + agg_in2_ids[n_aggs] = agg_ins2[n_aggs] ? agg_ins2[n_aggs]->id : RAY_OP_NONE; if (!agg_ins2[n_aggs]) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("domain", "select by: failed to compile binary aggregation second argument"); } if (agg_ins2[n_aggs]->out_type > 0 && !agg_type_admitted(op, agg_ins2[n_aggs]->out_type)) { @@ -9528,11 +10306,14 @@ ray_t* ray_select(ray_t** args, int64_t n) { uint16_t hop = resolve_agg_opcode(he[0]->i64); agg_ops[n_aggs] = hop; agg_ins[n_aggs] = compile_expr_dag(g, he[1]); + agg_in_ids[n_aggs] = agg_ins[n_aggs] ? agg_ins[n_aggs]->id : RAY_OP_NONE; if (!agg_ins[n_aggs]) { for (int ci = 0; ci < n_compound; ci++) ray_release(compound_rw[ci]); + ray_t* cerr = graph_take_compile_err(g); ray_graph_free(g); ray_release(tbl); - scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("domain", "select by: failed to compile aggregation argument"); + scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); + return cerr ? cerr : ray_error("domain", "select by: failed to compile aggregation argument"); } if (agg_ins[n_aggs]->out_type > 0 && !agg_type_admitted(hop, agg_ins[n_aggs]->out_type)) { @@ -9543,10 +10324,12 @@ ray_t* ray_select(ray_t** args, int64_t n) { scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("type", "select by: aggregation does not admit input type %s", ray_type_name(in_t)); } agg_ins2[n_aggs] = NULL; + agg_in2_ids[n_aggs] = RAY_OP_NONE; agg_k[n_aggs] = 0; if (agg_is_binary_agg(hop)) { /* hidden_agg_shape_ok admitted two bare column arguments */ agg_ins2[n_aggs] = compile_expr_dag(g, he[2]); + agg_in2_ids[n_aggs] = agg_ins2[n_aggs] ? agg_ins2[n_aggs]->id : RAY_OP_NONE; if (!agg_ins2[n_aggs]) { for (int ci = 0; ci < n_compound; ci++) ray_release(compound_rw[ci]); @@ -9610,6 +10393,15 @@ ray_t* ray_select(ray_t** args, int64_t n) { n_aggs++; } + /* Re-resolve the collected op pointers: every compile above may have + * moved g->nodes. */ + for (int64_t ki = 0; ki < n_keys; ki++) + key_ops[ki] = &g->nodes[key_ids[ki]]; + for (int64_t ai = 0; ai < n_aggs; ai++) { + agg_ins[ai] = &g->nodes[agg_in_ids[ai]]; + agg_ins2[ai] = agg_in2_ids[ai] != RAY_OP_NONE ? &g->nodes[agg_in2_ids[ai]] : NULL; + } + if (n_aggs > 0 || n_nonaggs > 0) { if (n_aggs > 0) { if (has_agg_k || has_binary_agg) { @@ -10380,13 +11172,27 @@ ray_t* ray_select(ray_t** args, int64_t n) { if (nc_max < 1) nc_max = 1; ray_t* colops_hdr = NULL; ray_op_t** col_ops = (ray_op_t**)scratch_alloc(&colops_hdr, - (size_t)nc_max * sizeof(ray_op_t*)); + (size_t)nc_max * (sizeof(ray_op_t*) + sizeof(int64_t) + sizeof(uint32_t))); if (!col_ops) { ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("oom", NULL); } int64_t nc = 0; int use_eval_fallback = 0; + /* Projections see the projections before them: once an output + * expression has compiled, its alias is published on the graph + * (g->sel_alias_*), where a later projection's name reference + * or literal symbol finds it ahead of the source table's + * columns. The binding is made AFTER the expression compiles, + * so an alias that shadows a source column still reads the + * source in its own definition and the new value in every + * later one. where:, by: and the sort keys are compiled + * outside this window and stay alias-blind. */ + int64_t* alias_syms = (int64_t*)(col_ops + nc_max); + uint32_t* alias_ids = (uint32_t*)(alias_syms + nc_max); + g->sel_alias_syms = alias_syms; + g->sel_alias_ids = alias_ids; + g->sel_alias_n = 0; for (int64_t i = 0; i + 1 < dict_n; i += 2) { int64_t kid = dict_elems[i]->i64; if (kid == from_id || kid == where_id || kid == by_id || kid == take_id || kid == asc_id || kid == desc_id || kid == nearest_id) continue; @@ -10395,9 +11201,16 @@ ray_t* ray_select(ray_t** args, int64_t n) { use_eval_fallback = 1; break; } + alias_syms[nc] = kid; + alias_ids[nc] = col_ops[nc]->id; nc++; + g->sel_alias_n = (int)nc; } + g->sel_alias_syms = NULL; + g->sel_alias_ids = NULL; + g->sel_alias_n = 0; if (use_eval_fallback) { + if (g->compile_err) { ray_release(g->compile_err); g->compile_err = NULL; } /* The fallback evaluates projections directly over `tbl`, * bypassing the DAG's ray_execute — so a WHERE clause (wired * into `root` as ray_filter) would be silently ignored. @@ -10472,6 +11285,24 @@ ray_t* ray_select(ray_t** args, int64_t n) { scratch_free(colops_hdr); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return ray_error("length", "select: output column length %lld does not match %lld", (long long)col_len, (long long)out_len); } + /* Later projections see this one: bind it in the table + * the remaining expressions are evaluated over, in place + * of a source column of the same name. A column of + * another length (`distinct`) cannot be a row of that + * table and stays unbound. */ + if (col_len == nrows) { + ray_t* bound = select_fallback_bind_alias(tbl, kid, col); + if (!bound || RAY_IS_ERR(bound)) { + ray_release(col); + ray_release(result); + if (nearest_handle_owned) ray_release(nearest_handle_owned); + if (nearest_query_owned) ray_free_raw(nearest_query_owned); + ray_graph_free(g); + scratch_free(colops_hdr); + scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return bound ? bound : ray_error("oom", NULL); + } + tbl = bound; + } result = ray_table_add_col(result, kid, col); ray_release(col); if (RAY_IS_ERR(result)) { @@ -10878,12 +11709,24 @@ ray_t* ray_select(ray_t** args, int64_t n) { ray_env_set_local(hidden_agg_names[hi], ray_table_get_col_idx(result, hbase + hi)); + int64_t n_groups = ray_table_nrows(result); for (int ci = 0; ci < n_compound && !cerr; ci++) { ray_t* v = ray_eval(compound_rw[ci]); - if (!v || RAY_IS_ERR(v)) + if (!v || RAY_IS_ERR(v)) { cerr = v ? v : ray_error("domain", "select by: failed to evaluate aggregate expression"); - else comp_cols[ci] = v; + } else if (ray_is_lazy(v) && + (!(v = ray_lazy_materialize(v)) || RAY_IS_ERR(v))) { + cerr = v ? v : ray_error("domain", + "select by: failed to evaluate aggregate expression"); + } else if (!(ray_is_vec(v) || v->type == RAY_LIST) || + ray_len(v) != n_groups) { + ray_t* nm = ray_sym_str(compound_names[ci]); + cerr = ray_error("domain", + "select by: output `%.*s` did not evaluate to one value per group", + nm ? (int)ray_str_len(nm) : 1, nm ? ray_str_ptr(nm) : "?"); + ray_release(v); + } else comp_cols[ci] = v; } ray_env_pop_scope(); } @@ -10911,14 +11754,22 @@ ray_t* ray_select(ray_t** args, int64_t n) { if (nt && !RAY_IS_ERR(nt)) { ray_release(result); result = nt; - } else if (nt) { - ray_release(nt); + } else { + /* An output that did not evaluate to one value per + * group (a scalar, say) cannot be appended: report it + * rather than hand back the hidden columns. */ + cerr = nt ? nt : ray_error("oom", NULL); } } for (int ci = 0; ci < n_compound; ci++) if (comp_cols[ci]) ray_release(comp_cols[ci]); scratch_free(cc_hdr); n_compound = 0; + if (cerr) { + ray_release(result); + ray_release(tbl); + scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); return cerr; + } } } @@ -12828,7 +13679,7 @@ ray_t* ray_update(ray_t** args, int64_t n) { /* Evaluate expression on sub-table via DAG */ ray_graph_t* ug = ray_graph_new(sub_tbl); ray_op_t* expr_op = compile_expr_dag(ug, agg_expr); - if (!expr_op) { ray_graph_free(ug); ray_release(sub_tbl); ray_release(out_col); UPDATE_BY_CLEANUP_COLS(); ray_release(groups); ray_release(tbl); DICT_VIEW_CLOSE(updv); return ray_error("domain", "update by: failed to compile aggregate expression"); } + if (!expr_op) { ray_t* cerr = graph_take_compile_err(ug); ray_graph_free(ug); ray_release(sub_tbl); ray_release(out_col); UPDATE_BY_CLEANUP_COLS(); ray_release(groups); ray_release(tbl); DICT_VIEW_CLOSE(updv); return cerr ? cerr : ray_error("domain", "update by: failed to compile aggregate expression"); } expr_op = ray_optimize(ug, expr_op); ray_t* agg_result = ray_execute(ug, expr_op); ray_graph_free(ug); @@ -13498,7 +14349,7 @@ ray_t* ray_update(ray_t** args, int64_t n) { ray_t* update_expr = dict_elems[d + 1]; ray_graph_t* ug = ray_graph_new(tbl); ray_op_t* expr_op = compile_expr_dag(ug, update_expr); - if (!expr_op) { ray_release(result); ray_release(tbl); ray_graph_free(ug); DICT_VIEW_CLOSE(upda); return ray_error("domain", "update: failed to compile new column expression"); } + if (!expr_op) { ray_t* cerr = graph_take_compile_err(ug); ray_release(result); ray_release(tbl); ray_graph_free(ug); DICT_VIEW_CLOSE(upda); return cerr ? cerr : ray_error("domain", "update: failed to compile new column expression"); } expr_op = ray_optimize(ug, expr_op); ray_t* expr_vec = ray_execute(ug, expr_op); ray_graph_free(ug); @@ -18402,7 +19253,7 @@ static ray_t* asof_eval_select_segment(ray_t* dict, ray_t* parted_tbl, int32_t s ray_retain(keys); ray_t* ndict = ray_dict_new(keys, nvals); /* consumes both */ if (!ndict || RAY_IS_ERR(ndict)) return ndict ? ndict : ray_error("oom", NULL); - ray_t* r = ray_select(&ndict, 1); + ray_t* r = ray_select_impl(&ndict, 1, true); ray_release(ndict); return r; } @@ -18942,7 +19793,7 @@ static ray_t* asof_carry_recompute(ray_t* qtab, ray_t* template, ray_t* dict = ray_dict_new(keys, vals); /* consumes keys + vals */ if (!dict || RAY_IS_ERR(dict)) return dict ? dict : ray_error("oom", NULL); - ray_t* raw = ray_select(&dict, 1); + ray_t* raw = ray_select_impl(&dict, 1, true); ray_release(dict); if (!raw || RAY_IS_ERR(raw)) return raw ? raw : ray_error("type", NULL); diff --git a/test/rfl/agg/parted_f64_agg.rfl b/test/rfl/agg/parted_f64_agg.rfl index e8562613..71d53fd8 100644 --- a/test/rfl/agg/parted_f64_agg.rfl +++ b/test/rfl/agg/parted_f64_agg.rfl @@ -101,16 +101,18 @@ (max (at Pi 'v)) -- 50 ;; ─── parted I64 sum wrap via raw-column expression fallback ────────── -;; `(first (list c))` deliberately defeats expr_compile today, leaving the -;; quoted raw column visible to the runtime agg_parted_sum path. If the +;; `(first (list …))` deliberately defeats expr_compile today, leaving the +;; raw parted column visible to the runtime agg_parted_sum path. If the ;; planner learns this shape later, this test may need a different wrapper. +;; (The column is taken with `at`: inside a select's per-row evaluation a +;; literal column name now stands for the row's cell, not the column.) (.sys.exec "rm -rf /tmp/rfl_agg_pwrap") (set Tw1 (table [v] (list [9223372036854775807]))) (set Tw2 (table [v] (list [9223372036854775807]))) (.db.splayed.set "/tmp/rfl_agg_pwrap/1/t/" Tw1) (.db.splayed.set "/tmp/rfl_agg_pwrap/2/t/" Tw2) (set Pw (.db.parted.get "/tmp/rfl_agg_pwrap/" 't)) -(at (select {r: ((fn [c] (sum (first (list c)))) 'v) from: Pw}) 'r) -- [-2 -2] +(sum (first (list (at Pw 'v)))) -- -2 ;; ─── parted DATE sum type error (agg_parted_sum line 129) ──────────── ;; DATE base type is explicitly excluded from sum — hits the type error path. diff --git a/test/rfl/regress/issue_617.rfl b/test/rfl/regress/issue_617.rfl new file mode 100644 index 00000000..9de7ab13 --- /dev/null +++ b/test/rfl/regress/issue_617.rfl @@ -0,0 +1,158 @@ +;; #617 — a select projection sees the projections before it. +;; +;; A projection is bound under its alias once its expression has compiled, +;; so every later projection can name it, bare or as a literal symbol. The +;; binding is made after the expression, so an alias that shadows a source +;; column still reads the source in its own definition and the new value in +;; every later one; where:, by: and the sort keys are compiled earlier and do +;; not see aliases. In a grouped select an alias of a per-row expression +;; is substituted where it is named; an alias that is one value per group +;; (an aggregate, or built on one) is read by reference from the group +;; result, so a chain of such outputs costs one column each, not a copy of +;; every expression before it. Arithmetic on a symbol inside a select +;; raises the type error it raises outside — it used to compute on the +;; symbol's interned id. +;; +;; Lines marked "guard" passed before the change and pin behaviour the change +;; must not alter. +(set t (table [sym price volume] (list [AAPL GOOG MSFT] [150 280 420] [500 400 900]))) + +;; the reported shape: bare name and literal symbol +(at (select {from: t sym: sym notional: (* price volume) nn: (+ notional 1)}) 'nn) -- [75001 112001 378001] +(at (select {from: t sym: sym notional: (* price volume) nn: (+ 'notional 1)}) 'nn) -- [75001 112001 378001] +(cols (select {from: t sym: sym notional: (* price volume) nn: (+ notional 1)})) -- [sym notional nn] + +;; a chain of aliases, each built on the previous +(at (select {from: t a: (* price 2) b: (+ a 1) c: (* b a)}) 'c) -- [90300 314160 706440] + +;; an alias that shadows a source column: its own definition reads the +;; source, every later projection reads the new column — bare or literal +(at (select {from: t price: (* price 2) p2: (+ price 1)}) 'price) -- [300 560 840] +(at (select {from: t price: (* price 2) p2: (+ price 1)}) 'p2) -- [301 561 841] +(at (select {from: t price: (* price 2) p2: (+ 'price 1)}) 'p2) -- [301 561 841] +;; guard: before the alias is defined the name is still the source column +(at (select {from: t p2: (+ price 1) price: (* price 2)}) 'p2) -- [151 281 421] +;; a lambda defined outside the select keeps its own free names: `rate` in +;; the body is the global, not the output of the select that calls it +(set rate 2) +(set scale (fn [x] (* x rate))) +(at (select {from: t rate: (+ price 1) v: (scale volume)}) 'v) -- [1000 800 1800] +(at (select {from: t rate: (+ price 1) r: (reverse volume) v: (scale volume)}) 'v) -- [1000 800 1800] + +;; guard: where: does not see aliases +(select {from: t notional: (* price volume) where: (> notional 100000)}) !- schema + +;; guard: a literal symbol that names no column and no alias stays a constant +(at (select {from: t o: 'price tag: 'nope}) 'tag) -- ['nope 'nope 'nope] +;; guard: a lambda formal is never captured by a literal, alias or not +(at (select {from: t o: ((fn [price] 'price) 999)}) 'o) -- [150 280 420] +(at (select {from: t n: (* price 2) f: ((fn [n] (+ n 100)) 1)}) 'f) -- [101 101 101] + +;; a projection beside a whole-column verb (evaluated outside the compiled +;; path) still sees the earlier aliases, shadowing included, as a literal too +(at (select {from: t a: (* price 2) r: (reverse volume) b: (+ a 1)}) 'b) -- [301 561 841] +(at (select {from: t a: (* price 2) r: (reverse volume) b: (+ 'a 1)}) 'b) -- [301 561 841] +(at (select {from: t r: (reverse volume) b: (+ 'price 1)}) 'b) -- [151 281 421] +(at (select {from: t price: (* price 2) r: (reverse volume) b: (+ price 1)}) 'b) -- [301 561 841] +(at (select {from: t a: (* price 2) d: (distinct sym) b: (+ a 1)}) 'b) -- [301 561 841] + +;; the number of projections does not limit what a later one can see, and +;; an inlined lambda with several formals still fits beside them +(set w (table [x y] (list [1 2 3] [10 20 30]))) +(set f5 (fn [a b c d e] (+ a (+ b (+ c (+ d e)))))) +(at (select {from: w c0: (+ x 0) c1: (+ x 1) c2: (+ x 2) c3: (+ x 3) c4: (+ x 4) c5: (+ x 5) c6: (+ x 6) c7: (+ x 7) c8: (+ x 8) c9: (+ x 9) c10: (+ x 10) c11: (+ x 11) c12: (+ x 12) c13: (+ x 13) c14: (+ x 14) c15: (+ x 15) c16: (+ x 16) c17: (+ x 17) c18: (+ x 18) c19: (+ x 19) c20: (+ x 20) c21: (+ x 21) c22: (+ x 22) c23: (+ x 23) c24: (+ x 24) c25: (+ x 25) c26: (+ x 26) c27: (+ x 27) c28: (+ x 28) c29: (+ x 29) c30: (+ x 30) c31: (+ x 31) c32: (+ x 32) c33: (+ x 33) c34: (+ x 34) z: (f5 x y x y c34)}) 'z) -- [57 80 103] +(count (cols (select {from: w c0: (+ x 0) c1: (+ x 1) c2: (+ x 2) c3: (+ x 3) c4: (+ x 4) c5: (+ x 5) c6: (+ x 6) c7: (+ x 7) c8: (+ x 8) c9: (+ x 9) c10: (+ x 10) c11: (+ x 11) c12: (+ x 12) c13: (+ x 13) c14: (+ x 14) c15: (+ x 15) c16: (+ x 16) c17: (+ x 17) c18: (+ x 18) c19: (+ x 19) c20: (+ x 20) c21: (+ x 21) c22: (+ x 22) c23: (+ x 23) c24: (+ x 24) c25: (+ x 25) c26: (+ x 26) c27: (+ x 27) c28: (+ x 28) c29: (+ x 29) c30: (+ x 30) c31: (+ x 31) c32: (+ x 32) c33: (+ x 33) c34: (+ x 34) z: (f5 x y x y c34)}))) -- 36 + +;; grouped: an output built on an aggregate alias, bare and literal, chained +(set g (table [sym price volume] (list [AAPL GOOG MSFT AAPL] [150 280 420 10] [500 400 900 100]))) +(at (select {from: g by: sym notional: (sum (* price volume)) nn: (+ notional 1)}) 'nn) -- [76001 112001 378001] +(at (select {from: g by: sym notional: (sum (* price volume)) nn: (+ 'notional 1)}) 'nn) -- [76001 112001 378001] +(at (select {from: g by: sym n: (count price) m: (* n 10) k: (+ m n)}) 'k) -- [22 11 11] +(cols (select {from: g by: sym n: (count price) m: (* n 10) k: (+ m n)})) -- [sym n m k] +;; a condition over an aggregate alias is decided per group +(at (select {from: g by: sym n: (count price) f: (if (> n 1) 1 0)}) 'f) -- [1 0 0] +(cols (select {from: g by: sym n: (count price) f: (if (> n 1) 1 0)})) -- [sym n f] +;; a chain where every output names the two before it: by reference, so the +;; expression does not double at each step (a copy of it crashed the process) +(at (select {from: g by: sym a0: (sum price) a1: (sum volume) a2: (+ a1 a0) a3: (+ a2 a1) a4: (+ a3 a2) a5: (+ a4 a3) a6: (+ a5 a4) a7: (+ a6 a5) a8: (+ a7 a6) a9: (+ a8 a7) a10: (+ a9 a8) a11: (+ a10 a9) a12: (+ a11 a10) a13: (+ a12 a11) a14: (+ a13 a12) a15: (+ a14 a13) a16: (+ a15 a14) a17: (+ a16 a15) a18: (+ a17 a16) a19: (+ a18 a17) a20: (+ a19 a18) a21: (+ a20 a19)}) 'a21) -- [7650000 6272600 12692700] +;; an output may combine an aggregate with an alias; outputs keep the +;; dict's order around a derived one; sort keys and take: may name one +(at (select {from: g by: sym n: (count price) f: (+ (sum volume) n)}) 'f) -- [602 401 901] +(cols (select {from: g by: sym n: (count price) f: (+ n 1) v: (sum volume)})) -- [sym n f v] +(at (select {from: g by: sym n: (count price) f: (+ n 1) desc: f take: 2}) 'sym) -- [AAPL GOOG] +;; a whole-column verb over an aggregate alias: one value per group still +(at (select {from: g by: sym n: (count price) f: (deltas n)}) 'f) -- [0Nl -1 0] +(at (select {from: g by: sym f: (deltas (sum price))}) 'f) -- [0Nl 120 140] +;; an inline lambda sees the alias in a grouped select as in an ungrouped one +(at (select {from: g by: sym n: (count price) f: ((fn [x] (+ x n)) 1)}) 'f) -- [3 2 2] +;; a per-row column beside a per-group value has no single shape — named +;; directly or carried in by an alias of a per-row expression +(select {from: g by: sym n: (count price) f: ((fn [x] (* x n)) price)}) !- domain +(select {from: g by: sym p2: (* price 2) n: (count price) f: (+ n p2)}) !- domain +;; aggregates inside a derived output are computed by the engine: any +;; number of them, of any shape, and inside a lambda body too +(at (select {from: g by: sym n: (count price) f: (+ n (+ (count (distinct price)) (+ (count (distinct price)) (+ (count (distinct price)) (+ (count (distinct price)) (+ (count (distinct price)) (+ (count (distinct price)) (+ (count (distinct price)) (count (distinct price))))))))))}) 'f) -- [18 9 9] +(at (select {from: g by: sym n: (count price) f: ((fn [x] (+ x (sum price))) n)}) 'f) -- [162 281 421] +;; a nested select inside a per-row output binds whole columns, whatever its +;; table and however few rows it has: the outer row never indexes into it +(set t2 (table [c d] (list [1 2 3 4 5] [10 20 30 40 50]))) +(first (at (select {from: t r: (reverse volume) b: (sum (at (select {from: t2 rr: (reverse 'd)}) 'rr))}) 'b)) -- 150 +(set t3 (table [c] (list [1 1]))) +(at (select {from: t r: (reverse volume) b: (count (select {from: t3 where: (== 'c 1)}))}) 'b) -- [2 2 2] +(at (select {from: t r: (reverse volume) b: (+ (count (select {from: t where: (> 'price 100)})) 'price)}) 'b) -- [153 283 423] +;; the same condition written out: it used to pick one branch for every +;; group; with constant branches the output went missing and the hidden +;; count appeared under a made-up name +(at (select {from: g by: sym f: (if (> (count price) 1) (count price) 0)}) 'f) -- [2 0 0] +(cols (select {from: g by: sym f: (if (> (count price) 1) 1 0)})) -- [sym f] +;; `try` evaluates its body in its own scope: decided per group as well +(at (select {from: g by: sym n: (count price) f: (try (+ n 1) 0)}) 'f) -- [3 2 2] +;; an output that is not one value per group is reported, not dropped +(select {from: g by: sym f: (at (sum price) 0)}) !- domain +;; grouped shadowing: outside an aggregate the alias wins ... +(at (select {from: g by: sym price: (sum price) p2: (+ price 1)}) 'p2) -- [161 281 421] +;; ... but inside an aggregate's argument a name that is a source column +;; stays the source column — an aggregate consumes rows, an alias is one +;; value per group (guard: this is what the query always meant) +(at (select {from: g by: sym price: (sum price) mx: (max price) av: (avg price)}) 'mx) -- [150 280 420] +(at (select {from: g by: sym price: (sum price) mx: (max price) av: (avg price)}) 'av) -- [80.0 280.0 420.0] +;; an alias that shadows nothing resolves inside an aggregate too +(at (select {from: g by: sym vals: price m: (max vals)}) 'm) -- [150 280 420] +;; a count over an expression (through an alias or written out) counts rows; +;; the written-out form used to crash the engine +(at (select {from: g by: sym p2: (* price 2) n: (count p2)}) 'n) -- [2 1 1] +(at (select {from: g by: sym n: (count (* price 2)) s: (sum price)}) 'n) -- [2 1 1] +;; a group key is one value per group: it may stand beside an alias +(at (select {from: g by: sym n: (count price) f: (if (== sym 'AAPL) n 0)}) 'f) -- [2 0 0] +(at (select {from: g by: [sym volume] n: (count price) f: (+ n volume)}) 'f) -- [501 401 901 101] +;; an aggregate alias has no rows left to aggregate — nor has the same +;; aggregate written out, which used to fold the whole table into every group +(select {from: g by: sym s: (sum price) mx: (max s)}) !- domain +(select {from: g by: sym mx: (max (sum price))}) !- domain +(select {from: g by: sym mx: (max (+ (sum price) 1))}) !- domain +;; guard: an aggregate over a whole-column verb is not a nested aggregate +(at (select {from: g by: sym c: (count (distinct price))}) 'c) -- [2 1 1] +;; the alias is not visible to an outer where:, but a nested select is +(at (select {from: (select {from: g by: sym n: (count price) nn: (+ n 1)}) where: (> nn 1)}) 'nn) -- [3 2 2] +;; a group-key alias is an alias like any other +(at (select {from: g by: sym s: sym n: (count s)}) 'n) -- [2 1 1] +;; guard: the lambda's own formal wins inside the lambda +(at (select {from: g by: sym n: (count price) f: ((fn [n] (+ n 100)) 1)}) 'f) -- [101 101 101] + +;; arithmetic on a symbol is a type error inside a select, as outside +(select {from: t nn: (+ 'nope 1)}) !- type +(select {from: t nn: (+ sym 1)}) !- type +(select {from: t nn: (* 'sym 2)}) !- type +(select {from: g by: sym s: (sum (+ sym 1))}) !- type +(update {from: t x: (+ sym 1)}) !- type +;; guard: inside a branch of `if` the arithmetic compiles as before — the +;; branch is evaluated element-wise and only read where it is selected +(at (select {from: t s: (if (> price 0) price (+ sym 1))}) 's) -- [150 280 420] +(at (select {from: g by: sym s: (sum (if (> price 0) price (+ sym 1)))}) 's) -- [160 280 420] +(select {from: t where: (< (+ sym 1) 5)}) !- type +;; guard +(select {from: g by: sym s: (sum price) x: (+ sym 1)}) !- type +(+ 'nope 1) !- type +;; guard: comparisons and membership on symbols are untouched +(count (select {from: t where: (== sym 'AAPL)})) -- 1 +(count (select {from: t where: (in sym [AAPL GOOG])})) -- 2 From d896a120bdca3a4f4bc7993b472b9d76df55bfa3 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Fri, 25 Sep 2026 18:57:52 +0300 Subject: [PATCH 27/51] perf(like): one column kernel for the direct builtin and the executor, one row pass (#623) --- src/ops/internal.h | 2 + src/ops/string.c | 565 +++++++++------------------------ src/ops/strop.c | 161 +--------- test/rfl/strop/like_vector.rfl | 48 +++ 4 files changed, 215 insertions(+), 561 deletions(-) create mode 100644 test/rfl/strop/like_vector.rfl diff --git a/src/ops/internal.h b/src/ops/internal.h index 2bb8dd9e..a1e1eeba 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -1615,6 +1615,8 @@ ray_t* exec_date_trunc(ray_graph_t* g, ray_op_t* op); /* ── string_exec.c ── */ ray_t* exec_like(ray_graph_t* g, ray_op_t* op); +/* Shared LIKE kernel over a STR/SYM column (see string.c); selection may be NULL. */ +ray_t* ray_like_vec(ray_t* input, ray_t* pat_v, ray_t* selection); ray_t* exec_ilike(ray_graph_t* g, ray_op_t* op); ray_t* exec_string_unary(ray_graph_t* g, ray_op_t* op); ray_t* exec_strlen(ray_graph_t* g, ray_op_t* op); diff --git a/src/ops/string.c b/src/ops/string.c index 7711a36b..501e8943 100644 --- a/src/ops/string.c +++ b/src/ops/string.c @@ -44,208 +44,85 @@ #define LIKE_PAR_MIN_ROWS_STR 200000 #define LIKE_PAR_MIN_ROWS_SYM 100000 -/* Pattern-resolve worker for the SYM-LIKE fast path. Runs over a - * range of sym_ids; for each marked-as-seen sid, runs the matcher and - * writes the answer to lut[sid]. Pure read-only on the inputs after - * the seen-mark phase, so workers are independent. */ +/* SYM-LIKE row worker: one pass over the rows. lut[sid] is 0 while the + * symbol is unresolved, 1 for "no match", 2 for "match". A worker that + * meets an unresolved sid runs the matcher and publishes the answer, so the + * pattern runs once per distinct symbol the rows actually name — no seen + * pass, no sweep of the dictionary (which is the whole runtime symbol table, + * usually far larger than the column's vocabulary). Two workers may resolve + * the same sid at the same moment: they store the same value, the stores + * are relaxed atomics, and the duplicate work is bounded by the worker + * count. Serially this is the two-load-one-store loop the direct builtin + * always ran; the fused predicate (fused_pred.c) keeps its LUT the same way. + * Width-specialised on the SYM dictionary width. */ typedef struct { - ray_t** sym_strings; /* runtime-domain snapshot, or NULL */ - /* sym_strings == NULL ⇒ FILE-domain column (sym-domain Phase 2): - * resolve each sid through `dom` instead. The LUT is sized by - * that domain's count at setup (fused_group.c precedent). */ + const void* base; + uint8_t* dst; + uint8_t* lut; /* [dict_n], zeroed: 0 unresolved, 1 no, 2 yes */ + uint64_t dict_n; + int sym_w; + ray_t** sym_strings; /* runtime-domain snapshot, or NULL */ + /* sym_strings == NULL ⇒ FILE-domain column: resolve each sid through + * `dom`; when pinned, `raw` reads the file prefix straight from the + * mapping (no atom materialisation, no lock). */ struct ray_sym_domain_s* dom; - /* FILE domain, pinned: the file prefix read straight from the mapping - * (no atom materialisation, no lock); positions past raw.count still - * resolve through the domain. */ ray_sym_domain_raw_t raw; bool raw_ok; - uint8_t* seen; - uint8_t* lut; const ray_glob_compiled_t* pc; bool use_simple; const char* pat_str; size_t pat_len; -} like_resolve_ctx_t; - -/* Worker for the SYM-LIKE seen-mark phase. Marks `seen[sid] = 1` for - * every row's sym_id. Multiple workers can target the same byte, so - * the store is via __atomic_store_n with relaxed ordering — same - * machine code as a plain byte store on x86, but standard-defined - * (plain non-atomic concurrent writes are UB even when the value is - * idempotent). Width-specialised on the SYM dictionary width. */ -typedef struct { - const void* base; - uint8_t* seen; - uint64_t dict_n; - int sym_w; - /* Optional rowsel — when non-NULL, skip rows already filtered out - * by an earlier WHERE conjunct. Reduces seen[] population to just - * the surviving rows' sym_ids and short-circuits phase 2 work - * (resolve runs over the smaller seen set). */ + /* What an unknown symbol answers — an id past the dictionary, or one + * with no string — is the pattern matched against "", as for an atom + * whose string cannot be resolved. */ + uint8_t empty_match; + /* Optional rowsel — when non-NULL, rows filtered out earlier are left + * untouched (rowsel_refine reads pred[r] for surviving rows only). */ const uint8_t* sel_flg; const uint32_t* sel_offs; const uint16_t* sel_idx; uint32_t sel_n_segs; - int64_t total_rows; -} like_seen_ctx_t; - -/* Macro to mark seen[sid] for a single row at index `r`. The store - * is __atomic_store_n with relaxed ordering — workers can race on the - * same byte (different rows can resolve to the same dict ID), and the - * relaxed atomic gives UB-free semantics with the same x86 codegen as - * a plain byte store. */ -#define LIKE_SEEN_MARK(SID) \ - __atomic_store_n(&seen[(SID)], (uint8_t)1, __ATOMIC_RELAXED) -#define LIKE_SEEN_MARK_ROW(W) do { \ - if ((W) == RAY_SYM_W8) { \ - uint64_t sid = ((const uint8_t*)x->base)[r]; \ - if (sid < dict_n) LIKE_SEEN_MARK(sid); \ - } else if ((W) == RAY_SYM_W16) { \ - uint64_t sid = ((const uint16_t*)x->base)[r]; \ - if (sid < dict_n) LIKE_SEEN_MARK(sid); \ - } else if ((W) == RAY_SYM_W32) { \ - uint64_t sid = ((const uint32_t*)x->base)[r]; \ - if (sid < dict_n) LIKE_SEEN_MARK(sid); \ - } else { \ - int64_t sid = ((const int64_t*)x->base)[r]; \ - if ((uint64_t)sid < dict_n) LIKE_SEEN_MARK(sid); \ - } \ -} while (0) - -static void like_seen_fn(void* vctx, uint32_t worker_id, - int64_t start, int64_t end) { - (void)worker_id; - like_seen_ctx_t* x = (like_seen_ctx_t*)vctx; - uint8_t* seen = x->seen; - uint64_t dict_n = x->dict_n; - int sym_w = x->sym_w; - - /* Selection-aware path: walk per morsel segment and only mark - * surviving rows. When the segment is fully out, skip its row - * range entirely; when fully in, run the dense width-typed loop; - * MIX builds the per-segment in-bitmap and probes per row. */ - if (x->sel_flg) { - const uint8_t* flg = x->sel_flg; - const uint32_t* offs = x->sel_offs; - const uint16_t* lidx = x->sel_idx; - uint32_t seg_lo = (uint32_t)(start / RAY_MORSEL_ELEMS); - uint32_t seg_hi = (uint32_t)((end + RAY_MORSEL_ELEMS - 1) / RAY_MORSEL_ELEMS); - if (seg_hi > x->sel_n_segs) seg_hi = x->sel_n_segs; - for (uint32_t seg = seg_lo; seg < seg_hi; seg++) { - int64_t s_lo = (int64_t)seg * RAY_MORSEL_ELEMS; - int64_t s_hi = s_lo + RAY_MORSEL_ELEMS; - if (s_lo < start) s_lo = start; - if (s_hi > end) s_hi = end; - uint8_t f = flg[seg]; - if (f == RAY_SEL_NONE) continue; - if (f == RAY_SEL_ALL) { - for (int64_t r = s_lo; r < s_hi; r++) LIKE_SEEN_MARK_ROW(sym_w); - continue; - } - uint8_t in_seg[RAY_MORSEL_ELEMS / 8] = {0}; - uint32_t off = offs[seg]; - uint32_t cnt = offs[seg + 1] - off; - for (uint32_t i = 0; i < cnt; i++) { - uint16_t loc = lidx[off + i]; - in_seg[loc >> 3] |= (uint8_t)(1u << (loc & 7)); - } - int64_t base = (int64_t)seg * RAY_MORSEL_ELEMS; - for (int64_t r = s_lo; r < s_hi; r++) { - uint16_t loc = (uint16_t)(r - base); - if (!(in_seg[loc >> 3] & (1u << (loc & 7)))) continue; - LIKE_SEEN_MARK_ROW(sym_w); - } - } - return; - } - - /* No selection: dense width-typed loop. */ - switch (sym_w) { - case RAY_SYM_W8: { - const uint8_t* d = (const uint8_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - if (sid < dict_n) LIKE_SEEN_MARK(sid); - } - break; - } - case RAY_SYM_W16: { - const uint16_t* d = (const uint16_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - if (sid < dict_n) LIKE_SEEN_MARK(sid); - } - break; - } - case RAY_SYM_W32: { - const uint32_t* d = (const uint32_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - if (sid < dict_n) LIKE_SEEN_MARK(sid); - } - break; - } - case RAY_SYM_W64: - default: { - const int64_t* d = (const int64_t*)x->base; - for (int64_t i = start; i < end; i++) { - int64_t sid = d[i]; - if ((uint64_t)sid < dict_n) LIKE_SEEN_MARK(sid); - } - break; - } - } +} like_rows_ctx_t; + +static inline uint8_t like_rows_resolve(const like_rows_ctx_t* x, uint64_t sid) { + uint8_t st = __atomic_load_n(&x->lut[sid], __ATOMIC_RELAXED); + if (st) return (uint8_t)(st - 1); + const char* sp = NULL; + size_t sl = 0; + if (x->raw_ok && (int64_t)sid < x->raw.count) { + sp = ray_sym_domain_raw_str(&x->raw, (int64_t)sid, &sl); + } else { + ray_t* str = x->sym_strings ? x->sym_strings[sid] + : ray_sym_domain_str(x->dom, (int64_t)sid); + if (str) { sp = ray_str_ptr(str); sl = ray_str_len(str); } + } + uint8_t m = x->empty_match; + if (sp) + m = (x->use_simple ? ray_glob_match_compiled(x->pc, sp, sl) + : ray_glob_match(sp, sl, x->pat_str, x->pat_len)) ? 1 : 0; + __atomic_store_n(&x->lut[sid], (uint8_t)(m + 1), __ATOMIC_RELAXED); + return m; } -#undef LIKE_SEEN_MARK_ROW -/* Worker for the SYM-LIKE row-projection phase. Reads the per-sid - * answer from `lut[]` and writes into the per-row bool destination. - * Workers write to disjoint slices of `dst`, so no synchronisation is - * needed. Width-specialised on the SYM dictionary width. */ -typedef struct { - const void* base; - uint8_t* dst; - const uint8_t* lut; - uint64_t dict_n; - int sym_w; - /* Optional rowsel — when non-NULL, leave dst[r] untouched for rows - * already filtered out (those positions don't matter to the caller - * since rowsel_refine only reads pred[r] for surviving rows). */ - const uint8_t* sel_flg; - const uint32_t* sel_offs; - const uint16_t* sel_idx; - uint32_t sel_n_segs; -} like_proj_ctx_t; - -#define LIKE_PROJ_SET_ROW(W) do { \ - if ((W) == RAY_SYM_W8) { \ - uint64_t sid = ((const uint8_t*)x->base)[r]; \ - dst[r] = (sid < dict_n) ? lut[sid] : 0; \ - } else if ((W) == RAY_SYM_W16) { \ - uint64_t sid = ((const uint16_t*)x->base)[r]; \ - dst[r] = (sid < dict_n) ? lut[sid] : 0; \ - } else if ((W) == RAY_SYM_W32) { \ - uint64_t sid = ((const uint32_t*)x->base)[r]; \ - dst[r] = (sid < dict_n) ? lut[sid] : 0; \ - } else { \ - int64_t sid = ((const int64_t*)x->base)[r]; \ - dst[r] = ((uint64_t)sid < dict_n) ? lut[sid] : 0; \ - } \ -} while (0) - -static void like_proj_fn(void* vctx, uint32_t worker_id, +#define LIKE_ROW(LOAD) do { \ + uint64_t sid = (uint64_t)(LOAD); \ + dst[r] = (sid < dict_n) ? like_rows_resolve(x, sid) : x->empty_match; \ + } while (0) +#define LIKE_ROW_W(W) do { \ + if ((W) == RAY_SYM_W8) LIKE_ROW(((const uint8_t*)x->base)[r]); \ + else if ((W) == RAY_SYM_W16) LIKE_ROW(((const uint16_t*)x->base)[r]); \ + else if ((W) == RAY_SYM_W32) LIKE_ROW(((const uint32_t*)x->base)[r]); \ + else LIKE_ROW(((const int64_t*)x->base)[r]); \ + } while (0) + +static void like_rows_fn(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { (void)worker_id; - like_proj_ctx_t* x = (like_proj_ctx_t*)vctx; + like_rows_ctx_t* x = (like_rows_ctx_t*)vctx; uint8_t* dst = x->dst; - const uint8_t* lut = x->lut; - uint64_t dict_n = x->dict_n; - int sym_w = x->sym_w; - - /* Selection-aware path: only project rows still in the rowsel. - * Rows filtered out by an earlier WHERE conjunct have dst[r] - * undefined — that's fine because ray_rowsel_refine ignores them - * when chaining the next selection. */ + const uint64_t dict_n = x->dict_n; + const int sym_w = x->sym_w; + if (x->sel_flg) { const uint8_t* flg = x->sel_flg; const uint32_t* offs = x->sel_offs; @@ -261,7 +138,7 @@ static void like_proj_fn(void* vctx, uint32_t worker_id, uint8_t f = flg[seg]; if (f == RAY_SEL_NONE) continue; if (f == RAY_SEL_ALL) { - for (int64_t r = s_lo; r < s_hi; r++) LIKE_PROJ_SET_ROW(sym_w); + for (int64_t r = s_lo; r < s_hi; r++) LIKE_ROW_W(sym_w); continue; } uint8_t in_seg[RAY_MORSEL_ELEMS / 8] = {0}; @@ -275,7 +152,7 @@ static void like_proj_fn(void* vctx, uint32_t worker_id, for (int64_t r = s_lo; r < s_hi; r++) { uint16_t loc = (uint16_t)(r - base); if (!(in_seg[loc >> 3] & (1u << (loc & 7)))) continue; - LIKE_PROJ_SET_ROW(sym_w); + LIKE_ROW_W(sym_w); } } return; @@ -284,40 +161,29 @@ static void like_proj_fn(void* vctx, uint32_t worker_id, switch (sym_w) { case RAY_SYM_W8: { const uint8_t* d = (const uint8_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - dst[i] = (sid < dict_n) ? lut[sid] : 0; - } + for (int64_t r = start; r < end; r++) LIKE_ROW(d[r]); break; } case RAY_SYM_W16: { const uint16_t* d = (const uint16_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - dst[i] = (sid < dict_n) ? lut[sid] : 0; - } + for (int64_t r = start; r < end; r++) LIKE_ROW(d[r]); break; } case RAY_SYM_W32: { const uint32_t* d = (const uint32_t*)x->base; - for (int64_t i = start; i < end; i++) { - uint64_t sid = d[i]; - dst[i] = (sid < dict_n) ? lut[sid] : 0; - } + for (int64_t r = start; r < end; r++) LIKE_ROW(d[r]); break; } case RAY_SYM_W64: default: { const int64_t* d = (const int64_t*)x->base; - for (int64_t i = start; i < end; i++) { - int64_t sid = d[i]; - dst[i] = ((uint64_t)sid < dict_n) ? lut[sid] : 0; - } + for (int64_t r = start; r < end; r++) LIKE_ROW(d[r]); break; } } } -#undef LIKE_PROJ_SET_ROW +#undef LIKE_ROW_W +#undef LIKE_ROW /* Worker for the RAY_STR-LIKE parallel path. Each task scans its * row range against the (pre-compiled) glob pattern; rows are @@ -353,30 +219,6 @@ static int64_t parted_row_count(ray_t* input) { return total; } -static void like_resolve_fn(void* ctx, uint32_t worker_id, - int64_t start, int64_t end) { - (void)worker_id; - like_resolve_ctx_t* x = (like_resolve_ctx_t*)ctx; - for (int64_t sid = start; sid < end; sid++) { - if (!x->seen[sid]) continue; - const char* sp; - size_t sl; - if (x->raw_ok && sid < x->raw.count) { - sp = ray_sym_domain_raw_str(&x->raw, sid, &sl); - } else { - ray_t* str = x->sym_strings ? x->sym_strings[sid] - : ray_sym_domain_str(x->dom, sid); - if (!str) { x->lut[sid] = 0; continue; } - sp = ray_str_ptr(str); - sl = ray_str_len(str); - } - x->lut[sid] = (x->use_simple - ? ray_glob_match_compiled(x->pc, sp, sl) - : ray_glob_match(sp, sl, x->pat_str, x->pat_len)) - ? 1 : 0; - } -} - static void exec_like_parted_str(ray_t* input, uint8_t* dst, const ray_glob_compiled_t* pc, bool use_simple, @@ -402,8 +244,7 @@ static void exec_like_parted_str(ray_t* input, uint8_t* dst, .pat_str = pat_str, .pat_len = pat_len, }; - if (pool && seg_len >= LIKE_PAR_MIN_ROWS_STR && - ray_pool_total_workers(pool) >= 2) { + if (ray_pool_par_dispatch_ok(pool, seg_len, LIKE_PAR_MIN_ROWS_STR)) { ray_pool_dispatch(pool, str_like_par_fn, &lctx, seg_len); } else { str_like_par_fn(&lctx, 0, 0, seg_len); @@ -416,7 +257,7 @@ static void exec_like_parted_sym(ray_t* input, uint8_t* dst, const ray_glob_compiled_t* pc, bool use_simple, const char* pat_str, size_t pat_len, - int64_t total_len) { + uint8_t empty_match, int64_t total_len) { ray_t** segs = (ray_t**)ray_data(input); /* Cell ids are positions in the COLUMN's domain (sym-domain * Phase 2). All partitions of a parted SYM column share ONE @@ -441,83 +282,41 @@ static void exec_like_parted_sym(ray_t* input, uint8_t* dst, dict_n = (dn > 0 && dn <= (int64_t)UINT32_MAX) ? (uint32_t)dn : 0; } ray_t* lut_hdr = NULL; - ray_t* seen_hdr = NULL; - uint8_t* lut = NULL; - uint8_t* seen = NULL; - if (dict_n > 0) { - lut = (uint8_t*)scratch_alloc (&lut_hdr, (size_t)dict_n); - seen = (uint8_t*)scratch_calloc(&seen_hdr, (size_t)dict_n); - } + uint8_t* lut = dict_n > 0 ? (uint8_t*)scratch_calloc(&lut_hdr, (size_t)dict_n) : NULL; ray_pool_t* pool = ray_pool_get(); - if (lut && seen) { - for (int64_t s = 0; s < input->len; s++) { - ray_t* seg = segs[s]; - if (!seg) continue; - int64_t seg_len = seg->len; - like_seen_ctx_t sctx = { - .base = ray_data(seg), - .seen = seen, - .dict_n = (uint64_t)dict_n, - .sym_w = (int)(seg->attrs & RAY_SYM_W_MASK), - .sel_flg = NULL, - .sel_offs = NULL, - .sel_idx = NULL, - .sel_n_segs = 0, - .total_rows = seg_len, - }; - if (pool && seg_len >= LIKE_PAR_MIN_ROWS_SYM && - ray_pool_total_workers(pool) >= 2) { - ray_pool_dispatch(pool, like_seen_fn, &sctx, seg_len); - } else { - like_seen_fn(&sctx, 0, 0, seg_len); - } - } - - like_resolve_ctx_t rctx = { - .sym_strings = sym_strings, .dom = dom, .seen = seen, .lut = lut, - .pc = pc, .use_simple = use_simple, - .pat_str = pat_str, .pat_len = pat_len, - }; - rctx.raw_ok = dom ? ray_sym_domain_raw_pin(dom, &rctx.raw) : false; - if (pool && (int64_t)dict_n >= 16384) { - ray_pool_dispatch(pool, like_resolve_fn, &rctx, (int64_t)dict_n); - } else { - like_resolve_fn(&rctx, 0, 0, (int64_t)dict_n); - } - if (rctx.raw_ok) ray_sym_domain_raw_unpin(dom); - + if (lut) { + /* One LUT for every segment: the partitions share the domain, so a + * symbol resolved in one segment is known in the next. */ + ray_sym_domain_raw_t raw; + bool raw_ok = dom ? ray_sym_domain_raw_pin(dom, &raw) : false; int64_t out_off = 0; for (int64_t s = 0; s < input->len; s++) { ray_t* seg = segs[s]; if (!seg) continue; int64_t seg_len = seg->len; - like_proj_ctx_t pctx = { - .base = ray_data(seg), - .dst = dst + out_off, - .lut = lut, - .dict_n = (uint64_t)dict_n, - .sym_w = (int)(seg->attrs & RAY_SYM_W_MASK), - .sel_flg = NULL, - .sel_offs = NULL, - .sel_idx = NULL, - .sel_n_segs = 0, + like_rows_ctx_t rctx = { + .base = ray_data(seg), .dst = dst + out_off, .lut = lut, + .dict_n = (uint64_t)dict_n, + .sym_w = (int)(seg->attrs & RAY_SYM_W_MASK), + .sym_strings = sym_strings, .dom = dom, .raw = raw, .raw_ok = raw_ok, + .pc = pc, .use_simple = use_simple, + .pat_str = pat_str, .pat_len = pat_len, + .empty_match = empty_match, }; - if (pool && seg_len >= LIKE_PAR_MIN_ROWS_SYM && - ray_pool_total_workers(pool) >= 2) { - ray_pool_dispatch(pool, like_proj_fn, &pctx, seg_len); + if (ray_pool_par_dispatch_ok(pool, seg_len, LIKE_PAR_MIN_ROWS_SYM)) { + ray_pool_dispatch(pool, like_rows_fn, &rctx, seg_len); } else { - like_proj_fn(&pctx, 0, 0, seg_len); + like_rows_fn(&rctx, 0, 0, seg_len); } out_off += seg_len; } + if (raw_ok) ray_sym_domain_raw_unpin(dom); scratch_free(lut_hdr); - scratch_free(seen_hdr); return; } if (lut_hdr) scratch_free(lut_hdr); - if (seen_hdr) scratch_free(seen_hdr); int64_t out_off = 0; for (int64_t s = 0; s < input->len; s++) { @@ -529,7 +328,7 @@ static void exec_like_parted_sym(ray_t* input, uint8_t* dst, ray_t* str = dom ? ray_sym_domain_str(dom, sym_id) : (sym_strings && (uint64_t)sym_id < (uint64_t)dict_n) ? sym_strings[sym_id] : NULL; - if (!str) { dst[out_off + i] = 0; continue; } + if (!str) { dst[out_off + i] = empty_match; continue; } const char* sp = ray_str_ptr(str); size_t sl = ray_str_len(str); dst[out_off + i] = (use_simple @@ -571,30 +370,17 @@ static ray_t* exec_like_input(ray_graph_t* g, ray_op_t* input_op) { return exec_node(g, input_op); } -ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { - ray_t* input = exec_like_input(g, op_child(g, op, 0)); - ray_t* pat_v = exec_node(g, op_child(g, op, 1)); - if (!input || RAY_IS_ERR(input)) { if (pat_v && !RAY_IS_ERR(pat_v)) ray_release(pat_v); return input; } - if (!pat_v || RAY_IS_ERR(pat_v)) { ray_release(input); return pat_v; } - +/* LIKE over a whole STR / SYM column (flat or parted): the parallel kernel + * shared by the DAG executor (exec_like) and the direct builtin (ray_like_fn), + * so `(like col pat)` and `where: (like col pat)` run the same code — one + * pattern resolve per distinct symbol, the row passes spread over the worker + * pool. `selection` is the executor's rowsel, or NULL for every row. + * Borrows `input` and `pat_v`; returns an owned BOOL vector. */ +ray_t* ray_like_vec(ray_t* input, ray_t* pat_v, ray_t* selection) { int8_t in_type = input->type; bool in_parted = RAY_IS_PARTED(in_type); int8_t base_type = in_parted ? (int8_t)RAY_PARTED_BASETYPE(in_type) : in_type; - /* Shapes this executor doesn't own — scalar subjects, list-of-atom - * columns (the legacy STRL splayed-load shape), unsupported types — - * delegate to the direct builtin BEFORE reading input->len or - * allocating: an atom's SSO bytes alias ->len, so the result alloc - * below would request a garbage capacity, and the old fallthrough - * memset made unsupported predicates silently match nothing. */ - bool vec_shape = in_parted ? (base_type == RAY_STR || RAY_IS_SYM(base_type)) - : (in_type == RAY_STR || RAY_IS_SYM(in_type)); - if (!vec_shape) { - ray_t* r = ray_like_fn(input, pat_v); - ray_release(input); ray_release(pat_v); - return r; - } - /* Get pattern string */ const char* pat_str = ray_str_ptr(pat_v); size_t pat_len = ray_str_len(pat_v); @@ -605,20 +391,19 @@ ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { * for the very common `*literal*` shape. */ ray_glob_compiled_t pc = ray_glob_compile(pat_str, pat_len); bool use_simple = pc.shape != RAY_GLOB_SHAPE_NONE; + uint8_t empty_match = (use_simple ? ray_glob_match_compiled(&pc, "", 0) + : ray_glob_match("", 0, pat_str, pat_len)) ? 1 : 0; int64_t len = in_parted ? parted_row_count(input) : input->len; ray_t* result = ray_vec_new(RAY_BOOL, len); - if (!result || RAY_IS_ERR(result)) { - ray_release(input); ray_release(pat_v); - return result; - } + if (!result || RAY_IS_ERR(result)) return result; /* input/pat_v are borrowed */ result->len = len; uint8_t* dst = (uint8_t*)ray_data(result); if (in_parted && base_type == RAY_STR) { exec_like_parted_str(input, dst, &pc, use_simple, pat_str, pat_len); } else if (in_parted && RAY_IS_SYM(base_type)) { - exec_like_parted_sym(input, dst, &pc, use_simple, pat_str, pat_len, len); + exec_like_parted_sym(input, dst, &pc, use_simple, pat_str, pat_len, empty_match, len); } else if (in_type == RAY_STR) { /* Parallel substring/glob match over RAY_STR. Wide text scans * over URL/title-like columns are memory-bandwidth bound; the @@ -637,33 +422,19 @@ ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { .pat_len = pat_len, }; ray_pool_t* str_pool = ray_pool_get(); - if (str_pool && len >= LIKE_PAR_MIN_ROWS_STR && ray_pool_total_workers(str_pool) >= 2) { + if (ray_pool_par_dispatch_ok(str_pool, len, LIKE_PAR_MIN_ROWS_STR)) { ray_pool_dispatch(str_pool, str_like_par_fn, &lctx, len); } else { str_like_par_fn(&lctx, 0, 0, len); } } else if (RAY_IS_SYM(in_type)) { - /* Dictionary-cached fast path. - * - * Three-phase pipeline: - * (1) seen-mark — single sequential row scan that flips a - * byte in `seen[]` for every referenced sym_id. Cheap; - * just sets a byte per row. - * (2) parallel pattern resolve — partition the dict_n range - * across pool workers; for each sid where seen[sid]==1, - * run the matcher and store the answer in lut[sid]. - * (3) parallel row projection — every row reads lut[sid_i]. - * - * Splitting the resolve from the row scan lets phase (2) drive - * the pattern matcher (memmem on long URL strings) across the - * worker pool. ray_sym_count is the GLOBAL dictionary so for - * a low-card column like BrowserCountry phase (1) keeps the - * resolve work bounded to that column's actual sym_ids. */ + /* Dictionary-cached path: the pattern runs once per distinct symbol + * the rows name, in the same parallel pass that writes the rows + * (like_rows_fn). Cell ids are positions in the COLUMN's domain. + * Runtime domain: borrow the global string snapshot (lock-free per + * sid). FILE domain: LUT sized by the domain's count, strings from + * the pinned mapping. */ const void* base = ray_data(input); - /* Cell ids are positions in the COLUMN's domain (sym-domain - * Phase 2). Runtime: borrow the global snapshot. FILE: LUT - * sized by the domain's count, resolved via ray_sym_domain_str - * (fused_group precedent). Pre-flip: always runtime. */ struct ray_sym_domain_s* dom = ray_sym_vec_domain(input); ray_t** sym_strings = NULL; uint32_t dict_n = 0; @@ -675,93 +446,41 @@ ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { dict_n = (dn > 0 && dn <= (int64_t)UINT32_MAX) ? (uint32_t)dn : 0; } ray_t* lut_hdr = NULL; - ray_t* seen_hdr = NULL; - uint8_t* lut = NULL; - uint8_t* seen = NULL; - if (dict_n > 0) { - lut = (uint8_t*)scratch_alloc (&lut_hdr, (size_t)dict_n); - seen = (uint8_t*)scratch_calloc(&seen_hdr, (size_t)dict_n); - } - if (lut && seen) { - int sym_w = (int)(input->attrs & RAY_SYM_W_MASK); - - ray_pool_t* pool = ray_pool_get(); - /* Pass 1: mark used sym_ids. Parallelised because for - * high-cardinality text columns the seen- - * mark scan was a 5 ms-class serial pass. Multiple workers - * may write 1 to the same byte concurrently — the value is - * idempotent so the race is benign. */ - like_seen_ctx_t sctx = { - .base = base, - .seen = seen, - .dict_n = (uint64_t)dict_n, - .sym_w = sym_w, - .sel_flg = NULL, - .sel_offs = NULL, - .sel_idx = NULL, - .sel_n_segs = 0, - .total_rows = len, - }; - if (g->selection) { - ray_rowsel_t* sm = ray_rowsel_meta(g->selection); - sctx.sel_flg = ray_rowsel_flags(g->selection); - sctx.sel_offs = ray_rowsel_offsets(g->selection); - sctx.sel_idx = ray_rowsel_idx(g->selection); - sctx.sel_n_segs = sm->n_segs; - } - if (pool && len >= LIKE_PAR_MIN_ROWS_SYM && ray_pool_total_workers(pool) >= 2) { - ray_pool_dispatch(pool, like_seen_fn, &sctx, len); - } else { - like_seen_fn(&sctx, 0, 0, len); - } - - /* Pass 2: parallel pattern resolve over the dict range. */ - like_resolve_ctx_t rctx = { - .sym_strings = sym_strings, .dom = dom, .seen = seen, .lut = lut, + uint8_t* lut = dict_n > 0 ? (uint8_t*)scratch_calloc(&lut_hdr, (size_t)dict_n) : NULL; + if (lut) { + like_rows_ctx_t rctx = { + .base = base, .dst = dst, .lut = lut, .dict_n = (uint64_t)dict_n, + .sym_w = (int)(input->attrs & RAY_SYM_W_MASK), + .sym_strings = sym_strings, .dom = dom, .pc = &pc, .use_simple = use_simple, .pat_str = pat_str, .pat_len = pat_len, + .empty_match = empty_match, }; + if (selection) { + ray_rowsel_t* sm = ray_rowsel_meta(selection); + rctx.sel_flg = ray_rowsel_flags(selection); + rctx.sel_offs = ray_rowsel_offsets(selection); + rctx.sel_idx = ray_rowsel_idx(selection); + rctx.sel_n_segs = sm->n_segs; + } rctx.raw_ok = dom ? ray_sym_domain_raw_pin(dom, &rctx.raw) : false; - if (pool && (int64_t)dict_n >= 16384) { - ray_pool_dispatch(pool, like_resolve_fn, &rctx, (int64_t)dict_n); + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, len, LIKE_PAR_MIN_ROWS_SYM)) { + ray_pool_dispatch(pool, like_rows_fn, &rctx, len); } else { - like_resolve_fn(&rctx, 0, 0, (int64_t)dict_n); + like_rows_fn(&rctx, 0, 0, len); } if (rctx.raw_ok) ray_sym_domain_raw_unpin(dom); - - /* Pass 3: row projection — gather lut[sid] into the per-row - * bool dst. Parallelised because it's a 5 M-row pass (~5 ms - * serial on a W64 SYM column). Width-specialised in the - * worker fn so the inner load is a typed pointer dereference. */ - like_proj_ctx_t pctx = { - .base = base, - .dst = dst, - .lut = lut, - .dict_n = (uint64_t)dict_n, - .sym_w = sym_w, - .sel_flg = sctx.sel_flg, - .sel_offs = sctx.sel_offs, - .sel_idx = sctx.sel_idx, - .sel_n_segs = sctx.sel_n_segs, - }; - if (pool && len >= LIKE_PAR_MIN_ROWS_SYM && ray_pool_total_workers(pool) >= 2) { - ray_pool_dispatch(pool, like_proj_fn, &pctx, len); - } else { - like_proj_fn(&pctx, 0, 0, len); - } - scratch_free(lut_hdr); - scratch_free(seen_hdr); } else { /* OOM building the LUT: fall back to per-row scan. */ if (lut_hdr) scratch_free(lut_hdr); - if (seen_hdr) scratch_free(seen_hdr); for (int64_t i = 0; i < len; i++) { int64_t sym_id = ray_read_sym(base, i, in_type, input->attrs); ray_t* s = dom ? ray_sym_domain_str(dom, sym_id) : (sym_strings && (uint64_t)sym_id < (uint64_t)dict_n) ? sym_strings[sym_id] : NULL; - if (!s) { dst[i] = 0; continue; } + if (!s) { dst[i] = empty_match; continue; } const char* sp = ray_str_ptr(s); size_t sl = ray_str_len(s); dst[i] = (use_simple @@ -771,6 +490,34 @@ ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { } } + return result; +} + +ray_t* exec_like(ray_graph_t* g, ray_op_t* op) { + ray_t* input = exec_like_input(g, op_child(g, op, 0)); + ray_t* pat_v = exec_node(g, op_child(g, op, 1)); + if (!input || RAY_IS_ERR(input)) { if (pat_v && !RAY_IS_ERR(pat_v)) ray_release(pat_v); return input; } + if (!pat_v || RAY_IS_ERR(pat_v)) { ray_release(input); return pat_v; } + + int8_t in_type = input->type; + bool in_parted = RAY_IS_PARTED(in_type); + int8_t base_type = in_parted ? (int8_t)RAY_PARTED_BASETYPE(in_type) : in_type; + + /* Shapes this executor doesn't own — scalar subjects, list-of-atom + * columns (the legacy STRL splayed-load shape), unsupported types — + * delegate to the direct builtin BEFORE reading input->len or + * allocating: an atom's SSO bytes alias ->len, so the result alloc + * below would request a garbage capacity, and the old fallthrough + * memset made unsupported predicates silently match nothing. */ + bool vec_shape = in_parted ? (base_type == RAY_STR || RAY_IS_SYM(base_type)) + : (in_type == RAY_STR || RAY_IS_SYM(in_type)); + if (!vec_shape) { + ray_t* r = ray_like_fn(input, pat_v); + ray_release(input); ray_release(pat_v); + return r; + } + + ray_t* result = ray_like_vec(input, pat_v, g->selection); ray_release(input); ray_release(pat_v); return result; } diff --git a/src/ops/strop.c b/src/ops/strop.c index 224b978c..3aba1645 100644 --- a/src/ops/strop.c +++ b/src/ops/strop.c @@ -938,158 +938,15 @@ ray_t* ray_like_fn(ray_t* x, ray_t* pattern) { return make_bool(m ? 1 : 0); } - /* Vector: map over elements */ - if (ray_is_vec(x) && (x->type == RAY_SYM || x->type == RAY_STR)) { - int64_t n = ray_len(x); - ray_t* result = ray_vec_new(RAY_BOOL, n); - if (RAY_IS_ERR(result)) return result; - result->len = n; - uint8_t* out = (uint8_t*)ray_data(result); - - if (x->type == RAY_SYM) { - /* SYM column is dictionary-encoded with adaptive widths - * (W8/W16/W32/W64). Two bugs to avoid: - * (a) Reading the column as int64_t* is wrong for any - * width != W64 — must use ray_read_sym. - * (b) ray_sym_str returns a borrowed pointer; releasing - * it would decrement the global sym table entry. - * - * Fast path: a SYM column with N rows references at most - * D = ray_sym_count() distinct sym_ids. Build a - * sym_id → bool LUT with a "seen" bitmap so each sym_id - * runs the glob matcher at most once. For LIKE on URL - * (1.7M unique values, 5M rows) this turns an O(n_rows) - * pattern-scan into O(n_distinct + n_rows) — the second - * pass is a single byte load + table lookup per row. */ - const void* base = ray_data(x); - int8_t in_type = x->type; - uint8_t in_attrs = x->attrs; - - /* The global sym table can be much larger than the set of - * IDs this column references (e.g. BrowserCountry with 54 - * uniques in a process that's also loaded URL with 1.7M - * uniques). Lazy-resolve via the seen bitmap so we only - * match against sym_ids actually touched. ray_sym_strings_borrow - * snapshots the strings array under one lock so each lookup - * is a plain pointer load. */ - /* Cell ids are positions in the COLUMN's domain (sym-domain - * Phase 2). Runtime: borrow the global snapshot (below). - * FILE: size the LUT by the domain's count and resolve via - * ray_sym_domain_str (fused_group precedent). Pre-flip - * this is always the runtime branch. */ - struct ray_sym_domain_s* dom = ray_sym_vec_domain(x); - ray_t** sym_strings = NULL; - uint32_t dict_n = 0; - if (dom == ray_sym_runtime_domain()) { - dom = NULL; - ray_sym_strings_borrow(&sym_strings, &dict_n); - } else { - int64_t dn = ray_sym_domain_count(dom); - dict_n = (dn > 0 && dn <= (int64_t)UINT32_MAX) ? (uint32_t)dn : 0; - } - ray_t* lut_hdr = NULL; - ray_t* seen_hdr = NULL; - uint8_t* lut = NULL; - uint8_t* seen = NULL; - if (dict_n > 0) { - lut = (uint8_t*)scratch_alloc (&lut_hdr, (size_t)dict_n); - seen = (uint8_t*)scratch_calloc(&seen_hdr, (size_t)dict_n); - } - if (lut && seen) { - /* Out-of-range sym IDs ("unknown" or "missing") match - * the pattern as if the string were empty — same - * semantics as the OOM fallback at line 332 below and - * the atom case at line 217 above (sym_str==NULL → ""). - * Compute the empty-string match once so the row pass - * can short-circuit out-of-range sids without taking - * the matcher per row. */ - uint8_t empty_match = (use_simple - ? ray_glob_match_compiled(&pc, "", 0) - : ray_glob_match("", 0, pat, pat_len)) - ? 1 : 0; - - /* First pass: discover the unique sym_ids referenced and - * resolve each pattern match exactly once. Second pass: - * width-specialised LUT projection so the per-row loop - * is a tight gather. */ - int sym_w = (int)(in_attrs & RAY_SYM_W_MASK); - #define DICT_PASS(LOAD) \ - for (int64_t i = 0; i < n; i++) { \ - int64_t sid = (LOAD); \ - if ((uint64_t)sid >= (uint64_t)dict_n) continue; \ - if (!seen[sid]) { \ - ray_t* s = sym_strings ? sym_strings[sid] \ - : ray_sym_domain_str(dom, sid); \ - const char* sp = s ? ray_str_ptr(s) : ""; \ - size_t sl = s ? ray_str_len(s) : 0; \ - lut[sid] = (use_simple \ - ? ray_glob_match_compiled(&pc, sp, sl)\ - : ray_glob_match(sp, sl, pat, pat_len)) \ - ? 1 : 0; \ - seen[sid] = 1; \ - } \ - } - #define ROW_PASS(LOAD) \ - for (int64_t i = 0; i < n; i++) { \ - int64_t sid = (LOAD); \ - out[i] = ((uint64_t)sid < (uint64_t)dict_n) \ - ? lut[sid] \ - : empty_match; \ - } - switch (sym_w) { - case RAY_SYM_W8: { - const uint8_t* d = (const uint8_t*)base; - DICT_PASS(d[i]) ROW_PASS(d[i]) break; - } - case RAY_SYM_W16: { - const uint16_t* d = (const uint16_t*)base; - DICT_PASS(d[i]) ROW_PASS(d[i]) break; - } - case RAY_SYM_W32: { - const uint32_t* d = (const uint32_t*)base; - DICT_PASS(d[i]) ROW_PASS(d[i]) break; - } - case RAY_SYM_W64: - default: { - const int64_t* d = (const int64_t*)base; - DICT_PASS(d[i]) ROW_PASS(d[i]) break; - } - } - #undef DICT_PASS - #undef ROW_PASS - scratch_free(lut_hdr); - scratch_free(seen_hdr); - } else { - /* OOM building the LUT: fall back to per-row scan. Still - * uses ray_read_sym for adaptive-width correctness. */ - if (lut_hdr) scratch_free(lut_hdr); - if (seen_hdr) scratch_free(seen_hdr); - for (int64_t i = 0; i < n; i++) { - int64_t sid = ray_read_sym(base, i, in_type, in_attrs); - ray_t* s = dom ? ray_sym_domain_str(dom, sid) - : (sym_strings && (uint64_t)sid < (uint64_t)dict_n) - ? sym_strings[sid] : NULL; - const char* sp = s ? ray_str_ptr(s) : ""; - size_t sl = s ? ray_str_len(s) : 0; - out[i] = (use_simple - ? ray_glob_match_compiled(&pc, sp, sl) - : ray_glob_match(sp, sl, pat, pat_len)) ? 1 : 0; - } - } - } else { - /* RAY_STR vector */ - for (int64_t i = 0; i < n; i++) { - size_t slen; - const char* s = ray_str_vec_get(x, i, &slen); - bool m = false; - if (s) { - m = use_simple ? ray_glob_match_compiled(&pc, s, slen) - : ray_glob_match(s, slen, pat, pat_len); - } - out[i] = m ? 1 : 0; - } - } - return result; + /* Whole STR / SYM column, flat or parted: the kernel the executor + * runs for `where: (like col pat)` — one pattern resolve per distinct + * symbol, row passes over the worker pool (ops/string.c). */ + if (ray_is_vec(x) && (x->type == RAY_SYM || x->type == RAY_STR)) + return ray_like_vec(x, pattern, NULL); + if (RAY_IS_PARTED(x->type)) { + int8_t base = (int8_t)RAY_PARTED_BASETYPE(x->type); + if (base == RAY_STR || RAY_IS_SYM(base)) + return ray_like_vec(x, pattern, NULL); } /* List of string/symbol atoms — the shape splayed string columns diff --git a/test/rfl/strop/like_vector.rfl b/test/rfl/strop/like_vector.rfl new file mode 100644 index 00000000..cb4ddae4 --- /dev/null +++ b/test/rfl/strop/like_vector.rfl @@ -0,0 +1,48 @@ +;; `like` over a whole column, called directly, runs the same kernel as +;; `select … where: (like …)`: one pattern match per distinct symbol, the row +;; pass over the worker pool. Sizes past the parallel thresholds (SYM +;; 100000 rows, STR 200000) so the pooled path runs; every answer must equal +;; the select form and the same match over the plain strings. +(set N 240000) +(set i (til N)) +;; low cardinality (8-bit dictionary width) and high cardinality (wider ids) +(set lo (as 'SYMBOL (map (fn [k] (format "site%.example.com" (% k 40))) i))) +(set hi (as 'SYMBOL (map (fn [k] (format "http://h%.example.com/p/%/%" (% k 700) (% (* 31 k) 977) k)) i))) +(set losv (as 'STR lo)) +(set hisv (as 'STR hi)) +(set T (table [lo hi losv hisv] (list lo hi losv hisv))) +(count T) -- 240000 +;; SYM, 8-bit ids: prefix / contains / exact / class / any / none +(all (== (like lo "site1*") (like losv "site1*"))) -- true +(== (sum (as 'I64 (like lo "site1*"))) (count (select {from: T where: (like lo "site1*")}))) -- true +(all (== (like lo "*7.example*") (like losv "*7.example*"))) -- true +(== (sum (as 'I64 (like lo "*7.example*"))) (count (select {from: T where: (like lo "*7.example*")}))) -- true +(all (== (like lo "site12.example.com") (like losv "site12.example.com"))) -- true +(sum (as 'I64 (like lo "site12.example.com"))) -- 6000 +(all (== (like lo "site[12]?.example.com") (like losv "site[12]?.example.com"))) -- true +(sum (as 'I64 (like lo "*"))) -- 240000 +(sum (as 'I64 (like lo "nothing*"))) -- 0 +;; SYM, wide ids +(all (== (like hi "http://h1*") (like hisv "http://h1*"))) -- true +(== (sum (as 'I64 (like hi "http://h1*"))) (count (select {from: T where: (like hi "http://h1*")}))) -- true +(all (== (like hi "*/p/5/*") (like hisv "*/p/5/*"))) -- true +(== (sum (as 'I64 (like hi "*/p/5/*"))) (count (select {from: T where: (like hi "*/p/5/*")}))) -- true +(all (== (like hi "*/p/5/[0-9]") (like hisv "*/p/5/[0-9]"))) -- true +(all (== (like hi "http://h12.example.com/p/492/*") (like hisv "http://h12.example.com/p/492/*"))) -- true +;; STR column past its own parallel threshold +(== (sum (as 'I64 (like hisv "*/p/5/*"))) (count (select {from: T where: (like hisv "*/p/5/*")}))) -- true +(== (sum (as 'I64 (like losv "site1*"))) (count (select {from: T where: (like losv "site1*")}))) -- true +;; the direct form and the select form agree with `==` on an exact pattern +(sum (as 'I64 (like hi "http://h3.example.com/p/93/3"))) -- 1 +;; a parted table: the select form runs the kernel per segment, the direct +;; form gets the column materialised flat — both must agree +(.sys.exec "rm -rf /tmp/rfl_like_vec_p") -- 0 +(set P1 (table [s] (list (as 'SYMBOL (map (fn [k] (format "a%" (% k 9))) (til 120000)))))) +(set P2 (table [s] (list (as 'SYMBOL (map (fn [k] (format "b%" (% k 9))) (til 120000)))))) +(.db.splayed.set "/tmp/rfl_like_vec_p/1/t/" P1) +(.db.splayed.set "/tmp/rfl_like_vec_p/2/t/" P2) +(set PT (.db.parted.get "/tmp/rfl_like_vec_p/" 't)) +(== (sum (as 'I64 (like (at PT 's) "a*"))) (count (select {from: PT where: (like s "a*")}))) -- true +(sum (as 'I64 (like (at PT 's) "a*"))) -- 120000 +(sum (as 'I64 (like (at PT 's) "?3"))) -- 26666 +(sum (as 'I64 (like (at PT 's) "*"))) -- 240000 From f35846a99e9e563de24c14a7ce7fd0030969c555 Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Fri, 25 Sep 2026 19:20:38 +0200 Subject: [PATCH 28/51] fix(system): report arity errors for extra arguments (#616) --- docs/docs/language/functions.md | 2 +- docs/docs/namespaces/sys.md | 10 +++++----- docs/docs/reference/all-functions.md | 2 +- src/lang/syscmd.c | 2 ++ src/ops/system.c | 3 ++- test/rfl/system/syscmd_coverage.rfl | 12 +++++++----- test/rfl/system/system_branch_cov2.rfl | 1 + 7 files changed, 19 insertions(+), 13 deletions(-) diff --git a/docs/docs/language/functions.md b/docs/docs/language/functions.md index 943e1b78..7580b8e6 100644 --- a/docs/docs/language/functions.md +++ b/docs/docs/language/functions.md @@ -516,7 +516,7 @@ spills exactly as it did before. | `.time.timer.del` | unary, restricted | Cancel a scheduled timer by id. Returns null. | `(.time.timer.del 0)` | | `.sys.args` | nullary | Application arguments as a typed dict (`user` subdict for post-`--` args) | `(.sys.args)` | | `env` | unary | List all global environment bindings | `(env 0)` | -| `.sys.build` | variadic | Build metadata: `version` + `build-date` | `(.sys.build)` | +| `.sys.build` | nullary | Build metadata: `version` + `build-date` | `(.sys.build)` | | `.sys.mem` | variadic | Memory allocator statistics (alloc/peak/slab) | `(.sys.mem)` | | `.sys.info` | variadic | System information (cores, page size, memory) | `(.sys.info)` | diff --git a/docs/docs/namespaces/sys.md b/docs/docs/namespaces/sys.md index f6eefbff..f1c8a1e8 100644 --- a/docs/docs/namespaces/sys.md +++ b/docs/docs/namespaces/sys.md @@ -10,18 +10,18 @@ Process-level introspection (build, memory, host info) and command-style operati | Function | Arity | Flags | Description | |---|---|---|---| | [`.sys.args`](#sys-args) | variadic | — | Command-line arguments as a typed dict. | -| [`.sys.build`](#sys-build) | variadic | — | Version + build date as a dict. | +| [`.sys.build`](#sys-build) | nullary | — | Version + build date as a dict. | | [`.sys.info`](#sys-info) | variadic | — | Host and process facts: cores, page size, total memory, pid, hostname. | | [`.sys.mem`](#sys-mem) | variadic | — | Allocator statistics. | | [`.sys.prof`](#sys-prof) | variadic | — | Last profiled query's per-step statistics as a table. | | [`.sys.querylog`](#sys-querylog) | variadic | — | Ambient per-query statistics ring as a table. | | [`.sys.querylog.enable`](#sys-querylog-enable) | variadic | restricted | Toggle query-statistics logging. | | [`.sys.gc`](#sys-gc) | variadic | — | Run allocator maintenance and return `0`. | -| [`.sys.env`](#sys-env) | variadic | — | Count or list of globally bound names. | +| [`.sys.env`](#sys-env) | variadic (0–1) | — | Count or list of globally bound names. | | [`.sys.exec`](#sys-exec) | variadic | restricted | Run a shell command; return its exit code, optionally with stdout. | | [`.sys.cmd`](#sys-cmd) | unary | restricted | Dispatch a colon-command string. | | [`.sys.listen`](#sys-listen) | unary | restricted | Bind an IPC listener on a TCP port. | -| [`.sys.timeit`](#sys-timeit) | variadic | — | Toggle / set the per-expression profiler. | +| [`.sys.timeit`](#sys-timeit) | variadic (0–1) | — | Toggle / set the per-expression profiler. | ## `.sys.args` { #sys-args } @@ -244,7 +244,7 @@ GC, not a tracing object collector. Returns `0`. ## `.sys.env` { #sys-env } -Signature: `(.sys.env)`. From a script / IPC context returns the **count** of globally bound names (i64). In a REPL context the same dispatcher prints one line per binding (name + type label) and returns null — the variadic registration accommodates both forms. +Signature: `(.sys.env [compat-arg])`. From a script / IPC context returns the **count** of globally bound names (i64). In a REPL context the same dispatcher prints one line per binding (name + type label) and returns null. At most one optional compatibility argument is accepted; additional arguments return an `arity` error. ```lisp (.sys.env) @@ -330,7 +330,7 @@ Errors: `type` (port not an int / not parseable from string), `domain` (port out ## `.sys.timeit` { #sys-timeit } -Signature: `(.sys.timeit [flag])`. Toggles the per-expression profiler. Calling with no argument flips the current state; passing `0` disables, anything non-zero enables. Returns the new state as `i64` (0/1). +Signature: `(.sys.timeit [flag])`. Toggles the per-expression profiler. Calling with no argument flips the current state; passing `0` disables, anything non-zero enables. Returns the new state as `i64` (0/1). More than one argument returns an `arity` error. ```lisp (.sys.timeit 1) ;; enable profiling diff --git a/docs/docs/reference/all-functions.md b/docs/docs/reference/all-functions.md index eba0d980..2c3e552e 100644 --- a/docs/docs/reference/all-functions.md +++ b/docs/docs/reference/all-functions.md @@ -655,7 +655,7 @@ System interaction, metaprogramming, diagnostics, and runtime inspection. | `.time.now` | variadic | — | Monotonic time in milliseconds | `(.time.now)` | | `.time.timer.set` | variadic | restricted | Schedule callback every `ms`, `num` times (0 = forever); returns id | `(.time.timer.set 1000 0 (fn [t] (println t)))` | | `.time.timer.del` | unary | restricted | Cancel a scheduled timer by id; returns null | `(.time.timer.del 0)` | -| `.sys.build` | variadic | — | Build metadata dict with `version` + `build-date` | `(.sys.build)` | +| `.sys.build` | nullary | — | Build metadata dict with `version` + `build-date` | `(.sys.build)` | | `.sys.mem` | variadic | — | Memory allocator statistics (alloc / peak / slab hits) | `(.sys.mem)` | | `.sys.prof` | variadic | — | Last profiled query's per-step statistics as a table (opt-in via `:t`) | `(.sys.prof)` | | `.sys.querylog` | variadic | — | Ambient per-query statistics ring as a table (opt-in via `-Q` / `.sys.querylog.enable`) | `(.sys.querylog)` | diff --git a/src/lang/syscmd.c b/src/lang/syscmd.c index 9fd7b44c..62f613c9 100644 --- a/src/lang/syscmd.c +++ b/src/lang/syscmd.c @@ -424,9 +424,11 @@ ray_t* ray_sys_listen_fn(ray_t* x) { return invoke_by_name("listen", x); } * matches `.sys.gc`'s convention and avoids the arity error users * would otherwise hit calling `(.sys.env)` with no args. */ ray_t* ray_sys_timeit_fn(ray_t** args, int64_t n) { + if (n > 1) return ray_error("arity", ".sys.timeit accepts at most one argument"); return invoke_by_name("timeit", n > 0 ? args[0] : RAY_NULL_OBJ); } ray_t* ray_sys_env_fn(ray_t** args, int64_t n) { + if (n > 1) return ray_error("arity", ".sys.env accepts at most one argument"); (void)args; return invoke_by_name("env", n > 0 ? args[0] : RAY_NULL_OBJ); } diff --git a/src/ops/system.c b/src/ops/system.c index 6d5a55cc..6c0a2627 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -1229,7 +1229,8 @@ ray_t* ray_env_fn(ray_t* x) { /* (.sys.build) -- return dict with internal build information */ ray_t* ray_internals_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".sys.build takes no arguments"); ray_t* keys = ray_sym_vec_new(RAY_SYM_W64, 2); if (RAY_IS_ERR(keys)) return keys; ray_t* vals = ray_list_new(2); diff --git a/test/rfl/system/syscmd_coverage.rfl b/test/rfl/system/syscmd_coverage.rfl index a69bf0af..656cda9f 100644 --- a/test/rfl/system/syscmd_coverage.rfl +++ b/test/rfl/system/syscmd_coverage.rfl @@ -113,17 +113,19 @@ (.sys.timeit) -- 1 (.sys.timeit) -- 0 -;; ────────────── ray_sys_timeit_fn: variadic with extra args ignored ────── -;; The variadic adapter (line 353) only forwards the first arg. +;; ────────────── ray_sys_timeit_fn: optional arg + arity guard ────────── +;; The variadic adapter accepts zero or one arg, and rejects extras before +;; changing profiler state. (.sys.timeit 1) -- 1 (.sys.timeit 0) -- 0 +(.sys.timeit 1 0) !- arity -;; ────────────── h_env: count > 0 + variadic with arg branch ────────────── -;; ray_sys_env_fn takes optional arg via variadic adapter (line 356); -;; arg is ignored by h_env (line 164). Both call shapes work. +;; ────────────── h_env: count > 0 + optional compatibility arg ────────── +;; ray_sys_env_fn accepts zero or one arg and rejects extras before h_env runs. (> (.sys.env) 0) -- true (> (.sys.env 1) 0) -- true (> (.sys.env "x") 0) -- true +(.sys.env 1 2) !- arity ;; ────────────── h_listen: domain / type / io branches ────────────── ;; The test runtime (test/main.c rfl_setup) registers a poll instance — diff --git a/test/rfl/system/system_branch_cov2.rfl b/test/rfl/system/system_branch_cov2.rfl index 73a2d595..eaf7ebdd 100644 --- a/test/rfl/system/system_branch_cov2.rfl +++ b/test/rfl/system/system_branch_cov2.rfl @@ -101,6 +101,7 @@ (type (.sys.build)) -- 'DICT (count (.sys.build)) -- 2 (type (at (.sys.build) 'version)) -- 'str +(.sys.build 1) !- arity ;; ══════════════════════════════════════════════════════════════════════ ;; ray_memstat_fn (.sys.mem) (lines 703-728) From fb426dcd45c06f72513d40205b544932e56bd3e2 Mon Sep 17 00:00:00 2001 From: Anton Kundenko Date: Fri, 25 Sep 2026 20:46:17 +0200 Subject: [PATCH 29/51] fix(in): admit the typed kernel for text columns, which it never reached (#607) --- src/ops/collection.c | 17 +++++++++++++++-- test/rfl/collection/in.rfl | 39 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 2 deletions(-) diff --git a/src/ops/collection.c b/src/ops/collection.c index b9ac009d..6b20c4b2 100644 --- a/src/ops/collection.c +++ b/src/ops/collection.c @@ -1374,8 +1374,21 @@ ray_t* ray_in_fn(ray_t* val, ray_t* vec) { * the WHERE kernel uses null-matches-nothing semantics — the * two must not be conflated. NULL result = unsupported shape * (STR etc.): fall through to the hashset probe below. */ - if (!ray_vec_may_have_nulls(val) && - !ray_vec_may_have_nulls(vec)) { + /* Admission check, so it uses the EXACT ray_vec_has_nulls rather + * than the conservative row-kernel gate. ray_vec_may_have_nulls + * returns true unconditionally for SYM and STR — their null is a + * payload value, not an attribute bit — so gating this on it made + * the branch unreachable for every text column, including the SYM + * verdict-LUT the kernel carries specifically for them. Every + * `in` over a symbol column fell through to the generic per-row + * hashset probe: ~4.8 ms against ~40 us for the equivalent `==` + * on a 351k-row column (#593). vec.h says as much where it + * defines the two — "paths requiring null-free data use has_nulls + * below". The exact check costs one pass and is not on a per-row + * path; a null-bearing operand still falls through, because the + * kernel's null-matches-nothing semantics differ from this path's + * null-equals-null. */ + if (!ray_vec_has_nulls(val) && !ray_vec_has_nulls(vec)) { ray_t* fast = ray_in_vec_exec(val, vec, false); if (fast) return fast; } diff --git a/test/rfl/collection/in.rfl b/test/rfl/collection/in.rfl index fbead052..026417e4 100644 --- a/test/rfl/collection/in.rfl +++ b/test/rfl/collection/in.rfl @@ -84,3 +84,42 @@ (in [1h 2h 3h] [3h 2h 1h]) -- [true true true] ;; Test with empty arrays (in (list) [1h 2h]) -- (list) + +;; ========== TEXT (SYM/STR) NULL SEMANTICS (#593) ========== +;; `in` admits the typed verdict-LUT kernel only when BOTH sides are exactly +;; null-free, because that kernel uses null-matches-nothing semantics while +;; this path is null-equals-null. The gate used to ask +;; ray_vec_may_have_nulls, which is unconditionally true for SYM and STR +;; (their null is a payload value, not an attribute bit) — so the kernel was +;; unreachable for text columns and every symbol `in` fell through to the +;; generic per-row hashset probe. These pin the semantics on both sides of +;; that gate so the admission check cannot drift back. + +;; null in the COLUMN, null needle — null equals null +(in (as 'SYM ["a" "" "b"]) (as 'SYM [""])) -- [false true false] +(in ["a" "" "b"] [""]) -- [false true false] + +;; null in the column, non-null needle — the null matches nothing else +(in (as 'SYM ["a" "" "b"]) (as 'SYM ["a"])) -- [true false false] + +;; null only in the NEEDLES: the column is null-free but the gate must still +;; reject, since a null needle needs null-equals-null against nothing +(in (as 'SYM ["a" "b"]) (as 'SYM ["" "a"])) -- [true false] + +;; nulls on both sides +(in (as 'SYM ["a" ""]) (as 'SYM ["" "z"])) -- [false true] + +;; null-free text: this is the shape that now reaches the kernel +(in (as 'SYM ["a" "b" "c"]) (as 'SYM ["b" "c"])) -- [false true true] +(in ["a" "b" "c"] ["b" "c"]) -- [false true true] + +;; empty needle set over a null-free symbol column +(in (as 'SYM ["a" "b"]) (as 'SYM [])) -- [false false] + +;; a single needle is the degenerate case the issue reported: same answer as +;; the equality it should cost the same as +(in (as 'SYM ["a" "b" "a"]) (as 'SYM ["a"])) -- [true false true] +(== (as 'SYM ["a" "b" "a"]) 'a) -- [true false true] + +;; duplicate needles must not double-count or change the verdict +(in (as 'SYM ["a" "b"]) (as 'SYM ["a" "a" "a"])) -- [true false] From db062c1b5d542f15718f8b86d4717b36068b2fd1 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sat, 26 Sep 2026 21:22:44 +0300 Subject: [PATCH 30/51] perf: shared-node memo, STR descriptor views, fused top-k on nullable like, bulk derived key, parallel dense grouping (#626) --- src/ops/agg_engine.c | 25 +- src/ops/exec.c | 121 +++++++ src/ops/fused_pred.c | 35 +- src/ops/fused_pred.h | 3 + src/ops/fused_topk.c | 20 +- src/ops/group.c | 54 +++- src/ops/internal.h | 13 + src/ops/ops.h | 17 + src/ops/pivot.c | 240 +++++++++++++- src/ops/query.c | 301 +++++++++++++++--- src/ops/string.c | 267 +++++++--------- test/rfl/fused/topk_like_nullable_nowhere.rfl | 37 +++ test/rfl/query/shared_node_memo.rfl | 32 ++ test/rfl/strop/substr_if_str_views.rfl | 33 ++ 14 files changed, 962 insertions(+), 236 deletions(-) create mode 100644 test/rfl/fused/topk_like_nullable_nowhere.rfl create mode 100644 test/rfl/query/shared_node_memo.rfl create mode 100644 test/rfl/strop/substr_if_str_views.rfl diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index 9f19eb1a..81ffe665 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -4645,15 +4645,34 @@ static int64_t agg_index_winner(void* raw, const int64_t* rows, int64_t count) { #undef INDEX_FIRST_VALID } int64_t best = -1; + /* A FILE domain's entries are compared off the mapped vocabulary: no + * atom per row — the lazily built atoms of a wide vocabulary cost far + * more than the compare and stay for the domain's lifetime. Positions + * past the file prefix, and the runtime domain, go through the atoms. */ + struct ray_sym_domain_s* dom = ray_sym_vec_domain(c->src); + ray_sym_domain_raw_t vocab; + bool vocab_ok = ray_sym_domain_raw_pin(dom, &vocab); for (int64_t j = 0; j < count; j++) { if ((j & 65535) == 0 && ray_interrupted()) return -1; int64_t r = rows[c->kind == OP_LAST ? count - 1 - j : j]; if (ray_vec_is_null(c->src, r)) continue; if (best < 0) best = r; if (c->kind == OP_FIRST || c->kind == OP_LAST) break; - ray_t* x = ray_group_sym_read(&c->symbols, ray_sym_vec_domain(c->src), ray_read_sym(data, r, c->src->type, c->src->attrs)); - ray_t* y = ray_group_sym_read(&c->symbols, ray_sym_vec_domain(c->src), ray_read_sym(data, best, c->src->type, c->src->attrs)); - int cmp = ray_str_cmp(x, y); + int64_t ir = ray_read_sym(data, r, c->src->type, c->src->attrs); + int64_t ib = ray_read_sym(data, best, c->src->type, c->src->attrs); + int cmp; + if (vocab_ok && ir >= 0 && ib >= 0 && ir < vocab.count && ib < vocab.count) { + size_t lr, lb; + const char* pr = ray_sym_domain_raw_str(&vocab, ir, &lr); + const char* pb = ray_sym_domain_raw_str(&vocab, ib, &lb); + size_t m = lr < lb ? lr : lb; + cmp = m ? memcmp(pr, pb, m) : 0; + if (cmp == 0) cmp = (lr > lb) - (lr < lb); + } else { + ray_t* x = ray_group_sym_read(&c->symbols, dom, ir); + ray_t* y = ray_group_sym_read(&c->symbols, dom, ib); + cmp = ray_str_cmp(x, y); + } if (c->kind == OP_MIN ? cmp < 0 : cmp > 0) best = r; } return best; diff --git a/src/ops/exec.c b/src/ops/exec.c index de07dbbe..fe5dd93b 100644 --- a/src/ops/exec.c +++ b/src/ops/exec.c @@ -1839,9 +1839,117 @@ static ray_t* exec_elementwise_tree(ray_graph_t* g, ray_op_t* root) { return result; } +/* Nodes whose result may be shared between consumers: pure, vector-valued + * operators. Structural ops (scan, filter, group, sort, join …) either + * carry side state (g->selection) or return their input, and stay out. */ +static inline bool op_memoizable(uint16_t o) { + switch (o) { + case OP_IF: case OP_LIKE: case OP_ILIKE: case OP_UPPER: case OP_LOWER: + case OP_STRLEN: case OP_SUBSTR: case OP_REPLACE: case OP_TRIM: + case OP_CONCAT: case OP_STR_FIND: case OP_EXTRACT: case OP_DATE_TRUNC: + case OP_IN: case OP_NOT_IN: + return true; + default: + return op_is_elementwise(o); + } +} + +/* Count each node's consumers over the in_id edges (plus one for the + * root's return) and arm the memo when some memoizable node has more than + * one. The memo keeps its own ref on every stored value until + * exec_memo_end: a consumer that receives an input with rc == 1 may reuse + * the buffer in place, and a fused window or a sibling still holding a raw + * pointer into that buffer would then read the overwritten values — so no + * shared value is ever handed out as the sole reference. Shared + * intermediates therefore live to the end of the execution; before this a + * shared node was recomputed per consumer instead. Values belong to the + * table the memo was armed over (memo_table): while g->table is swapped + * for a sub-table the memo is inert. Returns whether it armed; not + * re-entrant on purpose — a nested execution of the same graph leaves the + * outer memo alone and must not tear it down. */ +static bool exec_memo_begin(ray_graph_t* g, ray_op_t* root) { + if (!g || g->memo_uses || g->node_count == 0) return false; + uint32_t nc = g->node_count; + ray_t* hdr = NULL; + char* mem = (char*)scratch_calloc(&hdr, (size_t)nc * (sizeof(ray_t*) + sizeof(uint32_t))); + if (!mem) return false; + ray_t** vals = (ray_t**)mem; + uint32_t* uses = (uint32_t*)(mem + (size_t)nc * sizeof(ray_t*)); + for (uint32_t i = 0; i < nc; i++) { + ray_op_t* n = &g->nodes[i]; + if (n->flags & OP_FLAG_DEAD) continue; + for (uint8_t k = 0; k < n->arity && k < 2; k++) + if (n->in_id[k] != RAY_OP_NONE && n->in_id[k] < nc) uses[n->in_id[k]]++; + /* Operands kept in the ext node: the third input of if / substr / + * replace, the trailing arguments of concat. */ + if (n->opcode == OP_IF || n->opcode == OP_SUBSTR || n->opcode == OP_REPLACE) { + ray_op_ext_t* e = find_ext(g, n->id); + if (e && e->third_in < nc) uses[e->third_in]++; + } else if (n->opcode == OP_CONCAT) { + ray_op_ext_t* e = find_ext(g, n->id); + if (e) { + int n_args = (int)e->sym; + const uint32_t* trail = (const uint32_t*)((const char*)(e + 1)); + for (int a = 2; a < n_args; a++) + if (trail[a - 2] < nc) uses[trail[a - 2]]++; + } + } + } + if (root && root->id < nc) uses[root->id]++; + bool any = false; + for (uint32_t i = 0; i < nc && !any; i++) + any = uses[i] > 1 && op_memoizable(g->nodes[i].opcode); + if (!any) { scratch_free(hdr); return false; } + g->memo_vals = vals; g->memo_uses = uses; g->memo_n = nc; g->memo_hdr = hdr; + g->memo_table = g->table; + return true; +} + +static void exec_memo_end(ray_graph_t* g) { + if (!g || !g->memo_uses) return; + for (uint32_t i = 0; i < g->memo_n; i++) + if (g->memo_vals[i]) ray_release(g->memo_vals[i]); + scratch_free(g->memo_hdr); + g->memo_vals = NULL; g->memo_uses = NULL; g->memo_n = 0; g->memo_hdr = NULL; + g->memo_table = NULL; +} + +/* A sub-evaluation over another table (an `if` branch over its compacted + * rows) gets a memo of its own: the outer memo is set aside — its values + * belong to the outer table — and a fresh one is armed over the current + * g->table with the sub-root's census. Pop tears the inner memo down and + * puts the outer one back. */ +void ray_exec_memo_push(ray_graph_t* g, ray_op_t* root, ray_exec_memo_save_t* save) { + save->vals = g->memo_vals; save->uses = g->memo_uses; save->n = g->memo_n; + save->hdr = g->memo_hdr; save->table = g->memo_table; + g->memo_vals = NULL; g->memo_uses = NULL; g->memo_n = 0; g->memo_hdr = NULL; + g->memo_table = NULL; + (void)exec_memo_begin(g, root); +} + +void ray_exec_memo_pop(ray_graph_t* g, const ray_exec_memo_save_t* save) { + exec_memo_end(g); + g->memo_vals = save->vals; g->memo_uses = save->uses; g->memo_n = save->n; + g->memo_hdr = save->hdr; g->memo_table = save->table; +} + ray_t* exec_node(ray_graph_t* g, ray_op_t* op) { if (!op) return ray_error("nyi", NULL); + /* Shared node already computed: hand out a ref (the memo's own ref goes + * at exec_memo_end). */ + bool memo_on = g->memo_uses && g->table == g->memo_table && + op->id < g->memo_n && op_memoizable(op->opcode); + if (memo_on && g->memo_vals[op->id]) { + ray_t* v = g->memo_vals[op->id]; + ray_retain(v); + /* The memo keeps its own ref until exec_memo_end: a consumer that + * reaches its input with rc == 1 may reuse the buffer in place, and + * another consumer may still hold a raw pointer into it. */ + return v; + } + bool memo_shared = memo_on && g->memo_uses[op->id] > 1; + /* Per-op cancellation checkpoint. Long fused pipelines iterate * exec_node many times; this catches Ctrl-C between operators * without adding cost to the per-row hot path. */ @@ -1879,6 +1987,15 @@ ray_t* exec_node(ray_graph_t* g, ray_op_t* op) { ray_t* _prof_result = exec_node_inner(g, op); tl_exec_depth--; + /* First consumer of a shared node: keep a ref for the others (a lazy + * value is not kept — materialising it is the consumer's business). */ + if (memo_shared && _prof_result && !RAY_IS_ERR(_prof_result) && + !ray_is_lazy(_prof_result) && g->memo_uses && g->table == g->memo_table && + !g->memo_vals[op->id]) { + ray_retain(_prof_result); + g->memo_vals[op->id] = _prof_result; + } + if (profiling) { ray_prof_span_t* ep = ray_profile_span_end(oname); if (ep) { @@ -3926,7 +4043,11 @@ static ray_t* ray_execute_inner(ray_graph_t* g, ray_op_t* root) { if (seg_count == 0 || !dag_can_stream(g, root)) { /* Non-parted table or DAG contains ops that need specialized merge: * use existing flat-materialization path. */ + /* Only the call that armed the memo tears it down: a nested + * execution of the same graph leaves the outer memo alone. */ + bool memo_armed = exec_memo_begin(g, root); ray_t* result = exec_node(g, root); + if (memo_armed) exec_memo_end(g); if (g->selection && result && !RAY_IS_ERR(result) && result->type == RAY_TABLE) { /* Projection-aware compaction: a select publishes the keys-only diff --git a/src/ops/fused_pred.c b/src/ops/fused_pred.c index 391d741a..94d30219 100644 --- a/src/ops/fused_pred.c +++ b/src/ops/fused_pred.c @@ -104,8 +104,8 @@ static int fp_atom_col_compatible(int8_t atom_type, int8_t col_type) { } /* Numeric, temporal, STR and GUID comparisons have an explicit typed leg - * with null-as-minimum ordering. SYM equality uses domain codes. LIKE/IN keep - * their stricter null-free admission because they use different evaluators. */ + * with null-as-minimum ordering. SYM equality uses domain codes. IN keeps + * its stricter null-free admission because it uses a different evaluator. */ static int fp_col_supported_op(const ray_t* col, int eq_or_ne) { if (!col) return 0; if (col->type >= RAY_BOOL && col->type <= RAY_TIMESTAMP) return 1; @@ -114,8 +114,8 @@ static int fp_col_supported_op(const ray_t* col, int eq_or_ne) { return !ray_vec_has_nulls(col); } -/* Strict form — no nullable column at all. Used by the shapes whose - * evaluator arm is not an equality compare (LIKE, IN). */ +/* Strict form — no nullable column at all. Used by the shape whose + * evaluator arm is not an equality compare (IN). */ static int fp_col_supported(const ray_t* col) { return col && !ray_vec_has_nulls(col); } @@ -208,7 +208,10 @@ static int fp_check_like(ray_t* expr, ray_t* tbl) { if (!fp_expr_const_str(elems[2])) return 0; if (tbl) { ray_t* col = ray_table_get_col(tbl, lhs->i64); - if (!col || !fp_col_supported(col)) return 0; + /* A null text cell is the empty string on every like path (the + * kernel and this evaluator both match the pattern against ""), + * so a nullable column is admitted. */ + if (!col) return 0; if (col->type != RAY_STR && col->type != RAY_SYM) return 0; } return 1; @@ -445,10 +448,13 @@ void fp_eval_cmp(const fp_cmp_t* p, int64_t start, int64_t end, for (int64_t r = 0; r < n; r++) { uint64_t sid = (uint64_t)read_by_esz(base, start + r, esz_l); if (sid >= lut_n || !lut) { - bits[r] = 0; + bits[r] = p->like_empty_match; continue; } - uint8_t state = lut[sid]; + /* The LUT is shared by the workers: relaxed atomics on its + * bytes (every writer stores the same answer for a cell, + * so a repeated resolve is the only cost of a race). */ + uint8_t state = __atomic_load_n(&lut[sid], __ATOMIC_RELAXED); if (!state) { const char* sp = NULL; size_t sl = 0; @@ -466,7 +472,7 @@ void fp_eval_cmp(const fp_cmp_t* p, int64_t start, int64_t end, : (uint8_t)ray_glob_match(sp, sl, p->pat_str, p->pat_len); } state = (uint8_t)(match ? 2 : 1); - lut[sid] = state; + __atomic_store_n(&lut[sid], state, __ATOMIC_RELAXED); } bits[r] = (uint8_t)(state == 2); } @@ -608,8 +614,8 @@ static inline uint8_t fp_eval_cmp_one(const fp_cmp_t* p, int64_t row) { if (p->col_type == RAY_SYM) { uint64_t sid = (uint64_t)read_by_esz(p->col_base, row, p->col_esz); if (sid >= p->like_lut_count || !p->like_lut) - return 0; - uint8_t state = p->like_lut[sid]; + return p->like_empty_match; + uint8_t state = __atomic_load_n(&p->like_lut[sid], __ATOMIC_RELAXED); if (!state) { /* NULL sym_strings ⇒ FILE-domain column (see fp_eval_cmp) */ const char* sp = NULL; @@ -629,7 +635,7 @@ static inline uint8_t fp_eval_cmp_one(const fp_cmp_t* p, int64_t row) { : (uint8_t)ray_glob_match(sp, sl, p->pat_str, p->pat_len); } state = (uint8_t)(match ? 2 : 1); - p->like_lut[sid] = state; + __atomic_store_n(&p->like_lut[sid], state, __ATOMIC_RELAXED); } return (uint8_t)(state == 2); } @@ -797,8 +803,10 @@ static int fp_compile_cmp(ray_graph_t* g, ray_op_t* pred_op, ray_t* tbl, return 0; } if (out->op == FP_LIKE) { + /* A nullable text column is admitted: a null cell is the empty + * string on every like path, matched against the pattern once + * here (like_empty_match). */ if (col->type != RAY_STR && col->type != RAY_SYM) return -1; - if (!fp_col_supported(col)) return -1; ray_t* cv_like = rext->literal; if (!cv_like || cv_like->type != -RAY_STR) return -1; out->col_type = col->type; @@ -810,6 +818,9 @@ static int fp_compile_cmp(ray_graph_t* g, ray_op_t* pred_op, ray_t* tbl, out->pat_str = ray_str_ptr(cv_like); out->pat_len = ray_str_len(cv_like); out->pat_compiled = ray_glob_compile(out->pat_str, out->pat_len); + out->like_empty_match = (out->pat_compiled.shape != RAY_GLOB_SHAPE_NONE) + ? (uint8_t)ray_glob_match_compiled(&out->pat_compiled, "", 0) + : (uint8_t)ray_glob_match("", 0, out->pat_str, out->pat_len); if (col->type == RAY_SYM) { /* Cell ids are positions in the COLUMN's domain. Runtime * domain: borrow the global string snapshot (lock-free per diff --git a/src/ops/fused_pred.h b/src/ops/fused_pred.h index d26ee6b2..be2fea7f 100644 --- a/src/ops/fused_pred.h +++ b/src/ops/fused_pred.h @@ -82,6 +82,9 @@ typedef struct { * read straight from the mapping (see ray_sym_domain_raw_pin). */ ray_sym_domain_raw_t like_raw; uint8_t like_raw_ok; + /* pattern vs "": the answer for a null text cell and for a cell id the + * LUT does not cover (the same rule as the bare like kernel). */ + uint8_t like_empty_match; struct ray_sym_domain_s* like_dom; } fp_cmp_t; diff --git a/src/ops/fused_topk.c b/src/ops/fused_topk.c index ddf2ddfc..6d44e49d 100644 --- a/src/ops/fused_topk.c +++ b/src/ops/fused_topk.c @@ -379,16 +379,20 @@ ray_t* ray_fused_topk_select(ray_t* tbl, ctx.k = k; ctx.tbl = tbl; - /* Compile the predicate via a temp graph just for the WHERE clause. */ + /* Compile the predicate via a temp graph just for the WHERE clause. No + * where: is a predicate with no children — every row passes — and the + * heap does the whole sort-and-take in one pass. */ ray_graph_t* g = ray_graph_new(tbl); if (!g) { fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } - ray_op_t* pred_dag = compile_expr_dag(g, where_expr); - if (!pred_dag) { ray_graph_free(g); fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } - if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { - fp_pred_cleanup(&ctx.pred); - ray_graph_free(g); - fpk_unpin_keys(ctx.keys, n_sort_keys); - return NULL; + if (where_expr) { + ray_op_t* pred_dag = compile_expr_dag(g, where_expr); + if (!pred_dag) { ray_graph_free(g); fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } + if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + fpk_unpin_keys(ctx.keys, n_sort_keys); + return NULL; + } } if (sym_needed) { diff --git a/src/ops/group.c b/src/ops/group.c index 8ebc0418..0e5448e5 100644 --- a/src/ops/group.c +++ b/src/ops/group.c @@ -162,13 +162,24 @@ static void reduce_acc_init(reduce_acc_t* acc) { static inline bool sym_lex_lt(struct ray_sym_domain_s* dom, int64_t a, int64_t b) { if (a == b) return false; - ray_t* sa = ray_sym_domain_str(dom, a); - ray_t* sb = ray_sym_domain_str(dom, b); - if (!sa || !sb) return a < b; - const char* pa = ray_str_ptr(sa); - const char* pb = ray_str_ptr(sb); - size_t la = ray_str_len(sa); - size_t lb = ray_str_len(sb); + const char* pa; const char* pb; + size_t la, lb; + /* A FILE domain's entries are read off the mapped vocabulary: no atom + * is materialised per compare (the lazily built atoms cost more than + * the compare itself on a wide vocabulary). Positions past the file + * prefix, and the runtime domain, resolve through the atoms as before. */ + ray_sym_domain_raw_t raw; + if (ray_sym_domain_raw_pin(dom, &raw) && a >= 0 && b >= 0 && + a < raw.count && b < raw.count) { + pa = ray_sym_domain_raw_str(&raw, a, &la); + pb = ray_sym_domain_raw_str(&raw, b, &lb); + } else { + ray_t* sa = ray_sym_domain_str(dom, a); + ray_t* sb = ray_sym_domain_str(dom, b); + if (!sa || !sb) return a < b; + pa = ray_str_ptr(sa); pb = ray_str_ptr(sb); + la = ray_str_len(sa); lb = ray_str_len(sb); + } size_t m = la < lb ? la : lb; int c = memcmp(pa, pb, m); if (c != 0) return c < 0; @@ -7848,6 +7859,7 @@ typedef struct { ray_t* rowsel; ray_t** sym_strings; /* borrowed sym snapshot for strlen-on-SYM aggs */ uint32_t sym_count; + int64_t da_n_scan; /* rows to scan (ranged tasks split it) */ } da_ctx_t; typedef struct { @@ -8890,6 +8902,18 @@ static void da_accum_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t en #undef DA_PF_DIST } +/* One task per accumulator (ray_pool_dispatch_n): task i scans the i-th + * of n_accums equal row ranges into accums[i], whichever worker runs it. */ +static void da_accum_task_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { + (void)worker_id; (void)end; + da_ctx_t* c = (da_ctx_t*)ctx; + int64_t k = c->n_accums, n = c->da_n_scan; + int64_t base = n / k, rem = n % k; + int64_t lo = start * base + (start < rem ? start : rem); + int64_t hi = lo + base + (start < rem ? 1 : 0); + da_accum_fn(ctx, (uint32_t)start, lo, hi); +} + /* Parallel DA merge: merge per-worker accumulators into accums[0] by * dispatching disjoint slot ranges across pool workers. */ typedef struct { @@ -12641,8 +12665,15 @@ da_path:; uint64_t max_workers = cells_per_worker ? (uint64_t)n_scan / cells_per_worker : 1; if (max_workers < 1) max_workers = 1; - if ((uint64_t)da_n_workers > max_workers) - da_n_workers = 1; + /* More workers than the budget allows: keep the budget's worth + * of accumulators and give each one a contiguous row range (one + * task per accumulator, whichever worker runs it) instead of + * collapsing to a serial scan of every row. */ + bool da_ranged = false; + if ((uint64_t)da_n_workers > max_workers) { + da_n_workers = (uint32_t)max_workers; + da_ranged = da_n_workers > 1; + } ray_t* accums_hdr; da_accum_t* accums = (da_accum_t*)scratch_calloc(&accums_hdr, @@ -12768,9 +12799,12 @@ da_path:; .n_slots = n_slots, .match_idx = match_idx, .rowsel = rowsel, + .da_n_scan = n_scan, }; - if (da_n_workers > 1) + if (da_ranged) + ray_pool_dispatch_n(da_pool, da_accum_task_fn, &da_ctx, da_n_workers); + else if (da_n_workers > 1) ray_pool_dispatch(da_pool, da_accum_fn, &da_ctx, n_scan); else da_accum_fn(&da_ctx, 0, 0, n_scan); diff --git a/src/ops/internal.h b/src/ops/internal.h index a1e1eeba..63d19d33 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -1603,6 +1603,19 @@ ray_t* exec_k_shortest(ray_graph_t* g, ray_op_t* op, /* ── pivot_exec.c ── */ ray_t* exec_if(ray_graph_t* g, ray_op_t* op); + +/* Shared-node memo around a sub-evaluation over a swapped g->table + * (exec.c): push sets the outer memo aside and arms one for the current + * table and sub-root; pop tears it down and restores the outer one. */ +typedef struct { + ray_t** vals; + uint32_t* uses; + uint32_t n; + ray_t* hdr; + ray_t* table; +} ray_exec_memo_save_t; +void ray_exec_memo_push(ray_graph_t* g, ray_op_t* root, ray_exec_memo_save_t* save); +void ray_exec_memo_pop(ray_graph_t* g, const ray_exec_memo_save_t* save); ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl); /* ── embedding_exec.c ── */ diff --git a/src/ops/ops.h b/src/ops/ops.h index 656b4fd5..d24ce6ae 100644 --- a/src/ops/ops.h +++ b/src/ops/ops.h @@ -572,6 +572,23 @@ typedef struct ray_graph { * that reject an expression outright (arithmetic on a symbol) stay * quiet inside an arm and let the arm compile as it always did. */ int if_arm_depth; + + /* Result memo for shared nodes (exec.c, exec_node): a node with more + * than one consumer in the graph is executed once and its result is + * handed to every consumer as a retained ref; the memo's own ref is + * released at exec_memo_end. Set up around the flat exec_node(root) call in + * ray_execute_inner, NULL otherwise. Without it a `let`-bound string + * expression used three times ran three times. */ + ray_t** memo_vals; + uint32_t* memo_uses; + uint32_t memo_n; + ray_t* memo_hdr; + /* The table the memo was armed over. A node evaluated while g->table + * is swapped for another (an `if` branch over its compacted rows, a + * filter's right-hand side over a sub-table, a window partition) has a + * value of that table's length, so the memo neither serves nor stores + * while g->table differs. */ + ray_t* memo_table; } ray_graph_t; /* ===== Morsel Iterator ===== */ diff --git a/src/ops/pivot.c b/src/ops/pivot.c index 0be4a557..e43b1f6e 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -252,7 +252,12 @@ static ray_t* if_eval_branch(ray_graph_t* g, ray_op_t* branch, g->table = sub; g->selection = NULL; + /* The branch's shared nodes get a memo over the branch's rows; the + * outer memo (values of the full table) is set aside meanwhile. */ + ray_exec_memo_save_t memo_save; + ray_exec_memo_push(g, branch, &memo_save); ray_t* value = exec_node(g, branch); + ray_exec_memo_pop(g, &memo_save); if (g->selection) { ray_release(g->selection); g->selection = NULL; @@ -501,6 +506,144 @@ static ray_t* if_scatter_str(ray_t* result, ray_t* value, int64_t* ids, return result; } +/* Descriptor scatter for the STR arm of the selected path. A side is a + * STR vector with one row per id (or as many rows as the table, indexed by + * the id), or a broadcast scalar. The result takes the side's 16-byte + * descriptor at ids[j] instead of appending its bytes row by row: pooled + * strings keep pointing into their pool, two different pools are laid end + * to end (the second side's offsets shift), a pooled scalar's bytes go in + * once. Rows not in either id list stay the null descriptor the caller + * zeroed. Returns NULL, the result untouched, for a side this cannot take + * (a SYM branch, a length that is neither) — the caller then falls back to + * the per-row scatter, which also reports the length error. */ +typedef struct { + ray_t* v; + int64_t* ids; + int64_t n; + bool scalar; + bool full; /* vector of nrows: row ids[j] */ + const ray_str_t* desc; + ray_t* pool; + const char* sp; /* scalar bytes */ + size_t sl; +} if_str_side_t; + +static bool if_str_side_init(ray_t* v, int64_t* ids, int64_t n, int64_t nrows, + if_str_side_t* s) { + memset(s, 0, sizeof(*s)); + s->ids = ids; s->n = n; + if (!v || n <= 0) return true; + s->v = v; + if (v->type == -RAY_STR) { + s->scalar = true; s->sp = ray_str_ptr(v); s->sl = ray_str_len(v); + return true; + } + if (v->type != RAY_STR) return false; + if (v->len == 1) { + s->scalar = true; + s->sp = ray_str_vec_get(v, 0, &s->sl); + if (!s->sp) { s->sp = ""; s->sl = 0; } + return true; + } + if (v->len == n) s->full = false; + else if (v->len == nrows) s->full = true; + else return false; + const char* bytes = NULL; + str_resolve(v, &s->desc, &bytes); + s->pool = str_vec_pool_obj(v); + if (s->pool && RAY_IS_ERR(s->pool)) return false; + return true; +} + +typedef struct { + const int64_t* ids; + const ray_str_t* src; + bool scalar; + bool full; + ray_str_t sd; /* the scalar's descriptor */ + uint32_t shift; + ray_str_t* dst; +} if_str_scatter_ctx_t; + +static void if_str_scatter_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const if_str_scatter_ctx_t* c = (const if_str_scatter_ctx_t*)vctx; + if (c->scalar) { + for (int64_t j = start; j < end; j++) c->dst[c->ids[j]] = c->sd; + return; + } + for (int64_t j = start; j < end; j++) { + ray_str_t d = c->full ? c->src[c->ids[j]] : c->src[j]; + if (c->shift && !ray_str_is_inline(&d)) d.pool_off += c->shift; + c->dst[c->ids[j]] = d; + } +} + +static void if_str_scatter_side(const if_str_side_t* s, uint32_t shift, uint32_t scalar_off, + ray_str_t* dst) { + if (!s->v) return; + if_str_scatter_ctx_t c = { + .ids = s->ids, .src = s->desc, .scalar = s->scalar, .full = s->full, + .shift = shift, .dst = dst, + }; + if (s->scalar) { + memset(&c.sd, 0, sizeof(c.sd)); + c.sd.len = (uint32_t)s->sl; + if (s->sl <= RAY_STR_INLINE_MAX) { + if (s->sl) memcpy(c.sd.data, s->sp, s->sl); + } else { + memcpy(c.sd.prefix, s->sp, 4); + c.sd.pool_off = scalar_off; + } + } + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, s->n, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, if_str_scatter_fn, &c, s->n); + else + if_str_scatter_fn(&c, 0, 0, s->n); +} + +static ray_t* if_scatter_str_desc(ray_t* result, + ray_t* then_v, int64_t* t_ids, int64_t t_n, + ray_t* else_v, int64_t* e_ids, int64_t e_n, + int64_t nrows) { + if_str_side_t t, e; + if (!if_str_side_init(then_v, t_ids, t_n, nrows, &t)) return NULL; + if (!if_str_side_init(else_v, e_ids, e_n, nrows, &e)) return NULL; + + ray_t* pt = (t.v && !t.scalar) ? t.pool : NULL; + ray_t* pe = (e.v && !e.scalar) ? e.pool : NULL; + bool t_big = t.v && t.scalar && t.sl > RAY_STR_INLINE_MAX; + bool e_big = e.v && e.scalar && e.sl > RAY_STR_INLINE_MAX; + uint32_t e_shift = 0, ts_off = 0, es_off = 0; + if (!t_big && !e_big && (pt == pe || !pt || !pe)) { + ray_t* shared = pt ? pt : pe; + if (shared) { ray_retain(shared); result->str_pool = shared; } + } else { + int64_t tl = pt ? pt->len : 0; + int64_t el = (pe && pe != pt) ? pe->len : 0; + if (tl < 0 || el < 0) return NULL; + uint64_t total = (uint64_t)tl + (uint64_t)el + + (t_big ? (uint64_t)t.sl : 0) + (e_big ? (uint64_t)e.sl : 0); + if (total > UINT32_MAX) return NULL; + ray_t* np = ray_alloc(total > 0 ? (size_t)total : 1); + if (!np || RAY_IS_ERR(np)) return NULL; + np->type = RAY_U8; + np->len = (int64_t)total; + char* dst = (char*)ray_data(np); + uint32_t off = 0; + if (tl) { memcpy(dst, ray_data(pt), (size_t)tl); off += (uint32_t)tl; } + if (pe && pe != pt && el) { memcpy(dst + off, ray_data(pe), (size_t)el); e_shift = off; off += (uint32_t)el; } + if (t_big) { memcpy(dst + off, t.sp, t.sl); ts_off = off; off += (uint32_t)t.sl; } + if (e_big) { memcpy(dst + off, e.sp, e.sl); es_off = off; off += (uint32_t)e.sl; } + result->str_pool = np; + } + ray_str_t* dst = (ray_str_t*)ray_data(result); + if_str_scatter_side(&t, 0, ts_off, dst); + if_str_scatter_side(&e, e_shift, es_off, dst); + return result; +} + /* Runtime-id translation for one SYM branch of an `if`. * * A branch that scans a FILE-domain column used to go through the @@ -772,9 +915,15 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { bool ok = true; if (out_type == RAY_STR) { - result = if_scatter_str(result, then_v, true_ids, true_count, nrows); - if (result && !RAY_IS_ERR(result)) - result = if_scatter_str(result, else_v, false_ids, false_count, nrows); + ray_t* fast = if_scatter_str_desc(result, then_v, true_ids, true_count, + else_v, false_ids, false_count, nrows); + if (fast) { + result = fast; + } else { + result = if_scatter_str(result, then_v, true_ids, true_count, nrows); + if (result && !RAY_IS_ERR(result)) + result = if_scatter_str(result, else_v, false_ids, false_count, nrows); + } } else if (out_type == RAY_SYM) { ok = if_scatter_sym(result, then_v, true_ids, true_count, nrows) && if_scatter_sym(result, else_v, false_ids, false_count, nrows); @@ -893,6 +1042,32 @@ static void if_fill_par_fn(void* ctx, uint32_t wid, int64_t start, int64_t end) if_fill_range((const if_fill_ctx_t*)ctx, start, end); } +/* Descriptor fill for the STR arm of exec_if_eager: dst[r] is the chosen + * side's descriptor; a pooled else-descriptor moves by e_shift when the two + * pools were laid end to end. Rows are independent; workers write disjoint + * ranges. */ +typedef struct { + const uint8_t* cond; + const ray_str_t* t; + const ray_str_t* e; + ray_str_t* dst; + uint32_t e_shift; +} if_str_desc_ctx_t; + +static void if_str_desc_fn(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { + (void)worker_id; + const if_str_desc_ctx_t* c = (const if_str_desc_ctx_t*)vctx; + for (int64_t r = start; r < end; r++) { + if (c->cond[r]) { + c->dst[r] = c->t[r]; + } else { + ray_str_t d = c->e[r]; + if (c->e_shift && !ray_str_is_inline(&d)) d.pool_off += c->e_shift; + c->dst[r] = d; + } + } +} + static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { /* cond = inputs[0], then = inputs[1], else_id stored in ext->third_in */ ray_t* cond_v = exec_node(g, op_child(g, op, 0)); @@ -959,27 +1134,64 @@ static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { uint8_t* cond_p = (uint8_t*)ray_data(cond_v); if (out_type == RAY_STR) { + /* Two STR vectors: the result is descriptors only. Each row takes + * its side's 16-byte descriptor; pooled strings keep pointing into + * their pool. One shared pool (or one side inline-only) is reused + * as is; two different pools are laid end to end in a new pool and + * the else side's offsets shift by the then pool's length. Nulls + * are empty descriptors and travel unchanged. No per-row append, + * no rehash; the fill runs on the worker pool. */ if (!then_scalar && !else_scalar && then_v->type == RAY_STR && else_v->type == RAY_STR && - len <= then_v->len && len <= else_v->len && - !ray_vec_may_have_nulls(then_v) && - !ray_vec_may_have_nulls(else_v)) { + len <= then_v->len && len <= else_v->len) { ray_t* then_pool = str_vec_pool_obj(then_v); ray_t* else_pool = str_vec_pool_obj(else_v); + const ray_str_t* t_desc = NULL; + const ray_str_t* e_desc = NULL; + const char* t_bytes = NULL; + const char* e_bytes = NULL; + str_resolve(then_v, &t_desc, &t_bytes); + str_resolve(else_v, &e_desc, &e_bytes); + bool ok = true; + uint32_t e_shift = 0; if (then_pool == else_pool || !then_pool || !else_pool) { ray_t* out_pool = then_pool ? then_pool : else_pool; if (out_pool && !RAY_IS_ERR(out_pool)) { ray_retain(out_pool); result->str_pool = out_pool; } - const ray_str_t* t_desc = NULL; - const ray_str_t* e_desc = NULL; - const char* unused_pool = NULL; - str_resolve(then_v, &t_desc, &unused_pool); - str_resolve(else_v, &e_desc, &unused_pool); - ray_str_t* dst = (ray_str_t*)ray_data(result); - for (int64_t i = 0; i < len; i++) - dst[i] = cond_p[i] ? t_desc[i] : e_desc[i]; + } else if (RAY_IS_ERR(then_pool) || RAY_IS_ERR(else_pool)) { + ok = false; + } else { + int64_t tl = then_pool->len, el = else_pool->len; + if (tl < 0 || el < 0 || (uint64_t)tl + (uint64_t)el > UINT32_MAX) { + ok = false; + } else { + ray_t* np = ray_alloc((size_t)(tl + el) > 0 ? (size_t)(tl + el) : 1); + if (!np || RAY_IS_ERR(np)) { + ok = false; + } else { + np->type = RAY_U8; + np->len = tl + el; + if (tl) memcpy(ray_data(np), t_bytes, (size_t)tl); + if (el) memcpy((char*)ray_data(np) + tl, e_bytes, (size_t)el); + result->str_pool = np; + e_shift = (uint32_t)tl; + } + } + } + if (ok) { + if_str_desc_ctx_t dctx = { + .cond = cond_p, .t = t_desc, .e = e_desc, + .dst = (ray_str_t*)ray_data(result), .e_shift = e_shift, + }; + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, len, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, if_str_desc_fn, &dctx, len); + else + if_str_desc_fn(&dctx, 0, 0, len); + if (ray_vec_may_have_nulls(then_v) || ray_vec_may_have_nulls(else_v)) + result->attrs |= RAY_ATTR_HAS_NULLS; ray_release(cond_v); ray_release(then_v); ray_release(else_v); return result; } diff --git a/src/ops/query.c b/src/ops/query.c index f5bc3071..a17e1581 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -3069,6 +3069,225 @@ static int64_t derived_key_chunk_rows(void) { #endif return DERIVED_KEY_CHUNK; } + +/* ---- chunk STR build: the entries at positions dv[lo, lo+n) as one STR + * vector over one pool. Lengths and byte pointers are read on the workers + * (the raw snapshot needs no lock); a position past the file prefix — a + * runtime-appended entry — is resolved through the domain on the calling + * thread. A serial prefix sum places the pooled rows, then the workers + * write the descriptors and copy the pooled bytes into disjoint ranges. */ +typedef struct { + const void* dv; + uint8_t dattrs; + int64_t lo; + const ray_sym_domain_raw_t* raw; + const char** ptr; + uint32_t* len; + uint32_t* off; + ray_str_t* dst; + char* pool; + atomic_int late; +} dk_chunk_ctx_t; + +static void dk_chunk_scan_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dk_chunk_ctx_t* c = (dk_chunk_ctx_t*)vctx; + int late = 0; + for (int64_t i = start; i < end; i++) { + int64_t pos = ray_read_sym(c->dv, c->lo + i, RAY_SYM, c->dattrs); + if (pos >= 0 && pos < c->raw->count) { + size_t sl = 0; + c->ptr[i] = ray_sym_domain_raw_str(c->raw, pos, &sl); + c->len[i] = (uint32_t)sl; + } else { + c->ptr[i] = NULL; + c->len[i] = UINT32_MAX; + late++; + } + } + if (late) atomic_fetch_add_explicit(&c->late, late, memory_order_relaxed); +} + +static void dk_chunk_fill_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_chunk_ctx_t* c = (const dk_chunk_ctx_t*)vctx; + for (int64_t i = start; i < end; i++) { + ray_str_t* d = &c->dst[i]; + uint32_t l = c->len[i]; + memset(d, 0, sizeof(*d)); + d->len = l; + if (l == 0) continue; + if (l <= RAY_STR_INLINE_MAX) { + memcpy(d->data, c->ptr[i], l); + } else { + memcpy(c->pool + c->off[i], c->ptr[i], l); + memcpy(d->prefix, c->ptr[i], 4); + d->pool_off = c->off[i]; + } + } +} + +static ray_t* derived_key_chunk_strs(const void* dv, uint8_t dattrs, int64_t lo, int64_t n, + const ray_sym_domain_raw_t* raw, + struct ray_sym_domain_s* dom) { + ray_t* aux_hdr = NULL; + size_t ptr_sz = (size_t)n * sizeof(const char*); + size_t u32_sz = (size_t)n * sizeof(uint32_t); + char* mem = (char*)scratch_alloc(&aux_hdr, ptr_sz + 2 * u32_sz); + if (!mem) return NULL; + dk_chunk_ctx_t c; + memset(&c, 0, sizeof(c)); + c.dv = dv; c.dattrs = dattrs; c.lo = lo; c.raw = raw; + c.ptr = (const char**)mem; + c.len = (uint32_t*)(mem + ptr_sz); + c.off = (uint32_t*)(mem + ptr_sz + u32_sz); + atomic_store_explicit(&c.late, 0, memory_order_relaxed); + + ray_pool_t* pool = ray_pool_get(); + bool par = ray_pool_par_dispatch_ok(pool, n, RAY_PARALLEL_THRESHOLD); + if (par) ray_pool_dispatch(pool, dk_chunk_scan_fn, &c, n); + else dk_chunk_scan_fn(&c, 0, 0, n); + if (atomic_load_explicit(&c.late, memory_order_relaxed)) { + for (int64_t i = 0; i < n; i++) { + if (c.len[i] != UINT32_MAX) continue; + int64_t pos = ray_read_sym(dv, lo + i, RAY_SYM, dattrs); + ray_t* a = ray_sym_domain_str(dom, pos); + size_t sl = a ? ray_str_len(a) : 0; + if (sl > UINT32_MAX) { scratch_free(aux_hdr); return NULL; } + c.ptr[i] = a ? ray_str_ptr(a) : ""; + c.len[i] = (uint32_t)sl; + } + } + uint64_t total = 0; + for (int64_t i = 0; i < n; i++) { + if (c.len[i] <= RAY_STR_INLINE_MAX) continue; + c.off[i] = (uint32_t)total; + total += c.len[i]; + if (total > UINT32_MAX) { scratch_free(aux_hdr); return NULL; } + } + ray_t* sv = ray_vec_new(RAY_STR, n); + if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); scratch_free(aux_hdr); return NULL; } + sv->len = n; + if (total > 0) { + ray_t* sp = ray_alloc((size_t)total); + if (!sp || RAY_IS_ERR(sp)) { ray_release(sv); scratch_free(aux_hdr); return NULL; } + sp->type = RAY_U8; + sp->len = (int64_t)total; + sv->str_pool = sp; + c.pool = (char*)ray_data(sp); + } + c.dst = (ray_str_t*)ray_data(sv); + if (par) ray_pool_dispatch(pool, dk_chunk_fill_fn, &c, n); + else dk_chunk_fill_fn(&c, 0, 0, n); + scratch_free(aux_hdr); + return sv; +} + +/* ---- chunk result interning: each distinct string of the chunk's key + * vector is interned once, all of them under one lock, and its id spread + * over the rows that hold it. The hashes are the intern table's own + * (ray_hash_bytes), computed on the workers; the dedupe is an + * open-addressing table over the distinct ordinals, on scratch. */ +typedef struct { + const ray_str_t* desc; + const char* pool; + uint32_t* hash; +} dk_hash_ctx_t; + +static void dk_hash_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_hash_ctx_t* c = (const dk_hash_ctx_t*)vctx; + for (int64_t i = start; i < end; i++) { + const ray_str_t* d = &c->desc[i]; + c->hash[i] = (uint32_t)ray_hash_bytes(ray_str_t_ptr(d, c->pool), d->len); + } +} + +static bool derived_key_intern_chunk(ray_t* kc, int64_t n, int64_t* out) { + const ray_str_t* desc = NULL; + const char* pool = NULL; + str_resolve(kc, &desc, &pool); + int64_t slots = 1024; + while (slots < 2 * n) slots <<= 1; + ray_t* hdr = NULL; + size_t hash_sz = (size_t)n * sizeof(uint32_t); + size_t rep_sz = (size_t)n * sizeof(int32_t); + size_t tab_sz = (size_t)slots * sizeof(int32_t); + size_t dstr_sz = (size_t)n * sizeof(const char*); + size_t dlen_sz = (size_t)n * sizeof(size_t); + size_t dhsh_sz = (size_t)n * sizeof(uint32_t); + size_t did_sz = (size_t)n * sizeof(int64_t); + /* One carve; the 8-byte arrays go first so every field stays aligned. */ + char* mem = (char*)scratch_alloc(&hdr, hash_sz + rep_sz + tab_sz + dstr_sz + dlen_sz + dhsh_sz + did_sz); + if (!mem) return false; + const char** dstr = (const char**)mem; mem += dstr_sz; + size_t* dlen = (size_t*)mem; mem += dlen_sz; + int64_t* did = (int64_t*)mem; mem += did_sz; + uint32_t* hash = (uint32_t*)mem; mem += hash_sz; + int32_t* rep = (int32_t*)mem; mem += rep_sz; + int32_t* tab = (int32_t*)mem; mem += tab_sz; + uint32_t* dhsh = (uint32_t*)mem; + memset(tab, 0xff, tab_sz); + + dk_hash_ctx_t hc = { .desc = desc, .pool = pool, .hash = hash }; + ray_pool_t* rp = ray_pool_get(); + if (ray_pool_par_dispatch_ok(rp, n, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(rp, dk_hash_fn, &hc, n); + else + dk_hash_fn(&hc, 0, 0, n); + + uint64_t mask = (uint64_t)slots - 1; + int64_t nd = 0; + for (int64_t i = 0; i < n; i++) { + uint32_t h = hash[i]; + const ray_str_t* d = &desc[i]; + const char* sp = ray_str_t_ptr(d, pool); + uint64_t s = ((uint64_t)h * 0x9E3779B97F4A7C15ull >> 32) & mask; + for (;;) { + int32_t r = tab[s]; + if (r < 0) { + tab[s] = (int32_t)nd; + rep[i] = (int32_t)nd; + dstr[nd] = sp; dlen[nd] = d->len; dhsh[nd] = h; + nd++; + break; + } + if (dhsh[r] == h && dlen[r] == d->len && + (d->len == 0 || memcmp(dstr[r], sp, d->len) == 0)) { + rep[i] = r; + break; + } + s = (s + 1) & mask; + } + } + if (ray_sym_intern_batch(dhsh, dstr, dlen, nd, did) < 0) { scratch_free(hdr); return false; } + for (int64_t i = 0; i < n; i++) out[i] = did[rep[i]]; + scratch_free(hdr); + return true; +} + +/* ---- the spread pass of derived_key_over_sym_domain: each row takes the + * key of its symbol's slot (workers, disjoint ranges). */ +typedef struct { + const void* cd; + uint8_t attrs; + int64_t dn; + const int32_t* pos; + const int64_t* key; /* spread: key per slot; NULL = write the slot */ + int64_t* out; +} dk_rows_ctx_t; + +static void dk_spread_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_rows_ctx_t* c = (const dk_rows_ctx_t*)vctx; + if (c->key) { + for (int64_t r = start; r < end; r++) + c->out[r] = c->key[c->pos[ray_read_sym(c->cd, r, RAY_SYM, c->attrs)]]; + } else { + for (int64_t r = start; r < end; r++) + c->out[r] = c->pos[ray_read_sym(c->cd, r, RAY_SYM, c->attrs)]; + } +} static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom_vec, struct ray_sym_domain_s* dom, int64_t du) { if (!dom || dom == ray_sym_runtime_domain() || du <= 0) return NULL; @@ -3102,21 +3321,8 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom const int64_t chunk = derived_key_chunk_rows(); for (int64_t lo = 0; lo < du; lo += chunk) { int64_t n = du - lo < chunk ? du - lo : chunk; - ray_t* sv = ray_vec_new(RAY_STR, n); - if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); goto fail; } - for (int64_t i = 0; i < n; i++) { - int64_t pos = ray_read_sym(dv, lo + i, dom_vec->type, dom_vec->attrs); - const char* sp = NULL; - size_t sl = 0; - if (pos >= 0 && pos < raw.count) { - sp = ray_sym_domain_raw_str(&raw, pos, &sl); - } else { - ray_t* a = ray_sym_domain_str(dom, pos); - if (a) { sp = ray_str_ptr(a); sl = ray_str_len(a); } - } - sv = ray_str_vec_append(sv, sp ? sp : "", sp ? sl : 0); - if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); goto fail; } - } + ray_t* sv = derived_key_chunk_strs(dv, dom_vec->attrs, lo, n, &raw, dom); + if (!sv) goto fail; ray_t* mini = ray_table_new(0); if (mini && !RAY_IS_ERR(mini)) mini = ray_table_add_col(mini, col_sym, sv); ray_release(sv); @@ -3134,13 +3340,7 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom if (!kc || RAY_IS_ERR(kc)) { if (kc) ray_error_free(kc); goto fail; } if (!ray_is_vec(kc) || kc->len != n) { ray_release(kc); goto fail; } if (kc->type == RAY_STR) { - for (int64_t i = 0; i < n; i++) { - size_t sl = 0; - const char* sp = ray_str_vec_get(kc, i, &sl); - int64_t id = ray_sym_intern(sp ? sp : "", sp ? sl : 0); - if (id < 0) { ray_release(kc); goto fail; } - kd[lo + i] = id; - } + if (!derived_key_intern_chunk(kc, n, kd + lo)) { ray_release(kc); goto fail; } } else if (RAY_IS_SYM(kc->type)) { /* The STR evaluation still produced symbols (e.g. a literal * symbol branch): take them cell by cell as runtime ids. */ @@ -3215,6 +3415,9 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { ray_sym_vec_adopt_domain(dom_vec, C); int64_t du = 0, du_max = nrows / 2; if (du_max > INT32_MAX) du_max = INT32_MAX; /* slots are int32 */ + /* Slots in first-seen row order: the interned key ids follow the slot + * order, and with them the order the groups come out in — the same + * order the row-wise evaluation gives. */ bool ok = true; for (int64_t r = 0; r < nrows; r++) { int64_t id = ray_read_sym(cd, r, C->type, C->attrs); @@ -3228,6 +3431,11 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { } if (!ok || du == 0) { ray_release(dom_vec); scratch_free(pos_hdr); return NULL; } dom_vec->len = du; + dk_rows_ctx_t rc; + memset(&rc, 0, sizeof(rc)); + rc.cd = cd; rc.attrs = C->attrs; rc.dn = dn; + ray_pool_t* rpool = ray_pool_get(); + bool rows_par = ray_pool_par_dispatch_ok(rpool, nrows, RAY_PARALLEL_THRESHOLD); /* Evaluate the expression over the du distinct symbols through the * same DAG compiler the row-wise key would take, against a one-column @@ -3254,20 +3462,37 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { if (!ray_is_vec(key_dom) || key_dom->len != du) { ray_release(key_dom); scratch_free(pos_hdr); return NULL; } } - /* Pass 2: spread by slot. */ - ray_t* ids = ray_vec_new(RAY_I64, nrows); - if (!ids || RAY_IS_ERR(ids)) { if (ids) ray_error_free(ids); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } - ids->len = nrows; - int64_t* idp = (int64_t*)ray_data(ids); - for (int64_t r = 0; r < nrows; r++) - idp[r] = pos[ray_read_sym(cd, r, C->type, C->attrs)]; - scratch_free(pos_hdr); - ray_t* spread = ray_at_fn(key_dom, ids); - ray_release(ids); - ray_release(key_dom); - if (spread && !RAY_IS_ERR(spread) && ray_is_lazy(spread)) spread = ray_lazy_materialize(spread); - if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); return NULL; } - if (!ray_is_vec(spread) || spread->len != nrows) { ray_release(spread); return NULL; } + /* Pass 2: spread by slot — each row takes the key of its symbol's slot, + * written straight into the result (no index vector, no gather). */ + if (!ray_is_vec(key_dom) || key_dom->len != du) { ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + ray_t* spread = NULL; + if (key_dom->type == RAY_SYM && (key_dom->attrs & RAY_SYM_W_MASK) == RAY_SYM_W64) { + spread = ray_sym_vec_new(RAY_SYM_W64, nrows); + if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + spread->len = nrows; + ray_sym_vec_adopt_domain(spread, key_dom); + rc.pos = pos; rc.key = (const int64_t*)ray_data(key_dom); rc.out = (int64_t*)ray_data(spread); + if (rows_par) ray_pool_dispatch(rpool, dk_spread_fn, &rc, nrows); + else dk_spread_fn(&rc, 0, 0, nrows); + scratch_free(pos_hdr); + ray_release(key_dom); + } else { + /* Any other key vector (the one-shot SYM evaluation's width, or a + * non-SYM result): index by slot and gather. */ + ray_t* ids = ray_vec_new(RAY_I64, nrows); + if (!ids || RAY_IS_ERR(ids)) { if (ids) ray_error_free(ids); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + ids->len = nrows; + rc.pos = pos; rc.key = NULL; rc.out = (int64_t*)ray_data(ids); + if (rows_par) ray_pool_dispatch(rpool, dk_spread_fn, &rc, nrows); + else dk_spread_fn(&rc, 0, 0, nrows); + scratch_free(pos_hdr); + spread = ray_at_fn(key_dom, ids); + ray_release(ids); + ray_release(key_dom); + if (spread && !RAY_IS_ERR(spread) && ray_is_lazy(spread)) spread = ray_lazy_materialize(spread); + if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); return NULL; } + if (!ray_is_vec(spread) || spread->len != nrows) { ray_release(spread); return NULL; } + } agg_route_note_key_domain(); return spread; } @@ -7135,8 +7360,8 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * intermediate filtered table materialised. Closes a large * latency gap on ORDER BY + LIMIT shapes that were previously * dominated by the filtered-table materialisation step. */ - if (where_expr && take_expr && has_sort && !by_expr && !nearest_expr) { - if (ray_fused_topk_supported(where_expr, tbl)) { + if (take_expr && has_sort && !by_expr && !nearest_expr) { + if (!where_expr || ray_fused_topk_supported(where_expr, tbl)) { /* Walk the dict and check: exactly one asc/desc clause naming * a single scalar column, take is an atom K, and every * output column is a -RAY_SYM source-column reference (no diff --git a/src/ops/string.c b/src/ops/string.c index 501e8943..52a792f6 100644 --- a/src/ops/string.c +++ b/src/ops/string.c @@ -987,171 +987,148 @@ static bool substr_scalar_arg(ray_t* v, int64_t* out) { } } -static ray_t* substr_str_scalar_view(ray_t* input, int64_t start, int64_t length) { - if (!input || input->type != RAY_STR) - return NULL; - - int64_t nrows = input->len; - ray_t* result = ray_vec_new(RAY_STR, nrows); - if (!result || RAY_IS_ERR(result)) return result ? result : ray_error("oom", NULL); - result->len = nrows; +/* True when a whole-column (scalar) start/length argument is null. Such an + * argument applies to every row, so the entire result is null — and a null + * integer scalar is INT64_MIN, which must never reach the `scalar - 1` + * subtraction in the row loop (signed-overflow UB). Only scalars are tested + * here; the per-row vector case is handled inline via ray_vec_is_null. */ +static bool substr_scalar_is_null(ray_t* v) { + if (!v) return false; + if (ray_is_atom(v)) return RAY_ATOM_IS_NULL(v); + if (ray_is_vec(v) && v->len == 1 && ray_vec_may_have_nulls(v)) + return ray_vec_is_null(v, 0); + return false; +} - const ray_str_t* src = NULL; - const char* pool = NULL; - str_resolve(input, &src, &pool); - ray_t* owner = (input->attrs & RAY_ATTR_SLICE) ? input->slice_parent : input; - ray_t* pool_obj = owner ? owner->str_pool : NULL; - if (pool_obj && !RAY_IS_ERR(pool_obj)) { - ray_retain(pool_obj); - result->str_pool = pool_obj; +/* Per-row start / length argument of substr: an atom, a 1-element vector + * (scalar), or a vector with one value per row (I64 / I32 / F64). `ok` is + * false for a null. */ +typedef struct { + int64_t scalar; + const int64_t* i64; + const int32_t* i32; + const double* f64; + ray_t* v; /* the vector, for null checks; NULL for scalars */ + bool all_null; +} substr_arg_t; + +static bool substr_arg_init(ray_t* v, int64_t nrows, substr_arg_t* a) { + memset(a, 0, sizeof(*a)); + int64_t sc; + if (substr_scalar_arg(v, &sc)) { a->scalar = sc; return true; } + if (substr_scalar_is_null(v)) { a->all_null = true; return true; } + if (!ray_is_vec(v) || v->len != nrows) return false; + a->v = v; + switch (v->type) { + case RAY_I64: a->i64 = (const int64_t*)ray_data(v); return true; + case RAY_I32: a->i32 = (const int32_t*)ray_data(v); return true; + case RAY_F64: a->f64 = (const double*)ray_data(v); return true; + default: return false; } +} - ray_str_t* dst = (ray_str_t*)ray_data(result); - for (int64_t i = 0; i < nrows; i++) { - const ray_str_t* s = &src[i]; - ray_str_t* d = &dst[i]; - memset(d, 0, sizeof(*d)); +static inline bool substr_arg_at(const substr_arg_t* a, int64_t r, int64_t* out) { + if (a->all_null) return false; + if (!a->v) { *out = a->scalar; return true; } + if (a->i64) { int64_t x = a->i64[r]; if (x == NULL_I64) return false; *out = x; return true; } + if (a->i32) { int32_t x = a->i32[r]; if (x == NULL_I32) return false; *out = (int64_t)x; return true; } + double d = a->f64[r]; + if (d != d) return false; + *out = (int64_t)d; + return true; +} - int64_t st = start - 1; +/* Substring of a STR column as descriptors over the column's own pool: + * an inline result copies its bytes, a longer one points into the parent + * pool at the shifted offset. No bytes are copied and no pool is built, + * so the pass is descriptor-bound and runs on the worker pool. A null + * start or length gives a null (empty) row; the empty result of a start + * past the end is an empty row. Returns NULL for shapes it does not take + * (the caller keeps the general loop). */ +typedef struct { + const ray_str_t* src; + const char* pool; + ray_str_t* dst; + substr_arg_t start; + substr_arg_t len; + _Atomic(uint32_t) any_null; + _Atomic(uint32_t) range_err; +} substr_view_ctx_t; + +static void substr_view_fn(void* vctx, uint32_t worker_id, int64_t lo, int64_t hi) { + (void)worker_id; + substr_view_ctx_t* c = (substr_view_ctx_t*)vctx; + bool null_seen = false, range_seen = false; + for (int64_t i = lo; i < hi; i++) { + ray_str_t* d = &c->dst[i]; + memset(d, 0, sizeof(*d)); + int64_t st, ln; + if (!substr_arg_at(&c->start, i, &st) || !substr_arg_at(&c->len, i, &ln)) { + null_seen = true; + continue; + } + const ray_str_t* s = &c->src[i]; int64_t sl = (int64_t)s->len; + st -= 1; /* 1-based → 0-based */ if (st < 0) st = 0; if (st >= sl) continue; - - int64_t ln = length; if (ln < 0 || ln > sl - st) ln = sl - st; if (ln <= 0) continue; - + const char* sp = ray_str_t_ptr(s, c->pool) + st; d->len = (uint32_t)ln; - if (!ray_str_is_inline(s) && !pool) { - ray_release(result); - return NULL; - } - const char* sp = ray_str_t_ptr(s, pool) + st; if (ln <= RAY_STR_INLINE_MAX) { memcpy(d->data, sp, (size_t)ln); - } else if (!ray_str_is_inline(s) && pool_obj) { - if ((uint64_t)s->pool_off + (uint64_t)st > UINT32_MAX) { - ray_release(result); - return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); - } + } else { + /* a pooled result needs a pooled source (an inline source is at + * most 12 bytes, so a longer result never comes from one) */ + if ((uint64_t)s->pool_off + (uint64_t)st > UINT32_MAX) { range_seen = true; d->len = 0; continue; } memcpy(d->prefix, sp, 4); d->pool_off = s->pool_off + (uint32_t)st; - ray_str_t_cache_hash(d, pool); - } else { - ray_release(result); - return NULL; + /* hash32 stays 0: a consumer that needs it computes it once + * (ray_str_t_hash32); hashing every substring here paid a pass + * over the bytes that most consumers never used. */ } } - - return result; -} - -static bool substr_len_at(ray_t* len_v, int64_t row, int64_t* out) { - if (!len_v || !out) return false; - if (ray_vec_may_have_nulls(len_v)) { - if (ray_vec_is_null(len_v, row)) return false; - } - switch (len_v->type) { - case RAY_I64: { - int64_t v = ((const int64_t*)ray_data(len_v))[row]; - if (v == NULL_I64) return false; - *out = v; - return true; - } - case RAY_I32: { - int32_t v = ((const int32_t*)ray_data(len_v))[row]; - if (v == NULL_I32) return false; - *out = (int64_t)v; - return true; - } - default: - return false; - } + if (null_seen) atomic_store_explicit(&c->any_null, 1, memory_order_relaxed); + if (range_seen) atomic_store_explicit(&c->range_err, 1, memory_order_relaxed); } -static ray_t* substr_str_scalar_start_len_view(ray_t* input, - int64_t start, - ray_t* len_v) { - if (!input || input->type != RAY_STR || !len_v) - return NULL; - if (len_v->type != RAY_I64 && len_v->type != RAY_I32) - return NULL; +static ray_t* substr_str_view(ray_t* input, ray_t* start_v, ray_t* len_v) { + if (!input || input->type != RAY_STR) return NULL; int64_t nrows = input->len; - if (len_v->len != nrows) - return NULL; + substr_view_ctx_t ctx; + memset(&ctx, 0, sizeof(ctx)); + if (!substr_arg_init(start_v, nrows, &ctx.start)) return NULL; + if (!substr_arg_init(len_v, nrows, &ctx.len)) return NULL; ray_t* result = ray_vec_new(RAY_STR, nrows); if (!result || RAY_IS_ERR(result)) return result ? result : ray_error("oom", NULL); result->len = nrows; - - const ray_str_t* src = NULL; - const char* pool = NULL; - str_resolve(input, &src, &pool); + str_resolve(input, &ctx.src, &ctx.pool); ray_t* owner = (input->attrs & RAY_ATTR_SLICE) ? input->slice_parent : input; ray_t* pool_obj = owner ? owner->str_pool : NULL; if (pool_obj && !RAY_IS_ERR(pool_obj)) { ray_retain(pool_obj); result->str_pool = pool_obj; } + ctx.dst = (ray_str_t*)ray_data(result); + atomic_store_explicit(&ctx.any_null, 0, memory_order_relaxed); + atomic_store_explicit(&ctx.range_err, 0, memory_order_relaxed); - ray_str_t* dst = (ray_str_t*)ray_data(result); - int64_t st0 = start - 1; - if (st0 < 0) st0 = 0; - for (int64_t i = 0; i < nrows; i++) { - ray_str_t* d = &dst[i]; - memset(d, 0, sizeof(*d)); - - int64_t ln = 0; - if (!substr_len_at(len_v, i, &ln)) { - ray_vec_set_null(result, i, true); - continue; - } - - const ray_str_t* s = &src[i]; - int64_t sl = (int64_t)s->len; - if (st0 >= sl) continue; - - if (ln < 0 || ln > sl - st0) ln = sl - st0; - if (ln <= 0) continue; - - d->len = (uint32_t)ln; - if (!ray_str_is_inline(s) && !pool) { - ray_release(result); - return NULL; - } - const char* sp = ray_str_t_ptr(s, pool) + st0; - if (ln <= RAY_STR_INLINE_MAX) { - memcpy(d->data, sp, (size_t)ln); - } else if (!ray_str_is_inline(s) && pool_obj) { - if ((uint64_t)s->pool_off + (uint64_t)st0 > UINT32_MAX) { - ray_release(result); - return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); - } - memcpy(d->prefix, sp, 4); - d->pool_off = s->pool_off + (uint32_t)st0; - ray_str_t_cache_hash(d, pool); - } else { - ray_release(result); - return NULL; - } + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, nrows, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, substr_view_fn, &ctx, nrows); + else + substr_view_fn(&ctx, 0, 0, nrows); + if (atomic_load_explicit(&ctx.range_err, memory_order_relaxed)) { + ray_release(result); + return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); } - + if (atomic_load_explicit(&ctx.any_null, memory_order_relaxed)) + result->attrs |= RAY_ATTR_HAS_NULLS; return result; } -/* True when a whole-column (scalar) start/length argument is null. Such an - * argument applies to every row, so the entire result is null — and a null - * integer scalar is INT64_MIN, which must never reach the `scalar - 1` - * subtraction in the row loop (signed-overflow UB). Only scalars are tested - * here; the per-row vector case is handled inline via ray_vec_is_null. */ -static bool substr_scalar_is_null(ray_t* v) { - if (!v) return false; - if (ray_is_atom(v)) return RAY_ATOM_IS_NULL(v); - if (ray_is_vec(v) && v->len == 1 && ray_vec_may_have_nulls(v)) - return ray_vec_is_null(v, 0); - return false; -} - ray_t* exec_substr(ray_graph_t* g, ray_op_t* op) { ray_t* input = exec_node(g, op_child(g, op, 0)); ray_t* start_v = exec_node(g, op_child(g, op, 1)); @@ -1167,26 +1144,14 @@ ray_t* exec_substr(ray_graph_t* g, ray_op_t* op) { bool is_str = (input->type == RAY_STR); if (is_str) { - int64_t s_const = 0, l_const = 0; - if (substr_scalar_arg(start_v, &s_const) && - substr_scalar_arg(len_v, &l_const)) { - ray_t* view = substr_str_scalar_view(input, s_const, l_const); - if (view) { - ray_release(input); - ray_release(start_v); - ray_release(len_v); - return view; - } - } - if (substr_scalar_arg(start_v, &s_const) && - !ray_is_atom(len_v) && len_v->len == nrows) { - ray_t* view = substr_str_scalar_start_len_view(input, s_const, len_v); - if (view) { - ray_release(input); - ray_release(start_v); - ray_release(len_v); - return view; - } + /* Descriptors over the column's own pool, for every scalar / per-row + * combination of start and length that reads as an integer. */ + ray_t* view = substr_str_view(input, start_v, len_v); + if (view) { + ray_release(input); + ray_release(start_v); + ray_release(len_v); + return view; } } diff --git a/test/rfl/fused/topk_like_nullable_nowhere.rfl b/test/rfl/fused/topk_like_nullable_nowhere.rfl new file mode 100644 index 00000000..ef9c53fa --- /dev/null +++ b/test/rfl/fused/topk_like_nullable_nowhere.rfl @@ -0,0 +1,37 @@ +;; The fused filter + top-k takes a `like` over a text column that holds +;; empty strings (a null text cell is the empty string on every like path), +;; and a sort + take with no where: at all — both used to fall back to a full +;; filter / sort of the table. Results are checked against the unfused +;; spelling. +(set N 120000) +(set i (til N)) +(set url (as 'SYMBOL (map (fn [k] (if (== 0 (% k 97)) (format "http://google.com/q/%" k) (if (== 0 (% k 41)) "" (format "http://site%.example.com/p/%" (% k 500) k)))) i))) +(set us (as 'STR url)) +(set ts (% (* i 7919) 100003)) +(set T (table [url us ts k] (list url us ts (% i 17)))) +(count (select {from: T where: (nil? url)})) -- 2896 +;; like on the nullable SYM column: fused select equals the direct like +(count (select {from: T where: (like url "*google*")})) -- 1238 +(== (count (select {from: T where: (like url "*google*")})) (sum (as 'I64 (like url "*google*")))) -- true +(== (count (select {from: T where: (like us "*google*")})) (sum (as 'I64 (like us "*google*")))) -- true +;; top-k over the like: same rows as sorting the filtered table +(set A (select {from: T where: (like url "*google*") asc: ts take: 10})) +(set B (select {from: (select {from: T where: (like url "*google*")}) asc: ts take: 10})) +(all (== (at A 'ts) (at B 'ts))) -- true +(all (== (at A 'url) (at B 'url))) -- true +(count (cols A)) -- 4 +;; the sort key need not be in the projection +(all (== (at (select {from: T u: url where: (like url "*google*") asc: ts take: 10}) 'u) (at B 'url))) -- true +(all (== (at (select {from: T u: us where: (like us "*google*") desc: ts take: 10}) 'u) (at (select {from: (select {from: T where: (like url "*google*")}) desc: ts take: 10}) 'us))) -- true +;; a pattern the empty string matches admits the null rows on both paths +(== (count (select {from: T where: (like url "*") asc: ts take: 200000})) (sum (as 'I64 (like url "*")))) -- true +(== (count (select {from: T where: (like url "*") asc: ts take: 200000})) 120000) -- true +;; and with a projection list +(all (== (at (select {from: T u: url t: ts where: (like us "*google*") asc: ts take: 10}) 't) (at B 'ts))) -- true +;; no where: at all — the whole table's top-k, ascending and descending +(all (== (at (select {from: T asc: ts take: 10}) 'ts) (at (select {from: (select {from: T asc: ts}) take: 10}) 'ts))) -- true +(all (== (at (select {from: T desc: ts take: 10}) 'ts) (at (select {from: (select {from: T desc: ts}) take: 10}) 'ts))) -- true +(first (at (select {from: T asc: ts take: 1}) 'ts)) -- 0 +(count (select {from: T desc: ts take: 25})) -- 25 +;; empty strings sort first as null text +(first (at (select {from: T asc: us take: 1}) 'us)) -- "" diff --git a/test/rfl/query/shared_node_memo.rfl b/test/rfl/query/shared_node_memo.rfl new file mode 100644 index 00000000..239a5f7e --- /dev/null +++ b/test/rfl/query/shared_node_memo.rfl @@ -0,0 +1,32 @@ +;; A DAG node with several consumers is executed once and its value shared +;; (exec_node memo). The answers must equal the per-consumer evaluation the +;; engine used to do, including when one consumer's kernel would have been +;; free to reuse the input buffer in place (rc == 1) while a sibling still +;; read it. +(set N 70000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 500) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 500) k))))) i))) +(set T (table [S] (list S))) +;; a let-bound string search used twice and three times equals its explicit form +(all (== (at (select {from: T x: (let p (str-find S "://") (+ p p))}) 'x) (at (select {from: T x: (+ (str-find S "://") (str-find S "://"))}) 'x))) -- true +(all (== (at (select {from: T x: (let p (str-find S "://") (+ p (+ p p)))}) 'x) (at (select {from: T x: (* 3 (str-find S "://"))}) 'x))) -- true +;; the shared node feeds both an arithmetic consumer and a comparison; the +;; comparison must see the original values +(set R (select {from: T s: S x1: (let p (str-find S "://") (as 'I64 (within p [4 5]))) x3: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (as 'I64 (not (nil? sl))))))) x4: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (as 'I64 (and (within p [4 5]) (not (nil? sl)))))))) pp: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (+ p (* 0 sl)))))) p0: (str-find S "://")})) +(count (select {from: R where: (!= x4 (* x1 x3))})) -- 0 +(count (select {from: R where: (!= pp p0)})) -- 0 +(sum (at R 'x4)) -- 57576 +;; the whole derived host expression equals its unshared spelling +(set host (fn [] (select {from: T k: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr S 1 4) "http") (not (nil? sl))) (substr r 1 sl) S)))))}))) +(count (distinct (at (host) 'k))) -- 511 +(sum (as 'I64 (strlen (at (host) 'k)))) -- 1099454 +;; a shared node used both outside an `if` and inside its branches: the +;; branch runs over its own rows only, so the value computed there must not +;; be handed to the consumer that runs over the whole table (and vice versa) +(set c (< (% i 7) 3)) +(set U (table [S c] (list S c))) +(all (== (at (select {from: U r: (let x (upper S) (if c (concat x "?") (concat x "!")))}) 'r) (at (select {from: U r: (if c (concat (upper S) "?") (concat (upper S) "!"))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (strlen S) (if c x (+ x 1000)))}) 'r) (at (select {from: U r: (if c (strlen S) (+ (strlen S) 1000))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (strlen S) (+ x (if c x (+ x 1000))))}) 'r) (at (select {from: U r: (+ (strlen S) (if c (strlen S) (+ (strlen S) 1000)))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (substr S 8 5) (if c x (concat x "-")))}) 'r) (at (select {from: U r: (if c (substr S 8 5) (concat (substr S 8 5) "-"))}) 'r))) -- true +(sum (at (select {from: U r: (let x (strlen S) (+ x (if c x (+ x 1000))))}) 'r)) -- 41272796 diff --git a/test/rfl/strop/substr_if_str_views.rfl b/test/rfl/strop/substr_if_str_views.rfl new file mode 100644 index 00000000..13774ac4 --- /dev/null +++ b/test/rfl/strop/substr_if_str_views.rfl @@ -0,0 +1,33 @@ +;; substr over a STR column with per-row start / length is a descriptor view +;; over the column's pool, and `if` over two STR columns is a descriptor pick +;; whatever pools the sides come from. Every answer is checked against the +;; per-row evaluation of the same expression. +(set N 120000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 500) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 500) k))))) i))) +(set st (% (* i 7) 19)) +(set ln (- (% (* i 5) 23) 3)) +(set stn (as 'I64 (map (fn [j] (if (== 0 (% j 13)) 0N (% (* j 7) 19))) i))) +(set kk (% i 17)) +(set T (table [S st ln stn k] (list S st ln stn kk))) +(set oracle (fn [f] (as 'STR (map f i)))) +;; vector start, whole tail +(all (== (at (select {from: T x: (substr S st -1)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) -1))))) -- true +;; vector start and vector length (negative lengths, past-the-end starts) +(all (== (at (select {from: T x: (substr S st ln)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) (at ln j)))))) -- true +;; a float start vector reads as its integer image +(all (== (at (select {from: T x: (substr S (as 'F64 st) -1)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) -1))))) -- true +;; null starts give empty (null) rows +(all (== (at (select {from: T x: (substr S stn 4)}) 'x) (oracle (fn [j] (if (nil? (at stn j)) "" (substr (at S j) (at stn j) 4)))))) -- true +;; scalar start and length, the old view shape +(all (== (at (select {from: T x: (substr S 5 7)}) 'x) (oracle (fn [j] (substr (at S j) 5 7))))) -- true +;; a substring of a substring stays a view; the pool is shared, not copied +(all (== (at (select {from: T x: (substr (substr S 9 -1) 2 5)}) 'x) (oracle (fn [j] (substr (substr (at S j) 9 -1) 2 5))))) -- true +;; if over two STR sides from different pools, with empty strings on both +(all (== (at (select {from: T x: (if (== 0 (% k 2)) (substr S 9 -1) S)}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 2)) (substr (at S j) 9 -1) (at S j)))))) -- true +;; if over two views of one pool +(all (== (at (select {from: T x: (if (== 0 (% k 3)) (substr S 5 -1) (substr S 2 6))}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 3)) (substr (at S j) 5 -1) (substr (at S j) 2 6)))))) -- true +;; the picked side's empty strings are nulls of the result +(count (select {from: (select {from: T x: (if (== 0 (% k 2)) (substr S 9 -1) S)}) where: (nil? x)})) -- 16410 +;; a scalar side keeps the per-row path +(all (== (at (select {from: T x: (if (== 0 (% k 3)) S "zzz")}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 3)) (at S j) "zzz"))))) -- true From d70bc38efd750e8e4fb39634200cfe772479ec6c Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sat, 26 Sep 2026 22:27:37 +0300 Subject: [PATCH 31/51] v2.9.1 (#625) (#629) From e04b972f78080c4ceefeacd436add48df8ec7742 Mon Sep 17 00:00:00 2001 From: Anton Date: Sat, 26 Sep 2026 22:14:19 +0200 Subject: [PATCH 32/51] docs(release): merge master back into dev after each release A merged release PR leaves a commit on master that dev lacks, so the next release PR is blocked as out of date with the base, and after a squash it also shows conflicts that aren't real. Add the back-merge as step 5. --- RELEASE.md | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/RELEASE.md b/RELEASE.md index 76421ebe..24813304 100644 --- a/RELEASE.md +++ b/RELEASE.md @@ -29,6 +29,26 @@ truth for the version** — no version literal is ever hand-edited in source. - packages `rayforce-X.Y.Z--.tar.gz` + a `.sha256` checksum, - publishes a GitHub Release with a feature-oriented changelog and the artifacts. +5. **Merge `master` back into `dev`** right after the release PR merges: + + ```sh + git fetch origin + git switch dev && git pull --ff-only + git merge --no-ff -s ours origin/master \ + -m "chore: merge master back into dev after the vX.Y.Z release (#N)" + git diff origin/dev --stat # must print nothing + git push origin dev + ``` + + Merging the release PR leaves a squash or merge commit on `master` that `dev` + doesn't contain. `master` requires branches to be up to date before merging, + so without this step the next release PR is blocked as "out of date with the + base". After a squash, it also shows conflicts that aren't real. Copying the + release commit onto `dev` doesn't help: only a merge makes `master`'s commit + an ancestor of `dev`. `-s ours` is correct because `dev` already contains + everything the release shipped, so the empty `git diff` proves the merge + changed no files. If `master` ever carries a hotfix that isn't on `dev`, + drop `-s ours` and do a normal merge instead. That's the whole ritual. **Never edit the version in source to make a release** — the tag is authoritative. From f10a9c673a31a12460c6862057c4cdc34b2ec893 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sun, 27 Sep 2026 16:44:29 +0300 Subject: [PATCH 33/51] perf(select): whole-table count/min/max/sum/avg from chunk-zone metadata MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A select of whole-table aggregates with no other clause scanned every row, although a stored column already carries per-chunk min/max in its chunk-zone index. - The chunk-zone index of an integer or temporal column also records, per chunk, the sum of its non-null values (int64 with wraparound, as the engine sums) and their count, in a fourth child vector. Indexes written before hold zero in that slot and load without it. - `sum` of an integer column reads the per-chunk sums; `avg` does too when no partial sum can leave double's integer range, so the value is exactly the row-wise one (otherwise it scans as before). - `(select {from: T a: (sum c) b: (min d) …})` with no where/by/take/ sort, every output a count / min / max / sum / avg of a plain column the metadata answers, is built from the metadata in O(chunks). Any other output leaves the query to the planner. Test: store/splayed_zone_aggs (each answer, nulls and types equal to the same query over an in-memory copy; float sums, a filter and an avg past 2^53 take the usual path and still agree; scalar sum/avg). Co-Authored-By: Claude Opus 5.5 --- src/ops/agg.c | 48 +++++++++++++++ src/ops/idxop.c | 34 ++++++++++- src/ops/idxop.h | 12 ++++ src/ops/query.c | 89 ++++++++++++++++++++++++++++ src/ops/system.c | 1 + test/rfl/store/splayed_zone_aggs.rfl | 33 +++++++++++ 6 files changed, 215 insertions(+), 2 deletions(-) create mode 100644 test/rfl/store/splayed_zone_aggs.rfl diff --git a/src/ops/agg.c b/src/ops/agg.c index 51b0138d..670a835d 100644 --- a/src/ops/agg.c +++ b/src/ops/agg.c @@ -404,6 +404,39 @@ static ray_t* agg_pair_vec(ray_t* x, ray_t* y, uint16_t op) { return ray_f64(ray_f64_fin(num / sqrt(dx * dy))); } +/* Whole-column sum / non-null count of an integer column from its + * chunk-zone index (per-chunk int64 sums with wraparound and non-null + * counts), in O(n_chunks). `exact_f64` reports whether every partial sum + * an accumulation in double could meet stays below 2^53 in magnitude, i.e. + * the sum converted to double equals any double accumulation of the rows. + * Returns false when the column has no such index for its current length. */ +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64) { + if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return false; + ray_index_t* ix = ray_index_payload(x->index); + if (ix->built_for_len != x->len || ix->u.chunk_zone.is_f64 || !ix->u.chunk_zone.aggs) + return false; + uint32_t n = ix->u.chunk_zone.n_chunks; + if (ix->u.chunk_zone.aggs->len != 2 * (int64_t)n) return false; + const int64_t* ag = (const int64_t*)ray_data(ix->u.chunk_zone.aggs); + const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); + const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + uint64_t sum = 0; + int64_t nn = 0; + double bound = 0.0; + for (uint32_t g = 0; g < n; g++) { + sum += (uint64_t)ag[g]; + nn += ag[n + g]; + if (ag[n + g] > 0) { + double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); + bound += (a > b ? a : b) * (double)ag[n + g]; + } + } + *sum_out = (int64_t)sum; + *nn_out = nn; + if (exact_f64) *exact_f64 = bound < 9007199254740992.0; /* 2^53 */ + return true; +} + ray_t* ray_sum_fn(ray_t* x) { if (ray_is_lazy(x)) return ray_lazy_append(x, OP_SUM); if (RAY_IS_PARTED(x->type)) return agg_parted_sum(x); @@ -417,6 +450,11 @@ ray_t* ray_sum_fn(ray_t* x) { /* Canonical admission: numeric + TIME (duration); DATE/TIMESTAMP are * absolute points and SYM/STR/GUID are non-numeric → type error. */ if (!agg_type_admitted(OP_SUM, x->type)) return ray_error("type", "sum expects a numeric or time-duration vector, got %s", ray_type_name(x->type)); + /* Integer columns with per-chunk sums in their zone index. */ + if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { + int64_t zs, zn; + if (ray_zone_int_sum(x, &zs, &zn, NULL)) return make_i64(zs); + } /* Narrow/temporal types need specific return constructors that the * DAG executor doesn't provide — use scalar path for these. TIMESTAMP * is rejected by agg_type_admitted() above, so only the duration-like @@ -570,6 +608,16 @@ ray_t* ray_avg_fn(ray_t* x) { /* Canonical admission: numeric + temporal (→ F64); SYM/STR/GUID are * non-numeric → type error (the DAG path otherwise averaged raw ids). */ if (!agg_type_admitted(OP_AVG, x->type)) return ray_error("type", "avg expects a numeric or temporal vector, got %s", ray_type_name(x->type)); + /* Integer columns with per-chunk sums: exact when no partial sum can + * leave double's integer range (then every accumulation order in + * double gives the same value). */ + if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { + int64_t zs, zn; bool exact = false; + if (ray_zone_int_sum(x, &zs, &zn, &exact) && exact) { + if (zn == 0) return ray_typed_null(-RAY_F64); + return make_f64((double)zs / (double)zn); + } + } AGG_VEC_VIA_DAG(x, ray_avg); } if (!is_list(x)) return ray_error("type", "avg expects a numeric vector, atom, or list, got %s", ray_type_name(x->type)); diff --git a/src/ops/idxop.c b/src/ops/idxop.c index 6927a52d..c9228432 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -376,9 +376,12 @@ void ray_index_release_payload(ray_index_t* ix) { ray_release(ix->u.chunk_zone.maxs); if (ix->u.chunk_zone.null_bits && !RAY_IS_ERR(ix->u.chunk_zone.null_bits)) ray_release(ix->u.chunk_zone.null_bits); + if (ix->u.chunk_zone.aggs && !RAY_IS_ERR(ix->u.chunk_zone.aggs)) + ray_release(ix->u.chunk_zone.aggs); ix->u.chunk_zone.mins = NULL; ix->u.chunk_zone.maxs = NULL; ix->u.chunk_zone.null_bits = NULL; + ix->u.chunk_zone.aggs = NULL; break; case RAY_IDX_PART: if (ix->u.part.keys && !RAY_IS_ERR(ix->u.part.keys)) ray_release(ix->u.part.keys); @@ -409,7 +412,7 @@ int ray_index_child_blocks(const ray_index_t* ix, ray_t** out, int cap) { case RAY_IDX_BLOOM: c[0] = ix->u.bloom.bits; break; case RAY_IDX_CHUNK_ZONE: c[0] = ix->u.chunk_zone.mins; c[1] = ix->u.chunk_zone.maxs; - c[2] = ix->u.chunk_zone.null_bits; + c[2] = ix->u.chunk_zone.null_bits; c[3] = ix->u.chunk_zone.aggs; break; case RAY_IDX_PART: c[0] = ix->u.part.keys; c[1] = ix->u.part.starts; c[2] = ix->u.part.lens; @@ -458,6 +461,8 @@ void ray_index_retain_payload(ray_index_t* ix) { ray_retain(ix->u.chunk_zone.maxs); if (ix->u.chunk_zone.null_bits && !RAY_IS_ERR(ix->u.chunk_zone.null_bits)) ray_retain(ix->u.chunk_zone.null_bits); + if (ix->u.chunk_zone.aggs && !RAY_IS_ERR(ix->u.chunk_zone.aggs)) + ray_retain(ix->u.chunk_zone.aggs); break; case RAY_IDX_PART: if (ix->u.part.keys && !RAY_IS_ERR(ix->u.part.keys)) ray_retain(ix->u.part.keys); @@ -598,6 +603,8 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, int64_t s = (int64_t)g * csz; int64_t e = s + csz; if (e > n) e = n; int64_t mn = INT64_MAX, mx = INT64_MIN; + uint64_t sum = 0; /* wraps like the engine's int64 sum */ + int64_t nn = 0; bool any_null = false; for (int64_t i = s; i < e; i++) { if (ray_vec_is_null(v, i)) { any_null = true; continue; } @@ -611,6 +618,13 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, } if (val < mn) mn = val; if (val > mx) mx = val; + sum += (uint64_t)val; + nn++; + } + if (ix->u.chunk_zone.aggs) { + int64_t* ag = (int64_t*)ray_data(ix->u.chunk_zone.aggs); + ag[g] = (int64_t)sum; + ag[n_chunks + g] = nn; } /* Empty (all-null) chunks keep mn=INT64_MAX / mx=INT64_MIN so * the reduce path's min(mins[*]) / max(maxs[*]) ignores them. */ @@ -807,6 +821,12 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2) { ix->u.chunk_zone.mins = mins; ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; + if (!ix->u.chunk_zone.is_f64) { + ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } + aggs->len = 2 * (int64_t)n_chunks; + ix->u.chunk_zone.aggs = aggs; + } ray_err_t err = chunk_zone_scan(v, ix); if (err != RAY_OK) { @@ -857,6 +877,12 @@ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2) { ix->u.chunk_zone.mins = mins; ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; + if (!ix->u.chunk_zone.is_f64) { + ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } + aggs->len = 2 * (int64_t)n_chunks; + ix->u.chunk_zone.aggs = aggs; + } ray_err_t err = chunk_zone_scan(v, ix); if (err != RAY_OK) { @@ -980,7 +1006,11 @@ static int idx_child_slots(ray_index_t* ix, ray_t** slots[4]) { case RAY_IDX_CHUNK_ZONE: slots[n++] = &ix->u.chunk_zone.mins; slots[n++] = &ix->u.chunk_zone.maxs; - slots[n++] = &ix->u.chunk_zone.null_bits; break; + slots[n++] = &ix->u.chunk_zone.null_bits; + /* 4th slot added after the first on-disk generation: older regions + * hold zero there (the payload is zeroed at alloc), which maps to + * NULL — no aggregates, nothing else changes. */ + slots[n++] = &ix->u.chunk_zone.aggs; break; case RAY_IDX_PART: slots[n++] = &ix->u.part.keys; slots[n++] = &ix->u.part.starts; slots[n++] = &ix->u.part.lens; break; diff --git a/src/ops/idxop.h b/src/ops/idxop.h index 2c937844..97ff52dc 100644 --- a/src/ops/idxop.h +++ b/src/ops/idxop.h @@ -165,6 +165,12 @@ typedef struct { uint8_t chunk_log2; /* chunk size = 1 << chunk_log2 (default 16 → 64 K rows) */ uint8_t is_f64; uint8_t _pad[2]; + /* Integer / temporal zones only (NULL for float zones and for + * indexes written before it existed): RAY_I64 vec of + * 2 * n_chunks — [0, n) the chunk's non-null values summed with + * int64 wraparound, [n, 2n) its non-null row count. Whole-column + * sum / count / avg answer from it in O(n_chunks). */ + ray_t* aggs; } chunk_zone; struct { /* RAY_IDX_PART */ ray_t* keys; /* distinct partition values, in ascending block order */ @@ -271,6 +277,12 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2); * compute an index for persistence without COWing a shared column. */ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2); +/* Whole-column sum (int64 wraparound) and non-null count of an integer + * column from its chunk-zone per-chunk aggregates; false when the column + * carries none for its current length. exact_f64 (optional): whether the + * sum is exactly what a double accumulation of the rows gives. */ +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64); + /* Build a RAY_IDX_DICT (codes + distinct values) for STR vector `v` WITHOUT * attaching it — standalone RAY_INDEX object (caller releases / attaches). * Returns RAY_ERR_NYI for non-STR. Used at splayed save to persist the dict diff --git a/src/ops/query.c b/src/ops/query.c index a17e1581..e7c5ce19 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -528,6 +528,77 @@ static uint16_t resolve_agg_opcode(int64_t sym_id) { return 0; } +/* See the call site in ray_select_impl. NULL: some output is not answerable + * from metadata (the caller plans the query as usual). */ +static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t dict_n, + int64_t from_id) { + int64_t nrows = ray_table_nrows(tbl); + int64_t n_out = 0; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + if (dict_elems[i]->i64 == from_id) continue; + ray_t* e = dict_elems[i + 1]; + if (!e || e->type != RAY_LIST || ray_len(e) != 2) return NULL; + ray_t** el = (ray_t**)ray_data(e); + if (el[0]->type != -RAY_SYM || (el[0]->attrs & ATTR_QUOTED)) return NULL; + if (el[1]->type != -RAY_SYM || (el[1]->attrs & ATTR_QUOTED)) return NULL; + ray_t* col = ray_table_get_col(tbl, el[1]->i64); + if (!col || !ray_is_vec(col) || RAY_IS_PARTED(col->type) || col->type == RAY_MAPCOMMON || + (col->attrs & RAY_ATTR_SLICE) || col->len != nrows) + return NULL; + uint16_t op = resolve_agg_opcode(el[0]->i64); + bool int_col = col->type == RAY_I64 || col->type == RAY_I32 || + col->type == RAY_I16 || col->type == RAY_U8; + int64_t zs, zn; bool exact = false; + switch (op) { + case OP_COUNT: break; + case OP_MIN: case OP_MAX: { + if (ray_index_kind(col) != RAY_IDX_CHUNK_ZONE) return NULL; + ray_index_t* ix = ray_index_payload(col->index); + if (ix->built_for_len != col->len || ix->u.chunk_zone.is_f64) return NULL; + break; + } + case OP_SUM: + if (!int_col || !ray_zone_int_sum(col, &zs, &zn, NULL)) return NULL; + break; + case OP_AVG: + if (!int_col || !ray_zone_int_sum(col, &zs, &zn, &exact) || !exact) return NULL; + break; + default: return NULL; + } + n_out++; + } + if (n_out == 0) return NULL; + + ray_t* res = ray_table_new(n_out); + if (!res || RAY_IS_ERR(res)) return NULL; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid == from_id) continue; + ray_t** el = (ray_t**)ray_data(dict_elems[i + 1]); + ray_t* col = ray_table_get_col(tbl, el[1]->i64); + uint16_t op = resolve_agg_opcode(el[0]->i64); + ray_t* atom = op == OP_COUNT ? ray_i64(nrows) + : op == OP_MIN ? ray_min_fn(col) + : op == OP_MAX ? ray_max_fn(col) + : op == OP_SUM ? ray_sum_fn(col) + : ray_avg_fn(col); + if (!atom || RAY_IS_ERR(atom) || !ray_is_atom(atom)) { + if (atom && RAY_IS_ERR(atom)) ray_error_free(atom); else if (atom) ray_release(atom); + ray_release(res); + return NULL; + } + ray_t* v = ray_vec_new(-atom->type, 1); + if (v && !RAY_IS_ERR(v)) v = ray_vec_append(v, &atom->i64); + if (v && !RAY_IS_ERR(v) && RAY_ATOM_IS_NULL(atom)) ray_vec_set_null(v, 0, true); + ray_release(atom); + if (!v || RAY_IS_ERR(v)) { ray_release(res); return NULL; } + res = ray_table_add_col(res, kid, v); + ray_release(v); + if (!res || RAY_IS_ERR(res)) return NULL; + } + return res; +} + static bool agg_name_is_percentile(int64_t sym_id) { ray_t* s = ray_sym_str(sym_id); return s && ray_str_len(s) == 10 && @@ -7316,6 +7387,24 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (n_out == 0 && !where_expr && !by_expr && !take_expr && !has_sort && !nearest_expr) { DICT_VIEW_CLOSE(dv); return tbl; } + /* Whole-table aggregates answered from column metadata: + * `(select {from: T a: (sum c) b: (min d) …})` with no other clause, each + * output one of count / min / max / sum / avg over a plain column that + * the metadata answers exactly — count is the row count; min/max read + * the chunk-zone extrema of an integer or temporal column; sum and avg + * read its per-chunk sums (avg only when no partial sum can leave + * double's integer range, so the value is the row-wise one). Any + * output the metadata cannot answer leaves the query to the planner. */ + if (!where_expr && !by_expr && !take_expr && !has_sort && !nearest_expr && n_out > 0 && + tbl->type == RAY_TABLE) { + ray_t* meta_res = select_aggs_from_metadata(tbl, dict_elems, dict_n, from_id); + if (meta_res) { + ray_release(tbl); + DICT_VIEW_CLOSE(dv); + return meta_res; + } + } + /* Streaming parted ORDER BY: `(select {…} from: PARTED asc/desc: KEY)` * with no by:/take:/nearest: over a table whose partitions are already * internally sorted on KEY and whose partition key-ranges are globally diff --git a/src/ops/system.c b/src/ops/system.c index 6c0a2627..5bff55c8 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -680,6 +680,7 @@ static bool objsize_push_index_children(ray_objsize_walk_t* w, ray_index_t* ix) OBJSIZE_PUSH(ix->u.chunk_zone.mins); OBJSIZE_PUSH(ix->u.chunk_zone.maxs); OBJSIZE_PUSH(ix->u.chunk_zone.null_bits); + OBJSIZE_PUSH(ix->u.chunk_zone.aggs); break; case RAY_IDX_PART: OBJSIZE_PUSH(ix->u.part.keys); OBJSIZE_PUSH(ix->u.part.starts); diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl new file mode 100644 index 00000000..84794e30 --- /dev/null +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -0,0 +1,33 @@ +;; Whole-table aggregates over a splayed table answered from the chunk-zone +;; metadata (per-chunk min/max, per-chunk sums and non-null counts): every +;; answer must equal the same query over an in-memory copy of the data, +;; nulls and types included. Four chunks of 64k rows. +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv") -- 0 +(set n 200000) +(set i (til n)) +(set M (table [a16 a32 a64 big d f] (list (as 'I16 (% i 3000)) (as 'I32 (- (% (* i 7) 100000) 50000)) (- i 100000) (* (- i 100000) 92233720368547) (as 'DATE (% i 900)) (/ (as 'F64 i) 7.0)))) +(.csv.write M "rf_test_zone_aggs.csv") -- 0 +;; blank the first field of every row whose a16 value ends in 0: nulls +(.sys.exec "sed -i 's/^\\([0-9]*\\)0,/,/' rf_test_zone_aggs.csv") -- 0 +(set T (.csv.splayed [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv" "rf_test_zone_aggs/")) +(set Tm (.csv.read [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv")) +(at (.idx.info (at T 'a16)) 'kind) -- 'chunk_zone +(set q (fn [t] (select {from: t s16: (sum a16) c: (count a16) av: (avg a16) s32: (sum a32) av32: (avg a32) s64: (sum a64) av64: (avg a64) mn: (min d) mx: (max d) m16: (min a16) x32: (max a32)}))) +(set R (q T)) +(set O (q Tm)) +(cols R) -- (cols O) +(all (map (fn [c] (== (at (at R c) 0) (at (at O c) 0))) (cols R))) -- true +(all (map (fn [c] (== (type (at R c)) (type (at O c)))) (cols R))) -- true +;; the pinned values +(at (at R 's16) 0) -- (sum (at Tm 'a16)) +(at (at R 'mn) 0) -- 2000.01.01 +(at (at R 'c) 0) -- 200000 +;; sums past double's integer range: avg is left to the row-wise path and +;; still agrees; a float column and a filter take the usual path too +(at (select {from: T a: (avg big)}) 'a) -- (at (select {from: Tm a: (avg big)}) 'a) +(at (select {from: T a: (sum a16) b: (sum f)}) 'b) -- (at (select {from: Tm a: (sum a16) b: (sum f)}) 'b) +(at (select {from: T a: (sum a16) where: (> a32 0)}) 'a) -- (at (select {from: Tm a: (sum a16) where: (> a32 0)}) 'a) +;; scalar forms read the same metadata +(sum (at T 'a32)) -- (sum (at Tm 'a32)) +(avg (at T 'a16)) -- (avg (at Tm 'a16)) +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv") -- 0 From a2049e70ae0e602fe2cf97bb888f283b1c12a48b Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sun, 27 Sep 2026 17:18:29 +0300 Subject: [PATCH 34/51] =?UTF-8?q?fix(index):=20mapped=20index=20regions=20?= =?UTF-8?q?=E2=80=94=20bounds-checked=20children,=20never=20released=20on?= =?UTF-8?q?=20drop?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - A column's inline index region is mapped with the column file, and its child vectors are addressed by region-relative offsets. The offsets were trusted as written. A region re-saved by a binary that knows fewer child slots keeps a stale offset in the slot it did not write (the new chunk-zone aggregates slot), pointing at or past the end of the file; the free path then sized its munmap from it and unmapped a page it did not own. ray_index_inline_map now takes the region size and admits only children that lie inside it; an out-of-bounds or malformed aggregates child is dropped, any other one leaves the column unindexed. - Copies of a mapped column borrow its index without a reference. Dropping the index from such a copy (a delete or an in-place write on it) released the index anyway, freeing the loaded column's children in place: its min/max then read NULL and crashed. A mapped index is now only detached from the vector; the mapping's owner unmaps it. Co-Authored-By: Claude Opus 5.5 --- src/ops/idxop.c | 39 +++++++++++++++++++++++++++++++++++---- src/ops/idxop.h | 2 +- src/store/col.c | 3 ++- src/vec/vec.c | 9 +++++++-- 4 files changed, 45 insertions(+), 8 deletions(-) diff --git a/src/ops/idxop.c b/src/ops/idxop.c index c9228432..7e8a560a 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -1083,7 +1083,9 @@ void ray_index_inline_write(uint8_t* dst, const ray_index_t* ix) { * points at the start of the index region within the column's file mapping. * Returns NULL for a stale layout generation or a payload-size mismatch — * the caller loads the column unindexed (the index is rebuildable). */ -ray_t* ray_index_inline_map(uint8_t* region) { +ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size) { + int64_t head = IDX_ALIGN32(32 + (int64_t)sizeof(ray_index_t)); + if (region_size < head) return NULL; ray_t* idx = (ray_t*)region; if (idx->order != RAY_IDX_FORMAT_MAJOR) return NULL; if (idx->len != (int64_t)sizeof(ray_index_t)) return NULL; @@ -1092,8 +1094,32 @@ ray_t* ray_index_inline_map(uint8_t* region) { int nch = idx_child_slots(ix, slots); for (int i = 0; i < nch; i++) { int64_t o = (int64_t)(intptr_t)(*slots[i]); - *slots[i] = o ? (ray_t*)(region + o) : NULL; + ray_t* c = NULL; + /* A child must lie inside the region: a region re-saved by a + * binary that knows fewer child slots keeps a stale offset in a + * slot it did not write. */ + if (o >= head && o <= region_size - 32) { + ray_t* cand = (ray_t*)(region + o); + int64_t esz = ray_elem_size(cand->type); + if (cand->len >= 0 && esz > 0 && + cand->len <= (region_size - o - 32) / esz) + c = cand; + } + if (o && !c) { + /* The chunk-zone aggregates are optional; any other child out + * of bounds means the region cannot be trusted. */ + if (ix->kind == RAY_IDX_CHUNK_ZONE && slots[i] == &ix->u.chunk_zone.aggs) { + *slots[i] = NULL; + continue; + } + return NULL; + } + *slots[i] = c; } + if (ix->kind == RAY_IDX_CHUNK_ZONE && ix->u.chunk_zone.aggs && + (ix->u.chunk_zone.aggs->type != RAY_I64 || ix->u.chunk_zone.is_f64 || + ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks)) + ix->u.chunk_zone.aggs = NULL; ix->markers |= RAY_MARK_MMAP; idx->mmod = 1; return idx; @@ -2612,7 +2638,12 @@ ray_t* ray_index_drop(ray_t** vp) { * ray_alloc_copy (rc>1). Don't clobber the snapshot in that case — * the other holder still reads it. See vec_drop_index_inplace for * the same pattern. */ - bool shared = ray_atomic_load(&idx->rc) > 1; + /* A mapped index (mmod 1) rides the column file's mapping: copies of + * the column borrow it without a reference, and the mapping's owner + * unmaps it. Dropping it from a vector only detaches it — the + * snapshot stays for the other holders and nothing is released. */ + bool mapped = idx->mmod == 1; + bool shared = mapped || ray_atomic_load(&idx->rc) > 1; if (shared) { ray_index_retain_saved(ix); } @@ -2628,7 +2659,7 @@ ray_t* ray_index_drop(ray_t** vp) { /* Release the index. Per-kind children are released by the RAY_INDEX * branch of ray_release_owned_refs (added in heap.c). */ - ray_release(idx); + if (!mapped) ray_release(idx); return v; } diff --git a/src/ops/idxop.h b/src/ops/idxop.h index 97ff52dc..076a7747 100644 --- a/src/ops/idxop.h +++ b/src/ops/idxop.h @@ -300,7 +300,7 @@ ray_t* ray_index_attach_built(ray_t** vp, ray_t* idx); * in place and return the RAY_INDEX object (flagged RAY_MARK_MMAP). */ int64_t ray_index_inline_size(const ray_index_t* ix); void ray_index_inline_write(uint8_t* dst, const ray_index_t* ix); -ray_t* ray_index_inline_map(uint8_t* region); +ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size); /* Drop any attached index from *vp. No-op if none. Restores the * pre-attach aux state byte-for-byte. Returns *vp. */ diff --git a/src/store/col.c b/src/store/col.c index 27a6b751..c38a523e 100644 --- a/src/store/col.c +++ b/src/store/col.c @@ -1748,7 +1748,8 @@ static ray_t* col_mmap_impl(const char* path, struct ray_sym_domain_s* dom, * column's single mapping. ray_free reads the full mapping size from the * reserved _idx_pad slot to munmap the whole region (payload + index). */ if (cm.has_index) { - ray_t* idx = ray_index_inline_map((uint8_t*)cm.mapped + cm.index_offset); + ray_t* idx = ray_index_inline_map((uint8_t*)cm.mapped + cm.index_offset, + (int64_t)cm.mapped_size - (int64_t)cm.index_offset); if (idx) { /* NULL = stale index layout generation: load unindexed */ ray_t* r = ray_index_attach_built(&vec, idx); if (r && !RAY_IS_ERR(r)) vec = r; diff --git a/src/vec/vec.c b/src/vec/vec.c index f5be8d98..5c17b526 100644 --- a/src/vec/vec.c +++ b/src/vec/vec.c @@ -102,7 +102,12 @@ static inline void vec_drop_index_inplace(ray_t* v) { if (!(v->attrs & RAY_ATTR_HAS_INDEX)) return; ray_t* idx = v->index; ray_index_t* ix = ray_index_payload(idx); - bool shared = ray_atomic_load(&idx->rc) > 1; + /* A mapped index (mmod 1) rides the column file's mapping: copies of + * the column borrow it without a reference, and the mapping's owner + * unmaps it. Dropping it from a vector only detaches it — the + * snapshot stays for the other holders and nothing is released. */ + bool mapped = idx->mmod == 1; + bool shared = mapped || ray_atomic_load(&idx->rc) > 1; if (shared) { /* Take our own retained references to the saved-pointer slots @@ -120,7 +125,7 @@ static inline void vec_drop_index_inplace(ray_t* v) { ix->saved_attrs = 0; } v->attrs &= (uint8_t)~RAY_ATTR_HAS_INDEX; - ray_release(idx); + if (!mapped) ray_release(idx); } /* -------------------------------------------------------------------------- From f5ee037aa47352a9ae5466872c301f74643862b8 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sun, 27 Sep 2026 17:18:29 +0300 Subject: [PATCH 35/51] fix(agg): zone min/max tell all-null from the non-null counts; empty tables keep the planner - The zone min/max treated INT64_MAX / INT64_MIN as "no non-null value", so a column of int64 maxima answered null. When the zone carries non-null counts they decide; min/max also require the zone's extrema to be present. - The metadata select no longer answers an empty table (its count differed from the planner's there). Tests: store/splayed_zone_aggs (edit through a copy of a loaded column then aggregate the original; an all-INT64_MAX column; an empty filter). Co-Authored-By: Claude Opus 5.5 --- src/ops/agg.c | 18 ++++++++++++++---- src/ops/query.c | 4 +++- test/rfl/store/splayed_zone_aggs.rfl | 16 ++++++++++++++++ 3 files changed, 33 insertions(+), 5 deletions(-) diff --git a/src/ops/agg.c b/src/ops/agg.c index 670a835d..138afb65 100644 --- a/src/ops/agg.c +++ b/src/ops/agg.c @@ -645,7 +645,7 @@ ray_t* ray_min_fn(ray_t* x) { * (mutation paths call ray_index_drop). */ if (ray_index_kind(x) == RAY_IDX_CHUNK_ZONE) { ray_index_t* ix = ray_index_payload(x->index); - if (ix->built_for_len == x->len) { + if (ix->built_for_len == x->len && ix->u.chunk_zone.mins) { uint32_t n_chunks = ix->u.chunk_zone.n_chunks; if (ix->u.chunk_zone.is_f64) { const double* mins = (const double*)ray_data(ix->u.chunk_zone.mins); @@ -659,7 +659,12 @@ ray_t* ray_min_fn(ray_t* x) { int64_t mn = INT64_MAX; for (uint32_t g = 0; g < n_chunks; g++) if (mins[g] < mn) mn = mins[g]; - if (mn == INT64_MAX) return ray_typed_null(-x->type); + /* All-null is what the non-null counts say when the + * zone has them; the sentinel alone cannot tell a + * column of INT64_MAX values from an empty one. */ + int64_t zs_, zn_; + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + if (have_nn ? zn_ == 0 : mn == INT64_MAX) return ray_typed_null(-x->type); /* Preserve the column's storage width on the result. */ switch (x->type) { case RAY_BOOL: return ray_bool((bool)mn); @@ -700,7 +705,7 @@ ray_t* ray_max_fn(ray_t* x) { if (ray_is_vec(x)) { if (ray_index_kind(x) == RAY_IDX_CHUNK_ZONE) { ray_index_t* ix = ray_index_payload(x->index); - if (ix->built_for_len == x->len) { + if (ix->built_for_len == x->len && ix->u.chunk_zone.maxs) { uint32_t n_chunks = ix->u.chunk_zone.n_chunks; if (ix->u.chunk_zone.is_f64) { const double* maxs = (const double*)ray_data(ix->u.chunk_zone.maxs); @@ -714,7 +719,12 @@ ray_t* ray_max_fn(ray_t* x) { int64_t mx = INT64_MIN; for (uint32_t g = 0; g < n_chunks; g++) if (maxs[g] > mx) mx = maxs[g]; - if (mx == INT64_MIN) return ray_typed_null(-x->type); + /* All-null is what the non-null counts say when the + * zone has them; the sentinel alone cannot tell a + * column of INT64_MIN values from an empty one. */ + int64_t zs_, zn_; + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + if (have_nn ? zn_ == 0 : mx == INT64_MIN) return ray_typed_null(-x->type); switch (x->type) { case RAY_BOOL: return ray_bool((bool)mx); case RAY_U8: return ray_u8((uint8_t)mx); diff --git a/src/ops/query.c b/src/ops/query.c index e7c5ce19..7c8c4686 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -533,6 +533,7 @@ static uint16_t resolve_agg_opcode(int64_t sym_id) { static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t dict_n, int64_t from_id) { int64_t nrows = ray_table_nrows(tbl); + if (nrows <= 0) return NULL; /* empty tables keep the planner's answers */ int64_t n_out = 0; for (int64_t i = 0; i + 1 < dict_n; i += 2) { if (dict_elems[i]->i64 == from_id) continue; @@ -554,7 +555,8 @@ static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t case OP_MIN: case OP_MAX: { if (ray_index_kind(col) != RAY_IDX_CHUNK_ZONE) return NULL; ray_index_t* ix = ray_index_payload(col->index); - if (ix->built_for_len != col->len || ix->u.chunk_zone.is_f64) return NULL; + if (ix->built_for_len != col->len || ix->u.chunk_zone.is_f64 || + !ix->u.chunk_zone.mins || !ix->u.chunk_zone.maxs) return NULL; break; } case OP_SUM: diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl index 84794e30..2356da4d 100644 --- a/test/rfl/store/splayed_zone_aggs.rfl +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -30,4 +30,20 @@ ;; scalar forms read the same metadata (sum (at T 'a32)) -- (sum (at Tm 'a32)) (avg (at T 'a16)) -- (avg (at Tm 'a16)) +;; editing a copy of a loaded column drops the copy's index; the loaded +;; column's own metadata stays intact and still answers +(set U (delete {from: T where: (== a64 5)})) +(set w (at T 'a32)) +(alter 'w set 0 7) +(at (select {from: T m: (min a32) x: (max a32) s: (sum a32)}) 's) -- (at (select {from: Tm m: (min a32) x: (max a32) s: (sum a32)}) 's) +(at (select {from: T m: (min a32)}) 'm) -- (at (select {from: Tm m: (min a32)}) 'm) +;; a column whose every value is the int64 maximum is not "all null" +(.sys.exec "rm -rf rf_test_zone_max") -- 0 +(.db.splayed.set "rf_test_zone_max/" (table [m] (list (take [9223372036854775807] 70000)))) +(set TX (.db.splayed.get "rf_test_zone_max/")) +(at (select {from: TX a: (min m) b: (max m)}) 'a) -- [9223372036854775807] +(min (at TX 'm)) -- 9223372036854775807 +(.sys.exec "rm -rf rf_test_zone_max") -- 0 +;; an empty table keeps the planner's answers +(at (select {from: (select {from: T where: (< a64 -1000000)}) c: (count a16)}) 'c) -- (at (select {from: (select {from: Tm where: (< a64 -1000000)}) c: (count a16)}) 'c) (.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv") -- 0 From 82a5342a4c409c031d698c69a653f0455c485a09 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sun, 27 Sep 2026 19:10:24 +0300 Subject: [PATCH 36/51] test(store): portable in-place sed in splayed_zone_aggs BSD sed takes the argument after -i as the backup suffix, so the null injection failed on macOS; use -i.bak and remove the backup. Co-Authored-By: Claude Opus 5.5 --- test/rfl/store/splayed_zone_aggs.rfl | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl index 2356da4d..4716dc30 100644 --- a/test/rfl/store/splayed_zone_aggs.rfl +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -2,13 +2,13 @@ ;; metadata (per-chunk min/max, per-chunk sums and non-null counts): every ;; answer must equal the same query over an in-memory copy of the data, ;; nulls and types included. Four chunks of 64k rows. -(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv") -- 0 +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 (set n 200000) (set i (til n)) (set M (table [a16 a32 a64 big d f] (list (as 'I16 (% i 3000)) (as 'I32 (- (% (* i 7) 100000) 50000)) (- i 100000) (* (- i 100000) 92233720368547) (as 'DATE (% i 900)) (/ (as 'F64 i) 7.0)))) (.csv.write M "rf_test_zone_aggs.csv") -- 0 ;; blank the first field of every row whose a16 value ends in 0: nulls -(.sys.exec "sed -i 's/^\\([0-9]*\\)0,/,/' rf_test_zone_aggs.csv") -- 0 +(.sys.exec "sed -i.bak 's/^\\([0-9]*\\)0,/,/' rf_test_zone_aggs.csv") -- 0 (set T (.csv.splayed [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv" "rf_test_zone_aggs/")) (set Tm (.csv.read [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv")) (at (.idx.info (at T 'a16)) 'kind) -- 'chunk_zone @@ -46,4 +46,4 @@ (.sys.exec "rm -rf rf_test_zone_max") -- 0 ;; an empty table keeps the planner's answers (at (select {from: (select {from: T where: (< a64 -1000000)}) c: (count a16)}) 'c) -- (at (select {from: (select {from: Tm where: (< a64 -1000000)}) c: (count a16)}) 'c) -(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv") -- 0 +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 From 007663190d3e73f3fb2037d94b79f8f0e47c2d0b Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Sun, 27 Sep 2026 18:42:35 +0200 Subject: [PATCH 37/51] fix(system): reject arguments to gc (#615) * fix(system): reject arguments to gc * docs(memory): use zero-argument gc form * fix(system): report gc arity errors --- docs/docs/guides/memory.md | 2 +- src/ops/system.c | 3 ++- test/rfl/system/reserved_namespace.rfl | 8 ++++---- test/rfl/system/system_branch_cov.rfl | 4 ++-- 4 files changed, 9 insertions(+), 8 deletions(-) diff --git a/docs/docs/guides/memory.md b/docs/docs/guides/memory.md index a0050195..ba1780ea 100644 --- a/docs/docs/guides/memory.md +++ b/docs/docs/guides/memory.md @@ -385,7 +385,7 @@ This tells you the join needed about 1 GB of temporary memory beyond what was al | Tool | What It Does | When to Use | |---|---|---| | `(.sys.mem 0)` | Returns heap allocation statistics | Monitor memory usage, detect leaks | -| `(.sys.gc 0)` | Flushes caches, releases pages | Between heavy queries, before benchmarks | +| `(.sys.gc)` | Flushes caches, releases pages | Between heavy queries, before benchmarks | | `(.sys.info 0)` | Shows system and runtime info | Check total RAM, CPU count, OS details | | `(timeit expr)` | Measures execution time of one expression | Benchmark a specific operation | | `:timeit` | Toggles profiling for all REPL expressions | Interactive performance exploration | diff --git a/src/ops/system.c b/src/ops/system.c index 6c0a2627..96ef785c 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -873,7 +873,8 @@ ray_t* ray_mem_ts_fn(ray_t** args, int64_t n) { * pages/pools. Rayforce values are reference-counted, so this is allocator * GC rather than a tracing collector. Variadic to allow `(.sys.gc)`. */ ray_t* ray_gc_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".sys.gc takes no arguments"); ray_heap_gc(); /* Same statement-boundary rule as the REPL: an explicit maintenance * call is also a chance to notice the process has gone quiet. */ diff --git a/test/rfl/system/reserved_namespace.rfl b/test/rfl/system/reserved_namespace.rfl index fd0d7238..69a45144 100644 --- a/test/rfl/system/reserved_namespace.rfl +++ b/test/rfl/system/reserved_namespace.rfl @@ -5,10 +5,10 @@ ;; `'?` regardless of arity — so we exercise by calling, which is ;; what actually matters here.) ;; .sys.gc returns 0 on success — doubles as a binding-exists probe. -;; Registered as variadic so both (.sys.gc) and (.sys.gc 0) work; -;; other .sys.* info builtins follow the same convention. +;; Registered as variadic so the zero-argument call remains valid while +;; the implementation can reject unexpected arguments explicitly. (.sys.gc) -- 0 -(.sys.gc 0) -- 0 +(.sys.gc 0) !- arity (count (.sys.info)) -- 5 (count (.sys.mem)) -- 16 (count (.sys.build)) -- 2 @@ -116,7 +116,7 @@ internals !- name (del .sys.gc) !- reserve (del .os.getenv) !- reserve ;; Built-in still resolves after the blocked shadow attempts. -(.sys.gc 0) -- 0 +(.sys.gc) -- 0 (nil? .os.getenv) -- false ;; User-level dotted writes under a non-`.` root still work. (set myns.x 1) diff --git a/test/rfl/system/system_branch_cov.rfl b/test/rfl/system/system_branch_cov.rfl index bb67d9ed..d07fe3ae 100644 --- a/test/rfl/system/system_branch_cov.rfl +++ b/test/rfl/system/system_branch_cov.rfl @@ -180,8 +180,8 @@ ;; ray_gc_fn (line 539) ;; ══════════════════════════════════════════════════════════════════════ (.sys.gc) -- 0 -(.sys.gc 0) -- 0 -(.sys.gc "x") -- 0 +(.sys.gc 0) !- arity +(.sys.gc "x") !- arity ;; ────────────── teardown ────────────── (.sys.exec "rm -rf /tmp/rfl_sys_bc_*") From 1ae24358e469c2a9626ea62618eb2f9a282fb13f Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Sun, 27 Sep 2026 23:34:22 +0300 Subject: [PATCH 38/51] fix(string): a STR view keeps only the bytes it points at (#631) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(string): a STR view keeps only the bytes it points at The descriptor views introduced for substr and `if` over STR columns pinned more than they used. An `if` whose two sides came from different pools — including two columns of one table under a where: that kept a few rows — copied BOTH whole parent pools into the result (the compacted branch inputs share the full column pool, so a 10k-row result out of 500k rows carried the two columns' 43 MB); the eager arm did the same for an unfiltered `if`. A substr with a per-row start or length retained the parent pool even when every result fitted inline and nothing pointed into it. `if` now builds a pool of exactly the chosen rows' bytes whenever the sides come from different pools, a pooled scalar is involved, or the rows keep a small share (under an eighth) of one shared pool; one shared pool whose rows keep most of it is still pointed into as before, so a derived-key expression over one column stays copy-free. A substr view whose descriptors are all inline drops its pool reference; one that points into the pool stays a view (the column is alive anyway, and a sibling `if` over two such views can then pick either side without a copy). test/rfl/strop/str_view_pool_compact.rfl pins the retained bytes through direct-bytes: an `if` under a where over two columns, over two substring views, with a pooled scalar side, the eager `if` against the two-pool sum, and an inline substr view outliving its table. * test(string): cover the eager `if` over two STR pools The case labelled eager in str_view_pool_compact went through the selected path (a per-row condition column is always a row mask there), so the eager arm's two-pool offset/copy had no regression coverage. A scalar condition (a variable) is not a row mask: the selected path declines and the eager fill runs over two different pools. New cases take it at 500k rows (parallel fill) and 1000 rows (below the parallel threshold, serial fill), for both sides, check every row after all the sources are released, and bound the retained bytes to one side's pool. On the pre-fix tree the retention checks fail. The old case is relabelled as the per-row selected path it is. Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Anton Kundenko Co-authored-by: Claude Opus 5.5 --- src/ops/internal.h | 3 + src/ops/pivot.c | 167 ++++++++++++++++------- src/ops/string.c | 25 ++++ test/rfl/strop/str_view_pool_compact.rfl | 72 ++++++++++ 4 files changed, 215 insertions(+), 52 deletions(-) create mode 100644 test/rfl/strop/str_view_pool_compact.rfl diff --git a/src/ops/internal.h b/src/ops/internal.h index 63d19d33..64316f51 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -1604,6 +1604,9 @@ ray_t* exec_k_shortest(ray_graph_t* g, ray_op_t* op, /* ── pivot_exec.c ── */ ray_t* exec_if(ray_graph_t* g, ray_op_t* op); +/* Is a descriptor view worth rebuilding over its own bytes (string.c)? */ +bool ray_str_view_should_compact(uint64_t pooled_bytes, int64_t pool_len); + /* Shared-node memo around a sub-evaluation over a swapped g->table * (exec.c): push sets the outer memo aside and arms one for the current * table and sub-root; pop tears it down and restores the outer one. */ diff --git a/src/ops/pivot.c b/src/ops/pivot.c index e43b1f6e..ca058e5c 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -510,9 +510,10 @@ static ray_t* if_scatter_str(ray_t* result, ray_t* value, int64_t* ids, * STR vector with one row per id (or as many rows as the table, indexed by * the id), or a broadcast scalar. The result takes the side's 16-byte * descriptor at ids[j] instead of appending its bytes row by row: pooled - * strings keep pointing into their pool, two different pools are laid end - * to end (the second side's offsets shift), a pooled scalar's bytes go in - * once. Rows not in either id list stay the null descriptor the caller + * strings keep pointing into their pool when the rows keep most of it; + * otherwise (two different pools, a pooled scalar, or a small share of a + * big pool) the result gets a pool of exactly its own bytes. Rows not in + * either id list stay the null descriptor the caller * zeroed. Returns NULL, the result untouched, for a side this cannot take * (a SYM branch, a length that is neither) — the caller then falls back to * the per-row scatter, which also reports the length error. */ @@ -523,6 +524,7 @@ typedef struct { bool scalar; bool full; /* vector of nrows: row ids[j] */ const ray_str_t* desc; + const char* bytes; /* its pool's bytes (NULL when inline-only) */ ray_t* pool; const char* sp; /* scalar bytes */ size_t sl; @@ -548,21 +550,49 @@ static bool if_str_side_init(ray_t* v, int64_t* ids, int64_t n, int64_t nrows, if (v->len == n) s->full = false; else if (v->len == nrows) s->full = true; else return false; - const char* bytes = NULL; - str_resolve(v, &s->desc, &bytes); + str_resolve(v, &s->desc, &s->bytes); s->pool = str_vec_pool_obj(v); if (s->pool && RAY_IS_ERR(s->pool)) return false; return true; } +/* The side's descriptor for its j-th id. */ +static inline ray_str_t if_str_side_desc(const if_str_side_t* s, int64_t j) { + return s->full ? s->desc[s->ids[j]] : s->desc[j]; +} + +/* Bytes the side's pooled descriptors point at. */ +static uint64_t if_str_side_pooled_bytes(const if_str_side_t* s) { + if (!s->v || s->scalar) return 0; + uint64_t sum = 0; + for (int64_t j = 0; j < s->n; j++) { + ray_str_t d = if_str_side_desc(s, j); + if (!ray_str_is_inline(&d)) sum += d.len; + } + return sum; +} + +/* Assign compact offsets to the side's pooled descriptors from *run. */ +static void if_str_side_offsets(const if_str_side_t* s, uint32_t* newoff, uint64_t* run) { + for (int64_t j = 0; j < s->n; j++) { + ray_str_t d = if_str_side_desc(s, j); + if (ray_str_is_inline(&d)) continue; + newoff[j] = (uint32_t)*run; + *run += d.len; + } +} + typedef struct { const int64_t* ids; const ray_str_t* src; bool scalar; bool full; ray_str_t sd; /* the scalar's descriptor */ - uint32_t shift; ray_str_t* dst; + /* compaction: pooled bytes move to dst_bytes at newoff[j] */ + const uint32_t* newoff; + const char* src_bytes; + char* dst_bytes; } if_str_scatter_ctx_t; static void if_str_scatter_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { @@ -574,17 +604,20 @@ static void if_str_scatter_fn(void* vctx, uint32_t wid, int64_t start, int64_t e } for (int64_t j = start; j < end; j++) { ray_str_t d = c->full ? c->src[c->ids[j]] : c->src[j]; - if (c->shift && !ray_str_is_inline(&d)) d.pool_off += c->shift; + if (c->newoff && !ray_str_is_inline(&d)) { + memcpy(c->dst_bytes + c->newoff[j], c->src_bytes + d.pool_off, d.len); + d.pool_off = c->newoff[j]; + } c->dst[c->ids[j]] = d; } } -static void if_str_scatter_side(const if_str_side_t* s, uint32_t shift, uint32_t scalar_off, - ray_str_t* dst) { +static void if_str_scatter_side(const if_str_side_t* s, const uint32_t* newoff, char* dst_bytes, + uint32_t scalar_off, ray_str_t* dst) { if (!s->v) return; if_str_scatter_ctx_t c = { .ids = s->ids, .src = s->desc, .scalar = s->scalar, .full = s->full, - .shift = shift, .dst = dst, + .dst = dst, .newoff = newoff, .src_bytes = s->bytes, .dst_bytes = dst_bytes, }; if (s->scalar) { memset(&c.sd, 0, sizeof(c.sd)); @@ -615,32 +648,47 @@ static ray_t* if_scatter_str_desc(ray_t* result, ray_t* pe = (e.v && !e.scalar) ? e.pool : NULL; bool t_big = t.v && t.scalar && t.sl > RAY_STR_INLINE_MAX; bool e_big = e.v && e.scalar && e.sl > RAY_STR_INLINE_MAX; - uint32_t e_shift = 0, ts_off = 0, es_off = 0; - if (!t_big && !e_big && (pt == pe || !pt || !pe)) { - ray_t* shared = pt ? pt : pe; + uint64_t t_ref = if_str_side_pooled_bytes(&t); + uint64_t e_ref = if_str_side_pooled_bytes(&e); + ray_str_t* dst = (ray_str_t*)ray_data(result); + + /* One pool (or none) behind both sides, and the rows keep most of it: + * point into it as is. */ + bool one_pool = (pt == pe) || !pt || !pe; + ray_t* shared = pt ? pt : pe; + if (!t_big && !e_big && one_pool && + (!shared || !ray_str_view_should_compact(t_ref + e_ref, shared->len))) { if (shared) { ray_retain(shared); result->str_pool = shared; } - } else { - int64_t tl = pt ? pt->len : 0; - int64_t el = (pe && pe != pt) ? pe->len : 0; - if (tl < 0 || el < 0) return NULL; - uint64_t total = (uint64_t)tl + (uint64_t)el - + (t_big ? (uint64_t)t.sl : 0) + (e_big ? (uint64_t)e.sl : 0); - if (total > UINT32_MAX) return NULL; - ray_t* np = ray_alloc(total > 0 ? (size_t)total : 1); - if (!np || RAY_IS_ERR(np)) return NULL; - np->type = RAY_U8; - np->len = (int64_t)total; - char* dst = (char*)ray_data(np); - uint32_t off = 0; - if (tl) { memcpy(dst, ray_data(pt), (size_t)tl); off += (uint32_t)tl; } - if (pe && pe != pt && el) { memcpy(dst + off, ray_data(pe), (size_t)el); e_shift = off; off += (uint32_t)el; } - if (t_big) { memcpy(dst + off, t.sp, t.sl); ts_off = off; off += (uint32_t)t.sl; } - if (e_big) { memcpy(dst + off, e.sp, e.sl); es_off = off; off += (uint32_t)e.sl; } - result->str_pool = np; + if_str_scatter_side(&t, NULL, NULL, 0, dst); + if_str_scatter_side(&e, NULL, NULL, 0, dst); + return result; } - ray_str_t* dst = (ray_str_t*)ray_data(result); - if_str_scatter_side(&t, 0, ts_off, dst); - if_str_scatter_side(&e, e_shift, es_off, dst); + + /* Otherwise a pool of exactly the bytes the result points at: the two + * sides' pooled rows one after the other, then the pooled scalars. */ + uint64_t total = t_ref + e_ref + (t_big ? (uint64_t)t.sl : 0) + (e_big ? (uint64_t)e.sl : 0); + if (total > UINT32_MAX) return NULL; + int64_t n_off = (t.v && !t.scalar ? t.n : 0) + (e.v && !e.scalar ? e.n : 0); + ray_t* off_hdr = NULL; + uint32_t* newoff = (uint32_t*)scratch_alloc(&off_hdr, (size_t)(n_off > 0 ? n_off : 1) * sizeof(uint32_t)); + if (!newoff) return NULL; + uint32_t* t_off = (t.v && !t.scalar) ? newoff : NULL; + uint32_t* e_off = (e.v && !e.scalar) ? newoff + (t.v && !t.scalar ? t.n : 0) : NULL; + uint64_t run = 0; + if (t_off) if_str_side_offsets(&t, t_off, &run); + if (e_off) if_str_side_offsets(&e, e_off, &run); + uint32_t ts_off = 0, es_off = 0; + ray_t* np = ray_alloc(total > 0 ? (size_t)total : 1); + if (!np || RAY_IS_ERR(np)) { scratch_free(off_hdr); return NULL; } + np->type = RAY_U8; + np->len = (int64_t)total; + char* dst_bytes = (char*)ray_data(np); + if (t_big) { memcpy(dst_bytes + run, t.sp, t.sl); ts_off = (uint32_t)run; run += t.sl; } + if (e_big) { memcpy(dst_bytes + run, e.sp, e.sl); es_off = (uint32_t)run; run += e.sl; } + result->str_pool = np; + if_str_scatter_side(&t, t_off, dst_bytes, ts_off, dst); + if_str_scatter_side(&e, e_off, dst_bytes, es_off, dst); + scratch_free(off_hdr); return result; } @@ -1051,20 +1099,24 @@ typedef struct { const ray_str_t* t; const ray_str_t* e; ray_str_t* dst; - uint32_t e_shift; + /* two pools: the chosen row's pooled bytes move to dst_bytes at newoff[r] */ + const uint32_t* newoff; + const char* t_bytes; + const char* e_bytes; + char* dst_bytes; } if_str_desc_ctx_t; static void if_str_desc_fn(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { (void)worker_id; const if_str_desc_ctx_t* c = (const if_str_desc_ctx_t*)vctx; for (int64_t r = start; r < end; r++) { - if (c->cond[r]) { - c->dst[r] = c->t[r]; - } else { - ray_str_t d = c->e[r]; - if (c->e_shift && !ray_str_is_inline(&d)) d.pool_off += c->e_shift; - c->dst[r] = d; + ray_str_t d = c->cond[r] ? c->t[r] : c->e[r]; + if (c->newoff && !ray_str_is_inline(&d)) { + const char* src = c->cond[r] ? c->t_bytes : c->e_bytes; + memcpy(c->dst_bytes + c->newoff[r], src + d.pool_off, d.len); + d.pool_off = c->newoff[r]; } + c->dst[r] = d; } } @@ -1137,8 +1189,8 @@ static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { /* Two STR vectors: the result is descriptors only. Each row takes * its side's 16-byte descriptor; pooled strings keep pointing into * their pool. One shared pool (or one side inline-only) is reused - * as is; two different pools are laid end to end in a new pool and - * the else side's offsets shift by the then pool's length. Nulls + * as is; two different pools give a pool of exactly the chosen + * rows' bytes. Nulls * are empty descriptors and travel unchanged. No per-row append, * no rehash; the fill runs on the worker pool. */ if (!then_scalar && !else_scalar && @@ -1153,7 +1205,8 @@ static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { str_resolve(then_v, &t_desc, &t_bytes); str_resolve(else_v, &e_desc, &e_bytes); bool ok = true; - uint32_t e_shift = 0; + ray_t* off_hdr = NULL; + uint32_t* newoff = NULL; if (then_pool == else_pool || !then_pool || !else_pool) { ray_t* out_pool = then_pool ? then_pool : else_pool; if (out_pool && !RAY_IS_ERR(out_pool)) { @@ -1163,33 +1216,43 @@ static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { } else if (RAY_IS_ERR(then_pool) || RAY_IS_ERR(else_pool)) { ok = false; } else { - int64_t tl = then_pool->len, el = else_pool->len; - if (tl < 0 || el < 0 || (uint64_t)tl + (uint64_t)el > UINT32_MAX) { + /* Two pools: a pool of exactly the chosen rows' bytes (a + * serial pass assigns the offsets, the fill copies). */ + newoff = (uint32_t*)scratch_alloc(&off_hdr, (size_t)(len > 0 ? len : 1) * sizeof(uint32_t)); + if (!newoff) { ok = false; } else { - ray_t* np = ray_alloc((size_t)(tl + el) > 0 ? (size_t)(tl + el) : 1); + uint64_t run = 0; + for (int64_t r = 0; r < len; r++) { + const ray_str_t* d = cond_p[r] ? &t_desc[r] : &e_desc[r]; + if (ray_str_is_inline(d)) continue; + newoff[r] = (uint32_t)run; + run += d->len; + } + ray_t* np = (run <= UINT32_MAX) ? ray_alloc(run > 0 ? (size_t)run : 1) : NULL; if (!np || RAY_IS_ERR(np)) { ok = false; } else { np->type = RAY_U8; - np->len = tl + el; - if (tl) memcpy(ray_data(np), t_bytes, (size_t)tl); - if (el) memcpy((char*)ray_data(np) + tl, e_bytes, (size_t)el); + np->len = (int64_t)run; result->str_pool = np; - e_shift = (uint32_t)tl; } } + if (!ok && off_hdr) { scratch_free(off_hdr); off_hdr = NULL; newoff = NULL; } } if (ok) { if_str_desc_ctx_t dctx = { .cond = cond_p, .t = t_desc, .e = e_desc, - .dst = (ray_str_t*)ray_data(result), .e_shift = e_shift, + .dst = (ray_str_t*)ray_data(result), + .newoff = newoff, .t_bytes = t_bytes, .e_bytes = e_bytes, + .dst_bytes = result->str_pool ? (char*)ray_data(result->str_pool) : NULL, }; ray_pool_t* pool = ray_pool_get(); if (ray_pool_par_dispatch_ok(pool, len, RAY_PARALLEL_THRESHOLD)) ray_pool_dispatch(pool, if_str_desc_fn, &dctx, len); else if_str_desc_fn(&dctx, 0, 0, len); + if (off_hdr) scratch_free(off_hdr); if (ray_vec_may_have_nulls(then_v) || ray_vec_may_have_nulls(else_v)) result->attrs |= RAY_ATTR_HAS_NULLS; ray_release(cond_v); ray_release(then_v); ray_release(else_v); diff --git a/src/ops/string.c b/src/ops/string.c index 52a792f6..bf01958d 100644 --- a/src/ops/string.c +++ b/src/ops/string.c @@ -1053,12 +1053,14 @@ typedef struct { substr_arg_t len; _Atomic(uint32_t) any_null; _Atomic(uint32_t) range_err; + _Atomic(uint64_t) pooled_bytes; /* bytes the pooled results point at */ } substr_view_ctx_t; static void substr_view_fn(void* vctx, uint32_t worker_id, int64_t lo, int64_t hi) { (void)worker_id; substr_view_ctx_t* c = (substr_view_ctx_t*)vctx; bool null_seen = false, range_seen = false; + uint64_t pooled = 0; for (int64_t i = lo; i < hi; i++) { ray_str_t* d = &c->dst[i]; memset(d, 0, sizeof(*d)); @@ -1084,6 +1086,7 @@ static void substr_view_fn(void* vctx, uint32_t worker_id, int64_t lo, int64_t h if ((uint64_t)s->pool_off + (uint64_t)st > UINT32_MAX) { range_seen = true; d->len = 0; continue; } memcpy(d->prefix, sp, 4); d->pool_off = s->pool_off + (uint32_t)st; + pooled += (uint64_t)ln; /* hash32 stays 0: a consumer that needs it computes it once * (ray_str_t_hash32); hashing every substring here paid a pass * over the bytes that most consumers never used. */ @@ -1091,6 +1094,17 @@ static void substr_view_fn(void* vctx, uint32_t worker_id, int64_t lo, int64_t h } if (null_seen) atomic_store_explicit(&c->any_null, 1, memory_order_relaxed); if (range_seen) atomic_store_explicit(&c->range_err, 1, memory_order_relaxed); + if (pooled) atomic_fetch_add_explicit(&c->pooled_bytes, pooled, memory_order_relaxed); +} + +/* A view whose bytes are a small share (under an eighth) of the pool it + * points into is worth rebuilding over its own bytes: the copy costs the + * few bytes it keeps, the pool it would otherwise pin costs the rest. A + * larger share stays a view — the parent pool is usually alive anyway (a + * column, a sibling intermediate), and copying most of it would only add + * a second copy for the view's lifetime. */ +bool ray_str_view_should_compact(uint64_t pooled_bytes, int64_t pool_len) { + return pool_len > 0 && pooled_bytes * 8 < (uint64_t)pool_len; } static ray_t* substr_str_view(ray_t* input, ray_t* start_v, ray_t* len_v) { @@ -1114,6 +1128,7 @@ static ray_t* substr_str_view(ray_t* input, ray_t* start_v, ray_t* len_v) { ctx.dst = (ray_str_t*)ray_data(result); atomic_store_explicit(&ctx.any_null, 0, memory_order_relaxed); atomic_store_explicit(&ctx.range_err, 0, memory_order_relaxed); + atomic_store_explicit(&ctx.pooled_bytes, 0, memory_order_relaxed); ray_pool_t* pool = ray_pool_get(); if (ray_pool_par_dispatch_ok(pool, nrows, RAY_PARALLEL_THRESHOLD)) @@ -1126,6 +1141,16 @@ static ray_t* substr_str_view(ray_t* input, ray_t* start_v, ray_t* len_v) { } if (atomic_load_explicit(&ctx.any_null, memory_order_relaxed)) result->attrs |= RAY_ATTR_HAS_NULLS; + /* A result with no pooled descriptor (every substring fits inline) + * has nothing in the parent pool to keep alive. A view that does + * point into it stays a view: the column is alive anyway, and a + * sibling `if` over two such views can pick either side without + * copying (a compacted view would give it two different pools). */ + if (result->str_pool && + atomic_load_explicit(&ctx.pooled_bytes, memory_order_relaxed) == 0) { + ray_release(result->str_pool); + result->str_pool = NULL; + } return result; } diff --git a/test/rfl/strop/str_view_pool_compact.rfl b/test/rfl/strop/str_view_pool_compact.rfl new file mode 100644 index 00000000..5ed5eda9 --- /dev/null +++ b/test/rfl/strop/str_view_pool_compact.rfl @@ -0,0 +1,72 @@ +;; A STR result that is a view into another vector's pool keeps only the +;; bytes it points at: an `if` that kept a few rows of two big columns, or a +;; substring of a few bytes, must not hold (or copy) the whole parent pools. +;; direct-bytes counts the big (32 MB and up) pool allocations, so a result +;; that pins or copies a whole parent pool shows up there and one that keeps +;; its few bytes does not. +(set N 500000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) i))) +(set S2 (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) i))) +(set st (+ 1 (% i 50))) +(set c (< (% i 7) 3)) +(set T (table [S S2 st c] (list S S2 st c))) +(set direct (fn [] (do (.sys.gc) (at (.sys.mem) 'direct-bytes)))) +(set base (direct)) +(>= base 33554432) -- true +;; if over two pools under a where: keeping one row in fifty +(set R (select {from: T x: (if c S S2) where: (== st 1)})) +(count R) -- 10000 +(< (- (direct) base) 1000000) -- true +(all (== (at R 'x) (at (select {from: T x: (if c S S2) where: (== st 1)}) 'x))) -- true +(set R 0) +;; the same over two substring views +(set R (select {from: T x: (if c (substr S 2 30) (substr S2 2 30)) where: (== st 1)})) +(< (- (direct) base) 1000000) -- true +(at (at R 'x) 0) -- "ttps://host0.example.org/path/" +(set R 0) +;; a pooled scalar side under a where +(set R (select {from: T x: (if c S "a-rather-long-literal-string") where: (== st 1)})) +(< (- (direct) base) 1000000) -- true +(set R 0) +;; a per-row condition over two whole columns (the selected path — STR is +;; never routed to the eager arm when the condition is a column) copies only +;; the chosen rows' bytes: less than the two parent pools together +(set R (select {from: T x: (if c S S2)})) +(set grow (- (direct) base)) +(< grow 40000000) -- true +(all (map (fn [j] (== (at (at R 'x) j) (if (at c j) (at S j) (at S2 j)))) (til 20000))) -- true +(set R 0) +;; the eager arm: a scalar condition (a variable, not a column) is not a +;; row mask, so the selected path declines and the eager fill runs over two +;; different pools. It builds a pool of the chosen side's bytes only, and +;; the result keeps its values after every source is released — at 500k rows +;; (the fill runs on the worker pool) and at 1000 rows (below the parallel +;; threshold, serial fill) +(set pick true) +(set RE (select {from: T x: (if pick S S2)})) +(set grow (- (direct) base)) +(< grow 40000000) -- true +(set pick false) +(set RF (select {from: T x: (if pick S S2)})) +(set Ts (take T 1000)) +(set pick true) +(set RS (select {from: Ts x: (if pick S S2)})) +(set pick false) +(set RS2 (select {from: Ts x: (if pick S S2)})) +;; a substring of three bytes is inline everywhere: no pool at all, so the +;; result outlives the table without holding its bytes +(set R (select {from: T x: (substr S st 3)})) +(set T 0) (set S 0) (set S2 0) (set i 0) (set st 0) (set c 0) (set Ts 0) +;; the eager results still read their strings, every row checked +(set ix (til 500000)) +(all (== (at RE 'x) (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) ix)))) -- true +(all (== (at RF 'x) (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) ix)))) -- true +(all (== (at RS 'x) (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) (til 1000))))) -- true +(all (== (at RS2 'x) (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) (til 1000))))) -- true +;; each eager result holds one side's bytes, not both parents' pools +(set RE 0) (set RS 0) (set RS2 0) +(< (direct) 40000000) -- true +(set RF 0) +(< (direct) 1000000) -- true +(count R) -- 500000 From a73fab964165507c58df80b9393b657a75a5be45 Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Mon, 28 Sep 2026 11:53:00 +0200 Subject: [PATCH 39/51] fix(log): reject arguments to purge (#638) --- docs/docs/namespaces/log.md | 2 +- src/ops/journal.c | 3 ++- test/rfl/journal/ops_journal_purge.rfl | 1 + 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/docs/namespaces/log.md b/docs/docs/namespaces/log.md index fbde54b6..97240385 100644 --- a/docs/docs/namespaces/log.md +++ b/docs/docs/namespaces/log.md @@ -19,7 +19,7 @@ The journal is intentionally minimal: there's no per-entry timestamp or transact | [`.log.replay`](#log-replay) | unary | restricted | Replay a journal file; return entry count. | | [`.log.validate`](#log-validate) | unary | — | Scan a journal file; return `(chunks valid_bytes)`. | | [`.log.close`](#log-close) | variadic | restricted | Flush and close the active journal. | -| [`.log.purge`](#log-purge) | variadic | restricted | Close the active journal and delete all its files. | +| [`.log.purge`](#log-purge) | nullary | restricted | Close the active journal and delete all its files. | ## `.log.open` { #log-open } diff --git a/src/ops/journal.c b/src/ops/journal.c index fcf5d20d..85f40cc2 100644 --- a/src/ops/journal.c +++ b/src/ops/journal.c @@ -227,6 +227,7 @@ ray_t* ray_log_close_fn(ray_t** args, int64_t n) { * argument; acts on the journal .log.open/.write/.close target. Errors * with `domain` when no journal base is known (none ever opened). */ ray_t* ray_log_purge_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".log.purge takes no arguments"); return err_to_ray(ray_journal_purge(), "io"); } diff --git a/test/rfl/journal/ops_journal_purge.rfl b/test/rfl/journal/ops_journal_purge.rfl index f21da713..c570073e 100644 --- a/test/rfl/journal/ops_journal_purge.rfl +++ b/test/rfl/journal/ops_journal_purge.rfl @@ -41,6 +41,7 @@ ;; State was reset: a second purge has no base to act on -> domain error. (.log.purge) !- domain +(.log.purge 1) !- arity ;; ════════════════════════════════════════════════════════════════════════ ;; 2. The issue #279 workflow: open, write, close, then purge WITHOUT From 12502a6eb2199c99b338a1e05ca4477dcc6f046a Mon Sep 17 00:00:00 2001 From: Anton Kundenko Date: Mon, 28 Sep 2026 18:07:22 +0200 Subject: [PATCH 40/51] perf(vec): avoid rewriting copied text nulls during concat (#641) Co-authored-by: singaraiona <3473381+singaraiona@users.noreply.github.com> --- bench/concat_text_nulls.c | 45 +++++++++++++++++++ src/vec/vec.c | 34 ++++++--------- test/test_domain.c | 42 ++++++++++++++++++ test/test_vec.c | 91 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 192 insertions(+), 20 deletions(-) create mode 100644 bench/concat_text_nulls.c diff --git a/bench/concat_text_nulls.c b/bench/concat_text_nulls.c new file mode 100644 index 00000000..6ce308c9 --- /dev/null +++ b/bench/concat_text_nulls.c @@ -0,0 +1,45 @@ +/* Focused concat benchmark: build with release library, run before/after. */ +#define _POSIX_C_SOURCE 200809L +#include +#include "mem/heap.h" +#include "vec/vec.h" +#include +#include + +static double now(void) { + struct timespec t; + clock_gettime(CLOCK_MONOTONIC, &t); + return t.tv_sec + t.tv_nsec * 1e-9; +} + +int main(void) { + ray_heap_init(); + const int64_t n = 1000000; + const int reps = 40; + for (int kind = 0; kind < 2; kind++) { + for (int mode = 0; mode < 3; mode++) { + ray_t* a = ray_vec_new(kind ? RAY_SYM : RAY_STR, n); + if (!a || RAY_IS_ERR(a)) return 1; + if (kind) { + a->len = n; + for (int64_t i = 0; i < n; i++) ((int64_t*)ray_data(a))[i] = 1; + } else { + for (int64_t i = 0; i < n; i++) a = ray_str_vec_append(a, "abc", 3); + } + if (mode == 1) ray_vec_set_null(a, 0, true); + if (mode == 2) ray_vec_set_null(a, n - 1, true); + double start = now(); + for (int r = 0; r < reps; r++) { + ray_t* out = ray_vec_concat(a, a); + if (!out || RAY_IS_ERR(out) || out->len != 2 * n) return 2; + ray_release(out); + } + printf("%s %s rows=%lld reps=%d ms_per_concat=%.6f\n", + kind ? "SYM64" : "STR-inline", mode == 0 ? "no-null" : mode == 1 ? "first-null" : "last-null", + (long long)n, reps, (now() - start) * 1000 / reps); + ray_release(a); + } + } + ray_heap_destroy(); + return 0; +} diff --git a/src/vec/vec.c b/src/vec/vec.c index f5be8d98..16722e5c 100644 --- a/src/vec/vec.c +++ b/src/vec/vec.c @@ -515,22 +515,11 @@ ray_t* ray_vec_concat(ray_t* a, ray_t* b) { } } - /* Propagate null bitmaps from a and b. - * Slices don't carry RAY_ATTR_HAS_NULLS — check RAY_ATTR_SLICE too. */ - if (ray_vec_may_have_nulls(a) || - ray_vec_may_have_nulls(b)) { - for (int64_t i = 0; i < a->len; i++) { - if (ray_vec_is_null((ray_t*)a, i)) { - ray_err_t err = ray_vec_set_null_checked(result, i, true); - if (err != RAY_OK) { ray_release(result); return ray_error(ray_err_code_str(err), NULL); } - } - } - for (int64_t i = 0; i < b->len; i++) { - if (ray_vec_is_null((ray_t*)b, i)) { - ray_err_t err = ray_vec_set_null_checked(result, a->len + i, true); - if (err != RAY_OK) { ray_release(result); return ray_error(ray_err_code_str(err), NULL); } - } - } + /* Canonical empty payloads were copied above; only the null hint + * remains. Scan payloads (including slices), not input hint bits. */ + if (ray_vec_text_has_nulls(a) || ray_vec_text_has_nulls(b)) { + vec_drop_index_inplace(result); + result->attrs |= RAY_ATTR_HAS_NULLS; } return result; @@ -650,10 +639,15 @@ ray_t* ray_vec_concat(ray_t* a, ray_t* b) { (size_t)b->len * esz); } - /* Propagate null bitmaps from a and b. - * Slices don't carry RAY_ATTR_HAS_NULLS — check RAY_ATTR_SLICE too. */ - if (ray_vec_may_have_nulls(a) || - ray_vec_may_have_nulls(b)) { + /* SYM's zero id survives copying, widening and domain translation. + * Numeric sentinels retain their existing propagation path below. */ + if (result->type == RAY_SYM) { + if (ray_vec_text_has_nulls(a) || ray_vec_text_has_nulls(b)) { + vec_drop_index_inplace(result); + result->attrs |= RAY_ATTR_HAS_NULLS; + } + } else if (ray_vec_may_have_nulls(a) || + ray_vec_may_have_nulls(b)) { for (int64_t i = 0; i < a->len; i++) { if (ray_vec_is_null((ray_t*)a, i)) { ray_err_t err = ray_vec_set_null_checked(result, i, true); diff --git a/test/test_domain.c b/test/test_domain.c index ce7186ac..40dfca97 100644 --- a/test/test_domain.c +++ b/test/test_domain.c @@ -1804,6 +1804,47 @@ static test_result_t test_domain_runtime_lut(void) { PASS(); } +static test_result_t test_domain_concat_text_nulls(void) { + ray_sym_domain_t* dom = NULL; + int64_t pos_a = -1, pos_b = -1; + TEST_ASSERT_TRUE(build_divergent_qsym_fixture(&dom, &pos_a, &pos_b)); + /* Fixture includes the builtin vocabulary, so its positions need W16. */ + const uint8_t widths[] = {RAY_SYM_W16, RAY_SYM_W32, RAY_SYM_W64}; + for (int w = 0; w < 3; w++) { + ray_t* file = ray_sym_vec_new(widths[w], 4); + ray_sym_domain_release(file->sym_domain); + ray_sym_domain_retain(dom); + file->sym_domain = dom; + file->len = 4; + const int64_t vals[] = {pos_b, pos_a, 0, pos_b}; + for (int i = 0; i < 4; i++) ray_write_sym(ray_data(file), i, vals[i], RAY_SYM, file->attrs); + int64_t ids[] = {0, ray_sym_intern("dq_b", 4)}; + ray_t* runtime = ray_vec_from_raw(RAY_SYM, ids, 2); + ray_t* slice = ray_vec_slice(file, 1, 2); + for (int side = 0; side < 2; side++) { + ray_t* out = ray_vec_concat(side ? runtime : slice, side ? slice : runtime); + TEST_ASSERT_NOT_NULL(out); + TEST_ASSERT_FALSE(RAY_IS_ERR(out)); + TEST_ASSERT_EQ_PTR(ray_sym_vec_domain(out), ray_sym_runtime_domain()); + TEST_ASSERT_EQ_I(out->attrs & RAY_SYM_W_MASK, RAY_SYM_W64); + TEST_ASSERT_TRUE(out->attrs & RAY_ATTR_HAS_NULLS); + TEST_ASSERT_FALSE(out->attrs & (RAY_ATTR_SLICE | RAY_ATTR_HAS_INDEX)); + int64_t expected[] = {ray_sym_intern("dq_a", 4), 0, 0, ids[1]}; + for (int i = 0; i < 4; i++) { + int64_t value = expected[(i + (side ? 2 : 0)) % 4]; + TEST_ASSERT_EQ_I(((int64_t*)ray_data(out))[i], value); + TEST_ASSERT_EQ_I(ray_vec_is_null(out, i), value == 0); + } + ray_release(out); + } + ray_release(slice); ray_release(file); ray_release(runtime); + } + ray_sym_domain_release(dom); + unlink(TMP_DOM_QSYM_PATH); + unlink(TMP_DOM_QSYM_PATH ".lk"); + PASS(); +} + #define TMP_DOM_BADSYM_PATH "/tmp/rayforce_test_domain_badsym" /* Position-0 reservation: ray_sym_save-produced files carry "" at @@ -1986,6 +2027,7 @@ const test_entry_t domain_entries[] = { { "domain/parted_flatten_adopts", test_domain_parted_flatten_adopts, domain_rt_setup, domain_rt_teardown }, { "domain/str_eager_lockfree", test_domain_str_eager_lockfree, domain_setup, domain_teardown }, { "domain/raw_pin", test_domain_raw_pin, domain_setup, domain_teardown }, + { "domain/concat_text_nulls", test_domain_concat_text_nulls, domain_rt_setup, domain_rt_teardown }, { "domain/runtime_lut", test_domain_runtime_lut, domain_rt_setup, domain_rt_teardown }, { "domain/open_position0_validation", test_domain_open_position0_validation, domain_setup, domain_teardown }, { "domain/dict_upsert_file_keys", test_domain_dict_upsert_file_keys, domain_rt_setup, domain_rt_teardown }, diff --git a/test/test_vec.c b/test/test_vec.c index 458adf7c..986e537e 100644 --- a/test/test_vec.c +++ b/test/test_vec.c @@ -25,6 +25,7 @@ #include #include "mem/heap.h" #include "vec/vec.h" +#include "vec/str.h" #include "vec/embedding.h" #include "table/sym.h" #include "core/platform.h" @@ -1782,6 +1783,94 @@ static test_result_t test_vec_concat_str_null(void) { PASS(); } +/* Canonical text payloads must propagate even without HAS_NULLS. Exercise + * both operands, chunk/tail boundaries, all SYM width pairs and slices whose + * parent has nulls outside the selected range. */ +static test_result_t test_vec_concat_text_null_scan(void) { + const uint8_t widths[] = {RAY_SYM_W8, RAY_SYM_W16, RAY_SYM_W32, RAY_SYM_W64}; + const int64_t positions[] = {-1, 0, 255, 256, 258}; + for (int kind = 0; kind < 2; kind++) { + for (int wa = 0; wa < (kind ? 4 : 1); wa++) { + for (int wb = 0; wb < (kind ? 4 : 1); wb++) { + for (int side = 0; side < 2; side++) { + for (int p = 0; p < 5; p++) { + ray_t* v[2]; + for (int s = 0; s < 2; s++) { + v[s] = kind ? ray_sym_vec_new(widths[s ? wb : wa], 261) + : ray_vec_new(RAY_STR, 261); + TEST_ASSERT_NOT_NULL(v[s]); + TEST_ASSERT_FALSE(RAY_IS_ERR(v[s])); + if (kind) { + v[s]->len = 261; + for (int64_t i = 0; i < 261; i++) + ray_write_sym(ray_data(v[s]), i, 1, RAY_SYM, v[s]->attrs); + } else { + for (int64_t i = 0; i < 261; i++) + v[s] = ray_str_vec_append(v[s], "a pooled string longer than inline", 33); + } + ray_vec_set_null(v[s], 0, true); + ray_vec_set_null(v[s], 260, true); + if (s == side && positions[p] >= 0) + ray_vec_set_null(v[s], positions[p] + 1, true); + v[s]->attrs &= (uint8_t)~RAY_ATTR_HAS_NULLS; + } + ray_t* a = ray_vec_slice(v[0], 1, 259); + ray_t* b = ray_vec_slice(v[1], 1, 259); + ray_t* c = ray_vec_concat(a, b); + TEST_ASSERT_NOT_NULL(c); + TEST_ASSERT_FALSE(RAY_IS_ERR(c)); + TEST_ASSERT_EQ_I(c->len, 518); + TEST_ASSERT_EQ_I(!!(c->attrs & RAY_ATTR_HAS_NULLS), positions[p] >= 0); + TEST_ASSERT_FALSE(c->attrs & (RAY_ATTR_SLICE | RAY_ATTR_HAS_INDEX | RAY_ATTR_SORTED)); + /* Result owns its pool/domain independently of inputs. */ + ray_release(a); ray_release(b); + ray_release(v[0]); ray_release(v[1]); + for (int64_t i = 0; i < c->len; i++) { + bool is_null = positions[p] >= 0 && i == side * 259 + positions[p]; + TEST_ASSERT_EQ_I(ray_vec_is_null(c, i), is_null); + if (kind) { + TEST_ASSERT_EQ_I(ray_read_sym(ray_data(c), i, RAY_SYM, c->attrs), !is_null); + } else if (!is_null) { + size_t len; + const char* str = ray_str_vec_get(c, i, &len); + TEST_ASSERT_EQ_U(len, 33); + TEST_ASSERT_MEM_EQ(33, str, "a pooled string longer than inline"); + } else { + ray_str_t zero = {0}; + TEST_ASSERT_MEM_EQ(sizeof(zero), &((ray_str_t*)ray_data(c))[i], &zero); + } + } + ray_release(c); + } + } + } + } + } + PASS(); +} + +static test_result_t test_vec_concat_text_empty(void) { + for (int kind = 0; kind < 2; kind++) { + ray_t* empty = ray_vec_new(kind ? RAY_SYM : RAY_STR, 0); + ray_t* nulls = ray_vec_new(kind ? RAY_SYM : RAY_STR, 2); + nulls->len = 2; + ray_vec_set_null(nulls, 0, true); + ray_vec_set_null(nulls, 1, true); + nulls->attrs &= (uint8_t)~RAY_ATTR_HAS_NULLS; + for (int side = 0; side < 3; side++) { + ray_t* c = ray_vec_concat(side == 0 ? nulls : empty, side == 1 ? nulls : empty); + TEST_ASSERT_NOT_NULL(c); + TEST_ASSERT_FALSE(RAY_IS_ERR(c)); + TEST_ASSERT_EQ_I(c->len, side == 2 ? 0 : 2); + TEST_ASSERT_EQ_I(!!(c->attrs & RAY_ATTR_HAS_NULLS), side != 2); + for (int64_t i = 0; i < c->len; i++) TEST_ASSERT_TRUE(ray_vec_is_null(c, i)); + ray_release(c); + } + ray_release(nulls); ray_release(empty); + } + PASS(); +} + /* ---- str_vec_append: pool growth across many large strings ------------- */ static test_result_t test_str_vec_append_pool_grow(void) { @@ -2372,6 +2461,8 @@ const test_entry_t vec_entries[] = { { "vec/concat_empty", test_vec_concat_empty, vec_setup, vec_teardown }, { "vec/concat_str", test_vec_concat_str, vec_setup, vec_teardown }, { "vec/concat_str_null", test_vec_concat_str_null, vec_setup, vec_teardown }, + { "vec/concat_text_null_scan", test_vec_concat_text_null_scan, vec_setup, vec_teardown }, + { "vec/concat_text_empty", test_vec_concat_text_empty, vec_setup, vec_teardown }, { "vec/str_append_pool_grow", test_str_vec_append_pool_grow, vec_setup, vec_teardown }, { "vec/str_set_paths", test_str_vec_set_paths, vec_setup, vec_teardown }, { "vec/str_guards", test_str_vec_guards, vec_setup, vec_teardown }, From 338b9d67406ed098ea80560b13c267e79fcf016e Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Tue, 29 Sep 2026 08:30:31 +0200 Subject: [PATCH 41/51] fix(hnsw): reject invalid persisted graph ids (#642) --- src/store/hnsw.c | 56 +++++++++++++++++++++++++++++++++++++++++++ test/test_embedding.c | 29 +++++++++++++++++++++- 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/src/store/hnsw.c b/src/store/hnsw.c index a350c433..05ed802d 100644 --- a/src/store/hnsw.c +++ b/src/store/hnsw.c @@ -908,6 +908,57 @@ bool ray_hnsw_vec_size_valid(int64_t n_nodes, int32_t dim) { return (uint64_t)n_nodes <= SIZE_MAX / sizeof(float) / (uint64_t)dim; } +/* Validate the graph topology before making the loaded index available to + * search. All ids in these files are untrusted: a bad neighbor used to + * reach hnsw_greedy_closest() or hnsw_search_layer() becomes an unchecked + * offset into idx->vectors. The node_ids mapping is equally important — a + * duplicate, missing, or wrong-level entry can make a valid-looking layer + * resolve the wrong neighbor block. + */ +static bool hnsw_persisted_layers_valid(const ray_hnsw_t* idx) { + if (!idx || !idx->node_level || idx->n_nodes <= 0 || idx->n_layers <= 0) + return false; + + for (int64_t id = 0; id < idx->n_nodes; id++) { + if (idx->node_level[id] < 0 || idx->node_level[id] >= idx->n_layers) + return false; + } + + uint8_t* seen = (uint8_t*)ray_sys_alloc((size_t)idx->n_nodes); + if (!seen) return false; + + bool valid = true; + for (int32_t l = 0; l < idx->n_layers && valid; l++) { + const ray_hnsw_layer_t* layer = &idx->layers[l]; + int64_t expected = 0; + for (int64_t id = 0; id < idx->n_nodes; id++) + if (idx->node_level[id] >= l) expected++; + if (layer->n_nodes != expected || !layer->node_ids || !layer->neighbors) { + valid = false; + break; + } + + memset(seen, 0, (size_t)idx->n_nodes); + for (int64_t i = 0; i < layer->n_nodes && valid; i++) { + int64_t id = layer->node_ids[i]; + if (id < 0 || id >= idx->n_nodes || idx->node_level[id] < l || seen[id]) { + valid = false; + break; + } + seen[id] = 1; + } + + size_t nb_count = (size_t)layer->n_nodes * (size_t)layer->M_max; + for (size_t i = 0; i < nb_count && valid; i++) { + int64_t id = layer->neighbors[i]; + if (id != -1 && (id < 0 || id >= idx->n_nodes)) valid = false; + } + } + + ray_sys_free(seen); + return valid; +} + static ray_hnsw_t* hnsw_load_impl(const char* dir, bool use_mmap) { if (!dir) return NULL; (void)use_mmap; /* mmap optimization deferred — both paths read into memory */ @@ -1001,6 +1052,11 @@ static ray_hnsw_t* hnsw_load_impl(const char* dir, bool use_mmap) { fclose(f); } + if (!hnsw_persisted_layers_valid(idx)) { + ray_hnsw_free(idx); + return NULL; + } + /* Read vectors */ snprintf(path, sizeof(path), "%s/hnsw_vectors.bin", dir); f = fopen(path, "rb"); diff --git a/test/test_embedding.c b/test/test_embedding.c index 89037ec4..091e2c52 100644 --- a/test/test_embedding.c +++ b/test/test_embedding.c @@ -36,6 +36,7 @@ #include "lang/internal.h" #include "lang/format.h" #include "store/hnsw.h" +#include "store/fileio.h" #include #include #include @@ -1005,6 +1006,32 @@ static test_result_t test_hnsw_vec_size_valid_guard(void) { PASS(); } +/* A persisted neighbor id is an array index during greedy descent. A + * malformed file must be rejected by both load entry points before a query + * can turn it into an out-of-bounds read / process crash. */ +static test_result_t test_hnsw_load_rejects_bad_neighbor(void) { + const char* dir = "/tmp/ray_hnsw_bad_neighbor"; + float vecs[2 * 2] = { 1.0f, 0.0f, 0.0f, 1.0f }; + ray_hnsw_t* idx = ray_hnsw_build(vecs, 2, 2, RAY_HNSW_L2, 4, 50); + TEST_ASSERT_NOT_NULL(idx); + TEST_ASSERT_EQ_I(ray_hnsw_save(idx, dir), RAY_OK); + ray_hnsw_free(idx); + + char path[256]; + snprintf(path, sizeof(path), "%s/hnsw_layer_0.bin", dir); + FILE* f = fopen(path, "r+b"); + TEST_ASSERT_NOT_NULL(f); + /* Layer metadata is two int64 values; overwrite the first neighbor. */ + TEST_ASSERT_EQ_I(fseek(f, (long)(2 * sizeof(int64_t)), SEEK_SET), 0); + int64_t bad_id = 1000000000; + TEST_ASSERT_EQ_U(fwrite(&bad_id, sizeof(bad_id), 1, f), 1); + TEST_ASSERT_EQ_I(fclose(f), 0); + + TEST_ASSERT_NULL(ray_hnsw_load(dir)); + TEST_ASSERT_NULL(ray_hnsw_mmap(dir)); + PASS(); +} + /* Trigger the maxheap_sift_down / results-replacement path in hnsw_search_layer. * * The replacement branch (lines 342-344) fires when: @@ -2002,6 +2029,7 @@ const test_entry_t embedding_entries[] = { { "embedding/hnsw_mmap_load", test_hnsw_mmap_load, emb_setup, emb_teardown }, { "embedding/hnsw_build_overflow_rejected", test_hnsw_build_overflow_rejected, emb_setup, emb_teardown }, { "embedding/hnsw_vec_size_valid_guard", test_hnsw_vec_size_valid_guard, emb_setup, emb_teardown }, + { "embedding/hnsw_load_rejects_bad_neighbor", test_hnsw_load_rejects_bad_neighbor, emb_setup, emb_teardown }, { "embedding/hnsw_search_sift_down", test_hnsw_search_sift_down, emb_setup, emb_teardown }, /* rerank coverage (S7) */ @@ -2051,4 +2079,3 @@ const test_entry_t embedding_entries[] = { { NULL, NULL, NULL, NULL }, }; - From 69b45584f5a79965f770f27f12d81debba5056ec Mon Sep 17 00:00:00 2001 From: Evgen Belozerov Date: Tue, 29 Sep 2026 10:08:29 +0200 Subject: [PATCH 42/51] fix(log): reject arguments to roll (#639) --- docs/docs/namespaces/log.md | 2 +- src/ops/journal.c | 3 ++- test/rfl/journal/ops_journal_wrappers.rfl | 1 + 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/docs/namespaces/log.md b/docs/docs/namespaces/log.md index 97240385..7a366e97 100644 --- a/docs/docs/namespaces/log.md +++ b/docs/docs/namespaces/log.md @@ -15,7 +15,7 @@ The journal is intentionally minimal: there's no per-entry timestamp or transact | [`.log.write`](#log-write) | unary | — | Append a serialised expression to the open journal. | | [`.log.sync`](#log-sync) | variadic | — | `fsync` the journal. | | [`.log.snapshot`](#log-snapshot) | variadic | restricted | Write a snapshot of current state and roll to a fresh segment. | -| [`.log.roll`](#log-roll) | variadic | restricted | Close the active segment and start a new one. | +| [`.log.roll`](#log-roll) | nullary | restricted | Close the active segment and start a new one. | | [`.log.replay`](#log-replay) | unary | restricted | Replay a journal file; return entry count. | | [`.log.validate`](#log-validate) | unary | — | Scan a journal file; return `(chunks valid_bytes)`. | | [`.log.close`](#log-close) | variadic | restricted | Flush and close the active journal. | diff --git a/src/ops/journal.c b/src/ops/journal.c index 85f40cc2..2495014c 100644 --- a/src/ops/journal.c +++ b/src/ops/journal.c @@ -198,7 +198,8 @@ ray_t* ray_log_validate_fn(ray_t* path) { } ray_t* ray_log_roll_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".log.roll takes no arguments"); if (!ray_journal_is_open()) return ray_error("domain", ".log.roll: no journal open"); return err_to_ray(ray_journal_roll(), "io"); diff --git a/test/rfl/journal/ops_journal_wrappers.rfl b/test/rfl/journal/ops_journal_wrappers.rfl index 9826f149..9592562f 100644 --- a/test/rfl/journal/ops_journal_wrappers.rfl +++ b/test/rfl/journal/ops_journal_wrappers.rfl @@ -54,6 +54,7 @@ ;; closed → err_to_ray RAY_OK arm → null. ;; ════════════════════════════════════════════════════════════════════════ (.log.roll) !- domain +(.log.roll 1) !- arity (.log.snapshot) !- domain (nil? (.log.sync)) -- true (nil? (.log.close)) -- true From 85d80eef7d0e1a3450137cb44b5b682cf216c8c8 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 11:09:34 +0300 Subject: [PATCH 43/51] perf(expr): 32 registers per fused expression; nulls produced mid-program, narrow scratch (#636) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(expr): 32 registers / 96 instructions per fused expression A filter of four conditions, one of them a `within`, needs more than 16 expression registers; the compiler then bailed and the whole predicate was evaluated column at a time over every row, without the chunk-zone morsel decisions. A grouping under `(and (== CounterID 62) (within EventDate [..]) (== IsRefresh 0) (== DontCountHits 0))` on a 100M-row table took 71 ms warm / 540 ms cold, against 24 ms for the same filter with three conditions. The limits are now 32 registers and 96 instructions; the per-task scratch is sized to the registers the expression uses instead of the maximum. The same query: 24 ms warm, 130 ms cold, same answer. Test: expr/wide_predicate (four and six conditions with arithmetic and `within`, a grouping under the four-condition filter, a wide computed column — each against the conditions applied one at a time; these shapes bailed on the register limit before). Co-Authored-By: Claude Opus 5.5 * fix(expr): nulls produced mid-program survive; narrow scratch widened Two wrong answers in the fused expression compiler, both reachable with more registers (the wider programs that now compile are exactly the shapes that chain a null-producing op into further arithmetic): - A null produced by an instruction — x/0, overflow to Inf, sqrt(<0), |INT64_MIN| — was only flagged when that instruction was the LAST one. `(abs (/ h b))`, `(as 'I64 (/ h b))`, `(+ (div h b) 1)` and any wider chain came back with the sentinel lanes but without HAS_NULLS, so sums and averages were poisoned (0Nf) or, after a non-null-aware i64 kernel, garbage. Registers now carry the origin of their nullability: a column-derived flag keeps the conservative HAS_NULLS attr; an op-generated one makes every downstream kernel null-aware and the output is scanned for sentinels, whatever position the producer had. A pure-finite result still leaves HAS_NULLS unset. - An I32/I16 narrowing cast or a comparison result used as an operand of an i64 op, a comparison or AND/OR was read as 8-byte lanes: `(- a (as 'I32 -1))` gave [15 -4294967275 30], `(+ (> a 15) 1)` gave [65793 3 3]. Such sources are widened to the lane type the kernel reads (the widening casts map the narrow null sentinels); a promotion that cannot be placed bails to the planner instead of running the mismatched kernel. Co-Authored-By: Claude Fable 5.1 --------- Co-authored-by: Claude Opus 5.5 Co-authored-by: Anton Kundenko --- src/ops/expr.c | 215 ++++++++++++++--------------- src/ops/internal.h | 6 +- test/rfl/expr/narrow_binary.rfl | 17 +++ test/rfl/expr/null_propagation.rfl | 34 +++++ test/rfl/expr/wide_predicate.rfl | 33 +++++ test/test_expr_null.c | 22 +-- 6 files changed, 206 insertions(+), 121 deletions(-) create mode 100644 test/rfl/expr/wide_predicate.rfl diff --git a/src/ops/expr.c b/src/ops/expr.c index 81a76cbc..a3519208 100644 --- a/src/ops/expr.c +++ b/src/ops/expr.c @@ -584,6 +584,7 @@ static uint8_t expr_ensure_type(ray_expr_t* out, uint8_t src, int8_t target) { out->regs[r].kind = REG_SCRATCH; out->regs[r].type = target; out->regs[r].nullable = out->regs[src].nullable; + out->regs[r].null_src = out->regs[src].null_src; out->n_regs++; out->n_scratch++; out->ins[out->n_ins++] = (expr_ins_t){ @@ -593,6 +594,34 @@ static uint8_t expr_ensure_type(ray_expr_t* out, uint8_t src, int8_t target) { return r; } +/* Can this instruction write a null sentinel into its destination even + * when every source lane is a real value? F64: any op that ray_f64_fin / + * a zero-divisor test canonicalizes to NULL_F64 (overflow, x/0, sqrt(<0), + * log(<=0), ...). I64: zero-divisor IDIV/MOD and the INT64_MIN overflow of + * NEG/ABS. Such a destination is nullable for every downstream instruction + * (so CAST F64->I64, I64 arithmetic and MIN2/MAX2 pick their null-aware + * kernels) and, when nothing upstream is a nullable column, the output flag + * comes from a precise sentinel scan rather than the conservative attr. */ +static bool expr_op_generates_null(uint16_t op, int8_t ot, int8_t t1, bool binary) { + if (ot == RAY_F64) { + switch (op) { + case OP_ADD: case OP_SUB: case OP_MUL: + case OP_DIV: case OP_IDIV: case OP_MOD: case OP_POW: + case OP_SQRT: case OP_LOG: case OP_EXP: + case OP_SIN: case OP_ASIN: case OP_COS: case OP_ACOS: + case OP_TAN: case OP_ATAN: case OP_RECIPROCAL: + return true; + default: + return false; /* NEG/ABS/CEIL/FLOOR/ROUND/CAST/MIN2/MAX2: finite -> finite */ + } + } + if (ot == RAY_I64) { + if (binary) return op == OP_DIV || op == OP_IDIV || op == OP_MOD; + return (op == OP_NEG || op == OP_ABS) && t1 == RAY_I64; + } + return false; +} + /* Which (opcode, dst-type, src1-type) shapes have null-aware kernel * variants? Landing per-family: * Task 5: CAST shapes + F64 arithmetic (IEEE-propagating, no variant needed) @@ -804,6 +833,7 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].type = (base == RAY_F64 || base == RAY_F32) ? RAY_F64 : RAY_I64; out->regs[r].nullable = col_nulls; + out->regs[r].null_src = col_nulls; out->has_parted = true; } else { out->regs[r].col_type = col->type; @@ -815,6 +845,7 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].type = (col->type == RAY_F64 || col->type == RAY_F32) ? RAY_F64 : RAY_I64; out->regs[r].nullable = col_nulls; + out->regs[r].null_src = col_nulls; } } else if (node->opcode == OP_CONST) { ray_op_ext_t* ext = find_ext(g, node->id); @@ -878,37 +909,57 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { else ot = RAY_I64; - /* Type promotion: ensure both sources match for the operation. - * Skip for OP_CAST — the instruction itself IS the conversion. */ + /* Type promotion: every source of a non-CAST instruction + * must sit in the lane type its kernel reads. Scratch + * registers may hold I32/I16 (narrowing CAST results) or + * BOOL (comparison results): an I64 or comparison kernel + * reading such a buffer as 8-byte lanes returns garbage, so + * widen them here (the CAST kernels map the narrow null + * sentinels). A promotion that cannot be placed (register + * or instruction budget) bails instead of running the + * mismatched kernel. Skip for OP_CAST — the instruction + * itself IS the conversion. */ +#define EXPR_PROMOTE(sreg, T) do { \ + (sreg) = expr_ensure_type(out, (sreg), (T)); \ + if (out->regs[(sreg)].type != (T)) \ + EXPR_BAIL(EXPR_BAIL_REGS); \ + } while (0) if (op == OP_CAST) { /* No promotion needed; CAST handles the conversion */ - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_F64 && s2 != 0xFF) { - /* Arithmetic with f64 output — promote i64 inputs to f64 */ - s1 = expr_ensure_type(out, s1, RAY_F64); - s2 = expr_ensure_type(out, s2, RAY_F64); - r = out->n_regs; /* re-read after possible CAST inserts */ - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_F64 && s2 == 0xFF) { - /* Unary f64 — promote input */ - s1 = expr_ensure_type(out, s1, RAY_F64); - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_BOOL && s2 != 0xFF && t1 != t2) { - /* Comparison with mixed types — promote both to f64 */ - int8_t pt = (t1 == RAY_F64 || t2 == RAY_F64) ? RAY_F64 : RAY_I64; - s1 = expr_ensure_type(out, s1, pt); - s2 = expr_ensure_type(out, s2, pt); - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); + } else if (ot == RAY_F64) { + /* f64 arithmetic / unary math — promote i64 inputs to f64 */ + EXPR_PROMOTE(s1, RAY_F64); + if (s2 != 0xFF) EXPR_PROMOTE(s2, RAY_F64); + } else if (ot == RAY_I64) { + /* i64 arithmetic / NEG / ABS / SIGNUM — widen narrow and + * BOOL sources. An F64 source stays F64: only SIGNUM + * reaches here with one, and its kernel reads doubles. */ + if (out->regs[s1].type != RAY_F64) EXPR_PROMOTE(s1, RAY_I64); + if (s2 != 0xFF && out->regs[s2].type != RAY_F64) EXPR_PROMOTE(s2, RAY_I64); + } else if (ot == RAY_BOOL && s2 != 0xFF && + ((op >= OP_EQ && op <= OP_GE) || op == OP_AND || op == OP_OR)) { + /* Comparison / AND / OR — both sides in one lane type: + * BOOL stays BOOL only when both sides are BOOL. */ + int8_t pt = (t1 == RAY_F64 || t2 == RAY_F64) ? RAY_F64 + : (t1 == RAY_BOOL && t2 == RAY_BOOL) ? RAY_BOOL : RAY_I64; + if (pt != RAY_BOOL) { + EXPR_PROMOTE(s1, pt); + EXPR_PROMOTE(s2, pt); + } } +#undef EXPR_PROMOTE + r = out->n_regs; /* re-read after possible CAST inserts */ + if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); /* Compute nullability from the FINAL (post-promotion) s1/s2. * Inserted CASTs inherit nullable from their source (Step 3), * so the promoted regs already carry the right flag. */ bool in_null = out->regs[s1].nullable || (s2 != 0xFF && out->regs[s2].nullable); + bool in_src = out->regs[s1].null_src || + (s2 != 0xFF && out->regs[s2].null_src); + bool gen_null = expr_op_generates_null(op, ot, out->regs[s1].type, + s2 != 0xFF); bool ins_null_aware = false; bool dst_nullable = false; if (in_null) { @@ -959,7 +1010,12 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].kind = REG_SCRATCH; out->regs[r].type = ot; - out->regs[r].nullable = dst_nullable; + /* nullable: lanes may hold a sentinel, propagated from a + * source or produced by this op. null_src: that possibility + * traces to a nullable column (conservative output attr); + * without it the output flag comes from a precise scan. */ + out->regs[r].nullable = dst_nullable || gen_null; + out->regs[r].null_src = dst_nullable && in_src; out->n_scratch++; if (out->n_ins >= EXPR_MAX_INS) EXPR_BAIL(EXPR_BAIL_INS); @@ -1863,8 +1919,9 @@ static void expr_full_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t e /* Per-worker scratch buffers (heap-allocated via arena, morsel-sized) */ ray_t* scratch_hdr = NULL; + /* one morsel buffer per register the expression uses */ char* scratch_mem = (char*)scratch_alloc(&scratch_hdr, - (size_t)EXPR_MAX_REGS * EXPR_MORSEL * 8); + (size_t)(expr->n_regs ? expr->n_regs : 1) * EXPR_MORSEL * 8); if (!scratch_mem) return; void* scratch[EXPR_MAX_REGS]; for (uint8_t r = 0; r < expr->n_regs; r++) @@ -1941,57 +1998,27 @@ static void mark_i64_overflow_as_null(ray_t* result, int64_t off, int64_t len) { } } -/* The fused unary path may produce INT64_MIN via signed-overflow only for - * OP_NEG and OP_ABS over an i64 source (output type i64). Detect those - * shapes from the last instruction in the compiled expression. */ -static bool expr_last_op_overflows_i64(const ray_expr_t* expr) { - if (expr->out_type != RAY_I64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - if (last->opcode != OP_NEG && last->opcode != OP_ABS) return false; - if (last->src2 != 0xFF) return false; /* unary only */ - if (expr->regs[last->src1].type != RAY_I64) return false; - if (expr->regs[last->dst].type != RAY_I64) return false; - return true; -} - -/* The fused binary path writes NULL_I64 for an i64 DIV/IDIV/MOD whose - * divisor is zero (or the INT64_MIN/-1 overflow case) — null-model - * null handling requires HAS_NULLS set when that sentinel lands. When the - * output register is already marked `nullable` the conservative flag below - * covers it; this detector handles the case where it is not, reusing the - * mark_i64_overflow_as_null scan (which flips HAS_NULLS for any NULL_I64 - * lane). Detect the shape from the last instruction. */ -static bool expr_last_op_divmod_i64(const ray_expr_t* expr) { - if (expr->out_type != RAY_I64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - if (last->opcode != OP_DIV && last->opcode != OP_IDIV && - last->opcode != OP_MOD) return false; - if (last->src2 == 0xFF) return false; /* binary only */ - if (expr->regs[last->dst].type != RAY_I64) return false; - return true; -} - -/* Single-null float model: the fused F64 kernels canonicalize any non-finite - * result (overflow → ±Inf, div/mod-by-zero, sqrt(<0), log(≤0), exp(overflow)) - * to NULL_F64 in-buffer. Detect when the last instruction is such an F64 - * producer so the caller runs the cheap post-scan that flips HAS_NULLS for any - * 0Nf lane. Used to set HAS_NULLS conservatively from the op shape (no - * per-element scan — the scan is a full extra memory pass that regressed the - * hot float kernels ~50%); see the call site. Matches the fallback path's - * shape-based flagging so VM ≡ fallback. */ -static bool expr_last_op_produces_f64_null(const ray_expr_t* expr) { - if (expr->out_type != RAY_F64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - switch (last->opcode) { - case OP_ADD: case OP_SUB: case OP_MUL: - case OP_DIV: case OP_IDIV: case OP_MOD: case OP_POW: - case OP_SQRT: case OP_LOG: case OP_EXP: - case OP_SIN: case OP_ASIN: case OP_COS: case OP_ACOS: - case OP_TAN: case OP_ATAN: case OP_RECIPROCAL: - return true; - default: - return false; /* NEG/ABS/CEIL/FLOOR/ROUND/CAST/MIN2/MAX2: finite→finite */ - } +/* HAS_NULLS on a fused output. A register whose nullability traces to a + * nullable column gets the conservative attr — REQUIRED, not cosmetic: + * group.c feeds this vec to aggregates whose check-free fast path is gated + * on the attr; a missing attr with sentinel lanes = wrong aggregates. A + * register that is nullable only because some instruction in the program + * can produce a sentinel (x/0, overflow -> Inf, sqrt(<0), |INT64_MIN|, ...) + * is scanned instead, so a pure-finite result keeps HAS_NULLS unset — + * critical because this output is often an input to the NEXT op / + * aggregate, and a spurious HAS_NULLS would force that consumer onto the + * slow null-aware path (measured: conservative flagging regressed chained + * float kernels catastrophically by poisoning inputs). The scan looks at + * the whole program's generators, not just the last instruction: the + * sentinel survives every downstream null-aware kernel (abs, +, cast, ...). + * Produced once (post-join of all morsels), so the pass is not the per-op + * hot loop. Mirrors the fallback's shape-based flagging (VM == fallback). */ +static void expr_flag_output_nulls(const ray_expr_t* expr, ray_t* out, int64_t nrows) { + uint8_t o = expr->out_reg; + if (!expr->regs[o].nullable) return; + if (expr->regs[o].null_src) { out->attrs |= RAY_ATTR_HAS_NULLS; return; } + if (expr->out_type == RAY_I64) mark_i64_overflow_as_null(out, 0, nrows); + else if (expr->out_type == RAY_F64) mark_f64_nonfinite_as_null(out, 0, nrows); } /* Evaluate compiled expression over parted (segmented) columns. @@ -2063,24 +2090,7 @@ static ray_t* expr_eval_full_parted(const ray_expr_t* expr, int64_t nrows) { global_off += seg_len; } - if (expr_last_op_overflows_i64(expr) || expr_last_op_divmod_i64(expr)) - mark_i64_overflow_as_null(out, 0, nrows); - /* Single-null float model: flip HAS_NULLS PRECISELY if an F64 producer - * canonicalized a non-finite result to NULL_F64 in-buffer. Scan-based (not - * conservative-by-shape) so a pure-finite result keeps HAS_NULLS unset — - * critical because this fused output is often an input to the NEXT op / - * aggregate, and a spurious HAS_NULLS would force that consumer onto the - * slow null-aware path (measured: conservative flagging regressed chained - * float kernels catastrophically by poisoning inputs). This output is - * produced once (post-join of all morsels) so the single pass here is not - * the per-op hot loop; mirrors the i64-overflow mark above. */ - if (expr_last_op_produces_f64_null(expr)) - mark_f64_nonfinite_as_null(out, 0, nrows); - /* Conservative "may contain nulls" — REQUIRED, not cosmetic: group.c - * feeds this vec to aggregates whose check-free fast path is gated on - * the attr; a missing attr with sentinel lanes = wrong aggregates. */ - if (expr->regs[expr->out_reg].nullable) - out->attrs |= RAY_ATTR_HAS_NULLS; + expr_flag_output_nulls(expr, out, nrows); return out; } @@ -2104,24 +2114,7 @@ ray_t* expr_eval_full(const ray_expr_t* expr, int64_t nrows) { else expr_full_fn(&ctx, 0, 0, nrows); - if (expr_last_op_overflows_i64(expr) || expr_last_op_divmod_i64(expr)) - mark_i64_overflow_as_null(out, 0, nrows); - /* Single-null float model: flip HAS_NULLS PRECISELY if an F64 producer - * canonicalized a non-finite result to NULL_F64 in-buffer. Scan-based (not - * conservative-by-shape) so a pure-finite result keeps HAS_NULLS unset — - * critical because this fused output is often an input to the NEXT op / - * aggregate, and a spurious HAS_NULLS would force that consumer onto the - * slow null-aware path (measured: conservative flagging regressed chained - * float kernels catastrophically by poisoning inputs). This output is - * produced once (post-join of all morsels) so the single pass here is not - * the per-op hot loop; mirrors the i64-overflow mark above. */ - if (expr_last_op_produces_f64_null(expr)) - mark_f64_nonfinite_as_null(out, 0, nrows); - /* Conservative "may contain nulls" — REQUIRED, not cosmetic: group.c - * feeds this vec to aggregates whose check-free fast path is gated on - * the attr; a missing attr with sentinel lanes = wrong aggregates. */ - if (expr->regs[expr->out_reg].nullable) - out->attrs |= RAY_ATTR_HAS_NULLS; + expr_flag_output_nulls(expr, out, nrows); return out; } diff --git a/src/ops/internal.h b/src/ops/internal.h index 64316f51..3a271d27 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -809,8 +809,8 @@ extern uint64_t ray_join_nullfree_keys; extern bool ray_agg_engine_v2; /* route OP_GROUP through v2 agg engine; default ON (agg_engine.c) */ void ray_expr_stats_init(void); -#define EXPR_MAX_REGS 16 -#define EXPR_MAX_INS 48 +#define EXPR_MAX_REGS 32 +#define EXPR_MAX_INS 96 #define EXPR_MORSEL RAY_MORSEL_ELEMS typedef struct { @@ -837,6 +837,8 @@ typedef struct { uint8_t col_attrs; /* column attrs — RAY_SYM width (REG_SCAN only) */ bool is_parted; /* true if this SCAN refs a parted column */ bool nullable; /* lanes may contain NULL_I64 / NaN */ + bool null_src; /* that nullability traces to a nullable column + * (else: only op-generated sentinels) */ const void* data; /* column data pointer (REG_SCAN only) */ ray_t* col_obj; /* source column vec (REG_SCAN, non-parted) — * carries the chunk-zone index for zone-skip */ diff --git a/test/rfl/expr/narrow_binary.rfl b/test/rfl/expr/narrow_binary.rfl index c9280abf..5f3f77ec 100644 --- a/test/rfl/expr/narrow_binary.rfl +++ b/test/rfl/expr/narrow_binary.rfl @@ -237,3 +237,20 @@ ;; null is the least value in narrow-int comparison: -1 is NOT < null (< [-1h] [0Nh]) -- [false] (type (< [-1h] [0Nh])) -- 'B8 + +;; =================================================================== +;; Fused select: an I32/I16 narrowing cast or a comparison result used +;; as an operand of an i64 op (or of another comparison) is widened to +;; i64 lanes before the kernel reads it — the 4-, 2- or 1-byte scratch +;; buffer must never be read as 8-byte lanes. +;; =================================================================== +(set NB (table [a b c] (list [14 20 30] [1 2 3] [0Nl 2 3]))) +(at (select {from: NB z: (- a (as 'I32 -1))}) 'z) -- [15 21 31] +(at (select {from: NB z: (+ a (as 'I32 -1))}) 'z) -- [13 19 29] +(at (select {from: NB z: (neg (as 'I32 b))}) 'z) -- [-1 -2 -3] +(at (select {from: NB z: (abs (as 'I16 (neg b)))}) 'z) -- [1 2 3] +(at (select {from: NB z: (+ (> a 15) 1)}) 'z) -- [1 2 2] +(at (select {from: NB z: (< (as 'I32 b) (as 'I32 a))}) 'z) -- [true true true] +(count (select {from: NB where: (< (as 'I32 a) (as 'I16 25))})) -- 2 +;; the narrow null sentinel widens to the i64 null +(at (select {from: NB z: (+ a (as 'I32 c))}) 'z) -- [0Nl 22 33] diff --git a/test/rfl/expr/null_propagation.rfl b/test/rfl/expr/null_propagation.rfl index 837ac824..e9061584 100644 --- a/test/rfl/expr/null_propagation.rfl +++ b/test/rfl/expr/null_propagation.rfl @@ -133,3 +133,37 @@ ;; Arithmetic: null DOES propagate (count (+ [0Nl 1 2] 1)) -- 3 (sum (map nil? (+ [0Nl 1 2] 1))) -- 1 + +;; =================================================================== +;; Fused select: a null produced in the middle of the program (x/0, +;; overflow, |INT64_MIN|) survives every downstream op — abs/neg/floor, +;; cast to i64, further i64 arithmetic — and the output column carries +;; HAS_NULLS, so aggregates skip those lanes instead of being poisoned. +;; =================================================================== +(set i (til 200000)) +(set NT (table [h b a c] (list (- (% i 300) 150) (% i 13) (% i 97) (% i 11)))) +(set H (at NT 'h)) (set B (at NT 'b)) (set A (at NT 'a)) (set C (at NT 'c)) +;; every 13th b is zero +(count (where (== B 0))) -- 15385 +(count (where (nil? (at (select {from: NT x: (abs (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (neg (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (floor (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (as 'I64 (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (+ (as 'I64 (/ h b)) 1)}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (+ (div h b) 1)}) 'x)))) -- 15385 +;; sums equal the column-at-a-time answers (nulls skipped, not summed) +(sum (at (select {from: NT x: (abs (/ h b))}) 'x)) -- (sum (abs (/ H B))) +(sum (at (select {from: NT x: (as 'I64 (/ h b))}) 'x)) -- (sum (as 'I64 (/ H B))) +(sum (at (select {from: NT x: (+ (div h b) 1)}) 'x)) -- (sum (+ (div H B) 1)) +;; the same through a program wider than 16 registers +(set WX (at (select {from: NT x: (abs (+ (/ h b) (+ a (+ c (+ a (+ c (+ a (+ c (+ a (+ c (+ a c)))))))))))}) 'x)) +(count (where (nil? WX))) -- 15385 +(sum WX) -- (sum (abs (+ (/ H B) (+ A (+ C (+ A (+ C (+ A (+ C (+ A (+ C (+ A C)))))))))))) +;; grouped aggregates over the fused column match a pre-materialized one +(set G1 (select {from: NT by: {k: c} s: (sum (abs (/ h b))) m: (avg (abs (/ h b)))})) +(set NX (table [c x] (list C (abs (/ H B))))) +(set G2 (select {from: NX by: {k: c} s: (sum x) m: (avg x)})) +(at G1 's) -- (at G2 's) +(at G1 'm) -- (at G2 'm) +;; a null-free program keeps HAS_NULLS off (no spurious null-aware path) +(count (where (nil? (at (select {from: NT x: (abs (+ h 1.5))}) 'x)))) -- 0 diff --git a/test/rfl/expr/wide_predicate.rfl b/test/rfl/expr/wide_predicate.rfl new file mode 100644 index 00000000..5c2ce066 --- /dev/null +++ b/test/rfl/expr/wide_predicate.rfl @@ -0,0 +1,33 @@ +;; Filters and computed columns wider than 16 expression registers compile +;; to one fused expression (EXPR_MAX_REGS) instead of falling back to +;; column-at-a-time evaluation. Every answer is checked against the same +;; conditions combined step by step. +(set n 200000) +(set i (til n)) +(set T (table [a b c d e f] (list (% i 97) (% i 13) (% (* i 7) 1000) (as 'DATE (% i 60)) (% i 2) (% i 3)))) +;; four conditions, one a `within` (the shape that used to exceed the limit) +(set W4 (select {from: T where: (and (== a 5) (within d [2000.01.10 2000.01.20]) (== e 0) (== f 1))})) +(set S1 (select {from: T where: (== a 5)})) +(set S2 (select {from: S1 where: (within d [2000.01.10 2000.01.20])})) +(set S3 (select {from: S2 where: (== e 0)})) +(set S4 (select {from: S3 where: (== f 1)})) +(count W4) -- (count S4) +(at W4 'c) -- (at S4 'c) +;; six conditions with arithmetic on both sides +(set W6 (select {from: T where: (and (> (+ a b) 20) (< (* c 2) 1500) (within d [2000.01.05 2000.02.19]) (!= e 1) (>= (- f 1) 0) (<= (+ a (* 2 b)) 100))})) +(set R6 (select {from: T where: (> (+ a b) 20)})) +(set R6 (select {from: R6 where: (< (* c 2) 1500)})) +(set R6 (select {from: R6 where: (within d [2000.01.05 2000.02.19])})) +(set R6 (select {from: R6 where: (!= e 1)})) +(set R6 (select {from: R6 where: (>= (- f 1) 0)})) +(set R6 (select {from: R6 where: (<= (+ a (* 2 b)) 100)})) +(count W6) -- (count R6) +(sum (at W6 'c)) -- (sum (at R6 'c)) +;; a grouping under a four-condition filter (grouped by a computed key) +(set G4 (select {from: T by: {k: (xbar c 100)} n: (count a) where: (and (== a 5) (within d [2000.01.10 2000.01.20]) (== e 0) (== f 1))})) +(set GS (select {from: S4 by: {k: (xbar c 100)} n: (count a)})) +(at (select {from: G4 asc: k}) 'n) -- (at (select {from: GS asc: k}) 'n) +(at (select {from: G4 asc: k}) 'k) -- (at (select {from: GS asc: k}) 'k) +;; a wide computed column +(set X (select {from: T x: (+ (+ (+ (* a 3) (* b 5)) (+ (* c 7) (- a b))) (+ (+ (- c a) (+ b c)) (* (+ a 1) (+ b 2))))})) +(at (at X 'x) 12345) -- (+ (+ (+ (* 26 3) (* 8 5)) (+ (* 415 7) (- 26 8))) (+ (+ (- 415 26) (+ 8 415)) (* (+ 26 1) (+ 8 2)))) diff --git a/test/test_expr_null.c b/test/test_expr_null.c index 89f7fa67..7a2601fe 100644 --- a/test/test_expr_null.c +++ b/test/test_expr_null.c @@ -324,18 +324,24 @@ static test_result_t test_nullfree_promotion_invariance(void) { ray_expr_t ex; TEST_ASSERT(expr_compile(g, tbl, build_i64_plus_f64(g), &ex), "null-free promotion compiles"); - /* No INPUT is nullable, so the compiler marks no instruction null_aware and - * no reg nullable. Single-null float model: an F64 producer that may yield - * 0Nf from finite inputs (overflow) gets HAS_NULLS via a PRECISE post-scan - * in expr_eval_full (expr_last_op_produces_f64_null + - * mark_f64_nonfinite_as_null) — NOT the compile-time nullable flag — so - * this null-free-promotion compile invariant is preserved unchanged. */ + /* No INPUT is nullable, so the compiler marks no instruction null_aware + * and no reg null_src (column-derived nullability). Single-null float + * model: the F64 ADD may yield 0Nf from finite inputs (overflow), so its + * destination IS nullable — downstream kernels would have to honour the + * sentinel — but HAS_NULLS on the output comes from a PRECISE post-scan + * in expr_eval_full (expr_flag_output_nulls + mark_f64_nonfinite_as_null), + * NOT from a conservative flag, so a finite result stays null-free. */ for (uint8_t i = 0; i < ex.n_ins; i++) TEST_ASSERT(ex.ins[i].null_aware == 0, "no null_aware on null-free promotion"); for (uint8_t r2 = 0; r2 < ex.n_regs; r2++) - TEST_ASSERT(!ex.regs[r2].nullable, - "no nullable regs on null-free promotion"); + TEST_ASSERT(!ex.regs[r2].null_src, + "no column-derived nullable regs on null-free promotion"); + for (uint8_t r2 = 0; r2 < ex.n_regs; r2++) + if (ex.regs[r2].kind != REG_SCRATCH || r2 == ex.out_reg) continue; + else TEST_ASSERT(!ex.regs[r2].nullable, "promotion cast is not nullable"); + TEST_ASSERT(ex.regs[ex.out_reg].nullable, + "F64 ADD destination is nullable (overflow -> 0Nf generator)"); ray_graph_free(g); ray_release(tbl); ray_sym_destroy(); ray_heap_destroy(); From aa51f704a7b6be59c61befddecca992f9b69dd98 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 11:10:48 +0300 Subject: [PATCH 44/51] perf(select): filtered positional take stops early; sort by an unprojected column (#634) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(select): filtered positional take stops once the answer is known `(select {… from: T where: take: K})` with no ordering evaluated the predicate over the whole table, gathered every passing row and cut K afterwards; `take: -K` did the same for the last rows. The fused path now answers it: the table is walked in 64k-row chunks from the end the answer comes from, one pool task per chunk in that order, each worker keeping at most |K| passing row ids. A worker that holds |K| publishes the row beyond which nothing can be part of the answer, and every later chunk returns at once. The lists are merged by row id and only the |K| rows are gathered. `asc: key take: K` on a column marked sorted (no nulls) takes the same path: its first K passing rows are the K smallest, ties in table order. Test: fused/fused_take_early_stop (first and last K over many chunks, AND predicates, fewer matches than K, none, all columns, aliases, K larger than one worker's share, the sorted-key form). Co-Authored-By: Claude Opus 5.5 * fix(select): sort by a source column the projection does not output `(select {a: a from: T asc: c})` failed with `nyi` whenever the sort did not go through the fused top-k: the sort runs over the projected table and looked the key up there. A sort key that names a source column no output carries is now projected too and dropped from the result after the sort (an output with the same name is used as before). Test: query/sort_key_not_projected (no filter, a filter the fused path does not take, with and without take, two keys, computed outputs, a key that is also an output). Co-Authored-By: Claude Opus 5.5 * fix(select): hidden sort key vs a renamed output; alter set clears sorted - An output that renames the key column (`{b: a … asc: a}`) is a bare scan the projection names by its source column; the key was added as a hidden column under the same name and the drop took the output with it. Such a scan now counts as the key being present, and hidden keys are dropped by position (they are projected last). - `alter … set` wrote into a vector marked sorted and kept the marker; the ascending take on a sorted key trusts it. The marker is cleared on write. - Without a sort, the positional take path admits only a literal K, so a `take:` expression is evaluated once, by the general path. Co-Authored-By: Claude Opus 5.5 * fix(select): fused take/top-k leave alias-chained projections to the planner Projections resolve left to right: in `{b: a z: b …}` the output z reads the alias b (column a), not the source column b. The fused positional take and the fused top-k gather source columns by name, so z came back as column b — `take: 2` gave [30 10] instead of [3 1], and the sorted top-k had the same wrong answer before this branch. When an output names an alias bound by an earlier output (other than itself), the fused paths now decline and the planner resolves the projections. Tests: fused/fused_take_early_stop (positive and negative take, sorted take with and without a filter, an identity alias that stays fused). Co-Authored-By: Claude Opus 5.5 * fix(select): sort keys bind to projected columns by position, name source columns The sort of a projection resolved its keys by NAME over the projected table, whose columns are named by source column for scans and `_e` for expressions. Three wrong outcomes followed: - a hidden key whose source column is called `_e0` collided with the first computed output: `{z: (+ a 10) asc: _e0}` sorted by z; - `{b: a asc: b}` treated the output alias b as the key, found no column b in the projection (that scan is named a) and failed with `nyi`; - hidden keys were capped at 16, so a 17-key sort failed with `nyi`. A sort key now names a SOURCE column, like where: and by:. In the planner it binds to the projected column that scans it — an output, or the hidden key appended for it — and exec_sort reads a key that is one of the SELECT's columns by position, so no name lookup can pick another column. Output aliases are not consulted. The hidden-key capacity is one slot per sort key name. Co-Authored-By: Claude Fable 5.1 --------- Co-authored-by: Claude Opus 5.5 Co-authored-by: Anton Kundenko --- src/ops/fused_topk.c | 207 ++++++++++++++++++++++ src/ops/fused_topk.h | 17 ++ src/ops/query.c | 139 ++++++++++++++- src/ops/sort.c | 24 ++- src/ops/tblop.c | 3 + test/rfl/fused/fused_take_early_stop.rfl | 59 ++++++ test/rfl/query/sort_key_not_projected.rfl | 60 +++++++ 7 files changed, 502 insertions(+), 7 deletions(-) create mode 100644 test/rfl/fused/fused_take_early_stop.rfl create mode 100644 test/rfl/query/sort_key_not_projected.rfl diff --git a/src/ops/fused_topk.c b/src/ops/fused_topk.c index 6d44e49d..1c216006 100644 --- a/src/ops/fused_topk.c +++ b/src/ops/fused_topk.c @@ -49,6 +49,7 @@ #include "core/pool.h" #include +#include /* Use the same predicate-shape detector as fused_group. Single comparison * or AND of comparisons against literals on flat int/temporal/SYM columns. */ @@ -486,3 +487,209 @@ ray_t* ray_fused_topk_select(ray_t* tbl, } return result; } + +/* ───── Fused filter + positional take ──────────────────────────────── + * Chunks of FTK_CHUNK_ROWS rows, numbered from the end the answer comes + * from (row 0 for the first K, the last row for the last |K|), one pool + * task per chunk in that order — the workers sweep the table from that + * end together. Each worker appends passing rows to its own list until + * it holds |K|; those |K| bound the answer, so it publishes the |K|-th + * row as the cutoff: no row beyond it can be among the first |K| passing + * rows of the table, and every later chunk returns at once. The lists + * are merged by row id at the end and the |K| nearest the scanned end + * are gathered. + * ──────────────────────────────────────────────────────────────────── */ + +#define FTK_CHUNK_ROWS (64 * 1024) + +typedef struct { + fp_pred_t pred; + int64_t nrows; + int64_t k; /* |K| */ + bool from_end; + int64_t* rows; /* [nw * k] per-worker row ids, in scan order */ + int32_t* rows_n; /* [nw] */ + _Atomic(int64_t) cutoff; /* forward: rows >= cutoff are out; backward: rows <= cutoff */ +} ftk_ctx_t; + +static inline bool ftk_beyond(const ftk_ctx_t* c, int64_t row) { + int64_t cut = atomic_load_explicit(&c->cutoff, memory_order_relaxed); + return c->from_end ? row <= cut : row >= cut; +} + +static void ftk_publish(ftk_ctx_t* c, int64_t row) { + int64_t cur = atomic_load_explicit(&c->cutoff, memory_order_relaxed); + for (;;) { + bool tighter = c->from_end ? row > cur : row < cur; + if (!tighter) return; + if (atomic_compare_exchange_weak_explicit(&c->cutoff, &cur, row, + memory_order_relaxed, memory_order_relaxed)) + return; + } +} + +static void ftk_task_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { + (void)end; + ftk_ctx_t* c = (ftk_ctx_t*)raw; + int32_t k = (int32_t)c->k; + int64_t* my = &c->rows[(size_t)worker_id * (size_t)k]; + int32_t n = c->rows_n[worker_id]; + if (n >= k) return; /* this worker's list is complete */ + + /* chunk `start` counted from the scanned end */ + int64_t lo, hi; + if (!c->from_end) { + lo = start * FTK_CHUNK_ROWS; + hi = lo + FTK_CHUNK_ROWS; + if (hi > c->nrows) hi = c->nrows; + } else { + hi = c->nrows - start * FTK_CHUNK_ROWS; + lo = hi - FTK_CHUNK_ROWS; + if (lo < 0) lo = 0; + } + if (ftk_beyond(c, c->from_end ? hi - 1 : lo)) return; + + uint8_t bits[RAY_MORSEL_ELEMS]; + if (!c->from_end) { + for (int64_t row = lo; row < hi && n < k; ) { + if (ftk_beyond(c, row)) break; + int64_t mend = row + RAY_MORSEL_ELEMS; + if (mend > hi) mend = hi; + fp_eval_pred(&c->pred, row, mend, bits); + for (int64_t r = 0; r < mend - row && n < k; r++) + if (bits[r]) my[n++] = row + r; + row = mend; + } + } else { + for (int64_t mend = hi; mend > lo && n < k; ) { + if (ftk_beyond(c, mend - 1)) break; + int64_t row = mend - RAY_MORSEL_ELEMS; + if (row < lo) row = lo; + fp_eval_pred(&c->pred, row, mend, bits); + for (int64_t r = mend - row - 1; r >= 0 && n < k; r--) + if (bits[r]) my[n++] = row + r; + mend = row; + } + } + c->rows_n[worker_id] = n; + if (n >= k) ftk_publish(c, my[k - 1]); +} + +ray_t* ray_fused_take_select(ray_t* tbl, + ray_t* where_expr, + int64_t k, + const int64_t* out_col_syms, + const int64_t* out_alias_syms, + uint32_t n_out) +{ + if (!tbl || tbl->type != RAY_TABLE || !where_expr || k == 0 || n_out == 0) return NULL; + if (k == INT64_MIN) return NULL; + bool from_end = k < 0; + if (from_end) k = -k; + if (k > FPK_MAX_K) return NULL; + int64_t nrows = ray_table_nrows(tbl); + if (nrows <= 0 || k >= nrows) return NULL; + + for (uint32_t c = 0; c < n_out; c++) { + ray_t* col = ray_table_get_col(tbl, out_col_syms[c]); + if (!col) return NULL; + int8_t ot = col->type; + if (RAY_IS_PARTED(ot) || ot == RAY_MAPCOMMON) return NULL; + if (!ray_is_vec(col)) return NULL; + } + + ftk_ctx_t ctx; + memset(&ctx, 0, sizeof(ctx)); + ctx.nrows = nrows; + ctx.k = k; + ctx.from_end = from_end; + atomic_store_explicit(&ctx.cutoff, from_end ? -1 : INT64_MAX, memory_order_relaxed); + + ray_graph_t* g = ray_graph_new(tbl); + if (!g) return NULL; + ray_op_t* pred_dag = compile_expr_dag(g, where_expr); + if (!pred_dag) { ray_graph_free(g); return NULL; } + if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + + ray_pool_t* pool = ray_pool_get(); + uint32_t nw = pool ? ray_pool_total_workers(pool) : 1; + ray_t* rows_hdr = NULL; + ray_t* n_hdr = NULL; + ctx.rows = (int64_t*)scratch_alloc(&rows_hdr, (size_t)nw * (size_t)k * sizeof(int64_t)); + ctx.rows_n = (int32_t*)scratch_calloc(&n_hdr, (size_t)nw * sizeof(int32_t)); + if (!ctx.rows || !ctx.rows_n) { + if (rows_hdr) scratch_free(rows_hdr); + if (n_hdr) scratch_free(n_hdr); + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + + int64_t n_chunks = (nrows + FTK_CHUNK_ROWS - 1) / FTK_CHUNK_ROWS; + if (ray_pool_par_dispatch_ok(pool, n_chunks, 2)) + ray_pool_dispatch_n(pool, ftk_task_fn, &ctx, (uint32_t)n_chunks); + else + for (int64_t t = 0; t < n_chunks; t++) ftk_task_fn(&ctx, 0, t, t + 1); + + /* Merge: each list is in scan order; pick the row nearest the scanned + * end across lists k times, then present in table order. */ + int64_t out[FPK_MAX_K]; + int32_t out_n = 0; + ray_t* pos_hdr = NULL; + int32_t* pos = (int32_t*)scratch_calloc(&pos_hdr, (size_t)nw * sizeof(int32_t)); + if (!pos) { + scratch_free(rows_hdr); scratch_free(n_hdr); + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + while (out_n < (int32_t)k) { + int64_t best = -1; + uint32_t bw = 0; + for (uint32_t w = 0; w < nw; w++) { + if (pos[w] >= ctx.rows_n[w]) continue; + int64_t r = ctx.rows[(size_t)w * (size_t)k + (size_t)pos[w]]; + if (best < 0 || (from_end ? r > best : r < best)) { best = r; bw = w; } + } + if (best < 0) break; + pos[bw]++; + out[out_n++] = best; + } + scratch_free(pos_hdr); + scratch_free(rows_hdr); + scratch_free(n_hdr); + if (from_end) { + for (int32_t i = 0, j = out_n - 1; i < j; i++, j--) { + int64_t t = out[i]; out[i] = out[j]; out[j] = t; + } + } + + ray_t* result = ray_table_new(n_out); + if (!result || RAY_IS_ERR(result)) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return result ? result : ray_error("oom", NULL); + } + int build_ok = 1; + for (uint32_t c = 0; c < n_out; c++) { + int64_t cs = out_col_syms[c]; + int64_t alias = out_alias_syms ? out_alias_syms[c] : cs; + ray_t* src = ray_table_get_col(tbl, cs); + if (!src) { build_ok = 0; break; } + ray_t* col = gather_by_idx(src, out, out_n); + if (!col || RAY_IS_ERR(col)) { build_ok = 0; break; } + result = ray_table_add_col(result, alias, col); + ray_release(col); + } + ray_graph_free(g); + fp_pred_cleanup(&ctx.pred); + if (!build_ok) { + ray_release(result); + return ray_error("schema", NULL); + } + return result; +} diff --git a/src/ops/fused_topk.h b/src/ops/fused_topk.h index df3a2683..4b164ae4 100644 --- a/src/ops/fused_topk.h +++ b/src/ops/fused_topk.h @@ -72,6 +72,23 @@ int ray_fused_topk_supported(ray_t* where_expr, ray_t* tbl); * * Returns NULL on shape miss (errors during predicate compile etc.) so * the caller can fall back to the unfused FILTER + SORT_TAKE path. */ +/* Fused filter + positional take: `(select {cols… from: T where: + * take: K})` with no ordering, K > 0 (the first K rows that pass, in + * table order) or K < 0 (the last |K|). The scan stops as soon as the + * answer is known: the table is walked in chunks from the near end, each + * worker keeps at most |K| row ids, and a worker that has |K| publishes + * the row beyond which nothing can still be part of the answer. Also + * serves `asc: key take: K` when the key column carries RAY_ATTR_SORTED + * and no nulls (the first K passing rows are then the K smallest, ties in + * table order — what the stable sort returns). + * Returns NULL on a shape the path does not take (caller falls back). */ +ray_t* ray_fused_take_select(ray_t* tbl, + ray_t* where_expr, + int64_t k, + const int64_t* out_col_syms, + const int64_t* out_alias_syms, + uint32_t n_out); + ray_t* ray_fused_topk_select(ray_t* tbl, ray_t* where_expr, const int64_t* sort_key_syms, diff --git a/src/ops/query.c b/src/ops/query.c index a17e1581..0a1ce142 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -1018,6 +1018,38 @@ static ray_op_t* sel_alias_lookup(ray_graph_t* g, int64_t sym) { return NULL; } +/* The op a sort key binds to. When the sort runs over a projection + * (`root` is its SELECT), the key is the projected column that scans the + * source column — an output, or the hidden key appended for it — so the + * sort reads it by POSITION and no name lookup over the projected table + * (scans named by source, expressions `_e`) can pick another column. + * Otherwise a fresh scan resolved by name, as before. */ +static ray_op_t* select_sort_key_op(ray_graph_t* g, ray_op_t* root, int64_t sym, const char* name) { + if (root && root->opcode == OP_SELECT) { + ray_op_ext_t* se = find_ext(g, root->id); + if (se && se->base.opcode == OP_SELECT) + for (uint32_t c = 0; c < se->sort.n_cols; c++) { + ray_op_t* col = &g->nodes[se->sort.columns[c]]; + if (col->opcode != OP_SCAN) continue; + ray_op_ext_t* ce = find_ext(g, col->id); + if (ce && ce->base.opcode == OP_SCAN && ce->sym == sym) return col; + } + } + return ray_scan(g, name); +} + +/* Index of the first projected column that is a bare scan of source + * column `sym`, or -1. Used to bind a sort key to the projected column + * that already carries it. */ +static int64_t select_scan_output(ray_graph_t* g, ray_op_t** col_ops, int64_t nc, int64_t sym) { + for (int64_t c = 0; c < nc; c++) { + if (!col_ops[c] || col_ops[c]->opcode != OP_SCAN) continue; + ray_op_ext_t* ce = find_ext(g, col_ops[c]->id); + if (ce && ce->base.opcode == OP_SCAN && ce->sym == sym) return c; + } + return -1; +} + /* Takes the error compile_expr_dag left on the graph (see ops.h), or * NULL. Callers that report a compile failure use it so the message * names the actual problem when there is one. */ @@ -7360,7 +7392,7 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * intermediate filtered table materialised. Closes a large * latency gap on ORDER BY + LIMIT shapes that were previously * dominated by the filtered-table materialisation step. */ - if (take_expr && has_sort && !by_expr && !nearest_expr) { + if (take_expr && (has_sort || where_expr) && !by_expr && !nearest_expr) { if (!where_expr || ray_fused_topk_supported(where_expr, tbl)) { /* Walk the dict and check: exactly one asc/desc clause naming * a single scalar column, take is an atom K, and every @@ -7409,6 +7441,16 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * column. The dict key is the alias the result publishes; * the value names the source column to gather from. */ if (n_out_syms >= 255) { bad_clause = 1; break; } + /* Projections resolve left to right: a name bound by an + * earlier output is that output's value, not the source + * column. The fused paths gather source columns, so such + * a shape is left to the planner. */ + if (v && v->type == -RAY_SYM && !(v->attrs & ATTR_QUOTED)) { + bool alias_ref = false; + for (uint8_t o = 0; o < n_out_syms; o++) + if (out_aliases[o] == v->i64 && out_syms[o] != v->i64) alias_ref = true; + if (alias_ref) { bad_clause = 1; break; } + } if (v && v->type == -RAY_SYM && !(v->attrs & ATTR_QUOTED)) { ray_t* oc = ray_table_get_col(tbl, v->i64); if (!oc) { bad_clause = 1; break; } @@ -7457,13 +7499,43 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (!kc) bad_clause = 1; } } - if (!bad_clause && n_sort_keys > 0 && n_out_syms > 0) { + /* Without a sort the positional path takes only a literal K: + * an expression is left to the general path, which evaluates + * it exactly once. */ + if (n_sort_keys == 0 && !(ray_is_atom(take_expr) && + (take_expr->type == -RAY_I64 || take_expr->type == -RAY_I32))) + bad_clause = 1; + if (!bad_clause && n_out_syms > 0) { ray_t* tv = ray_eval(take_expr); if (tv && !RAY_IS_ERR(tv) && ray_is_atom(tv) && (tv->type == -RAY_I64 || tv->type == -RAY_I32)) { int64_t k = (tv->type == -RAY_I64) ? tv->i64 : tv->i32; ray_release(tv); - if (k > 0 && k <= FPK_MAX_K && k < ray_table_nrows(tbl)) { + /* Positional take under a filter, and an ascending + * take on a column known to be sorted (no nulls): the + * answer is the first |k| passing rows from one end of + * the table, found without scanning the rest. */ + bool sorted_asc = false; + if (n_sort_keys == 1 && !sort_descs[0]) { + ray_t* kc = ray_table_get_col(tbl, sort_key_syms[0]); + sorted_asc = kc && (kc->attrs & RAY_ATTR_SORTED) && + !(kc->attrs & RAY_ATTR_SLICE) && + !ray_vec_may_have_nulls(kc); + } + if (where_expr && (n_sort_keys == 0 || (sorted_asc && k > 0)) && + k != 0 && k != INT64_MIN && + (k < 0 ? -k : k) <= FPK_MAX_K && + (k < 0 ? -k : k) < ray_table_nrows(tbl)) { + ray_t* res = ray_fused_take_select(tbl, where_expr, k, + out_syms, out_aliases, + n_out_syms); + if (res && !RAY_IS_ERR(res)) { + ray_release(tbl); + DICT_VIEW_CLOSE(dv); return res; + } + if (res && RAY_IS_ERR(res)) ray_release(res); + } + if (n_sort_keys > 0 && k > 0 && k <= FPK_MAX_K && k < ray_table_nrows(tbl)) { ray_t* res = ray_fused_topk_select(tbl, where_expr, sort_key_syms, sort_descs, @@ -7502,6 +7574,10 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * FLAT path so nothing there changes. */ bool parted_bydict_deferred = false; bool computed_single_key = false; /* by: is one expression, compiled or const */ + /* Sort keys that are source columns no output scans: carried through + * the projection (appended after the outputs) so the sort can read + * them, dropped from the result after it ran. */ + int n_hidden_sort = 0; ray_t* deferred_bydict = NULL; int64_t deferred_nk = 0; int64_t dep_key_base_sym = -1; @@ -11394,6 +11470,15 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * eval fallback via the width check below; now only the other * use_eval_fallback trigger (a failed compile) applies. */ int64_t nc_max = select_output_count(dict_elems, dict_n); + /* plus one slot per sort key name: any of them may have to be + * carried through the projection as a hidden column */ + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* val = dict_elems[i + 1]; + if (val->type == -RAY_SYM) nc_max += 1; + else if (ray_is_vec(val) && val->type == RAY_SYM) nc_max += ray_len(val); + } if (nc_max < 1) nc_max = 1; ray_t* colops_hdr = NULL; ray_op_t** col_ops = (ray_op_t**)scratch_alloc(&colops_hdr, @@ -11434,6 +11519,35 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { g->sel_alias_syms = NULL; g->sel_alias_ids = NULL; g->sel_alias_n = 0; + /* Sort keys name SOURCE columns (like where: and by:, they are + * alias-blind). A key some output scans bare is read from that + * output (the sort binds to the projected column by position, + * see exec_sort); a key no output scans is projected too, after + * the outputs, and dropped from the result afterwards. Output + * aliases are not consulted: `{b: a ... asc: b}` sorts by the + * source column b, not by the output that renames a. */ + if (!use_eval_fallback && has_sort) { + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* val = dict_elems[i + 1]; + int64_t nk = val->type == -RAY_SYM ? 1 + : (ray_is_vec(val) && val->type == RAY_SYM) ? val->len : 0; + for (int64_t k = 0; k < nk; k++) { + int64_t ks = val->type == -RAY_SYM ? val->i64 : sym_cell_runtime_id(val, k); + if (select_scan_output(g, col_ops, nc, ks) >= 0) continue; + if (!ray_table_get_col(tbl, ks) || nc >= nc_max) continue; + ray_t* nm = ray_sym_str(ks); + ray_op_t* sc = nm ? ray_scan(g, ray_str_ptr(nm)) : NULL; + if (!sc) continue; + col_ops[nc] = sc; + alias_syms[nc] = ks; + alias_ids[nc] = sc->id; + nc++; + n_hidden_sort++; + } + } + } if (use_eval_fallback) { if (g->compile_err) { ray_release(g->compile_err); g->compile_err = NULL; } /* The fallback evaluates projections directly over `tbl`, @@ -11587,14 +11701,14 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (val->type == -RAY_SYM) { /* Single column name */ ray_t* s = ray_sym_str(val->i64); - sort_keys[n_sort] = ray_scan(g, ray_str_ptr(s)); + sort_keys[n_sort] = select_sort_key_op(g, root, val->i64, ray_str_ptr(s)); sort_descs[n_sort] = is_desc; n_sort++; } else if (ray_is_vec(val) && val->type == RAY_SYM) { /* Multiple column names — cell-data via the vec's domain */ for (int64_t c = 0; c < val->len; c++) { ray_t* s = ray_sym_vec_cell(val, c); - sort_keys[n_sort] = ray_scan(g, ray_str_ptr(s)); + sort_keys[n_sort] = select_sort_key_op(g, root, sym_cell_runtime_id(val, c), ray_str_ptr(s)); sort_descs[n_sort] = is_desc; n_sort++; } @@ -11697,6 +11811,21 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { /* Optimize and execute */ root = ray_optimize(g, root); ray_t* result = ray_execute(g, root); + if (n_hidden_sort > 0 && result && !RAY_IS_ERR(result)) { + if (ray_is_lazy(result)) result = ray_lazy_materialize(result); + if (result && !RAY_IS_ERR(result) && result->type == RAY_TABLE) { + /* the hidden keys were projected last: drop the trailing + * columns (by position — an output may carry the same name) */ + int64_t rc = ray_table_ncols(result); + int64_t keep_n = rc - n_hidden_sort; + ray_t* kept = keep_n >= 0 ? ray_table_new(keep_n > 0 ? keep_n : 1) : NULL; + for (int64_t c = 0; kept && !RAY_IS_ERR(kept) && c < keep_n; c++) + kept = ray_table_add_col(kept, ray_table_col_name(result, c), + ray_table_get_col_idx(result, c)); + if (kept && !RAY_IS_ERR(kept)) { ray_release(result); result = kept; } + else if (kept) ray_release(kept); + } + } /* A computed key takes its column name from its op's ext sym, which is * not a name (a const node's slot holds the literal; an expression node's * ext resolves to whatever shares its id) — name it the way the diff --git a/src/ops/sort.c b/src/ops/sort.c index b835f71c..3e8a02ed 100644 --- a/src/ops/sort.c +++ b/src/ops/sort.c @@ -3572,6 +3572,26 @@ ray_t* ray_sort(ray_t** cols, uint8_t* descs, uint8_t* nulls_first, return result; } +/* The column of `tbl` a scan sort key reads. When the sort runs directly + * over a SELECT and the key is one of that SELECT's columns, bind by + * POSITION: the projection names scans by their source column and + * expressions `_e`, so a hidden key or a renamed output can share a + * name with another projected column and a name lookup would read the + * wrong one. Any other key resolves by name. Borrowed. */ +static ray_t* sort_key_scan_col(ray_graph_t* g, ray_op_t* op, ray_t* tbl, + uint32_t key_id, int64_t key_sym) { + ray_op_t* child = g ? op_child(g, op, 0) : NULL; + if (child && child->opcode == OP_SELECT) { + ray_op_ext_t* se = find_ext(g, child->id); + if (se && se->base.opcode == OP_SELECT && + (int64_t)se->sort.n_cols == ray_table_ncols(tbl)) + for (uint32_t c = 0; c < se->sort.n_cols; c++) + if (se->sort.columns[c] == key_id) + return ray_table_get_col_idx(tbl, c); + } + return ray_table_get_col(tbl, key_sym); +} + ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { if (!tbl || RAY_IS_ERR(tbl)) return tbl; @@ -3603,7 +3623,7 @@ ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { ray_op_t* key_op = op_node(g, ext->sort.columns[k]); ray_op_ext_t* key_ext = find_ext(g, key_op->id); if (key_ext && key_ext->base.opcode == OP_SCAN) { - key_cols[k] = ray_table_get_col(tbl, key_ext->sym); + key_cols[k] = sort_key_scan_col(g, op, tbl, key_op->id, key_ext->sym); if (!key_cols[k]) { all_scan = 0; break; } } else { all_scan = 0; @@ -3655,7 +3675,7 @@ ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { ray_op_t* key_op = op_node(g, ext->sort.columns[k]); ray_op_ext_t* key_ext = find_ext(g, key_op->id); if (key_ext && key_ext->base.opcode == OP_SCAN) { - sort_vecs[k] = ray_table_get_col(tbl, key_ext->sym); + sort_vecs[k] = sort_key_scan_col(g, op, tbl, key_op->id, key_ext->sym); } else { ray_t* saved = g->table; g->table = tbl; diff --git a/src/ops/tblop.c b/src/ops/tblop.c index 375d4ae3..2e65da07 100644 --- a/src/ops/tblop.c +++ b/src/ops/tblop.c @@ -1341,6 +1341,9 @@ ray_t* ray_alter_fn(ray_t** args, int64_t n) { return ray_alter_set_cow_fail(original_var, cow_result, idx, val, name_sym); } var = cow_result; + /* A value written in place can break the order a `sorted` marker + * promises (consumers trust it without re-checking). */ + var->attrs &= (uint8_t)~RAY_ATTR_SORTED; /* Validate idx shape + (for the atom case) bounds BEFORE we * touch any state. The accelerator-index drop below would diff --git a/test/rfl/fused/fused_take_early_stop.rfl b/test/rfl/fused/fused_take_early_stop.rfl new file mode 100644 index 00000000..afd1895f --- /dev/null +++ b/test/rfl/fused/fused_take_early_stop.rfl @@ -0,0 +1,59 @@ +;; Fused filter + positional take (src/ops/fused_topk.c, ray_fused_take_select). +;; +;; `(select {cols from: T where: pred take: K})` with no ordering answers +;; with the first K passing rows (K > 0) or the last |K| (K < 0) in table +;; order, and stops scanning once they are known. `asc: key take: K` on a +;; column marked sorted (no nulls) goes the same way. The table spans +;; many scan chunks so the per-worker lists and the cutoff are exercised; +;; every answer is known by construction (x = row, c = x mod 1000). + +(set N 1000000) +(set X (til N)) +(set TK (table [x c s] (list X (% X 1000) (.attr.set 'sorted (til N))))) + +;; first K passing rows +(at (select {x: x from: TK where: (== c 7) take: 10}) 'x) -- [7 1007 2007 3007 4007 5007 6007 7007 8007 9007] +;; last |K| passing rows, in table order +(at (select {x: x from: TK where: (== c 7) take: -10}) 'x) -- [990007 991007 992007 993007 994007 995007 996007 997007 998007 999007] +;; AND predicate +(at (select {x: x from: TK where: (and (== c 7) (> x 500000)) take: 3}) 'x) -- [500007 501007 502007] +(at (select {x: x from: TK where: (and (== c 7) (< x 500000)) take: -2}) 'x) -- [498007 499007] +;; fewer passing rows than K, and none +(at (select {x: x from: TK where: (and (== c 7) (< x 3000)) take: 10}) 'x) -- [7 1007 2007] +(count (select {x: x from: TK where: (> x 5000000) take: 10})) -- 0 +;; all columns, aliased column +(count (select {from: TK where: (== c 7) take: 4})) -- 4 +(at (select {y: x from: TK where: (== c 7) take: 2}) 'y) -- [7 1007] +;; K beyond one worker's chunk of matches: identical to filter-then-take +(set FA (at (select {x: x from: TK where: (== c 7) take: 700}) 'x)) +(set FB (take (at (select {x: x from: TK where: (== c 7)}) 'x) 700)) +(count FA) -- 700 +(sum (as 'I64 (== FA FB))) -- 700 +(set LA (at (select {x: x from: TK where: (== c 7) take: -700}) 'x)) +(set LB (take (at (select {x: x from: TK where: (== c 7)}) 'x) -700)) +(sum (as 'I64 (== LA LB))) -- 700 +;; ascending take on a sorted key column = first K passing rows +(at (select {x: x from: TK where: (== c 7) asc: s take: 5}) 'x) -- [7 1007 2007 3007 4007] +;; the same key without the sorted marker still orders correctly +(at (select {x: x from: TK where: (== c 7) asc: x take: 3}) 'x) -- [7 1007 2007] +;; descending take on the sorted column stays on the ordered path +(at (select {x: x from: TK where: (== c 7) desc: s take: 3}) 'x) -- [999007 998007 997007] +;; a value written with alter set can break a `sorted` marker's order: the +;; marker is cleared, and the ascending take orders by value again +(set sv (.attr.set 'sorted (til 10))) +(alter 'sv set 8 -3) +(set TS (table [x s c] (list (til 10) sv (% (til 10) 2)))) +(at (select {x: x from: TS where: (== c 0) asc: s take: 2}) 'x) -- [8 0] +;; a take given as an expression is evaluated once +(set tcnt 0) +(count (select {x: x from: TK where: (== c 7) take: (do (set tcnt (+ tcnt 1)) 9)})) -- 9 +tcnt -- 1 +;; projections resolve left to right: an output that names an earlier +;; output's alias reads that output's value, not the source column — the +;; fused paths gather source columns and leave such a shape to the planner +(set TA (table [a b c] (list [3 1 2 4] [30 10 20 40] [0 0 0 0]))) +(at (select {b: a z: b from: TA where: (== c 0) take: 2}) 'z) -- [3 1] +(at (select {b: a z: b from: TA where: (== c 0) take: -2}) 'z) -- [2 4] +(at (select {b: a z: b from: TA where: (== c 0) asc: a take: 2}) 'z) -- [1 2] +(at (select {b: a z: b from: TA asc: a take: 2}) 'z) -- [1 2] +(at (select {b: b z: b from: TA where: (== c 0) take: 2}) 'z) -- [30 10] diff --git a/test/rfl/query/sort_key_not_projected.rfl b/test/rfl/query/sort_key_not_projected.rfl new file mode 100644 index 00000000..91ac3221 --- /dev/null +++ b/test/rfl/query/sort_key_not_projected.rfl @@ -0,0 +1,60 @@ +;; A select that sorts by a source column it does not output. The key is +;; carried through the projection for the sort and dropped from the result; +;; before, the sort looked the key up in the projected table and failed with +;; `nyi` (only the fused top-k shapes, which read the source table, worked). +(set T (table [a b c s] (list (til 100) (% (til 100) 7) (reverse (til 100)) (as 'SYMBOL (map (fn [i] (format "x%" (% i 13))) (til 100)))))) + +;; plain sort, no filter, no take +(take (at (select {a: a from: T asc: c}) 'a) 5) -- [99 98 97 96 95] +(cols (select {a: a from: T asc: c})) -- ['a] +(count (select {a: a from: T asc: c})) -- 100 +;; a filter the fused path does not take, with and without take +(at (select {a: a from: T where: (> (+ b 1) 3) asc: c take: 3}) 'a) -- [97 96 95] +(cols (select {a: a from: T where: (> (+ b 1) 3) asc: c take: 3})) -- ['a] +(take (at (select {a: a from: T where: (> (+ b 1) 3) desc: c}) 'a) 4) -- [3 4 5 6] +(at (select {a: a from: T where: (in s ['x1 'x2]) desc: c take: 3}) 'a) -- [1 2 14] +;; two keys, neither projected +(at (select {a: a from: T where: (> (+ b 1) 3) asc: [b c] take: 4}) 'a) -- [94 87 80 73] +;; a computed output alongside +(at (select {a: a d: (+ a 1) from: T where: (> b 2) desc: c take: 2}) 'd) -- [4 5] +(cols (select {a: a d: (+ a 1) from: T where: (> b 2) desc: c take: 2})) -- ['a 'd] +;; a key that is also an output is not duplicated +(cols (select {a: a c: c from: T asc: c take: 2})) -- ['a 'c] +(at (select {a: a c: c from: T asc: c take: 2}) 'c) -- [0 1] +;; an output that renames the key column: the sort reads that output (the +;; key binds to the projected column that scans it), nothing extra is +;; added or dropped +(at (select {b: a from: T asc: a}) 'b) -- (til 100) +(take (at (select {b: a from: T desc: a}) 'b) 3) -- [99 98 97] +(take (at (select {b: a from: T where: (> (+ b 1) 3) desc: a}) 'b) 3) -- [97 96 95] +(select {x: c y: a from: T desc: a take: [1 2]}) -- (table [x y] (list [1 2] [98 97])) + +;; Sort keys name SOURCE columns, never output aliases: `b: a asc: b` sorts +;; by the source b (hidden, since no output scans it) and returns a under +;; the name b; a computed output under the key's name changes nothing +(set HB (table [a b] (list [3 1 2] [0 2 1]))) +(at (select {b: a from: HB asc: b}) 'b) -- [3 2 1] +(cols (select {b: a from: HB asc: b})) -- ['b] +(at (select {b: a from: HB desc: b take: 2}) 'b) -- [1 2] +(at (select {b: (+ a 1) from: HB asc: b}) 'b) -- [4 3 2] +(at (select {c: b from: HB asc: b}) 'c) -- [0 1 2] +;; the key binds to the projected column by position, so a hidden key whose +;; name collides with the projection's internal name for a computed output +;; (`_e`) still sorts by the source column — with and without take, one +;; and two keys, either direction +(set HC (table [a _e0 _e1] (list [3 1 2 5 4] [0 2 1 4 3] [1 1 0 0 1]))) +(at (select {z: (+ a 10) from: HC asc: _e0}) 'z) -- [13 12 11 14 15] +(cols (select {z: (+ a 10) from: HC asc: _e0})) -- ['z] +(at (select {z: (+ a 10) from: HC asc: _e0 take: 3}) 'z) -- [13 12 11] +(at (select {z: (+ a 10) from: HC asc: [_e1 _e0]}) 'z) -- [12 15 13 11 14] +(at (select {z: (+ a 10) from: HC asc: [_e1 _e0] take: 3}) 'z) -- [12 15 13] +(at (select {z: (+ a 10) w: (* a 2) from: HC desc: _e1 asc: _e0}) 'z) -- [13 11 14 12 15] +(at (select {z: (+ a 10) w: (* a 2) from: HC desc: _e1 asc: _e0 take: 2}) 'w) -- [6 2] +;; more hidden keys than the former fixed capacity of 16 +(set i (til 1000)) +(set T17 (table [a k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16] (list i (- 0 (% (* i 3) 5)) (- 0 (% (* i 4) 5)) (- 0 (% (* i 5) 5)) (- 0 (% (* i 6) 5)) (- 0 (% (* i 7) 5)) (- 0 (% (* i 8) 5)) (- 0 (% (* i 9) 5)) (- 0 (% (* i 10) 5)) (- 0 (% (* i 11) 5)) (- 0 (% (* i 12) 5)) (- 0 (% (* i 13) 5)) (- 0 (% (* i 14) 5)) (- 0 (% (* i 15) 5)) (- 0 (% (* i 16) 5)) (- 0 (% (* i 17) 5)) (- 0 (% (* i 18) 5)) (- 0 (% (* i 19) 5))))) +(set R17 (select {z: (+ a 10) from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16]})) +(set M17 (select {z: (+ a 10) k0: k0 k1: k1 k2: k2 k3: k3 k4: k4 k5: k5 k6: k6 k7: k7 k8: k8 k9: k9 k10: k10 k11: k11 k12: k12 k13: k13 k14: k14 k15: k15 k16: k16 from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16]})) +(cols R17) -- ['z] +(at R17 'z) -- (at M17 'z) +(at (select {z: (+ a 10) from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16] take: 5}) 'z) -- (take (at M17 'z) 5) From fdc1a4c33459f570f704a82f3030d4678ab41102 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 11:11:29 +0300 Subject: [PATCH 45/51] perf(group): partitioned dense accumulate; derived-key grouping over the column's distinct values (#633) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(group): partition the slots when the dense accumulate cannot afford one set per worker; pin the SYM vocabulary once per accumulate The direct-array group path keeps one accumulator set per worker and bounds that state by the row count. Past the bound it ran ranged tasks over as many sets as the budget allowed — one set, i.e. a serial scan of every row, whenever the slots times the per-group state exceeded the rows, which a wide key with a nullable min and a strlen average over a large table does at once (5M slots, three aggregates: budget 1 out of 24 workers). Instead the slots are partitioned: one parallel pass computes each row's group and buckets its row id by slot span per worker (no shared writes), then one task per span drains every worker's bucket for it into the single accumulator set. No two tasks touch a slot, nothing is merged, and the state is one set however many workers run; the buckets cost four bytes a row. Row-range tasks stay for the small-slot case and as the fallback when the buckets cannot be allocated. Inside the accumulate, strlen over a SYM column pinned the FILE-domain vocabulary snapshot and ran ray_vec_is_null per row, and the lexical min/max compare pinned it per compare. The snapshot is pinned once per SYM aggregate column for the whole accumulate; null is the id-0 check. test/rfl/group/dense_slot_split.rfl: 200k rows over 130k slots with a nullable min/max/sum/avg/first/last, strlen over a SYM column with empty cells, a two-key composite and a where: selection — each against the same grouping forced onto the hash path by one far-away key. * perf(group): decide a derived-key grouping over the column's distinct values when every aggregate reads that column A grouping keyed by an expression over one SYM column C whose aggregates are count(C), min(C), max(C), sum(strlen C) or avg(strlen C) is decided by the distinct values of C: a group's count is the sum of its values' row counts, its min the smallest of its values, its strlen sum the length-weighted count. It used to be evaluated per row — a spread key over every row and an accumulate of every row's length and lexical compare — while the key expression itself already ran once per distinct value. derived_key_vocab_aggs now groups the rows by C once (a column key and a count, the engine's cheapest grouping, under the select's where:), runs the key once per distinct value (derived_key_str_chunks) and rewrites the aggregates over the per-value table: count → sum of counts (null rows count, as before), avg(strlen) → weighted length sum over the non-null count, min/max → over the values (null skipped). Sort and take clauses on output aliases pass through; the outputs come back in the written order under the computed key's usual name. Any other aggregate, a runtime-domain column or a small table keeps the row path. Interning the key strings cached dotted segments for every new value — a host or URL interned every one of its dot-separated parts as a symbol too, which was most of the cold-run cost. ray_sym_intern_batch_no_split interns them as values; a value that later appears as an identifier gets its segments on that intern. The per-value length pass runs on the workers. An `if` whose two STR branches are cheap and total (scans, constants, substrings, case/trim/length, add/sub/mul, comparisons, and/or/not) takes the eager descriptor arm instead of compacting the rows for each branch. test/rfl/group/derived_key_vocab_aggs.rfl: the five aggregates with and without where:, against the row path (forced by one aggregate over another column), the same table in memory, the null group's avg and min, sort/take on an alias, and the key name yielding to an alias. * fix(group): keep FIRST/LAST groupings past the slot budget on the serial row-order scan Several FIRST/LAST aggregates share one first_row/last_row per slot (a documented limitation of da_accum_row), so the answer for a null-mixed pair depends on the order the rows arrive in; the slot-partitioned drain walks the workers' buckets, not the rows, and made `last` differ by worker count. Past the budget such a grouping stays on the single accumulator in row order. * fix(group): partitioned accumulate and vocabulary rewrite independent of core count The slot-partitioned accumulate scattered rows by pool morsel into per-worker buckets, so a slot received its rows in scheduler order and float sums (and var/stddev) rounded differently from run to run. The scatter now runs nw tasks over contiguous row ranges in order, bucket row = task, and the drain walks them in order: each slot gets its rows in table order, the serial scan's accumulation. The vocabulary rewrite took its group order from the per-value count grouping, whose order depends on the core count. The counts are now put in domain-position order first (bitmap rank + scatter), so the output order is the same at every core count. Tests: dense_slot_split (1e16 + 1 - 1e16 + 1 per group sums to 1.0 in table order), derived_key_vocab_aggs (unsorted output order pinned). Both fail with their fix disabled. Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- src/ops/group.c | 192 +++++++++- src/ops/pivot.c | 51 ++- src/ops/query.c | 409 +++++++++++++++++++++- src/table/sym.c | 21 ++ src/table/sym.h | 3 + test/rfl/group/dense_slot_split.rfl | 54 +++ test/rfl/group/derived_key_vocab_aggs.rfl | 59 ++++ 7 files changed, 771 insertions(+), 18 deletions(-) create mode 100644 test/rfl/group/dense_slot_split.rfl create mode 100644 test/rfl/group/derived_key_vocab_aggs.rfl diff --git a/src/ops/group.c b/src/ops/group.c index 0e5448e5..bb737c18 100644 --- a/src/ops/group.c +++ b/src/ops/group.c @@ -7860,6 +7860,11 @@ typedef struct { ray_t** sym_strings; /* borrowed sym snapshot for strlen-on-SYM aggs */ uint32_t sym_count; int64_t da_n_scan; /* rows to scan (ranged tasks split it) */ + /* Per-agg raw vocabulary snapshot of a FILE-domain SYM column, pinned + * once for the whole accumulate: strlen and lexical min/max read the + * entry off the mapping instead of pinning (and checking null) per row. */ + ray_sym_domain_raw_t* agg_raw; + uint8_t* agg_raw_ok; } da_ctx_t; typedef struct { @@ -8571,6 +8576,47 @@ static void scalar_accum_fn(void* ctx, uint32_t worker_id, int64_t start, int64_ * Fast path for SUM/AVG-only queries: eliminates op-code dispatch and da_read_val * dual-write overhead. The branch on c->all_sum is perfectly predicted (invariant * across all rows). */ +/* strlen of agg column a at row r. STR: the descriptor's length (a null + * is the empty string, length 0). SYM: null is id 0; a FILE-domain entry's + * length is the u32 prefix in the pinned mapping; anything else resolves + * as group_strlen_at_cached does. */ +static inline int64_t da_strlen_at(const da_ctx_t* c, uint32_t a, int64_t r) { + const ray_t* col = c->agg_cols[a]; + if (col->type == RAY_STR) { + const ray_str_t* elems; const char* pool; (void)pool; + str_resolve(col, &elems, &pool); + return (int64_t)elems[r].len; + } + if (col->type == RAY_SYM) { + int64_t sid = ray_read_sym(ray_data((ray_t*)col), r, RAY_SYM, col->attrs); + if (sid == 0) return 0; + if (c->agg_raw_ok && c->agg_raw_ok[a] && sid > 0 && sid < c->agg_raw[a].count) { + size_t sl; + (void)ray_sym_domain_raw_str(&c->agg_raw[a], sid, &sl); + return (int64_t)sl; + } + } + return group_strlen_at_cached(col, r, c->sym_strings, c->sym_count); +} + +/* Lexical x < y for two cells of agg column a_idx (a SYM column): the + * pinned mapping when both positions are in the file prefix, sym_lex_lt + * otherwise. */ +static inline bool da_sym_lex_lt(const da_ctx_t* c, uint32_t a_idx, int64_t x, int64_t y) { + if (x == y) return false; + if (c->agg_raw_ok && c->agg_raw_ok[a_idx] && x >= 0 && y >= 0 && + x < c->agg_raw[a_idx].count && y < c->agg_raw[a_idx].count) { + size_t lx, ly; + const char* px = ray_sym_domain_raw_str(&c->agg_raw[a_idx], x, &lx); + const char* py = ray_sym_domain_raw_str(&c->agg_raw[a_idx], y, &ly); + size_t m = lx < ly ? lx : ly; + int r = m ? memcmp(px, py, m) : 0; + if (r != 0) return r < 0; + return lx < ly; + } + return sym_lex_lt(ray_sym_vec_domain(c->agg_cols[a_idx]), x, y); +} + static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64_t r) { uint8_t n_aggs = c->n_aggs; acc->count[gid]++; @@ -8593,9 +8639,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 } if (!c->agg_ptrs[a]) continue; if (c->agg_strlen && c->agg_strlen[a]) { - acc->sum[idx].i = wrap_add_i64( - acc->sum[idx].i, - group_strlen_at_cached(c->agg_cols[a], r, c->sym_strings, c->sym_count)); + acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, da_strlen_at(c, a, r)); if (nn) nn[idx]++; } else if (f64m & ((uint64_t)1 << a)) { /* NaN payload = null, skip from sum. */ @@ -8646,8 +8690,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 fv = prod_val_f64(&c->agg_prod[a], r); iv = (int64_t)fv; } else if (c->agg_strlen && c->agg_strlen[a]) { - iv = group_strlen_at_cached(c->agg_cols[a], r, - c->sym_strings, c->sym_count); + iv = da_strlen_at(c, a, r); fv = (double)iv; } else { uint8_t attrs = c->agg_cols[a] ? c->agg_cols[a]->attrs : 0; @@ -8739,7 +8782,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 } else if (c->agg_types[a] == RAY_SYM && !int_null) { /* Lex compare for SYM; INT64_MAX = "not seen yet". */ if (acc->min_val[idx].i == INT64_MAX || - sym_lex_lt(ray_sym_vec_domain(c->agg_cols[a]), iv, acc->min_val[idx].i)) + da_sym_lex_lt(c, a, iv, acc->min_val[idx].i)) acc->min_val[idx].i = iv; } else if (!int_null) { if (iv < acc->min_val[idx].i) acc->min_val[idx].i = iv; @@ -8750,7 +8793,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 if (fv == fv && fv > acc->max_val[idx].f) acc->max_val[idx].f = fv; } else if (c->agg_types[a] == RAY_SYM && !int_null) { if (acc->max_val[idx].i == INT64_MIN || - sym_lex_gt(ray_sym_vec_domain(c->agg_cols[a]), iv, acc->max_val[idx].i)) + da_sym_lex_lt(c, a, acc->max_val[idx].i, iv)) acc->max_val[idx].i = iv; } else if (!int_null) { if (iv > acc->max_val[idx].i) acc->max_val[idx].i = iv; @@ -8902,6 +8945,98 @@ static void da_accum_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t en #undef DA_PF_DIST } +/* ---- slot-partitioned accumulate -------------------------------------- + * For a pool with more workers than the per-worker slot budget allows. + * Pass 1 (nw tasks, contiguous row ranges in order): each row's group is + * computed once and its row id appended to the bucket of (range, task) + * with task = gid / span — no shared writes. Pass 2 (one task per slot span, + * ray_pool_dispatch_n): task t drains every worker's bucket t into the + * single accumulator set; no two tasks touch a slot, so nothing is merged + * and the state is one set however many workers run. The buckets hold + * int32 row ids (the caller admits tables below INT32_MAX rows). */ +typedef struct { int32_t* data; int64_t len, cap; ray_t* hdr; } da_bucket_t; +typedef struct { + da_ctx_t* c; + da_bucket_t* buckets; /* [nw * k] */ + uint32_t nw, k, span; + int64_t n_scan; + _Atomic(int) oom; +} da_part_ctx_t; + +static bool da_bucket_push(da_bucket_t* b, int32_t v) { + if (b->len == b->cap) { + int64_t ncap = b->cap ? b->cap * 2 : 1024; + ray_t* nh = NULL; + int32_t* nd = (int32_t*)scratch_alloc(&nh, (size_t)ncap * sizeof(int32_t)); + if (!nd) return false; + if (b->len) memcpy(nd, b->data, (size_t)b->len * sizeof(int32_t)); + if (b->hdr) scratch_free(b->hdr); + b->data = nd; b->hdr = nh; b->cap = ncap; + } + b->data[b->len++] = v; + return true; +} + +/* Task `t` of nw scans the t-th contiguous row range into bucket row t: + * draining rows 0..nw-1 in order then hands each slot its rows in table + * order, so the accumulation (float sums included) is the serial scan's, + * whatever the scheduling. */ +static void da_part_scatter_fn(void* raw, uint32_t wid, int64_t task, int64_t task_end) { + (void)wid; (void)task_end; + da_part_ctx_t* p = (da_part_ctx_t*)raw; + da_ctx_t* c = p->c; + da_bucket_t* mine = &p->buckets[(size_t)task * p->k]; + int64_t start = p->n_scan * task / p->nw; + int64_t end = p->n_scan * (task + 1) / p->nw; + const int64_t* match_idx = c->match_idx; + for (int64_t i = start; i < end; i++) { + int64_t r = match_idx ? match_idx[i] : i; + if (!match_idx && c->rowsel && !group_rowsel_pass(c->rowsel, r)) continue; + uint32_t t = (uint32_t)da_composite_gid(c, r) / p->span; + if (!da_bucket_push(&mine[t], (int32_t)r)) { + atomic_store_explicit(&p->oom, 1, memory_order_relaxed); + return; + } + } +} + +static void da_part_drain_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + da_part_ctx_t* p = (da_part_ctx_t*)raw; + da_ctx_t* c = p->c; + da_accum_t* acc = &c->accums[0]; + uint32_t t = (uint32_t)start; + for (uint32_t w = 0; w < p->nw; w++) { + const da_bucket_t* b = &p->buckets[(size_t)w * p->k + t]; + for (int64_t j = 0; j < b->len; j++) { + int64_t r = b->data[j]; + da_accum_row(c, acc, da_composite_gid(c, r), r); + } + } +} + +/* Returns false (accumulators untouched) when the buckets could not be + * allocated; the caller then falls back to the plain scan. */ +static bool da_accum_partitioned(da_ctx_t* c, ray_pool_t* pool, uint32_t k, int64_t n_scan) { + uint32_t nw = ray_pool_total_workers(pool); + uint32_t span = (c->n_slots + k - 1) / k; + if (span == 0) span = 1; + k = (c->n_slots + span - 1) / span; + ray_t* bh = NULL; + da_bucket_t* buckets = (da_bucket_t*)scratch_calloc(&bh, (size_t)nw * k * sizeof(da_bucket_t)); + if (!buckets) return false; + da_part_ctx_t p = { .c = c, .buckets = buckets, .nw = nw, .k = k, .span = span, + .n_scan = n_scan }; + atomic_store_explicit(&p.oom, 0, memory_order_relaxed); + ray_pool_dispatch_n(pool, da_part_scatter_fn, &p, nw); + bool ok = atomic_load_explicit(&p.oom, memory_order_relaxed) == 0; + if (ok) ray_pool_dispatch_n(pool, da_part_drain_fn, &p, k); + for (size_t i = 0; i < (size_t)nw * k; i++) + if (buckets[i].hdr) scratch_free(buckets[i].hdr); + scratch_free(bh); + return ok; +} + /* One task per accumulator (ray_pool_dispatch_n): task i scans the i-th * of n_accums equal row ranges into accums[i], whichever worker runs it. */ static void da_accum_task_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { @@ -12670,9 +12805,26 @@ da_path:; * task per accumulator, whichever worker runs it) instead of * collapsing to a serial scan of every row. */ bool da_ranged = false; - if ((uint64_t)da_n_workers > max_workers) { - da_n_workers = (uint32_t)max_workers; - da_ranged = da_n_workers > 1; + uint32_t da_part_tasks = 0; + if ((uint64_t)da_n_workers > max_workers && da_has_first_last) { + /* Several FIRST/LAST aggregates share one first_row/last_row + * per slot (see da_accum_row): the answer for a null-mixed + * pair depends on the order the rows arrive in, so past the + * budget those keep the serial scan in row order. */ + da_n_workers = 1; + } else if ((uint64_t)da_n_workers > max_workers) { + if (n_slots >= 2 * da_n_workers && nrows <= INT32_MAX) { + /* Partition the SLOTS instead: one pass buckets the row + * ids by slot span, then one task per span drains its + * buckets into the single accumulator set. Twice as + * many spans as workers evens out skewed groups. */ + da_part_tasks = 2 * da_n_workers; + if (da_part_tasks > n_slots) da_part_tasks = n_slots; + da_n_workers = 1; + } else { + da_n_workers = (uint32_t)max_workers; + da_ranged = da_n_workers > 1; + } } ray_t* accums_hdr; @@ -12801,13 +12953,31 @@ da_path:; .rowsel = rowsel, .da_n_scan = n_scan, }; + /* Pin each SYM agg column's vocabulary once for the accumulate. */ + ray_t* agg_raw_hdr = NULL; + if (n_aggs > 0) { + char* rm = (char*)scratch_calloc(&agg_raw_hdr, + (size_t)n_aggs * (sizeof(ray_sym_domain_raw_t) + 1)); + if (rm) { + da_ctx.agg_raw = (ray_sym_domain_raw_t*)rm; + da_ctx.agg_raw_ok = (uint8_t*)(rm + (size_t)n_aggs * sizeof(ray_sym_domain_raw_t)); + for (uint32_t a = 0; a < n_aggs; a++) + if (agg_vecs[a] && agg_vecs[a]->type == RAY_SYM) + da_ctx.agg_raw_ok[a] = ray_sym_domain_raw_pin( + ray_sym_vec_domain(agg_vecs[a]), &da_ctx.agg_raw[a]) ? 1 : 0; + } + } - if (da_ranged) + if (da_part_tasks > 0) { + if (!da_accum_partitioned(&da_ctx, da_pool, da_part_tasks, n_scan)) + da_accum_fn(&da_ctx, 0, 0, n_scan); + } else if (da_ranged) ray_pool_dispatch_n(da_pool, da_accum_task_fn, &da_ctx, da_n_workers); else if (da_n_workers > 1) ray_pool_dispatch(da_pool, da_accum_fn, &da_ctx, n_scan); else da_accum_fn(&da_ctx, 0, 0, n_scan); + if (agg_raw_hdr) scratch_free(agg_raw_hdr); /* Merge target is always accums[0] */ da_accum_t* merged = &accums[0]; diff --git a/src/ops/pivot.c b/src/ops/pivot.c index ca058e5c..2084ec31 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -845,6 +845,42 @@ static bool if_branch_trivial(ray_graph_t* g, ray_op_t* op) { return false; } +/* A branch that is cheap to evaluate over ALL rows and total (no row can + * fail): scans, constants, substrings, string case/trim/length, add/sub/mul, + * comparisons, and/or/not, and `if` over such — a descriptor-only STR + * result then costs less than the selected path's compaction of the rows + * for each branch (a serial gather of the branch's columns) plus its + * scatter. Searches (str-find, like, replace, concat) and everything else + * are worth restricting to the rows that need them. */ +static bool if_branch_cheap(ray_graph_t* g, ray_op_t* op, int depth) { + if (!op || depth > 32) return false; + switch (op->opcode) { + case OP_ALIAS: case OP_MATERIALIZE: + return if_branch_cheap(g, op_child(g, op, 0), depth + 1); + case OP_SCAN: return true; + case OP_CONST: { + ray_op_ext_t* ext = find_ext(g, op->id); + return ext && ext->literal && ray_is_atom(ext->literal); + } + case OP_SUBSTR: case OP_IF: { + ray_op_ext_t* ext = find_ext(g, op->id); + if (!ext || ext->third_in >= g->node_count) return false; + for (int k = 0; k < op->arity && k < 2; k++) + if (!if_branch_cheap(g, op_child(g, op, k), depth + 1)) return false; + return if_branch_cheap(g, op_node(g, ext->third_in), depth + 1); + } + case OP_STRLEN: case OP_UPPER: case OP_LOWER: case OP_TRIM: + case OP_ADD: case OP_SUB: case OP_MUL: case OP_NEG: case OP_ABS: + case OP_EQ: case OP_NE: case OP_LT: case OP_LE: case OP_GT: case OP_GE: + case OP_AND: case OP_OR: case OP_NOT: case OP_ISNULL: + for (int k = 0; k < op->arity && k < 2; k++) + if (!if_branch_cheap(g, op_child(g, op, k), depth + 1)) return false; + return true; + default: + return false; + } +} + /* Is a trivial branch of static type `bt` filled CORRECTLY by the eager * elementwise path for result type `out`? Mixed numeric/string branch * combinations (e.g. `(if c n1 s2)` with I64 + STR) rely on the selected @@ -889,11 +925,16 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { * answer exactly as it would, or the same expression gets two types * depending on the worker count. Where it is not, this arm is the only * one there is and may report the type the rows actually have. */ - bool eager_possible = op->out_type != RAY_STR && - if_branch_trivial(g, then_op) && - if_branch_trivial(g, else_op) && - if_type_eager_ok(then_op->out_type, op->out_type) && - if_type_eager_ok(else_op->out_type, op->out_type); + bool eager_possible = (op->out_type != RAY_STR && + if_branch_trivial(g, then_op) && + if_branch_trivial(g, else_op) && + if_type_eager_ok(then_op->out_type, op->out_type) && + if_type_eager_ok(else_op->out_type, op->out_type)) || + /* two STR vector branches that are cheap and total: + * the eager arm picks descriptors over one pass */ + (op->out_type == RAY_STR && + then_op->out_type == RAY_STR && else_op->out_type == RAY_STR && + if_branch_cheap(g, then_op, 0) && if_branch_cheap(g, else_op, 0)); { ray_pool_t* rp = ray_pool_get(); if (rp && rp->n_workers > 0 && eager_possible) diff --git a/src/ops/query.c b/src/ops/query.c index 0a1ce142..22f17664 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -3292,7 +3292,10 @@ static bool derived_key_intern_chunk(ray_t* kc, int64_t n, int64_t* out) { s = (s + 1) & mask; } } - if (ray_sym_intern_batch(dhsh, dstr, dlen, nd, did) < 0) { scratch_free(hdr); return false; } + /* Key strings are values, not names: interned without the dotted- + * segment caching that a name with '.' in it gets (a host or URL would + * otherwise intern every one of its segments too). */ + if (ray_sym_intern_batch_no_split(dhsh, dstr, dlen, nd, did) < 0) { scratch_free(hdr); return false; } for (int64_t i = 0; i < n; i++) out[i] = did[rep[i]]; scratch_free(hdr); return true; @@ -3392,6 +3395,397 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom ray_sym_domain_raw_unpin(dom); return NULL; } + +/* ---- aggregates over the vocabulary ------------------------------------- + * A grouping whose key is derived from one SYM column C and whose every + * aggregate reads only C — count(C), min(C), max(C), sum/avg(strlen C) — + * is decided by the distinct values of C, not by the rows: a group's count + * is the sum of its values' row counts, its min the smallest of its values, + * its strlen sum the length-weighted count. So: one grouping of the rows by + * C itself (the engine's fastest shape: a column key, a count) gives the + * distinct values with their counts; the key expression runs once per + * distinct value (derived_key_str_chunks); and the aggregates are rewritten + * over that per-value table. Nothing per row is evaluated or accumulated + * beyond the count histogram. count(C) counts null rows too, avg(strlen C) + * divides by the non-null rows, min/max skip null — the row-wise rules. + * Returns NULL for every shape it does not take (the caller runs the row + * path). */ +enum { DKV_COUNT = 1, DKV_MIN, DKV_MAX, DKV_SUM_LEN, DKV_AVG_LEN }; +#define DKV_MAX_AGGS 32 + +static ray_t* dkv_sym(const char* name) { return ray_sym(ray_sym_intern(name, strlen(name))); } + +/* (head a) / (head a b) as an expression list; the arguments are consumed. */ +static ray_t* dkv_call(const char* head, ray_t* a, ray_t* b) { + ray_t* l = ray_list_new(b ? 3 : 2); + if (!l || RAY_IS_ERR(l)) { if (a) ray_release(a); if (b) ray_release(b); return NULL; } + ray_t* h = dkv_sym(head); + l = ray_list_append(l, h); ray_release(h); + l = ray_list_append(l, a); ray_release(a); + if (b) { l = ray_list_append(l, b); ray_release(b); } + if (!l || RAY_IS_ERR(l)) { if (l) ray_error_free(l); return NULL; } + return l; +} + +/* (select {keys[i]: vals[i] …}); keys are runtime sym ids, vals consumed. */ +static ray_t* dkv_select(const int64_t* keys, ray_t** vals, int64_t n) { + ray_t* kv = ray_sym_vec_new(RAY_SYM_W64, n); + ray_t* vl = ray_list_new(n); + if (!kv || RAY_IS_ERR(kv) || !vl || RAY_IS_ERR(vl)) { + for (int64_t i = 0; i < n; i++) if (vals[i]) ray_release(vals[i]); + if (kv && !RAY_IS_ERR(kv)) ray_release(kv); + if (vl && !RAY_IS_ERR(vl)) ray_release(vl); + return NULL; + } + kv->len = n; + for (int64_t i = 0; i < n; i++) { + ((int64_t*)ray_data(kv))[i] = keys[i]; + vl = ray_list_append(vl, vals[i]); + ray_release(vals[i]); + } + ray_t* d = ray_dict_new(kv, vl); + if (!d || RAY_IS_ERR(d)) { if (d) ray_error_free(d); return NULL; } + ray_t* sel = ray_list_new(2); + if (!sel || RAY_IS_ERR(sel)) { ray_release(d); return NULL; } + ray_t* h = dkv_sym("select"); + sel = ray_list_append(sel, h); ray_release(h); + sel = ray_list_append(sel, d); ray_release(d); + return sel; +} + +static bool dkv_is_col(ray_t* e, int64_t col) { + return e && e->type == -RAY_SYM && !(e->attrs & ATTR_QUOTED) && e->i64 == col; +} + +/* Which decomposable aggregate over C is `v`, or 0. */ +static int dkv_agg_kind(ray_t* v, int64_t col) { + if (!v || v->type != RAY_LIST || ray_len(v) != 2) return 0; + ray_t** e = (ray_t**)ray_data(v); + if (!e[0] || e[0]->type != -RAY_SYM) return 0; + ray_t* hs = ray_sym_str(e[0]->i64); + if (!hs) return 0; + const char* h = ray_str_ptr(hs); size_t hl = ray_str_len(hs); + bool is = false; + #define DKV_IS(lit) (hl == sizeof(lit) - 1 && memcmp(h, lit, hl) == 0) + if (DKV_IS("count") || DKV_IS("min") || DKV_IS("max")) { + if (!dkv_is_col(e[1], col)) return 0; + return DKV_IS("count") ? DKV_COUNT : DKV_IS("min") ? DKV_MIN : DKV_MAX; + } + is = DKV_IS("sum") || DKV_IS("avg"); + if (!is) return 0; + ray_t* a = e[1]; + if (!a || a->type != RAY_LIST || ray_len(a) != 2) return 0; + ray_t** ae = (ray_t**)ray_data(a); + if (!ae[0] || ae[0]->type != -RAY_SYM) return 0; + ray_t* as = ray_sym_str(ae[0]->i64); + if (!as || ray_str_len(as) != 6 || memcmp(ray_str_ptr(as), "strlen", 6) != 0) return 0; + if (!dkv_is_col(ae[1], col)) return 0; + return DKV_IS("sum") ? DKV_SUM_LEN : DKV_AVG_LEN; + #undef DKV_IS +} + +/* Per distinct value (workers): its non-null row count and its + * length-weighted row count, the length read off the pinned mapping. A + * position past the file prefix is marked (lenw = INT64_MIN) and resolved + * through the domain on the calling thread. */ +typedef struct { + const void* hc; + uint8_t attrs; + const int64_t* cnt; + int64_t* nn; + int64_t* lenw; + struct ray_sym_domain_s* dom; + ray_sym_domain_raw_t raw; + bool raw_ok; + atomic_int late; +} dkv_len_ctx_t; + +static void dkv_len_fn(void* vctx, uint32_t wid, int64_t lo, int64_t hi) { + (void)wid; + dkv_len_ctx_t* c = (dkv_len_ctx_t*)vctx; + int late = 0; + for (int64_t i = lo; i < hi; i++) { + int64_t pos = ray_read_sym(c->hc, i, RAY_SYM, c->attrs); + c->nn[i] = pos > 0 ? c->cnt[i] : 0; + if (pos <= 0) { c->lenw[i] = 0; continue; } + if (c->raw_ok && pos < c->raw.count) { + size_t l; (void)ray_sym_domain_raw_str(&c->raw, pos, &l); + c->lenw[i] = c->cnt[i] * (int64_t)l; + } else { + c->lenw[i] = INT64_MIN; + late++; + } + } + if (late) atomic_fetch_add_explicit(&c->late, late, memory_order_relaxed); +} + +/* H (distinct values of C with their counts) in C's domain-position order. + * The grouping that produced H emits in an order that depends on the core + * count; the rewrite's output order follows H, and a derived-key grouping + * on the row path comes out in the same order at every core count (the + * first-seen order of the key values, which follows the positions of the + * values they come from). Positions are distinct, so the order is the + * rank of each position among those present: a bitmap over the domain, + * block popcounts, then a scatter. Returns a new table, or NULL (H kept). */ +static ray_t* dkv_order_by_position(ray_t* H, int64_t dom_count) { + ray_t* Hc = ray_table_get_col_idx(H, 0); + ray_t* Hn = ray_table_get_col_idx(H, 1); + int64_t du = ray_table_nrows(H); + if (dom_count <= 0 || du <= 1) return NULL; + int64_t nw = (dom_count + 63) / 64; + int64_t nb = (nw + 63) / 64; /* rank blocks of 64 words */ + ray_t *bh = NULL, *rh = NULL; + uint64_t* bits = (uint64_t*)scratch_calloc(&bh, (size_t)nw * sizeof(uint64_t)); + int64_t* rank = (int64_t*)scratch_alloc(&rh, (size_t)(nb + 1) * sizeof(int64_t)); + if (!bits || !rank) { scratch_free(bh); scratch_free(rh); return NULL; } + const void* cd = ray_data(Hc); + for (int64_t i = 0; i < du; i++) { + int64_t pos = ray_read_sym(cd, i, RAY_SYM, Hc->attrs); + if (pos < 0 || pos >= dom_count) { scratch_free(bh); scratch_free(rh); return NULL; } + bits[pos >> 6] |= (uint64_t)1 << (pos & 63); + } + int64_t run = 0; + for (int64_t b = 0; b < nb; b++) { + rank[b] = run; + int64_t w1 = (b + 1) * 64 < nw ? (b + 1) * 64 : nw; + for (int64_t w = b * 64; w < w1; w++) run += __builtin_popcountll(bits[w]); + } + ray_t* nc = ray_sym_vec_new(Hc->attrs & RAY_SYM_W_MASK, du); + ray_t* nn = ray_vec_new(RAY_I64, du); + if (!nc || RAY_IS_ERR(nc) || !nn || RAY_IS_ERR(nn)) { + if (nc && !RAY_IS_ERR(nc)) ray_release(nc); + if (nn && !RAY_IS_ERR(nn)) ray_release(nn); + scratch_free(bh); scratch_free(rh); + return NULL; + } + ray_sym_vec_adopt_domain(nc, Hc); + nc->len = du; nn->len = du; + void* ncd = ray_data(nc); + int64_t* nnd = (int64_t*)ray_data(nn); + const int64_t* hnd = (const int64_t*)ray_data(Hn); + for (int64_t i = 0; i < du; i++) { + int64_t pos = ray_read_sym(cd, i, RAY_SYM, Hc->attrs); + int64_t w = pos >> 6; + int64_t r = rank[w >> 6]; + for (int64_t x = (w >> 6) * 64; x < w; x++) r += __builtin_popcountll(bits[x]); + r += __builtin_popcountll(bits[w] & (((uint64_t)1 << (pos & 63)) - 1)); + ray_write_sym(ncd, r, (uint64_t)pos, RAY_SYM, nc->attrs); + nnd[r] = hnd[i]; + } + if (Hc->attrs & RAY_ATTR_HAS_NULLS) nc->attrs |= RAY_ATTR_HAS_NULLS; + scratch_free(bh); scratch_free(rh); + ray_t* out = ray_table_new(2); + if (!out || RAY_IS_ERR(out)) { ray_release(nc); ray_release(nn); return NULL; } + out = ray_table_add_col(out, ray_table_col_name(H, 0), nc); + out = ray_table_add_col(out, ray_table_col_name(H, 1), nn); + ray_release(nc); ray_release(nn); + if (!out || RAY_IS_ERR(out)) return NULL; + return out; +} + +static ray_t* derived_key_vocab_aggs(ray_t* tbl, ray_t* by_expr, ray_t* where_expr, + ray_t** dict_elems, int64_t dict_n, + int64_t from_id, int64_t by_id, int64_t where_id, + int64_t take_id, int64_t asc_id, int64_t desc_id, + int64_t nearest_id) { + if (!by_expr || by_expr->type != RAY_LIST || !tbl) return NULL; + int64_t ref_syms[2]; + if (collect_col_refs(by_expr, tbl, ref_syms, 2, 0) != 1) return NULL; + int64_t col = ref_syms[0]; + int64_t bound[32]; + if (!derived_key_expr_ok(by_expr, tbl, col, bound, 0)) return NULL; + ray_t* C = ray_table_get_col(tbl, col); + int64_t nrows = ray_table_nrows(tbl); + if (!C || C->type != RAY_SYM || !ray_is_vec(C) || C->len != nrows || nrows < 65536) return NULL; + struct ray_sym_domain_s* dom = ray_sym_vec_domain(C); + if (!dom || dom == ray_sym_runtime_domain()) return NULL; + + /* Every output is one of the decomposable aggregates over C; sort/take + * clauses may only name output aliases. */ + int64_t alias[DKV_MAX_AGGS]; int kind[DKV_MAX_AGGS]; int n_aggs = 0; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid == from_id || kid == by_id || kid == where_id || kid == take_id || + kid == asc_id || kid == desc_id) continue; + if (kid == nearest_id) return NULL; + if (n_aggs >= DKV_MAX_AGGS) return NULL; + int k = dkv_agg_kind(dict_elems[i + 1], col); + if (!k) return NULL; + alias[n_aggs] = kid; kind[n_aggs] = k; n_aggs++; + } + if (n_aggs == 0) return NULL; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* v = dict_elems[i + 1]; + int64_t nk = (v && v->type == -RAY_SYM) ? 1 : (v && v->type == RAY_SYM) ? ray_len(v) : -1; + if (nk < 0) return NULL; + for (int64_t j = 0; j < nk; j++) { + int64_t sid = (v->type == -RAY_SYM) ? v->i64 : sym_cell_runtime_id(v, j); + bool ok = false; + for (int a = 0; a < n_aggs; a++) if (alias[a] == sid) ok = true; + if (!ok) return NULL; + } + } + + /* 1. the rows grouped by C itself: distinct values with their counts */ + int64_t s_t = ray_sym_intern("__dkv_t", 7), s_cnt = ray_sym_intern("__dkv_cnt", 9); + int64_t s_s = ray_sym_intern("__dkv_s", 7), s_k = ray_sym_intern("__dkv_k", 7); + int64_t s_ref = ray_sym_intern("__dkv_ref", 9), s_nn = ray_sym_intern("__dkv_nn", 8); + int64_t s_lenw = ray_sym_intern("__dkv_lenw", 10); + ray_t* H = NULL; + { + int64_t keys[4]; ray_t* vals[4]; int64_t n = 0; + keys[n] = from_id; vals[n++] = ray_sym(s_t); + keys[n] = by_id; vals[n++] = ray_sym(col); + keys[n] = s_cnt; vals[n++] = dkv_call("count", ray_sym(col), NULL); + if (where_expr) { keys[n] = where_id; ray_retain(where_expr); vals[n++] = where_expr; } + for (int64_t i = 0; i < n; i++) if (!vals[i]) { for (int64_t j = 0; j < n; j++) if (vals[j]) ray_release(vals[j]); return NULL; } + ray_t* q = dkv_select(keys, vals, n); + if (!q) return NULL; + if (ray_env_push_query_scope() != RAY_OK) { ray_release(q); return NULL; } + ray_env_set_query_local(s_t, tbl); + H = ray_eval(q); + ray_env_pop_scope(); + ray_release(q); + } + if (!H || RAY_IS_ERR(H) || H->type != RAY_TABLE || ray_table_ncols(H) != 2) { + if (H && !RAY_IS_ERR(H)) ray_release(H); else if (H) ray_error_free(H); + return NULL; + } + ray_t* Hc = ray_table_get_col_idx(H, 0); + ray_t* Hn = ray_table_get_col_idx(H, 1); + int64_t du = ray_table_nrows(H); + if (!Hc || !Hn || Hc->type != RAY_SYM || Hn->type != RAY_I64 || du <= 0 || + ray_sym_vec_domain(Hc) != dom) { ray_release(H); return NULL; } + { + ray_t* Ho = dkv_order_by_position(H, ray_sym_domain_count(dom)); + if (Ho) { + ray_release(H); + H = Ho; + Hc = ray_table_get_col_idx(H, 0); + Hn = ray_table_get_col_idx(H, 1); + } + } + + /* 2. the key once per distinct value */ + ray_t* key_dom = derived_key_str_chunks(by_expr, col, Hc, dom, du); + if (!key_dom || RAY_IS_ERR(key_dom)) { if (key_dom) ray_error_free(key_dom); ray_release(H); return NULL; } + + /* 3. the per-value table: key, count, the value, its non-null count, + * its length-weighted count */ + ray_t* S = NULL; + { + ray_t* nn = ray_vec_new(RAY_I64, du); + ray_t* lenw = ray_vec_new(RAY_I64, du); + if (!nn || RAY_IS_ERR(nn) || !lenw || RAY_IS_ERR(lenw)) { + if (nn && !RAY_IS_ERR(nn)) ray_release(nn); + if (lenw && !RAY_IS_ERR(lenw)) ray_release(lenw); + ray_release(key_dom); ray_release(H); return NULL; + } + nn->len = du; lenw->len = du; + dkv_len_ctx_t lc = { + .hc = ray_data(Hc), .attrs = Hc->attrs, .cnt = (const int64_t*)ray_data(Hn), + .nn = (int64_t*)ray_data(nn), .lenw = (int64_t*)ray_data(lenw), .dom = dom, + }; + lc.raw_ok = ray_sym_domain_raw_pin(dom, &lc.raw); + atomic_store_explicit(&lc.late, 0, memory_order_relaxed); + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, du, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, dkv_len_fn, &lc, du); + else + dkv_len_fn(&lc, 0, 0, du); + if (atomic_load_explicit(&lc.late, memory_order_relaxed)) { + /* positions past the file prefix: resolved through the domain on + * the calling thread */ + for (int64_t i = 0; i < du; i++) { + if (lc.lenw[i] != INT64_MIN) continue; + int64_t pos = ray_read_sym(lc.hc, i, RAY_SYM, lc.attrs); + ray_t* a = ray_sym_domain_str(dom, pos); + lc.lenw[i] = lc.cnt[i] * (a ? (int64_t)ray_str_len(a) : 0); + } + } + S = ray_table_new(5); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_k, key_dom); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_cnt, Hn); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_ref, Hc); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_nn, nn); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_lenw, lenw); + ray_release(nn); ray_release(lenw); + } + ray_release(key_dom); + ray_release(H); + if (!S || RAY_IS_ERR(S)) { if (S) ray_error_free(S); return NULL; } + + /* 4. the aggregates rewritten over the per-value table, sort/take as + * written */ + ray_t* R = NULL; + { + int64_t keys[DKV_MAX_AGGS + 8]; ray_t* vals[DKV_MAX_AGGS + 8]; int64_t n = 0; + bool bad = false; + for (int64_t i = 0; i + 1 < dict_n && !bad; i += 2) { + int64_t kid = dict_elems[i]->i64; + ray_t* v = NULL; + if (kid == from_id) v = ray_sym(s_s); + else if (kid == by_id) v = ray_sym(s_k); + else if (kid == where_id) continue; + else if (kid == take_id || kid == asc_id || kid == desc_id) { v = dict_elems[i + 1]; ray_retain(v); } + else { + int k = 0; + for (int a = 0; a < n_aggs; a++) if (alias[a] == kid) k = kind[a]; + switch (k) { + case DKV_COUNT: v = dkv_call("sum", ray_sym(s_cnt), NULL); break; + case DKV_MIN: v = dkv_call("min", ray_sym(s_ref), NULL); break; + case DKV_MAX: v = dkv_call("max", ray_sym(s_ref), NULL); break; + case DKV_SUM_LEN: v = dkv_call("sum", ray_sym(s_lenw), NULL); break; + case DKV_AVG_LEN: v = dkv_call("/", dkv_call("sum", ray_sym(s_lenw), NULL), + dkv_call("sum", ray_sym(s_nn), NULL)); break; + default: v = NULL; + } + } + if (!v) { bad = true; break; } + keys[n] = kid; vals[n++] = v; + } + if (bad) { for (int64_t j = 0; j < n; j++) ray_release(vals[j]); ray_release(S); return NULL; } + ray_t* q = dkv_select(keys, vals, n); + if (!q) { ray_release(S); return NULL; } + if (ray_env_push_query_scope() != RAY_OK) { ray_release(q); ray_release(S); return NULL; } + ray_env_set_query_local(s_s, S); + R = ray_eval(q); + ray_env_pop_scope(); + ray_release(q); + } + ray_release(S); + if (!R || RAY_IS_ERR(R)) { if (R) ray_error_free(R); return NULL; } + if (ray_is_lazy(R)) R = ray_lazy_materialize(R); + if (!R || RAY_IS_ERR(R) || R->type != RAY_TABLE || ray_table_ncols(R) == 0) { + if (R && !RAY_IS_ERR(R)) ray_release(R); else if (R) ray_error_free(R); + return NULL; + } + /* The engine emits a post-aggregate expression (the avg quotient) + * after the plain aggregates: put the outputs back in the order they + * were written, key first. */ + int64_t rc = ray_table_ncols(R); + if (rc != n_aggs + 1) { ray_release(R); return NULL; } + ray_t* O = ray_table_new(rc); + if (!O || RAY_IS_ERR(O)) { if (O) ray_error_free(O); ray_release(R); return NULL; } + O = ray_table_add_col(O, ray_table_col_name(R, 0), ray_table_get_col_idx(R, 0)); + for (int a = 0; a < n_aggs && O && !RAY_IS_ERR(O); a++) { + ray_t* col = ray_table_get_col(R, alias[a]); + if (!col) { ray_release(O); ray_release(R); return NULL; } + O = ray_table_add_col(O, alias[a], col); + } + ray_release(R); + if (!O || RAY_IS_ERR(O)) { if (O) ray_error_free(O); return NULL; } + /* the key column is named as the row path names a computed key */ + int64_t kname = derived_key_name(by_expr); + for (int64_t c = 1; c < rc; c++) + if (ray_table_col_name(O, c) == kname) { kname = ray_sym_intern("key", 3); break; } + ray_table_set_col_name(O, 0, kname); + agg_route_note_key_domain(); + return O; +} + static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { if (!by_expr || by_expr->type != RAY_LIST || !tbl) return NULL; int64_t ref_syms[2]; @@ -10472,7 +10866,18 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { } else { /* Single key expression. Over a lone SYM column evaluate it per * distinct symbol and feed the spread key as a constant node, - * named the way the eval-level path names a computed key. */ + * named the way the eval-level path names a computed key. When + * every aggregate reads that column too, the whole grouping is + * decided over the distinct values (derived_key_vocab_aggs). */ + { + ray_t* vres = derived_key_vocab_aggs(tbl, by_expr, where_expr, dict_elems, dict_n, + from_id, by_id, where_id, take_id, + asc_id, desc_id, nearest_id); + if (vres) { + ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); + return vres; + } + } ray_t* dom_key = derived_key_over_sym_domain(by_expr, tbl); if (dom_key) { key_ops[0] = ray_const_vec(g, dom_key); diff --git a/src/table/sym.c b/src/table/sym.c index 2843827d..7c55cdf1 100644 --- a/src/table/sym.c +++ b/src/table/sym.c @@ -824,6 +824,27 @@ int64_t ray_sym_intern_batch(const uint32_t* hashes, const char* const* strs, return 0; } +/* Intern n pre-hashed, already-deduplicated strings under one lock without + * caching dotted segments: for VALUES (a derived group key's strings — hosts, + * URLs), which are not namespace paths. Such a symbol is not dotted until + * the same string is interned as a name (the probe-hit path of + * sym_intern_nolock caches the segments then) or ray_sym_rebuild_segments + * runs — the contract of ray_sym_intern_no_split. Same ids as + * ray_sym_intern_batch. */ +int64_t ray_sym_intern_batch_no_split(const uint32_t* hashes, const char* const* strs, + const size_t* lens, int64_t n, int64_t* out_ids) { + if (!atomic_load_explicit(&g_sym_inited, memory_order_acquire)) return -1; + if (n <= 0) return 0; + sym_lock(); + for (int64_t i = 0; i < n; i++) { + int64_t id = sym_intern_nolock_noseg(hashes[i], strs[i], lens[i]); + if (id < 0) { sym_unlock(); return -1; } + out_ids[i] = id; + } + sym_unlock(); + return 0; +} + /* -------------------------------------------------------------------------- * ray_sym_intern_no_split — persistence-only bulk intern * -------------------------------------------------------------------------- */ diff --git a/src/table/sym.h b/src/table/sym.h index 15418a17..7ed5e0bf 100644 --- a/src/table/sym.h +++ b/src/table/sym.h @@ -129,6 +129,9 @@ int64_t ray_sym_intern_prehashed(uint32_t hash, const char* str, size_t len); * any intern failed, in which case out_ids is only partially filled. */ int64_t ray_sym_intern_batch(const uint32_t* hashes, const char* const* strs, const size_t* lens, int64_t n, int64_t* out_ids); +/* The same for VALUES: no dotted-segment caching (see sym.c). */ +int64_t ray_sym_intern_batch_no_split(const uint32_t* hashes, const char* const* strs, + const size_t* lens, int64_t n, int64_t* out_ids); /* Monotonic counter bumped by ray_sym_init and ray_sym_destroy. A cache * keyed on sym ids is valid only while this is unchanged: ids are stable diff --git a/test/rfl/group/dense_slot_split.rfl b/test/rfl/group/dense_slot_split.rfl new file mode 100644 index 00000000..229ff1b2 --- /dev/null +++ b/test/rfl/group/dense_slot_split.rfl @@ -0,0 +1,54 @@ +;; The direct-array group path with more workers than its slot budget allows: +;; the slots are split across tasks (each scans every row for its own range +;; of groups, one accumulator set). 200k rows over ~250k key slots with a +;; nullable min/sum/count keep cells * workers above the row count at any +;; worker count from two up. The oracle is the same grouping forced onto +;; the hash path by one far-away key (the range then exceeds the slot +;; budget); every group but that one must agree. +(set N 200000) +(set i (til N)) +(set k (% (* i 7919) 130003)) +(set s (as 'SYMBOL (map (fn [j] (if (== 0 (% j 13)) "" (format "host%.example.org" (% (* j 31) 9973)))) i))) +(set v (as 'I64 (map (fn [j] (if (== 0 (% j 7)) 0N (% (* j 17) 1000))) i))) +(set f (% (* i 3) 100)) +(set T (table [k s v f] (list k s v f))) +(set T2 (table [k s v f] (list (concat k [1000000000]) (concat s (as 'SYMBOL ["zz"])) (concat v [5]) (concat f [7])))) +(set R (select {from: T by: k c: (count s) m: (min s) x: (max s) sv: (sum v) av: (avg v) fi: (first f) la: (last f)})) +(set O (select {from: T2 by: k c: (count s) m: (min s) x: (max s) sv: (sum v) av: (avg v) fi: (first f) la: (last f) where: (< k 1000000000)})) +(count R) -- 130003 +(== (count R) (count O)) -- true +(set RS (select {from: R asc: k})) +(set OS (select {from: O asc: k})) +(all (== (at RS 'k) (at OS 'k))) -- true +(all (== (at RS 'c) (at OS 'c))) -- true +(all (== (as 'STR (at RS 'm)) (as 'STR (at OS 'm)))) -- true +(all (== (as 'STR (at RS 'x)) (as 'STR (at OS 'x)))) -- true +(all (== (at RS 'sv) (at OS 'sv))) -- true +(all (== (at RS 'av) (at OS 'av))) -- true +(all (== (at RS 'fi) (at OS 'fi))) -- true +(all (== (at RS 'la) (at OS 'la))) -- true +;; strlen over the SYM column with empty cells, and a two-key composite +(set R2 (select {from: T by: k l: (avg (strlen s)) c: (count k)})) +(set O2 (select {from: T2 by: k l: (avg (strlen s)) c: (count k) where: (< k 1000000000)})) +(all (== (at (select {from: R2 asc: k}) 'l) (at (select {from: O2 asc: k}) 'l))) -- true +(set R3 (select {from: T by: [k f] c: (count s) m: (min s) sv: (sum v)})) +(set O3 (select {from: T2 by: [k f] c: (count s) m: (min s) sv: (sum v) where: (< k 1000000000)})) +(== (count R3) (count O3)) -- true +(== (sum (at R3 'sv)) (sum (at O3 'sv))) -- true +(== (sum (as 'I64 (strlen (as 'STR (at R3 'm))))) (sum (as 'I64 (strlen (as 'STR (at O3 'm)))))) -- true +;; the same under a where: selection (the tasks walk the selection too) +(set R4 (select {from: T by: k c: (count s) m: (min s) sv: (sum v) where: (> f 9)})) +(set O4 (select {from: T2 by: k c: (count s) m: (min s) sv: (sum v) where: (and (> f 9) (< k 1000000000))})) +(== (count R4) (count O4)) -- true +(all (== (at (select {from: R4 asc: k}) 'sv) (at (select {from: O4 asc: k}) 'sv))) -- true +(all (== (as 'STR (at (select {from: R4 asc: k}) 'm)) (as 'STR (at (select {from: O4 asc: k}) 'm)))) -- true +;; float sums over the partitioned accumulate add each group's rows in +;; table order: 1e16 + 1 - 1e16 + 1 rounds to 1.0 in that order, whatever +;; the core count (a scheduler-dependent order gave 0.0 for some groups) +(set NF 200000) +(set iF (til NF)) +(set TF (table [k f s] (list (% iF 50000) (at [1e16 1.0 -1e16 1.0] (as 'I64 (floor (/ iF 50000)))) (as 'SYMBOL (map (fn [j] (format "h%" (% j 97))) iF))))) +(set RF (select {from: TF by: k l: (sum (strlen s)) sf: (sum f) c: (count f)})) +(distinct (at RF 'sf)) -- [1.0] +(count RF) -- 50000 + diff --git a/test/rfl/group/derived_key_vocab_aggs.rfl b/test/rfl/group/derived_key_vocab_aggs.rfl new file mode 100644 index 00000000..d5574041 --- /dev/null +++ b/test/rfl/group/derived_key_vocab_aggs.rfl @@ -0,0 +1,59 @@ +;; A grouping keyed by an expression over one FILE-domain SYM column whose +;; every aggregate reads that column — count, min, max, sum/avg of strlen — +;; is decided over the column's distinct values: the rows are grouped by the +;; column once (a count per value), the key runs once per value, and the +;; aggregates are rewritten over that. Every answer is checked against the +;; row-wise evaluation, which one aggregate over another column forces. +(.sys.exec "rm -rf /tmp/rfl_dkv/") -- 0 +(set N 200000) +(set i (til N)) +(set D 3000) +(set refs (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 70) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 70) k))))) (til D))) +(set ref (as 'SYMBOL (map (fn [k] (at refs (% (* k 7919) D))) i))) +(set v (% (* i 31) 1000)) +(set T (table [ref v] (list ref v))) +(.db.splayed.set "/tmp/rfl_dkv/" T) +(set F (.db.splayed.get "/tmp/rfl_dkv/")) +(set key (fn [t] (select {from: t by: (let p (str-find ref "://") (let s (substr ref (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr ref 1 4) "http") (not (nil? sl))) (substr r 1 sl) ref))))) l: (avg (strlen ref)) c: (count ref) mn: (min ref) mx: (max ref) sl: (sum (strlen ref)) where: (!= ref "")}))) +(set oracle (fn [t] (select {from: t by: (let p (str-find ref "://") (let s (substr ref (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr ref 1 4) "http") (not (nil? sl))) (substr r 1 sl) ref))))) l: (avg (strlen ref)) c: (count ref) mn: (min ref) mx: (max ref) sl: (sum (strlen ref)) sv: (sum v) where: (!= ref "")}))) +(set same (fn [R O col] (all (== (at (select {from: R asc: c}) col) (at (select {from: O asc: c}) col))))) +(set R (key F)) +(set O (oracle F)) +(count R) -- 80 +(== (count R) (count O)) -- true +(cols R) -- [p l c mn mx sl] +(type (at R 'l)) -- 'F64 +(type (at R 'mn)) -- 'SYM +(same R O 'c) -- true +(same R O 'l) -- true +(same R O 'sl) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'mn)) (as 'STR (at (select {from: O asc: c}) 'mn)))) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'mx)) (as 'STR (at (select {from: O asc: c}) 'mx)))) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'p)) (as 'STR (at (select {from: O asc: c}) 'p)))) -- true +;; the same table in memory (runtime domain) takes the row path and agrees +(set RM (key T)) +(== (count RM) (count R)) -- true +(same RM R 'c) -- true +(same RM R 'l) -- true +;; no where: the null group is there, its avg is the null float, its min the empty symbol +(set R2 (select {from: F by: (substr ref 1 4) c: (count ref) l: (avg (strlen ref)) mn: (min ref)})) +(set O2 (select {from: F by: (substr ref 1 4) c: (count ref) l: (avg (strlen ref)) mn: (min ref) sv: (sum v)})) +(count R2) -- 12 +(same R2 O2 'c) -- true +(all (== (nil? (at (select {from: R2 asc: c}) 'l)) (nil? (at (select {from: O2 asc: c}) 'l)))) -- true +(sum (at R2 'c)) -- 200000 +;; sort and take on an output alias +(set R3 (select {from: F by: (substr ref 1 4) c: (count ref) where: (!= ref "") desc: c take: 3})) +(set O3 (select {from: F by: (substr ref 1 4) c: (count ref) sv: (sum v) where: (!= ref "") desc: c take: 3})) +(all (== (at R3 'c) (at O3 'c))) -- true +(all (== (as 'STR (at R3 'ref)) (as 'STR (at O3 'ref)))) -- true +;; an alias that spells the source column's name renames the key to `key` +(cols (select {from: F by: (substr ref 1 4) ref: (count ref) where: (!= ref "")})) -- [key ref] +;; unsorted output: groups in the order their first value appears in the +;; column's vocabulary, the same at every core count (the per-value counts +;; come out of a grouping whose order depends on the core count, and are +;; put back in vocabulary order first) +(set RU (select {from: F by: (substr ref 1 4) c: (count ref)})) +(as 'STR (at RU 'ref)) -- ["" "http" "ab5" "ab0" "ab9" "ab4" "ab3" "ab8" "ab2" "ab1" "ab6" "ab7"] +(at RU 'c) -- [18203 164471 1733 1736 1733 1733 1734 1731 1731 1731 1732 1732] +(.sys.exec "rm -rf /tmp/rfl_dkv/") -- 0 From 41063a08e10684b43af9a81409daa16ca10a7e8d Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 11:11:49 +0300 Subject: [PATCH 46/51] perf(csv): parallel splayed load, batched symfile interning, parallel hash index build (#632) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(csv): scan, append and index the splayed load in parallel Streaming a CSV into a splayed table ran its per-chunk row scan, the per-column writes and the index build on the calling thread; only the field parse used the pool. - Row offsets for a chunk come from the parallel quote-parity scanner over a byte window sized from the running average row length, widened until it holds the chunk (falls back to the serial limited scan). - The chunk's columns are appended by one task per column; SYM cells go through a direct-mapped runtime-id to position cache in the writer, so only a value's first occurrence probes the symfile domain. - ray_splay_build_indexes builds one column per task. Co-Authored-By: Claude Fable 5.1 * perf(csv): intern chunk dictionaries straight into the symfile domain Every distinct string of a splayed load was interned twice, serially: once into the runtime symbol table by the chunk materialisation, then again into the table's symfile domain by the column writer, one cell at a time under the global domain spinlock — the two together were more than half of the load time, and the runtime table grew by the whole vocabulary for nothing. - csv_materialize_rows takes a target domain: SYM chunk columns are built over the symfile domain, and the writer copies their positions. - ray_sym_domain_intern_batch (domain.c): one batch per chunk over all SYM columns. The existing vocabulary is probed in parallel against a snapshot of the reverse index (replaced bucket tables are retired, never freed), misses are deduplicated per hash partition, and the new atoms are built in parallel into per-partition arena regions (ray_arena_alloc_raw / ray_arena_str_at) before the count is published. Index growth re-slots the old table instead of rehashing every atom; the symfile flush packs records into a 1 MB buffer. - SYM fields are hashed by the parse tasks; with a domain target each SYM column's dedupe is split by hash partition across tasks. - The window scan is preceded by a readahead hint and finished chunks are dropped from the mapping, so a long file does not stay resident. Co-Authored-By: Claude Fable 5.1 * perf(index): build hash indexes partition-parallel ray_index_attach_hash grouped the rows in one serial open-addressing pass; on a high-cardinality numeric column of a large table it was the longest single stretch of a splayed load, and running the columns as pool tasks left the build itself single-threaded. Numeric and SYM keys above 64k rows now build in partition-parallel passes: key words and hash-partition buckets over row ranges, a per-partition dedupe that marks each group's first row in a bitmap, then keys, counts, the row scatter and the key table (CAS on the empty slot) per partition. A group's number is the rank of its first row among the marked rows, so the layout is the serial walk's: groups in first-occurrence order, rows ascending inside a group. Every array a pass writes is faulted in beforehand from all workers in slices — a fresh mapping faulted at random from every worker serialised on the page-table locks. STR keys and small columns keep the serial walk. ray_splay_build_indexes defers the columns that qualify for a hash index out of the per-column tasks, builds them one after another with the parallel path and writes all deferred columns together. Co-Authored-By: Claude Fable 5.1 * chore(index,domain): cast the CAS targets in place cppcheck cannot parse a declaration of `_Atomic(T)*` type (AST broken on the `=`), which failed static analysis; the two batch inserts now cast at the compare-exchange, the form the rest of domain.c uses. Co-Authored-By: Claude Fable 5.1 * fix(domain): retire the reverse index on external extend; batch intern tests The batch intern probes a snapshot of a domain's reverse index outside the lock, relying on replaced tables being retired rather than freed. The external-growth path (a reader catching up with a symfile another process appended to) still freed the table, so a probe running at that moment could read freed memory. It now retires it like the rebuild does. A probe that meets an entry it cannot compare (no atom and no raw bytes, possible only if publishing the raw snapshot failed) no longer counts it as a miss: the whole batch then resolves under the lock, so no string is appended twice. The splayed writer's per-column dispatch goes through ray_pool_par_dispatch_ok like the other sites. The batch-intern header comment no longer claims appends in batch order: new strings get positions grouped by hash partition. Tests: domain/intern_batch (repeats across partitions, pre-interned hits keep their positions, repeat batch appends nothing, "" is 0, flush/reopen); csv_splayed_dedupe_overflow (140k distinct strings in one SYM column overflow a partition's dictionary in debug builds and take the row-by-row fallback); index/hash_large_parallel now creates the pool so the parallel build is the one tested. Co-Authored-By: Claude Opus 5.5 * fix(csv): split every window as the whole file; symfile and index bytes independent of cores A chunk's byte window chose the scanner's quote mode from its own bytes. In a file with quotes elsewhere, a window without any took the quote-free fast path, where a lone '\r' does not end a row: two rows merged and a value was lost (the serial walk and .csv.read split them). The window now scans in the file's quote mode. The symfile no longer depends on the worker count. The chunk batch is laid out the way the cell-by-cell writer meets the strings — columns in order, each column's strings by first occurrence (partition runs merged by the first row, recorded in the dictionary entry) — and the batch intern appends new strings in batch order. The hash index key table is filled in group order on the calling thread, as the serial build does, so the persisted index is the same bytes on any core count. A replaced reverse-index table is freed at once unless a batch intern is probing a snapshot outside the lock (counted under the lock), so a long-lived reader of a growing symfile keeps no extra tables. Tests: splay/csv_symfile_order (positions follow the writer order over three chunks, parallel batch and partitioned dedupe), and splay/csv_quote_mode_per_file (quotes only in chunk 0, a lone '\r' in chunk 2: same rows and values as .csv.read). Both fail with their fix disabled. Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Fable 5.1 Co-authored-by: Anton Kundenko --- src/io/csv.c | 438 ++++++++++++++++++-- src/mem/arena.c | 54 ++- src/mem/arena.h | 8 + src/ops/idxop.c | 352 +++++++++++++++- src/store/splay.c | 104 ++++- src/table/domain.c | 427 ++++++++++++++++++- src/table/domain.h | 11 + test/rfl/io/csv_splayed_dedupe_overflow.rfl | 27 ++ test/test_domain.c | 106 +++++ test/test_index.c | 80 ++++ test/test_splay.c | 139 +++++++ 11 files changed, 1671 insertions(+), 75 deletions(-) create mode 100644 test/rfl/io/csv_splayed_dedupe_overflow.rfl diff --git a/src/io/csv.c b/src/io/csv.c index 25302c1f..f4d583ef 100644 --- a/src/io/csv.c +++ b/src/io/csv.c @@ -151,6 +151,7 @@ static inline void scratch_free(ray_t* hdr) { typedef struct { const char* ptr; uint32_t len; + uint32_t hash; /* ray_hash_bytes of the field; filled for SYM columns */ } csv_strref_t; RAY_INLINE const char* scan_field(const char* p, const char* buf_end, @@ -952,9 +953,15 @@ static size_t csv_scan_split_at(const char* buf, size_t file_size, size_t s, /* Returns the row count (>=0), -1 if interrupted, or -2 when the parallel * path does not apply and the caller should run the serial scan. */ +/* force_quotes: the caller scans a byte window of a file and knows quotes + * occur somewhere in the file's data. The quote-aware state machine is + * then used even when this window holds none, so a window is split into + * rows exactly as the whole file would be (the quote-free fast path treats + * a lone '\r' and "\n\r" differently). */ static int64_t build_row_offsets_par(const char* buf, size_t buf_size, size_t data_offset, uint64_t prog_base, uint64_t prog_len, + bool force_quotes, int64_t** offsets_out, ray_t** hdr_out) { *offsets_out = NULL; *hdr_out = NULL; @@ -1023,7 +1030,7 @@ static int64_t build_row_offsets_par(const char* buf, size_t buf_size, total_q += quote_cnt[i]; total_term += term_cnt[i]; } - ctx.has_quotes = total_q != 0; + ctx.has_quotes = total_q != 0 || force_quotes; /* Now nudge the boundaries for the state machine pass 1 just chose. The * byte skipped is always '\n' or '\r', never '"', so the parities @@ -1201,7 +1208,7 @@ static int64_t build_row_offsets(const char* buf, size_t buf_size, * -2 means "not applicable" (small file, no pool, allocation refused) and * falls back to the serial scan, which defines the semantics. */ int64_t par = build_row_offsets_par(buf, buf_size, data_offset, - prog_base, prog_len, + prog_base, prog_len, false, offsets_out, hdr_out); if (par != -2) return par; @@ -1211,6 +1218,59 @@ static int64_t build_row_offsets(const char* buf, size_t buf_size, offsets_out, hdr_out); } +/* Row starts of the next `max_rows` rows from `data_offset`, found by the + * parallel scanner over a byte window instead of the serial walk: the + * window is sized from `avg_row` (bytes per row seen so far) with slack, and + * doubled when it holds fewer than max_rows + 1 row starts before the end + * of the file (the extra start proves the max_rows-th row is complete). + * Returns the row count, -1 if interrupted, or -2 when the parallel path + * does not apply (small window, no pool) — the caller then runs the serial + * limited scan. */ +static int64_t build_row_offsets_window(const char* buf, size_t buf_size, + size_t data_offset, int64_t max_rows, + size_t avg_row, bool data_has_quotes, + int64_t** offsets_out, ray_t** hdr_out, + size_t* next_offset_out) { + *offsets_out = NULL; *hdr_out = NULL; + if (next_offset_out) *next_offset_out = data_offset; + if (max_rows <= 0 || data_offset >= buf_size) return 0; + if (avg_row < 8) avg_row = 8; + size_t window = (size_t)max_rows * avg_row + (size_t)max_rows * avg_row / 4 + (64u << 10); + for (;;) { + size_t end = data_offset + window; + if (end > buf_size || end < data_offset) end = buf_size; + /* Readahead hint for the window about to be scanned: the scanner's + * tasks fault the pages in parallel, which a cold file serves best + * when the kernel already streams the range. */ + { + size_t ps = (size_t)sysconf(_SC_PAGESIZE); + size_t a = data_offset & ~(ps - 1); + madvise((void*)(buf + a), end - a, MADV_WILLNEED); + } + int64_t* offs = NULL; ray_t* hdr = NULL; + int64_t n = build_row_offsets_par(buf, end, data_offset, 0, 0, + data_has_quotes, &offs, &hdr); + if (n < 0) return n; /* -1 interrupted, -2 not applicable */ + if (n == 0) { scratch_free(hdr); return -2; } + if (n > max_rows) { + /* row max_rows - 1 ends before start max_rows: complete */ + if (next_offset_out) *next_offset_out = (size_t)offs[max_rows]; + *offsets_out = offs; *hdr_out = hdr; + return max_rows; + } + if (end == buf_size) { + /* every remaining row, the last one ended by the file */ + if (next_offset_out) *next_offset_out = buf_size; + *offsets_out = offs; *hdr_out = hdr; + return n; + } + /* the window held at most max_rows starts: widen and rescan */ + scratch_free(hdr); + if (window > SIZE_MAX / 2) return -2; + window *= 2; + } +} + static int64_t build_row_offsets_limited(const char* buf, size_t buf_size, size_t data_offset, int64_t max_rows, bool data_has_quotes, @@ -1381,6 +1441,7 @@ typedef struct { uint32_t len; const char* ptr; int64_t gid; /* global sym id, filled in step B */ + int64_t row; /* first row the string occurs on (domain batch order) */ } csv_dedup_ent_t; typedef struct { @@ -1438,8 +1499,32 @@ typedef struct { /* [n_cols] "this column wrote a canonical null", recorded where the value * is written. See csv_note_empty. */ bool* empties; + /* Target FILE domain (splayed save): the distinct strings are + * interned straight into the table's symfile domain, batched over + * every SYM column of the chunk, and the codes are positions in it. + * NULL: runtime symbol table, id-order-preserving serial walk. */ + struct ray_sym_domain_s* dom; + /* Hash partitions per SYM column (domain target only; 1 otherwise): + * dicts[i * n_part + p] holds the distinct strings of column cols[i] + * whose hash >> part_shift == p. Partitions of one column are + * deduplicated by independent tasks. */ + int n_part; + int part_shift; } csv_dedup_ctx_t; +static inline int csv_dedup_part(const csv_dedup_ctx_t* dd, uint32_t hash) { + return dd->n_part > 1 ? (int)(hash >> dd->part_shift) : 0; +} + +/* Every partition of column i deduplicated without overflow. */ +static inline bool csv_col_dict_ok(const csv_dedup_ctx_t* dd, int i) { + for (int p = 0; p < dd->n_part; p++) { + const csv_dedup_t* d = &dd->dicts[i * dd->n_part + p]; + if (!d->done || d->overflow) return false; + } + return true; +} + /* HAS_NULLS accounting for SYM/STR columns (step 9c). * * bfb5b380 made an empty SYM/STR cell a canonical null and, where the parse @@ -1460,10 +1545,14 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, (void)worker_id; (void)end_i; csv_dedup_ctx_t* ctx = (csv_dedup_ctx_t*)arg; csv_dedup_t* d = &ctx->dicts[start]; - const csv_strref_t* refs = ctx->str_refs[ctx->cols[start]]; + int col_i = (int)(start / ctx->n_part); + int part = (int)(start % ctx->n_part); + const csv_strref_t* refs = ctx->str_refs[ctx->cols[col_i]]; /* Local codes land in the destination id array and are replaced in place - * by step C, so the dedupe needs no per-row scratch of its own. */ - uint32_t* codes = (uint32_t*)ctx->col_data[ctx->cols[start]]; + * by step C, so the dedupe needs no per-row scratch of its own. With + * partitions each task owns the rows whose hash falls in its partition; + * partition 0 also writes the null codes. */ + uint32_t* codes = (uint32_t*)ctx->col_data[ctx->cols[col_i]]; int64_t n_rows = ctx->n_rows; if (!csv_dedup_grow(d) || !csv_dedup_grow_ents(d)) { @@ -1475,8 +1564,9 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, for (int64_t r = 0; r < n_rows; r++) { if (RAY_UNLIKELY((r & 1023) == 0 && ray_interrupted())) return; - if (refs[r].ptr == NULL) { codes[r] = 0; continue; } - uint32_t h = (uint32_t)ray_hash_bytes(refs[r].ptr, refs[r].len); + if (refs[r].ptr == NULL) { if (part == 0) codes[r] = 0; continue; } + uint32_t h = refs[r].hash; /* computed by the parse */ + if (csv_dedup_part(ctx, h) != part) continue; uint32_t mask = d->n_slots - 1; uint32_t j = h & mask; uint32_t found = 0; @@ -1507,6 +1597,7 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, e->len = refs[r].len; e->ptr = refs[r].ptr; e->gid = 0; + e->row = r; d->n_ents++; codes[r] = d->n_ents; /* code = entry index + 1 */ d->slots[j] = d->n_ents; @@ -1523,8 +1614,9 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, int64_t start, int64_t end_i) { (void)worker_id; (void)end_i; csv_dedup_ctx_t* ctx = (csv_dedup_ctx_t*)arg; - const csv_dedup_t* d = &ctx->dicts[start]; - if (d->overflow) return; /* step B already wrote real ids */ + if (!csv_col_dict_ok(ctx, (int)start)) return; /* step B already wrote real ids */ + const csv_dedup_t* dcol = &ctx->dicts[start * ctx->n_part]; + const csv_strref_t* refs = ctx->str_refs[ctx->cols[start]]; uint32_t* ids = (uint32_t*)ctx->col_data[ctx->cols[start]]; int64_t n_rows = ctx->n_rows; uint32_t empty = (uint32_t)ctx->empty_gid; @@ -1532,7 +1624,8 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, for (int64_t r = 0; r < n_rows; r++) { if (RAY_UNLIKELY((r & 1023) == 0 && ray_interrupted())) return; uint32_t code = ids[r]; - uint32_t id = code ? (uint32_t)d->ents[code - 1].gid : empty; + uint32_t id = code ? (uint32_t)dcol[csv_dedup_part(ctx, refs[r].hash)].ents[code - 1].gid + : empty; ids[r] = id; saw_null |= (id == 0); } @@ -1551,9 +1644,135 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, * empty string anyway, so collapsing them is the only deterministic answer the * parser can give. Empty fields are local code 0 out of the dedupe and are * mapped to that id by step C. */ +/* Step B for a FILE domain target: one batch for the whole chunk, laid out + * the way the cell-by-cell writer met the strings — columns in order, each + * column's strings by first occurrence — so the symfile gets the same + * positions whatever the worker count and however a column's dictionary + * was split into hash partitions. The domain probes the existing + * vocabulary in parallel and appends the new strings in batch order. A + * column whose dictionary overflowed contributes its rows directly (the + * batch dedupes them) and gets its ids written here. */ +/* Append column i's dictionary entries to out[] by first row. Each hash + * partition lists its strings in row order already, so this is a merge of + * n_part sorted runs (n_part is small). */ +static int64_t csv_col_ents_by_row(csv_dedup_ctx_t* dd, int i, csv_dedup_ent_t** out) { + int np = dd->n_part; + csv_dedup_t* d0 = &dd->dicts[i * np]; + if (np == 1) { + for (uint32_t e = 0; e < d0->n_ents; e++) out[e] = &d0->ents[e]; + return d0->n_ents; + } + uint32_t cur[64]; + for (int p = 0; p < np; p++) cur[p] = 0; + int64_t n = 0; + for (;;) { + int best = -1; + int64_t br = 0; + for (int p = 0; p < np; p++) { + if (cur[p] >= d0[p].n_ents) continue; + int64_t r = d0[p].ents[cur[p]].row; + if (best < 0 || r < br) { best = p; br = r; } + } + if (best < 0) break; + out[n++] = &d0[best].ents[cur[best]++]; + } + return n; +} + +static bool csv_intern_dicts_domain(csv_dedup_ctx_t* dd, int n_sym, + int64_t* col_max_ids, + uint64_t prog_base, uint64_t prog_len) { + struct ray_sym_domain_s* dom = dd->dom; + /* Position 0 of a symfile domain is "" (reserved on creation). */ + if (ray_sym_domain_intern(dom, "", 0) != 0) return false; + dd->empty_gid = 0; + + int64_t total = 0; + for (int i = 0; i < n_sym; i++) { + if (csv_col_dict_ok(dd, i)) { + for (int p = 0; p < dd->n_part; p++) total += dd->dicts[i * dd->n_part + p].n_ents; + } else { + total += dd->n_rows; + } + } + if (total == 0) { + if (prog_len) ray_progress_span_set(prog_base + prog_len); + return true; + } + + ray_t *hs = NULL, *hl = NULL, *hh = NULL, *hp = NULL, *ho = NULL; + const char** strs = (const char**)scratch_alloc(&hs, (size_t)total * sizeof(char*)); + size_t* lens = (size_t*)scratch_alloc(&hl, (size_t)total * sizeof(size_t)); + uint32_t* hashes = (uint32_t*)scratch_alloc(&hh, (size_t)total * sizeof(uint32_t)); + int64_t* pos = (int64_t*)scratch_alloc(&hp, (size_t)total * sizeof(int64_t)); + /* order[k]: the dictionary entry behind batch slot k (dict columns) */ + csv_dedup_ent_t** order = (csv_dedup_ent_t**)scratch_alloc(&ho, + (size_t)total * sizeof(csv_dedup_ent_t*)); + bool ok = strs && lens && hashes && pos && order; + + int64_t k = 0; + for (int i = 0; ok && i < n_sym; i++) { + if (csv_col_dict_ok(dd, i)) { + int64_t ne = csv_col_ents_by_row(dd, i, order + k); + for (int64_t e = 0; e < ne; e++, k++) { + strs[k] = order[k]->ptr; + lens[k] = order[k]->len; + hashes[k] = order[k]->hash; + } + } else { + const csv_strref_t* refs = dd->str_refs[dd->cols[i]]; + for (int64_t r = 0; r < dd->n_rows; r++) { + if (refs[r].ptr == NULL) continue; + strs[k] = refs[r].ptr; + lens[k] = refs[r].len; + hashes[k] = refs[r].hash; + k++; + } + } + } + if (ok) ok = ray_sym_domain_intern_batch(dom, k, strs, lens, hashes, pos); + + /* Hand the positions back in the same walk. */ + k = 0; + for (int i = 0; ok && i < n_sym; i++) { + int c = dd->cols[i]; + int64_t max_id = 0; + if (csv_col_dict_ok(dd, i)) { + int64_t ne = 0; + for (int p = 0; p < dd->n_part; p++) ne += dd->dicts[i * dd->n_part + p].n_ents; + for (int64_t e = 0; e < ne; e++, k++) { + int64_t id = pos[k]; + if (id < 0) { ok = false; id = 0; } + order[k]->gid = id; + if (id > max_id) max_id = id; + } + } else { + const csv_strref_t* refs = dd->str_refs[c]; + uint32_t* ids = (uint32_t*)dd->col_data[c]; + bool saw_null = false; + for (int64_t r = 0; r < dd->n_rows; r++) { + if (refs[r].ptr == NULL) { ids[r] = 0; saw_null = true; continue; } + int64_t id = pos[k++]; + if (id < 0) { ok = false; id = 0; } + ids[r] = (uint32_t)id; + saw_null |= (id == 0); + if (id > max_id) max_id = id; + } + csv_note_empty(dd->empties, c, saw_null); + } + if (col_max_ids) col_max_ids[c] = max_id; + } + + scratch_free(hs); scratch_free(hl); scratch_free(hh); scratch_free(hp); scratch_free(ho); + if (prog_len) ray_progress_span_set(prog_base + prog_len); + return ok; +} + static bool csv_intern_dicts(csv_dedup_ctx_t* dd, int n_sym, int64_t* col_max_ids, uint64_t prog_base, uint64_t prog_len) { + if (dd->dom) + return csv_intern_dicts_domain(dd, n_sym, col_max_ids, prog_base, prog_len); bool ok = true; /* Same first call, same reason, as the serial walk: sym 0 is reserved by @@ -1739,6 +1958,7 @@ typedef struct { int n_sym; bool* empties; /* [n_cols] SYM/STR column wrote a null */ bool intern_ok; + struct ray_sym_domain_s* dom; /* SYM target domain, NULL = runtime */ } csv_finalize_ctx_t; /* dispatch 1: [0, n_fill) fill a RAY_STR column, [n_fill, n_fill+n_sym) dedupe @@ -1768,7 +1988,7 @@ static void csv_finalize_task(void* arg, uint32_t worker_id, * prog_base/prog_len describe this phase's slice of the load's byte axis; pass * len 0 when no progress span is active (the streaming conversion path). */ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, - bool* fill_ok, int* sym_cols, csv_dedup_t* dicts, + bool* fill_ok, int* sym_cols, bool* empties, uint64_t prog_base, uint64_t prog_len) { int n_fill = 0, n_sym = 0; @@ -1778,7 +1998,31 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, else sym_cols[n_sym++] = c; } for (int i = 0; i < n_fill; i++) fill_ok[i] = true; - memset(dicts, 0, (size_t)n_sym * sizeof(csv_dedup_t)); + + ray_pool_t* pool = ray_pool_get(); + bool par = pool && ray_pool_total_workers(pool) >= 2; + + /* Partitions per SYM column. The runtime path keeps one dictionary per + * column (its id order is the serial walk's); a domain target has no + * order to keep, so the heavy columns are split by hash until the + * dedupe tasks cover the pool about twice over. */ + int n_part = 1; + if (ctx->dom && par && n_sym > 0) { + int64_t want = (int64_t)ray_pool_total_workers(pool) * 2 / n_sym; + while (n_part * 2 <= want && n_part < 16) n_part *= 2; + while (n_part > 1 && (int64_t)n_fill + (int64_t)n_sym * n_part > (int64_t)RAY_POOL_INIT_TASKS) + n_part /= 2; + } + int part_shift = 32; + for (int q = n_part; q > 1; q >>= 1) part_shift--; + + ray_t* dicts_hdr = NULL; + csv_dedup_t* dicts = NULL; + if (n_sym > 0) { + dicts = (csv_dedup_t*)scratch_calloc(&dicts_hdr, + (size_t)n_sym * (size_t)n_part * sizeof(csv_dedup_t)); + if (!dicts) return false; + } ctx->fill_cols = fill_cols; ctx->n_fill = n_fill; @@ -1793,6 +2037,9 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, ctx->dd.n_rows = ctx->n_rows; ctx->dd.empty_gid = 0; ctx->dd.empties = ctx->empties; + ctx->dd.dom = ctx->dom; + ctx->dd.n_part = n_part; + ctx->dd.part_shift = part_shift; /* Slice the phase: dedupe+fill is the bulk, the serial intern touches only * distinct strings, the remap is one linear pass per SYM column. Measured @@ -1800,10 +2047,8 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, uint64_t w1 = prog_len * 6 / 10; uint64_t w2 = prog_len / 10; - int64_t n_tasks = (int64_t)n_fill + (int64_t)n_sym; - ray_pool_t* pool = ray_pool_get(); - bool par = pool && ray_pool_total_workers(pool) >= 2 && - n_tasks > 0 && n_tasks <= (int64_t)RAY_POOL_INIT_TASKS; + int64_t n_tasks = (int64_t)n_fill + (int64_t)n_sym * n_part; + par = par && n_tasks > 0 && n_tasks <= (int64_t)RAY_POOL_INIT_TASKS; if (prog_len) ray_progress_span_phase("finalize", prog_base, w1); if (par) ray_pool_dispatch_n(pool, csv_finalize_task, ctx, (uint32_t)n_tasks); @@ -1830,12 +2075,14 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, if (ray_interrupted()) goto fail; } - for (int i = 0; i < n_sym; i++) csv_dedup_release(&dicts[i]); + for (int i = 0; i < n_sym * n_part; i++) csv_dedup_release(&dicts[i]); + scratch_free(dicts_hdr); for (int i = 0; i < n_fill; i++) if (!fill_ok[i]) return false; return true; fail: - for (int i = 0; i < n_sym; i++) csv_dedup_release(&dicts[i]); + for (int i = 0; i < n_sym * n_part; i++) csv_dedup_release(&dicts[i]); + scratch_free(dicts_hdr); return false; } @@ -2021,6 +2268,10 @@ static void csv_parse_fn(void* arg, uint32_t worker_id, } ctx->str_refs[c][row].ptr = fld; ctx->str_refs[c][row].len = (uint32_t)flen; + /* SYM columns are deduplicated by hash right after + * the parse; hash here while the bytes are hot. */ + if (ctx->resolved_types[c] == RAY_SYM) + ctx->str_refs[c][row].hash = (uint32_t)ray_hash_bytes(fld, flen); } break; } @@ -2202,6 +2453,8 @@ static bool csv_parse_serial(const char* buf, size_t buf_size, } str_refs[c][row].ptr = fld; str_refs[c][row].len = (uint32_t)flen; + if (resolved_types[c] == RAY_SYM) + str_refs[c][row].hash = (uint32_t)ray_hash_bytes(fld, flen); } break; } @@ -2479,7 +2732,8 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, const int64_t* row_offsets, int64_t n_rows, int ncols, char delimiter, const int64_t* col_name_ids, - const int8_t* resolved_types) { + const int8_t* resolved_types, + struct ray_sym_domain_s* sym_dom) { /* Defensive guard: RAY_CSV_AUTO_TAG must be resolved to a concrete width * before reaching this point (by csv_resolve_auto_in_place or * csv_resolve_auto_streamed). If a marker slips through, the resolution @@ -2505,6 +2759,11 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, for (int j = 0; j < c; j++) ray_release(col_vecs[j]); return NULL; } + if (type == RAY_SYM && sym_dom) { + /* Cells are positions in the target symfile domain. */ + ray_sym_domain_retain(sym_dom); + col_vecs[c]->sym_domain = sym_dom; + } col_vecs[c]->len = n_rows; col_data[c] = ray_data(col_vecs[c]); } @@ -2645,14 +2904,14 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, .col_vecs = col_vecs, .n_rows = n_rows, .sym_max_ids = sym_max_ids, + .dom = sym_dom, }; int fill_cols[CSV_MAX_COLS]; int sym_cols[CSV_MAX_COLS]; bool fill_ok[CSV_MAX_COLS]; - csv_dedup_t dicts[CSV_MAX_COLS]; /* No progress span on the conversion path — pass a zero-length slice. */ bool fin_ok = csv_finalize_run(&fctx, fill_cols, fill_ok, - sym_cols, dicts, col_wrote_null, 0, 0); + sym_cols, col_wrote_null, 0, 0); if (!fin_ok || ray_interrupted()) { csv_free_escaped_strrefs(str_ref_bufs, ncols, parse_types, n_rows, buf, file_size, row_done, col_had_escaped); @@ -2693,6 +2952,7 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, if (new_w >= RAY_SYM_W32) continue; ray_t* narrow = ray_sym_vec_new(new_w, n_rows); if (!narrow || RAY_IS_ERR(narrow)) continue; + ray_sym_vec_adopt_domain(narrow, col_vecs[c]); narrow->len = n_rows; const uint32_t* src = (const uint32_t*)col_data[c]; void* dst = ray_data(narrow); @@ -3123,9 +3383,8 @@ static ray_t* csv_read_named_opts_inner(const char* path, char delimiter, bool h int fill_cols[CSV_MAX_COLS]; int sym_cols[CSV_MAX_COLS]; bool fill_ok[CSV_MAX_COLS]; - csv_dedup_t dicts[CSV_MAX_COLS]; bool fin_ok = csv_finalize_run(&fctx, fill_cols, fill_ok, - sym_cols, dicts, col_wrote_null, + sym_cols, col_wrote_null, prog_parse_end, (uint64_t)file_size - prog_parse_end); if (!fin_ok || ray_interrupted()) { @@ -3283,7 +3542,15 @@ typedef struct { * fixed at W32: a streaming writer can't know the final vocabulary * before the last chunk, and W32 covers any STRL count. */ struct ray_sym_domain_s* dom; + /* runtime id -> domain position, direct-mapped: the chunk vecs are + * runtime-domain and a column's values repeat across rows and chunks, + * so a value is interned into the symfile's domain (a locked probe) the + * first time it is met and looked up here after. */ + int64_t* lut_id; /* [CSV_SPLAYED_LUT] runtime id per slot, -1 empty */ + uint32_t* lut_pos; /* [CSV_SPLAYED_LUT] position per slot */ } csv_splayed_col_writer_t; +#define CSV_SPLAYED_LUT_BITS 19 +#define CSV_SPLAYED_LUT (1u << CSV_SPLAYED_LUT_BITS) static ray_err_t csv_splayed_writer_open(csv_splayed_col_writer_t* w, const char* dir, int64_t name_id, @@ -3314,9 +3581,25 @@ static ray_err_t csv_splayed_writer_open(csv_splayed_col_writer_t* w, if (!w->fp) return RAY_ERR_IO; ray_t zero = {0}; if (fwrite(&zero, 1, 32, w->fp) != 32) return RAY_ERR_IO; + if (type == RAY_SYM) { + /* best effort: without the cache every cell probes the domain */ + w->lut_id = (int64_t*)ray_alloc_raw((size_t)CSV_SPLAYED_LUT * sizeof(int64_t)); + w->lut_pos = (uint32_t*)ray_alloc_raw((size_t)CSV_SPLAYED_LUT * sizeof(uint32_t)); + if (!w->lut_id || !w->lut_pos) { + ray_free_raw(w->lut_id); ray_free_raw(w->lut_pos); + w->lut_id = NULL; w->lut_pos = NULL; + } else { + memset(w->lut_id, 0xff, (size_t)CSV_SPLAYED_LUT * sizeof(int64_t)); + } + } return RAY_OK; } +static void csv_splayed_writer_drop_lut(csv_splayed_col_writer_t* w) { + ray_free_raw(w->lut_id); ray_free_raw(w->lut_pos); + w->lut_id = NULL; w->lut_pos = NULL; +} + static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, ray_t* col) { if (!w->fp || !col || RAY_IS_ERR(col)) return RAY_ERR_TYPE; @@ -3328,18 +3611,40 @@ static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, * resolve each cell through the chunk vec's own domain and * find-or-append into the target (distinct work rides the * write). The domain is flushed before the column files are - * committed (close), preserving the sym-first crash ordering. */ + * committed (close), preserving the sym-first crash ordering. + * A runtime-domain chunk vec goes through the id -> position + * cache: only a value's first encounter pays the domain probe. */ + bool direct = ray_sym_vec_domain(col) == w->dom; + bool cached = ray_sym_vec_domain(col) == ray_sym_runtime_domain() && w->lut_id; + const void* cd = ray_data(col); uint32_t buf[8192]; for (int64_t off = 0; off < n; ) { int64_t cnt = n - off; if (cnt > (int64_t)(sizeof(buf) / sizeof(buf[0]))) cnt = (int64_t)(sizeof(buf) / sizeof(buf[0])); for (int64_t i = 0; i < cnt; i++) { - ray_t* s = ray_sym_vec_cell(col, off + i); - if (!s) return RAY_ERR_CORRUPT; - int64_t pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), - ray_str_len(s)); - if (pos < 0) return RAY_ERR_OOM; + int64_t pos; + if (direct) { + /* Already encoded over the target domain. */ + pos = ray_read_sym(cd, off + i, RAY_SYM, col->attrs); + } else if (cached) { + int64_t id = ray_read_sym(cd, off + i, RAY_SYM, col->attrs); + uint32_t slot = (uint32_t)(((uint64_t)id * 0x9E3779B97F4A7C15ull) >> (64 - CSV_SPLAYED_LUT_BITS)); + if (w->lut_id[slot] == id) { + pos = w->lut_pos[slot]; + } else { + ray_t* s = ray_sym_str(id); + if (!s) return RAY_ERR_CORRUPT; + pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), ray_str_len(s)); + if (pos < 0) return RAY_ERR_OOM; + w->lut_id[slot] = id; w->lut_pos[slot] = (uint32_t)pos; + } + } else { + ray_t* s = ray_sym_vec_cell(col, off + i); + if (!s) return RAY_ERR_CORRUPT; + pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), ray_str_len(s)); + if (pos < 0) return RAY_ERR_OOM; + } buf[i] = (uint32_t)pos; /* Position 0 of any symfile is the empty string (domain.c * enforces that reservation on open), so a re-encoded cell is @@ -3365,7 +3670,29 @@ static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, return RAY_OK; } +/* Append task: column `start` of the chunk table into its writer. The + * first failure is kept (a later task cannot clear it). */ +typedef struct { + csv_splayed_col_writer_t* writers; + ray_t* tbl; + int ncols; + _Atomic(ray_err_t) err; +} csv_splayed_append_ctx_t; + +static void csv_splayed_append_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + csv_splayed_append_ctx_t* a = (csv_splayed_append_ctx_t*)raw; + if (atomic_load_explicit(&a->err, memory_order_relaxed) != RAY_OK) return; + ray_t* col = ray_table_get_col_idx(a->tbl, (int64_t)start); + ray_err_t e = csv_splayed_writer_append(&a->writers[start], col); + if (e != RAY_OK) { + ray_err_t ok = RAY_OK; + atomic_compare_exchange_strong_explicit(&a->err, &ok, e, memory_order_relaxed, memory_order_relaxed); + } +} + static ray_err_t csv_splayed_writer_close(csv_splayed_col_writer_t* w) { + csv_splayed_writer_drop_lut(w); if (!w->fp) return RAY_OK; ray_err_t err = RAY_OK; @@ -3397,6 +3724,7 @@ static ray_err_t csv_splayed_writer_close(csv_splayed_col_writer_t* w) { } static void csv_splayed_writer_abort(csv_splayed_col_writer_t* w) { + csv_splayed_writer_drop_lut(w); if (w->fp) fclose(w->fp); w->fp = NULL; remove(w->tmp_path); @@ -3630,16 +3958,27 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool size_t chunk_offset = data_offset; bool wrote_any = false; + size_t avg_row_bytes = 64; /* refined from every chunk scanned */ while (chunk_offset < file_size || !wrote_any) { ray_t* row_offsets_hdr = NULL; int64_t* row_offsets = NULL; size_t next_offset = chunk_offset; int64_t cnt = 0; if (chunk_offset < file_size) { - cnt = build_row_offsets_limited(buf, file_size, chunk_offset, - rows_per_chunk, data_has_quotes, - &row_offsets, - &row_offsets_hdr, &next_offset); + /* Parallel scan over a byte window sized from the rows seen so + * far; the serial walk remains the fallback and the semantics. */ + cnt = build_row_offsets_window(buf, file_size, chunk_offset, + rows_per_chunk, avg_row_bytes, + data_has_quotes, + &row_offsets, &row_offsets_hdr, + &next_offset); + if (cnt == -2) + cnt = build_row_offsets_limited(buf, file_size, chunk_offset, + rows_per_chunk, data_has_quotes, + &row_offsets, + &row_offsets_hdr, &next_offset); + if (cnt > 0 && next_offset > chunk_offset) + avg_row_bytes = (next_offset - chunk_offset) / (size_t)cnt; if (cnt <= 0) { scratch_free(row_offsets_hdr); err = (cnt < 0) ? RAY_ERR_CANCEL : RAY_ERR_IO; @@ -3649,7 +3988,7 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool ray_t* tbl = csv_materialize_rows(buf, file_size, row_offsets, cnt, ncols, delimiter, col_name_ids, - resolved_types); + resolved_types, sym_dom); scratch_free(row_offsets_hdr); if (!tbl || RAY_IS_ERR(tbl)) { err = (tbl && RAY_IS_ERR(tbl)) ? ray_err_from_obj(tbl) @@ -3658,14 +3997,31 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool break; } - for (int c = 0; c < ncols; c++) { - ray_t* col = ray_table_get_col_idx(tbl, c); - err = csv_splayed_writer_append(&writers[c], col); - if (err != RAY_OK) break; + /* One task per column: each writer owns its file, its cache and + * its symfile domain (the domain probe takes the domain lock, the + * symbol table read its own), so the columns of a chunk are + * encoded and written side by side. */ + { + csv_splayed_append_ctx_t actx = { .writers = writers, .tbl = tbl, + .ncols = ncols, .err = RAY_OK }; + ray_pool_t* wpool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(wpool, ncols, 2)) + ray_pool_dispatch_n(wpool, csv_splayed_append_task, &actx, (uint32_t)ncols); + else + for (int c = 0; c < ncols; c++) csv_splayed_append_task(&actx, 0, c, c + 1); + err = actx.err; } ray_release(tbl); if (err != RAY_OK) break; wrote_any = true; + /* The chunk's bytes are done with: drop them from the mapping so a + * long file does not pin its whole length in resident memory. */ + if (next_offset > chunk_offset) { + size_t ps = (size_t)sysconf(_SC_PAGESIZE); + size_t a = (chunk_offset + ps - 1) & ~(ps - 1); + size_t b = next_offset & ~(ps - 1); + if (b > a) madvise((void*)(buf + a), b - a, MADV_DONTNEED); + } if (cnt == 0) break; chunk_offset = next_offset; } @@ -3673,8 +4029,9 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool /* Flush the symfile BEFORE committing column files (writer_close * renames tmp → final): columns must never reference positions the * symfile doesn't persist (sym-first crash ordering). */ - if (err == RAY_OK && sym_dom) + if (err == RAY_OK && sym_dom) { err = ray_sym_domain_flush(sym_dom, false); + } for (int c = 0; c < ncols; c++) { ray_err_t cerr = (err == RAY_OK) ? csv_splayed_writer_close(&writers[c]) @@ -3886,7 +4243,8 @@ ray_err_t ray_csv_save_parted_named_opts(const char* path, char delimiter, bool } ray_t* tbl = csv_materialize_rows(buf, file_size, row_offsets, - cnt, ncols, delimiter, col_name_ids, resolved_types); + cnt, ncols, delimiter, col_name_ids, resolved_types, + NULL); if (!tbl || RAY_IS_ERR(tbl)) { err = (tbl && RAY_IS_ERR(tbl)) ? ray_err_from_obj(tbl) : RAY_ERR_OOM; diff --git a/src/mem/arena.c b/src/mem/arena.c index df44caf8..cc2e6e0b 100644 --- a/src/mem/arena.c +++ b/src/mem/arena.c @@ -109,26 +109,54 @@ ray_t* ray_arena_alloc(ray_arena_t* arena, size_t nbytes) { return v; } -ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { +size_t ray_arena_str_bytes(size_t len) { + if (len < 7) return 32; + /* [U8 header (32) | data (len+1) | pad to 32 | STR header (32)] */ + return (((32 + len + 1) + 31) & ~(size_t)31) + 32; +} + +void* ray_arena_alloc_raw(ray_arena_t* arena, size_t nbytes) { + if (!arena) return NULL; + if (nbytes > SIZE_MAX - (ARENA_ALIGN - 1)) return NULL; + size_t block_size = ARENA_ALIGN_UP(nbytes); + ray_arena_chunk_t* c = arena->chunks; + if (c->used + block_size > c->cap) { + size_t new_cap = arena->chunk_size; + if (block_size > new_cap) new_cap = ARENA_ALIGN_UP(block_size); + ray_arena_chunk_t* nc = arena_new_chunk(new_cap); + if (!nc) return NULL; + nc->next = arena->chunks; + arena->chunks = nc; + c = nc; + } + void* p = chunk_data(c) + c->used; + c->used += block_size; + return p; +} + +ray_t* ray_arena_str_at(void* at, const char* s, size_t len) { if (len < 7) { - /* SSO: bytes inline in the header (ray_arena_alloc zeroes it and sets - * RAY_ATTR_ARENA + rc=1). */ - ray_t* v = ray_arena_alloc(arena, 0); - if (!v) return NULL; + /* SSO: bytes inline in the header. */ + ray_t* v = (ray_t*)at; + memset(v, 0, 32); + v->attrs = RAY_ATTR_ARENA; + ray_atomic_store(&v->rc, 1); v->type = -RAY_STR; v->slen = (uint8_t)len; if (len > 0) memcpy(v->sdata, s, len); v->sdata[len] = '\0'; return v; } - /* Long string: fused single allocation for the U8 data vec + the STR atom. + /* Long string: fused single block for the U8 data vec + the STR atom. * Layout: [U8 ray_t header (32) | data (len+1) | pad to 32 | STR header (32)]. - * One arena_alloc instead of two. 32-byte arena alignment keeps the atom's - * obj pointer low byte out of is_sso()'s 1..7 SSO range. */ + * 32-byte arena alignment keeps the atom's obj pointer low byte out of + * is_sso()'s 1..7 SSO range. */ size_t data_size = len + 1; size_t chars_block = ((32 + data_size) + 31) & ~(size_t)31; /* align up to 32 */ - ray_t* chars = ray_arena_alloc(arena, chars_block); - if (!chars) return NULL; + ray_t* chars = (ray_t*)at; + memset(chars, 0, 32); + chars->attrs = RAY_ATTR_ARENA; + ray_atomic_store(&chars->rc, 1); chars->type = RAY_U8; chars->len = (int64_t)len; memcpy(ray_data(chars), s, len); @@ -143,6 +171,12 @@ ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { return v; } +ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { + void* at = ray_arena_alloc_raw(arena, ray_arena_str_bytes(len)); + if (!at) return NULL; + return ray_arena_str_at(at, s, len); +} + bool ray_arena_reserve(ray_arena_t* arena, size_t bytes) { if (!arena) return false; if (bytes == 0) return true; diff --git a/src/mem/arena.h b/src/mem/arena.h index 1cce80eb..d7861fd5 100644 --- a/src/mem/arena.h +++ b/src/mem/arena.h @@ -45,6 +45,14 @@ ray_t* ray_arena_alloc(ray_arena_t* arena, size_t nbytes); * heap (used by the global sym table and FILE sym domains). NULL on OOM. */ ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len); +/* Bulk string construction: reserve one raw region for many atoms, then + * build each atom in place (ray_arena_str == alloc_raw + str_at). The + * region is arena memory: 32-byte aligned, released with the arena. Lets + * a caller carve one region per worker and build atoms in parallel. */ +size_t ray_arena_str_bytes(size_t len); /* bytes one atom needs */ +void* ray_arena_alloc_raw(ray_arena_t* arena, size_t nbytes); /* NULL on OOM */ +ray_t* ray_arena_str_at(void* at, const char* s, size_t len); /* at: ray_arena_str_bytes(len) */ + /* Ensure the arena can serve subsequent allocations totalling at least * `bytes` without the head chunk needing to grow. If the head chunk has * enough free space already, this is a no-op; otherwise a new chunk with diff --git a/src/ops/idxop.c b/src/ops/idxop.c index 6927a52d..b4ec5f78 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -32,6 +32,8 @@ #include "ops/ops.h" #include "ops/rowsel.h" #include "ops/hash.h" /* ray_hash_bytes: STR hash-index key word */ +#include "core/pool.h" /* parallel hash-index build */ +#include "mem/sys.h" /* ray_sys_alloc: build scratch off the buddy heap */ #include #include #include @@ -1084,6 +1086,324 @@ ray_t* ray_index_inline_map(uint8_t* region) { * hit yields the contiguous ascending slice rows[offs[gid]..offs[gid+1]). * -------------------------------------------------------------------------- */ +/* Parallel build of the CSR hash layout for numeric / SYM keys. The + * result is the serial walk's: groups numbered by first occurrence, rows + * ascending inside a group, nulls excluded — assembled in + * partition-parallel passes: + * + * A row ranges: key word per row, rows bucketed by hash partition + * (a partition's rows stay ascending: the ranges are in row order); + * B per partition: open-addressing dedupe into local groups, each + * group's first row marked in a bitmap; + * C a group's number is the rank of its first row among all marked + * rows (block popcounts + one prefix) — exactly the order the + * serial walk assigns; + * D per partition: keys and counts into the group's slots; + * E per partition: row scatter (of[] as cursor); then the key table, + * filled in group order on the calling thread so its bytes match the + * serial build. + * + * Every array a pass writes at random is faulted in beforehand from all + * workers in slices: a fresh mapping faulted at random from every worker + * serialises on the page-table locks. Returns false (nothing allocated) + * when the pool cannot be used; the caller then runs the serial walk. */ +typedef struct { + ray_t* v; + const uint8_t* base; + int64_t n; + int n_tasks; + int n_part; + int part_shift; /* partition = mix64(key) >> shift */ + uint64_t* kw; /* [n] key word (non-null rows) */ + int64_t* pr; /* [n_keys] row ids grouped by partition */ + int64_t* lg; /* [n_keys] local group of pr[j] */ + int64_t* cnt; /* [n_tasks * n_part] rows per (task, partition) */ + int64_t* part_off; /* [n_part + 1] */ + int64_t* ng_p; /* [n_part] local groups per partition */ + int64_t* gfirst; /* [n_keys] first row per local group (partition-relative) */ + int64_t* gcount; /* [n_keys] rows per local group, then its global number */ + uint64_t* bits; /* [n/64 + 1] first-row marks */ + int64_t* blk_rank; /* [n/HP_BLOCK + 1] exclusive prefix of block popcounts */ + int64_t* gk; /* [n_groups] keys */ + int64_t* of; /* [n_groups + 1] */ + int64_t* rw; /* [n_keys] */ + int64_t* tbl; /* [cap] */ + uint64_t tmask; + _Atomic(bool) oom; +} hash_par_t; + +#define HP_BLOCK 4096 + +static inline int64_t hp_task_lo(const hash_par_t* h, int64_t t) { return h->n * t / h->n_tasks; } + +static void hp_pass_a(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t lo = hp_task_lo(h, start), hi = hp_task_lo(h, start + 1); + int64_t* cnt = h->cnt + start * h->n_part; + for (int64_t i = lo; i < hi; i++) { + if (ray_vec_is_null(h->v, i)) { h->kw[i] = 0; continue; } + uint64_t k = hash_row_key_word(h->v, h->base, i); + h->kw[i] = k; + cnt[mix64(k) >> h->part_shift]++; + } +} + +static void hp_pass_a2(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t lo = hp_task_lo(h, start), hi = hp_task_lo(h, start + 1); + int64_t* cur = h->cnt + start * h->n_part; /* now the write cursors */ + for (int64_t i = lo; i < hi; i++) { + if (ray_vec_is_null(h->v, i)) continue; + h->pr[cur[mix64(h->kw[i]) >> h->part_shift]++] = i; + } +} + +static void hp_pass_b(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p], hi = h->part_off[p + 1]; + int64_t cnt = hi - lo; + h->ng_p[p] = 0; + if (cnt == 0) return; + uint64_t cap = next_pow2((uint64_t)cnt * 2 + 1); + if (cap < 16) cap = 16; + int64_t* tab = (int64_t*)ray_sys_alloc((size_t)cap * sizeof(int64_t)); + if (!tab) { atomic_store_explicit(&h->oom, true, memory_order_relaxed); return; } + memset(tab, 0, (size_t)cap * sizeof(int64_t)); + uint64_t mask = cap - 1; + int64_t* gfirst = h->gfirst + lo; + int64_t* gcount = h->gcount + lo; + int64_t ng = 0; + for (int64_t j = lo; j < hi; j++) { + int64_t i = h->pr[j]; + uint64_t k = h->kw[i]; + uint64_t slot = mix64(k) & mask; + for (;;) { + int64_t g1 = tab[slot]; + if (g1 == 0) { + tab[slot] = ng + 1; + gfirst[ng] = i; + gcount[ng] = 1; + h->lg[j] = ng; + /* the partition's rows are ascending, so i is this group's + * first row; the word is shared with other partitions */ + atomic_fetch_or_explicit((_Atomic(uint64_t)*)&h->bits[i >> 6], + (uint64_t)1 << (i & 63), memory_order_relaxed); + ng++; + break; + } + if (h->kw[gfirst[g1 - 1]] == k) { h->lg[j] = g1 - 1; gcount[g1 - 1]++; break; } + slot = (slot + 1) & mask; + } + } + h->ng_p[p] = ng; + ray_sys_free(tab); +} + +/* per-block popcount of the first-row marks (prefixed serially after) */ +static void hp_pass_c(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t w0 = start * (HP_BLOCK / 64), w1 = w0 + HP_BLOCK / 64; + int64_t nw = (h->n + 63) / 64; + if (w1 > nw) w1 = nw; + int64_t c = 0; + for (int64_t w = w0; w < w1; w++) c += __builtin_popcountll(h->bits[w]); + h->blk_rank[start] = c; +} + +static inline int64_t hp_rank(const hash_par_t* h, int64_t i) { + int64_t blk = i / HP_BLOCK; + int64_t r = h->blk_rank[blk]; + int64_t w = i >> 6; + for (int64_t x = blk * (HP_BLOCK / 64); x < w; x++) r += __builtin_popcountll(h->bits[x]); + return r + __builtin_popcountll(h->bits[w] & (((uint64_t)1 << (i & 63)) - 1)); +} + +static void hp_pass_d(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p]; + const int64_t* gfirst = h->gfirst + lo; + int64_t* gcount = h->gcount + lo; + int64_t ng = h->ng_p[p]; + for (int64_t g = 0; g < ng; g++) { + int64_t G = hp_rank(h, gfirst[g]); + h->gk[G] = (int64_t)h->kw[gfirst[g]]; + h->of[G + 1] = gcount[g]; + gcount[g] = G; /* local -> global from here on */ + } +} + +static void hp_pass_e(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p], hi = h->part_off[p + 1]; + const int64_t* gmap = h->gcount + lo; + /* rows of a partition are ascending -> ascending inside each group; + * of[] doubles as the fill cursor (the caller shifts it back); the + * partition's groups are nobody else's, so the cursors are private */ + for (int64_t j = lo; j < hi; j++) + h->rw[h->of[gmap[h->lg[j]]]++] = h->pr[j]; +} + +/* Zero (and so fault in) up to 8 fresh regions in parallel slices: each + * task owns one contiguous slice per region, so the page faults spread + * over the workers without two of them ever meeting on a page. */ +typedef struct { void* p[8]; size_t bytes[8]; int n; int n_tasks; } hp_touch_t; +static void hp_touch_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hp_touch_t* t = (hp_touch_t*)raw; + for (int r = 0; r < t->n; r++) { + size_t lo = t->bytes[r] * (size_t)start / (size_t)t->n_tasks; + size_t hi = t->bytes[r] * (size_t)(start + 1) / (size_t)t->n_tasks; + if (hi > lo) memset((char*)t->p[r] + lo, 0, hi - lo); + } +} +static void hp_touch(ray_pool_t* pool, int n_tasks, hp_touch_t* t) { + t->n_tasks = n_tasks; + ray_pool_dispatch_n(pool, hp_touch_fn, t, (uint32_t)n_tasks); +} + +static bool hash_build_par(ray_t* v, ray_t** gkeys_out, ray_t** offs_out, + ray_t** rows_out, ray_t** table_out, uint64_t* mask_out, + int64_t* n_keys_out, int64_t* n_groups_out) { + int64_t n = v->len; + ray_pool_t* pool = ray_pool_get(); + if (!ray_pool_par_dispatch_ok(pool, n, 1 << 16)) return false; + int workers = (int)ray_pool_total_workers(pool); + + hash_par_t h; + memset(&h, 0, sizeof(h)); + h.v = v; h.base = (const uint8_t*)ray_data(v); h.n = n; + h.n_tasks = workers * 4; + if (h.n_tasks > 512) h.n_tasks = 512; + h.n_part = 2; + while (h.n_part < workers * 4 && h.n_part < 512) h.n_part <<= 1; + h.part_shift = 64; + for (int q = h.n_part; q > 1; q >>= 1) h.part_shift--; + + int64_t nw = (n + 63) / 64; + int64_t nblk = (n + HP_BLOCK - 1) / HP_BLOCK; + size_t cnt_b = (size_t)h.n_tasks * (size_t)h.n_part * sizeof(int64_t); + size_t po_b = (size_t)(h.n_part + 1) * sizeof(int64_t); + size_t bits_b = (size_t)(nw + 1) * sizeof(uint64_t); + h.kw = (uint64_t*)ray_sys_alloc((size_t)n * sizeof(uint64_t)); + h.cnt = (int64_t*)ray_sys_alloc(cnt_b); + h.part_off = (int64_t*)ray_sys_alloc(po_b); + h.ng_p = (int64_t*)ray_sys_alloc(po_b); + h.bits = (uint64_t*)ray_sys_alloc(bits_b); + h.blk_rank = (int64_t*)ray_sys_alloc((size_t)(nblk + 1) * sizeof(int64_t)); + ray_t *gkeys = NULL, *offs = NULL, *rows = NULL, *table = NULL; + bool ok = h.kw && h.cnt && h.part_off && h.ng_p && h.bits && h.blk_rank; + if (!ok) goto done; + memset(h.cnt, 0, cnt_b); memset(h.part_off, 0, po_b); memset(h.ng_p, 0, po_b); + memset(h.blk_rank, 0, (size_t)(nblk + 1) * sizeof(int64_t)); + { + hp_touch_t t = { .p = { h.kw, h.bits }, .bytes = { (size_t)n * sizeof(uint64_t), bits_b }, .n = 2 }; + hp_touch(pool, h.n_tasks, &t); + } + + /* A: key words + partition counts, then the bucketed row ids */ + ray_pool_dispatch_n(pool, hp_pass_a, &h, (uint32_t)h.n_tasks); + if (ray_interrupted()) { ok = false; goto done; } + { + int64_t run = 0; + for (int p = 0; p < h.n_part; p++) { + h.part_off[p] = run; + for (int t = 0; t < h.n_tasks; t++) { + int64_t c = h.cnt[(int64_t)t * h.n_part + p]; + h.cnt[(int64_t)t * h.n_part + p] = run; + run += c; + } + } + h.part_off[h.n_part] = run; + } + int64_t n_keys = h.part_off[h.n_part]; + size_t kb = (size_t)(n_keys > 0 ? n_keys : 1) * sizeof(int64_t); + h.pr = (int64_t*)ray_sys_alloc(kb); + h.lg = (int64_t*)ray_sys_alloc(kb); + h.gfirst = (int64_t*)ray_sys_alloc(kb); + h.gcount = (int64_t*)ray_sys_alloc(kb); + if (!h.pr || !h.lg || !h.gfirst || !h.gcount) { ok = false; goto done; } + { + hp_touch_t t = { .p = { h.pr, h.lg, h.gfirst, h.gcount }, .bytes = { kb, kb, kb, kb }, .n = 4 }; + hp_touch(pool, h.n_tasks, &t); + } + ray_pool_dispatch_n(pool, hp_pass_a2, &h, (uint32_t)h.n_tasks); + if (ray_interrupted()) { ok = false; goto done; } + + /* B: per-partition dedupe */ + ray_pool_dispatch_n(pool, hp_pass_b, &h, (uint32_t)h.n_part); + if (ray_interrupted() || atomic_load_explicit(&h.oom, memory_order_relaxed)) { ok = false; goto done; } + int64_t n_groups = 0; + for (int p = 0; p < h.n_part; p++) n_groups += h.ng_p[p]; + + /* C: first-occurrence numbering = rank of the group's first row */ + ray_pool_dispatch_n(pool, hp_pass_c, &h, (uint32_t)nblk); + { + int64_t run = 0; + for (int64_t b = 0; b < nblk; b++) { int64_t c = h.blk_rank[b]; h.blk_rank[b] = run; run += c; } + h.blk_rank[nblk] = run; + } + + gkeys = ray_vec_new(RAY_I64, n_groups > 0 ? n_groups : 1); + offs = ray_vec_new(RAY_I64, n_groups + 1); + rows = ray_vec_new(RAY_I64, n_keys > 0 ? n_keys : 1); + uint64_t cap = next_pow2((uint64_t)(n_groups < 4 ? 8 : 2 * n_groups)); + if (cap < 8) cap = 8; + table = ray_vec_new(RAY_I64, (int64_t)cap); + if (!gkeys || RAY_IS_ERR(gkeys) || !offs || RAY_IS_ERR(offs) || + !rows || RAY_IS_ERR(rows) || !table || RAY_IS_ERR(table)) { ok = false; goto done; } + gkeys->len = n_groups; offs->len = n_groups + 1; rows->len = n_keys; table->len = (int64_t)cap; + h.gk = (int64_t*)ray_data(gkeys); h.of = (int64_t*)ray_data(offs); + h.rw = (int64_t*)ray_data(rows); h.tbl = (int64_t*)ray_data(table); + h.tmask = cap - 1; + { + hp_touch_t t = { .p = { h.gk, h.of, h.rw, h.tbl }, + .bytes = { (size_t)(n_groups > 0 ? n_groups : 1) * sizeof(int64_t), + (size_t)(n_groups + 1) * sizeof(int64_t), kb, + (size_t)cap * sizeof(int64_t) }, .n = 4 }; + hp_touch(pool, h.n_tasks, &t); + } + + /* D: keys and counts; E: rows and the key table */ + ray_pool_dispatch_n(pool, hp_pass_d, &h, (uint32_t)h.n_part); + for (int64_t g = 0; g < n_groups; g++) h.of[g + 1] += h.of[g]; + ray_pool_dispatch_n(pool, hp_pass_e, &h, (uint32_t)h.n_part); + for (int64_t g = n_groups; g > 0; g--) h.of[g] = h.of[g - 1]; + h.of[0] = 0; + /* Key table filled in group order, exactly as the serial build does: + * the persisted index is then the same bytes whatever the core count. */ + for (int64_t g = 0; g < n_groups; g++) { + uint64_t slot = mix64((uint64_t)h.gk[g]) & h.tmask; + while (h.tbl[slot] != 0) slot = (slot + 1) & h.tmask; + h.tbl[slot] = g + 1; + } + if (ray_interrupted()) { ok = false; goto done; } + + *gkeys_out = gkeys; *offs_out = offs; *rows_out = rows; *table_out = table; + *mask_out = h.tmask; *n_keys_out = n_keys; *n_groups_out = n_groups; + +done: + if (!ok) { + if (gkeys && !RAY_IS_ERR(gkeys)) ray_release(gkeys); + if (offs && !RAY_IS_ERR(offs)) ray_release(offs); + if (rows && !RAY_IS_ERR(rows)) ray_release(rows); + if (table && !RAY_IS_ERR(table)) ray_release(table); + } + ray_sys_free(h.kw); ray_sys_free(h.pr); ray_sys_free(h.lg); + ray_sys_free(h.gfirst); ray_sys_free(h.gcount); ray_sys_free(h.cnt); + ray_sys_free(h.part_off); ray_sys_free(h.ng_p); ray_sys_free(h.bits); + ray_sys_free(h.blk_rank); + return ok; +} + ray_t* ray_index_attach_hash(ray_t** vp) { /* allow_str: keyed on a byte hash with payload-verified compares; * allow_sym: RAY_SYM uses domain ids. */ @@ -1092,6 +1412,34 @@ ray_t* ray_index_attach_hash(ray_t** vp) { bool is_str = (v->type == RAY_STR); int64_t n = v->len; + ray_t* table = NULL; + uint64_t mask = 0; + { + ray_t *pg = NULL, *po = NULL, *pr = NULL, *pt = NULL; + int64_t pk = 0, pn = 0; + uint64_t pm = 0; + if (!is_str && hash_build_par(v, &pg, &po, &pr, &pt, &pm, &pk, &pn)) { + if (ray_interrupted()) { + ray_release(pg); ray_release(po); ray_release(pr); ray_release(pt); + return ray_error("cancel", "interrupted"); + } + ray_t* idx = ray_index_alloc(RAY_IDX_HASH, v->type, n); + if (!idx || RAY_IS_ERR(idx)) { + ray_release(pg); ray_release(po); ray_release(pr); ray_release(pt); + return idx ? idx : ray_error("oom", NULL); + } + ray_index_t* ix = ray_index_payload(idx); + ix->u.hash.table = pt; + ix->u.hash.gkeys = pg; + ix->u.hash.offs = po; + ix->u.hash.rows = pr; + ix->u.hash.mask = pm; + ix->u.hash.n_keys = pk; + ix->u.hash.n_groups = pn; + ix->u.hash.order_sym = -1; + return attach_finalize(v, idx); + } + } /* Build-time capacity: sized by rows for O(1) inserts. */ uint64_t bcap = next_pow2((uint64_t)(n < 4 ? 8 : 2 * n)); if (bcap < 8) bcap = 8; @@ -1195,8 +1543,8 @@ ray_t* ray_index_attach_hash(ray_t** vp) { /* Attached/persisted bucket table: sized by DISTINCT keys. */ uint64_t cap = next_pow2((uint64_t)(n_groups < 4 ? 8 : 2 * n_groups)); if (cap < 8) cap = 8; - uint64_t mask = cap - 1; - ray_t* table = ray_vec_new(RAY_I64, (int64_t)cap); + mask = cap - 1; + table = ray_vec_new(RAY_I64, (int64_t)cap); if (!table || RAY_IS_ERR(table)) { ray_release(gkeys); ray_release(offs); ray_release(rows); return table ? table : ray_error("oom", NULL); diff --git a/src/store/splay.c b/src/store/splay.c index d59bd0b2..4fe29bc1 100644 --- a/src/store/splay.c +++ b/src/store/splay.c @@ -23,6 +23,8 @@ #include "splay.h" #include "core/runtime.h" +#include "core/pool.h" +#include "mem/sys.h" #include "store/col.h" #include "store/fileio.h" #include "store/serde.h" @@ -512,12 +514,16 @@ ray_t* ray_splay_load(const char* dir, const char* sym_path) { * rewrite), so a later mmap load gets the same block-skip an in-memory build * has. Best-effort and idempotent-ish: ray_col_append_index refuses a file * that is not exactly payload-sized (already indexed), so re-runs are no-ops. */ -void ray_splay_build_indexes(const char* dir, ray_t* tbl) { - if (!dir || !tbl || RAY_IS_ERR(tbl) || tbl->type != RAY_TABLE) return; - int64_t nc = ray_table_ncols(tbl); - for (int64_t c = 0; c < nc; c++) { +/* Index one column of a just-written splayed table (see + * ray_splay_build_indexes). Columns are independent — each reads and + * appends to its own file — so the caller runs one task per column. */ +/* deferred: when non-NULL and the column qualifies for a hash index, the + * computed zone is stored there instead of persisted and nothing is written; + * splay_persist_hash_or_zone finishes the column. */ +static void splay_build_index_col(const char* dir, ray_t* tbl, int64_t c, ray_t** deferred) { + { ray_t* col = ray_table_get_col_idx(tbl, c); - if (!col || RAY_IS_ERR(col)) continue; + if (!col || RAY_IS_ERR(col)) return; /* Explicit SYM index: a SYM column carrying a grouped (hash) index in * memory gets a hash index persisted inline — regardless of length (the @@ -557,7 +563,7 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { } } } - continue; + return; } /* Explicit STR index: a grouped / unique hash on a STR column is keyed @@ -575,10 +581,10 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { (void)ray_col_append_index(path, ray_index_payload(col->index), col->len, RAY_STR); } - continue; + return; } - if (col->len < (1 << 16)) continue; + if (col->len < (1 << 16)) return; /* STR columns get a dictionary (group on int codes); numeric/temporal * get the per-chunk min/max for block-skip. @@ -593,11 +599,18 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { ray_t* idx = (col->type == RAY_STR) ? ray_index_dict_compute(col) : ray_index_chunk_zone_compute(col, 16); - if (!idx || RAY_IS_ERR(idx)) { if (idx) ray_error_free(idx); continue; } + if (!idx || RAY_IS_ERR(idx)) { if (idx) ray_error_free(idx); return; } if (col->type != RAY_STR && ray_csv_hash_upgrade_check(col->type, col->len, ray_index_payload(idx))) { + if (deferred) { + /* Inside a per-column task: the hash build has its own + * parallel path that needs the pool, so hand the zone + * back and let the caller build the hash afterwards. */ + *deferred = idx; + return; + } ray_t* hi = ray_idx_hash_fn(col); if (hi && !RAY_IS_ERR(hi) && (hi->attrs & RAY_ATTR_HAS_INDEX)) { ray_release(idx); /* zone sacrificed for the hash */ @@ -612,7 +625,7 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { ray_index_payload(hi->index), hi->len, hi->type); } ray_release(hi); - continue; + return; } if (hi) { if (RAY_IS_ERR(hi)) ray_error_free(hi); else ray_release(hi); } /* Hash build failed — fall through and persist the zone. */ @@ -631,6 +644,77 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { } } +/* Persist a deferred column: its hash index when the build succeeded + * (hashed[c]), else the zone it was computed with (deferred[c]). */ +typedef struct { const char* dir; ray_t* tbl; ray_t** deferred; ray_t** hashed; } splay_index_ctx_t; + +static void splay_persist_deferred(splay_index_ctx_t* x, int64_t c) { + ray_t* col = ray_table_get_col_idx(x->tbl, c); + ray_t* nstr = ray_sym_str(ray_table_col_name(x->tbl, c)); + char path[1100]; + int n = (nstr && !RAY_IS_ERR(nstr)) + ? snprintf(path, sizeof(path), "%s/%.*s", x->dir, (int)ray_str_len(nstr), ray_str_ptr(nstr)) + : -1; + bool have_path = n > 0 && n < (int)sizeof(path); + ray_t* hi = x->hashed[c]; + if (hi) { + if (have_path) + (void)ray_col_append_index(path, ray_index_payload(hi->index), hi->len, hi->type); + ray_release(hi); + } else if (have_path) { + (void)ray_col_append_index(path, ray_index_payload(x->deferred[c]), col->len, col->type); + } + ray_release(x->deferred[c]); + x->hashed[c] = NULL; x->deferred[c] = NULL; +} + +static void splay_persist_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + splay_index_ctx_t* x = (splay_index_ctx_t*)raw; + if (x->deferred[start]) splay_persist_deferred(x, start); +} +static void splay_build_index_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + splay_index_ctx_t* c = (splay_index_ctx_t*)raw; + splay_build_index_col(c->dir, c->tbl, start, &c->deferred[start]); +} + +void ray_splay_build_indexes(const char* dir, ray_t* tbl) { + + if (!dir || !tbl || RAY_IS_ERR(tbl) || tbl->type != RAY_TABLE) return; + int64_t nc = ray_table_ncols(tbl); + if (nc <= 0) return; + /* One task per column: the zone / dictionary / hash builds are per-row + * scans of each column and used to run one after another on the + * calling thread — the longest serial stretch of a CSV → splayed load. */ + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, nc, 2)) { + /* Zones / dictionaries per column in parallel; the columns that + * qualify for a hash index come back deferred and are built one + * after another on this thread, each hash build parallel inside. */ + ray_t** deferred = (ray_t**)ray_sys_alloc((size_t)nc * 2 * sizeof(ray_t*)); + if (!deferred) { + for (int64_t c = 0; c < nc; c++) splay_build_index_col(dir, tbl, c, NULL); + return; + } + memset(deferred, 0, (size_t)nc * 2 * sizeof(ray_t*)); + splay_index_ctx_t ctx = { .dir = dir, .tbl = tbl, .deferred = deferred, .hashed = deferred + nc }; + ray_pool_dispatch_n(pool, splay_build_index_task, &ctx, (uint32_t)nc); + /* Hash builds one after another (each parallel inside), then the + * writes of all deferred columns together. */ + for (int64_t c = 0; c < nc; c++) { + if (!deferred[c]) continue; + ray_t* hi = ray_idx_hash_fn(ray_table_get_col_idx(tbl, c)); + if (hi && !RAY_IS_ERR(hi) && (hi->attrs & RAY_ATTR_HAS_INDEX)) ctx.hashed[c] = hi; + else if (hi) { if (RAY_IS_ERR(hi)) ray_error_free(hi); else ray_release(hi); } + } + ray_pool_dispatch_n(pool, splay_persist_task, &ctx, (uint32_t)nc); + ray_sys_free(deferred); + } else { + for (int64_t c = 0; c < nc; c++) splay_build_index_col(dir, tbl, c, NULL); + } +} + ray_t* ray_read_splayed(const char* dir, const char* sym_path) { return splay_load_impl(dir, sym_path, true); } diff --git a/src/table/domain.c b/src/table/domain.c index 0e9f5869..a7d13f77 100644 --- a/src/table/domain.c +++ b/src/table/domain.c @@ -63,6 +63,8 @@ #include "mem/arena.h" /* ray_arena_t / ray_arena_str — domain atom storage */ #include "store/fileio.h" /* flock + tmp/rename protocol for flush */ #include "ops/hash.h" /* ray_hash_bytes (same hash family as g_sym) */ +#include "core/pool.h" /* batch intern: parallel read-only probe */ +#include "sym.h" /* ray_sym_intern_prehashed (runtime fallback) */ #include #include #include @@ -177,6 +179,10 @@ struct ray_sym_domain_s { * inserts incrementally; growth rebuilds). Guarded by g_dom_lock. */ uint64_t* buckets; uint64_t bucket_mask; /* cap - 1; 0 = not built yet */ + /* Batch interns currently probing a snapshot of `buckets` outside the + * lock (guarded by g_dom_lock). While non-zero a replaced table is + * retired instead of freed. */ + int32_t batch_inflight; dom_retired_t* retired; /* replaced atom arrays + old LUTs */ @@ -292,6 +298,15 @@ static bool dom_retire(ray_sym_domain_t* d, void* p) { d->retired = r; return true; } + +/* Drop a replaced reverse-index table: freed at once unless a batch intern + * is probing a snapshot outside the lock, then retired with the domain. */ +static bool dom_drop_buckets_locked(ray_sym_domain_t* d, uint64_t* old) { + if (!old) return true; + if (d->batch_inflight == 0) { ray_sys_free(old); return true; } + return dom_retire(d, old); +} + /* Publish (map, offsets, count) as the domain's raw snapshot; the previous * record is retired. Called with the file's current image in place. */ static bool dom_publish_raw_snap(ray_sym_domain_t* d) { @@ -516,7 +531,17 @@ static bool dom_extend_from_file_locked(ray_sym_domain_t* d, size_t st_size) { * the grown count (OOB reads for lock-free consumers) — the * same corner dom_append_locked hits; mirror its loud abort * (the count is already published, there is no clean undo). */ - if (d->buckets) { ray_sys_free(d->buckets); d->buckets = NULL; d->bucket_mask = 0; } + if (d->buckets) { + /* A batch intern may be probing a snapshot of this + * table outside the lock. */ + if (!dom_drop_buckets_locked(d, d->buckets)) { + fprintf(stderr, "rayforce: sym domain '%s': OOM retiring " + "reverse index after external extend\n", + d->path ? d->path : "?"); + abort(); + } + d->buckets = NULL; d->bucket_mask = 0; + } int64_t* lut = atomic_load_explicit(&d->runtime_lut, memory_order_relaxed); if (lut) { if (!dom_retire(d, lut)) { @@ -763,15 +788,33 @@ static bool dom_build_index_locked(ray_sym_domain_t* d, int64_t extra) { if (!buckets) return false; uint64_t mask = cap - 1; - for (int64_t i = 0; i < count; i++) { - ray_t* a = dom_atom_at_locked(d, i); - if (!a) { ray_sys_free(buckets); return false; } - uint32_t h = (uint32_t)ray_hash_bytes(ray_str_ptr(a), ray_str_len(a)); - uint64_t slot = h & mask; - while (buckets[slot] != 0) slot = (slot + 1) & mask; - buckets[slot] = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)i + 1); + if (d->buckets) { + /* Growth: the old table covers every published entry and carries + * the hashes — re-slot its entries, no string hashing. */ + uint64_t old_cap = d->bucket_mask + 1; + for (uint64_t i = 0; i < old_cap; i++) { + uint64_t e = d->buckets[i]; + if (e == 0) continue; + uint64_t slot = (uint32_t)(e >> 32) & mask; + while (buckets[slot] != 0) slot = (slot + 1) & mask; + buckets[slot] = e; + } + } else { + for (int64_t i = 0; i < count; i++) { + ray_t* a = dom_atom_at_locked(d, i); + if (!a) { ray_sys_free(buckets); return false; } + uint32_t h = (uint32_t)ray_hash_bytes(ray_str_ptr(a), ray_str_len(a)); + uint64_t slot = h & mask; + while (buckets[slot] != 0) slot = (slot + 1) & mask; + buckets[slot] = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)i + 1); + } + } + /* A batch intern may be probing a snapshot of the old table outside + * the lock (see ray_sym_domain_intern_batch). */ + if (d->buckets && !dom_drop_buckets_locked(d, d->buckets)) { + ray_sys_free(buckets); + return false; } - ray_sys_free(d->buckets); d->buckets = buckets; d->bucket_mask = mask; return true; @@ -919,6 +962,345 @@ int64_t ray_sym_domain_intern(ray_sym_domain_t* dom, const char* str, size_t len return pos; } +/* ---- batch intern ---------------------------------------------------------- */ + +/* Read-only probe over a snapshot of the reverse index taken under the + * lock. Everything the snapshot points at outlives it: replaced bucket + * tables and atom arrays are retired, not freed, and the file prefix is + * read through the pinned raw snapshot. Entries appended after the + * snapshot (pos >= count) are ignored here and resolved under the lock. */ +typedef struct { + const uint64_t* buckets; + uint64_t mask; + ray_t* const* atoms; + int64_t count; + ray_sym_domain_raw_t raw; /* raw.count == 0: no file prefix */ + const char* const* strs; + const size_t* lens; + const uint32_t* hashes; + int64_t* out_pos; + /* misses, grouped by hash partition (hash >> part_shift) */ + int64_t* miss; /* [n_miss] batch indices */ + int64_t* uniq; /* [n_miss] first occurrences, per partition segment */ + int64_t* part_off; /* [n_part + 1] */ + int64_t* uniq_n; /* [n_part] distinct misses per partition */ + int64_t* bytes_p; /* [n_part] arena bytes the partition's atoms need */ + void** region; /* [n_part] arena region per partition */ + ray_t** atoms_w; /* current atom array (fill target) */ + /* Set when a probe met an entry it could not compare (no atom, no + * raw bytes): the misses are then resolved under the lock instead. */ + _Atomic(bool) unsure; + int part_shift; + _Atomic(bool) oom; +} dom_batch_ctx_t; + +static void dom_batch_probe_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dom_batch_ctx_t* b = (const dom_batch_ctx_t*)raw; + for (int64_t i = start; i < end; i++) { + uint32_t h = b->hashes[i]; + size_t len = b->lens[i]; + const char* s = b->strs[i]; + uint64_t slot = h & b->mask; + int64_t found = -1; + for (;;) { + uint64_t e = atomic_load_explicit((_Atomic(uint64_t)*)&b->buckets[slot], + memory_order_relaxed); + if (e == 0) break; + if ((uint32_t)(e >> 32) == h) { + int64_t pos = (int64_t)(uint32_t)e - 1; + if (pos < b->count) { + const char* p = NULL; + size_t l = 0; + ray_t* a = atomic_load_explicit((_Atomic(ray_t*)*)&b->atoms[pos], + memory_order_acquire); + if (a) { p = ray_str_ptr(a); l = ray_str_len(a); } + else if (pos < b->raw.count) p = ray_sym_domain_raw_str(&b->raw, pos, &l); + else atomic_store_explicit((_Atomic(bool)*)&b->unsure, true, memory_order_relaxed); + if (p && l == len && (len == 0 || memcmp(p, s, len) == 0)) { + found = pos; + break; + } + } + } + slot = (slot + 1) & b->mask; + } + b->out_pos[i] = found; + } +} + +/* Dedupe the misses of one hash partition among themselves: the first + * occurrence stays a miss (out_pos -1) and is listed in the partition's + * segment of `uniq`; a repeat records its representative as -(rep + 2). + * Partitions are disjoint by hash, so no two tasks ever see the same + * string. */ +static void dom_batch_dedupe_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dom_batch_ctx_t* b = (dom_batch_ctx_t*)raw; + for (int64_t p = start; p < end; p++) { + int64_t lo = b->part_off[p], hi = b->part_off[p + 1]; + int64_t cnt = hi - lo; + if (cnt == 0) { b->uniq_n[p] = 0; continue; } + uint64_t cap = 16; + while ((uint64_t)cnt * 2 > cap) cap <<= 1; + int64_t* tab = (int64_t*)ray_sys_alloc((size_t)cap * sizeof(int64_t)); + if (!tab) { atomic_store_explicit(&b->oom, true, memory_order_relaxed); b->uniq_n[p] = 0; continue; } + memset(tab, 0xff, (size_t)cap * sizeof(int64_t)); /* -1 = empty */ + uint64_t mask = cap - 1; + int64_t k = 0; + int64_t bytes = 0; + for (int64_t j = lo; j < hi; j++) { + int64_t i = b->miss[j]; + uint32_t h = b->hashes[i]; + uint64_t slot = h & mask; + int64_t rep = -1; + while (tab[slot] >= 0) { + int64_t r = tab[slot]; + if (b->hashes[r] == h && b->lens[r] == b->lens[i] && + (b->lens[i] == 0 || memcmp(b->strs[r], b->strs[i], b->lens[i]) == 0)) { + rep = r; + break; + } + slot = (slot + 1) & mask; + } + if (rep >= 0) { + b->out_pos[i] = -(rep + 2); + } else { + tab[slot] = i; + b->uniq[lo + k++] = i; + bytes += (int64_t)ray_arena_str_bytes(b->lens[i]); + } + } + b->uniq_n[p] = k; + b->bytes_p[p] = bytes; + ray_sys_free(tab); + } +} + +/* Append the partition's distinct strings: build the atoms in the + * partition's arena region at the positions the caller assigned (batch + * order), publish them in the reverse index (CAS on the empty slot keeps + * concurrent partitions from claiming one slot twice) and resolve the + * partition's repeats. Runs under the domain lock; the count is + * published by the caller once every partition is done. */ +static void dom_batch_insert_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dom_batch_ctx_t* b = (dom_batch_ctx_t*)raw; + for (int64_t p = start; p < end; p++) { + int64_t lo = b->part_off[p]; + int64_t hi = lo + b->uniq_n[p]; + char* at = (char*)b->region[p]; + for (int64_t j = lo; j < hi; j++) { + int64_t i = b->uniq[j]; + int64_t pos = b->out_pos[i]; /* assigned in batch order */ + ray_t* s = ray_arena_str_at(at, b->strs[i], b->lens[i]); + at += ray_arena_str_bytes(b->lens[i]); + b->atoms_w[pos] = s; + uint32_t h = b->hashes[i]; + uint64_t e = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)pos + 1); + uint64_t slot = h & b->mask; + for (;;) { + uint64_t cur = 0; + if (atomic_compare_exchange_strong_explicit((_Atomic(uint64_t)*)&b->buckets[slot], + &cur, e, memory_order_relaxed, memory_order_relaxed)) + break; + slot = (slot + 1) & b->mask; + } + } + for (int64_t j = lo; j < b->part_off[p + 1]; j++) { + int64_t i = b->miss[j]; + int64_t v = b->out_pos[i]; + if (v < -1) b->out_pos[i] = b->out_pos[-(v + 2)]; + } + } +} + +/* Serial fallback for the misses: find-or-append one by one under the lock + * (the domain changed under us, or the parallel path ran out of memory). */ +static bool dom_batch_append_serial_locked(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos) { + for (int64_t i = 0; i < n; i++) { + if (out_pos[i] >= 0) continue; + int64_t pos = dom_probe_locked(dom, hashes[i], strs[i], lens[i]); + if (pos < 0) pos = dom_append_locked(dom, hashes[i], strs[i], lens[i]); + if (pos < 0) return false; + out_pos[i] = pos; + } + return true; +} + +bool ray_sym_domain_intern_batch(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos) { + if (!dom || n < 0) return false; + if (n == 0) return true; + if (dom->kind == DOM_RUNTIME) { + for (int64_t i = 0; i < n; i++) { + int64_t id = ray_sym_intern_prehashed(hashes[i], strs[i], lens[i]); + if (id < 0) return false; + out_pos[i] = id; + } + return true; + } + + dom_batch_ctx_t b; + memset(&b, 0, sizeof(b)); + + dom_lock(); + int64_t count = atomic_load_explicit(&dom->count, memory_order_relaxed); + /* Headroom for the whole batch up front so no rebuild happens while + * the batch is in flight. */ + if (!dom->buckets || + (double)(count + n + 1) > 0.7 * (double)(dom->bucket_mask + 1)) { + if (!dom_build_index_locked(dom, n + 1)) { dom_unlock(); return false; } + } + if (count == 0) { + uint32_t h0 = (uint32_t)ray_hash_bytes("", 0); + if (dom_append_locked(dom, h0, "", 0) != 0) { dom_unlock(); return false; } + } + b.buckets = dom->buckets; + b.mask = dom->bucket_mask; + b.atoms = atomic_load_explicit(&dom->atoms, memory_order_acquire); + b.count = atomic_load_explicit(&dom->count, memory_order_acquire); + dom->batch_inflight++; + dom_unlock(); + + if (!ray_sym_domain_raw_pin(dom, &b.raw)) b.raw.count = 0; + b.strs = strs; b.lens = lens; b.hashes = hashes; b.out_pos = out_pos; + + ray_pool_t* pool = ray_pool_get(); + bool par = ray_pool_par_dispatch_ok(pool, n, 4096); + if (par) ray_pool_dispatch(pool, dom_batch_probe_fn, &b, n); + else dom_batch_probe_fn(&b, 0, 0, n); + + /* Misses, grouped by hash partition. */ + int n_part = 1; + if (par) { + int64_t want = (int64_t)ray_pool_total_workers(pool) * 4; + while (n_part < want && n_part < 1024) n_part <<= 1; + } + b.part_shift = 31; + for (int p = n_part; p > 2; p >>= 1) b.part_shift--; + if (n_part == 1) n_part = 2; /* keep the shift below the type width */ + int64_t n_miss = 0; + for (int64_t i = 0; i < n; i++) n_miss += (out_pos[i] < 0); + if (n_miss == 0) { + dom_lock(); + dom->batch_inflight--; + dom_unlock(); + return true; + } + + b.miss = (int64_t*)ray_sys_alloc((size_t)n_miss * sizeof(int64_t)); + b.uniq = (int64_t*)ray_sys_alloc((size_t)n_miss * sizeof(int64_t)); + b.part_off = (int64_t*)ray_sys_alloc((size_t)(n_part + 1) * sizeof(int64_t)); + b.uniq_n = (int64_t*)ray_sys_alloc((size_t)n_part * sizeof(int64_t)); + b.bytes_p = (int64_t*)ray_sys_alloc((size_t)n_part * sizeof(int64_t)); + b.region = (void**)ray_sys_alloc((size_t)n_part * sizeof(void*)); + bool ok = b.miss && b.uniq && b.part_off && b.uniq_n && b.bytes_p && b.region; + if (ok) { + memset(b.part_off, 0, (size_t)(n_part + 1) * sizeof(int64_t)); + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < 0) b.part_off[(hashes[i] >> b.part_shift) + 1]++; + for (int p = 0; p < n_part; p++) b.part_off[p + 1] += b.part_off[p]; + { + int64_t* fill = b.uniq_n; /* scratch cursor per partition */ + memcpy(fill, b.part_off, (size_t)n_part * sizeof(int64_t)); + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < 0) b.miss[fill[hashes[i] >> b.part_shift]++] = i; + } + if (par) ray_pool_dispatch_n(pool, dom_batch_dedupe_fn, &b, (uint32_t)n_part); + else dom_batch_dedupe_fn(&b, 0, 0, n_part); + if (atomic_load_explicit(&b.oom, memory_order_relaxed)) { + /* Undo the repeat marks; the serial path resolves everything. */ + for (int64_t i = 0; i < n; i++) if (out_pos[i] < -1) out_pos[i] = -1; + ok = false; + } + } + + dom_lock(); + dom->batch_inflight--; /* no probe outside the lock past this point */ + bool unchanged = ok && dom->buckets == b.buckets && + atomic_load_explicit(&dom->count, memory_order_relaxed) == b.count && + !atomic_load_explicit(&b.unsure, memory_order_relaxed); + if (unchanged) { + int64_t total = 0; + for (int p = 0; p < n_part; p++) total += b.uniq_n[p]; + int64_t base = b.count; + ok = base + total < (int64_t)UINT32_MAX; + /* Atom array: grow once by replacement (lock-free readers may hold + * the old pointer). */ + if (ok && base + total > dom->atoms_cap) { + int64_t ncap = dom->atoms_cap < 8 ? 8 : dom->atoms_cap; + while (ncap < base + total) ncap *= 2; + ray_t** narr = (ray_t**)ray_sys_alloc((size_t)ncap * sizeof(ray_t*)); + if (!narr) ok = false; + else { + ray_t** old = atomic_load_explicit(&dom->atoms, memory_order_relaxed); + if (base > 0) memcpy(narr, old, (size_t)base * sizeof(ray_t*)); + if (!dom_retire(dom, old)) { ray_sys_free(narr); ok = false; } + else { + atomic_store_explicit(&dom->atoms, narr, memory_order_release); + dom->atoms_cap = ncap; + } + } + } + if (ok) { + /* One arena region per partition; positions in partition order. + * Nothing is published until every partition has built its + * atoms, so a failed reservation costs only arena space. */ + /* New strings take positions in batch order (first occurrence), + * so the symfile does not depend on how the batch was split. */ + int64_t pos = base; + for (int64_t i = 0; i < n; i++) + if (out_pos[i] == -1) out_pos[i] = pos++; + for (int p = 0; p < n_part; p++) { + b.region[p] = NULL; + if (b.bytes_p[p] > 0) { + b.region[p] = ray_arena_alloc_raw(dom->arena, (size_t)b.bytes_p[p]); + if (!b.region[p]) { ok = false; break; } + } + } + if (ok) { + b.atoms_w = atomic_load_explicit(&dom->atoms, memory_order_relaxed); + if (par) ray_pool_dispatch_n(pool, dom_batch_insert_fn, &b, (uint32_t)n_part); + else dom_batch_insert_fn(&b, 0, 0, n_part); + atomic_store_explicit(&dom->count, pos, memory_order_release); + /* Same invalidation as dom_append_locked: the runtime LUT + * no longer covers the vocabulary. */ + int64_t* lut = atomic_load_explicit(&dom->runtime_lut, memory_order_relaxed); + if (lut) { + if (!dom_retire(dom, lut)) { + fprintf(stderr, "rayforce: sym domain '%s': OOM retiring runtime " + "LUT after batch append\n", dom->path ? dom->path : "?"); + abort(); + } + atomic_store_explicit(&dom->runtime_lut, NULL, memory_order_release); + } + } else { + /* Nothing published: undo the assigned positions and the + * repeat marks; the serial path resolves them again. */ + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < -1 || out_pos[i] >= base) out_pos[i] = -1; + } + } + } else { + for (int64_t i = 0; i < n; i++) if (out_pos[i] < -1) out_pos[i] = -1; + } + if (!unchanged || !ok) + ok = dom_batch_append_serial_locked(dom, n, strs, lens, hashes, out_pos); + dom_unlock(); + + ray_sys_free(b.miss); + ray_sys_free(b.uniq); + ray_sys_free(b.part_off); + ray_sys_free(b.uniq_n); + ray_sys_free(b.bytes_p); + ray_sys_free(b.region); + return ok; +} + int64_t ray_sym_domain_count(ray_sym_domain_t* dom) { if (!dom) return 0; if (dom->kind == DOM_RUNTIME) return (int64_t)ray_sym_count(); @@ -1013,18 +1395,37 @@ ray_err_t ray_sym_domain_flush(ray_sym_domain_t* dom, bool durable) { if (fwrite(&magic, 4, 1, f) != 1 || fwrite(&count, 8, 1, f) != 1) err = RAY_ERR_IO; written_size = 12; + /* Records are packed into a large buffer first: two stdio calls per + * entry dominated the flush of a big vocabulary. */ + enum { FLUSH_BUF = 1u << 20 }; + uint8_t* wb = (uint8_t*)ray_sys_alloc(FLUSH_BUF); + size_t wn = 0; + if (!wb) err = RAY_ERR_OOM; for (int64_t i = 0; err == RAY_OK && i < count; i++) { ray_t* s = atoms[i]; size_t slen = ray_str_len(s); if (slen > UINT32_MAX) { err = RAY_ERR_RANGE; break; } uint32_t len32 = (uint32_t)slen; - if (fwrite(&len32, 4, 1, f) != 1 || - (slen > 0 && fwrite(ray_str_ptr(s), 1, slen, f) != slen)) { - err = RAY_ERR_IO; - break; + if (wn + 4 + slen > FLUSH_BUF) { + if (wn && fwrite(wb, 1, wn, f) != wn) { err = RAY_ERR_IO; break; } + wn = 0; + } + if (4 + slen > FLUSH_BUF) { + /* Oversized record: straight through. */ + if (fwrite(&len32, 4, 1, f) != 1 || + fwrite(ray_str_ptr(s), 1, slen, f) != slen) { + err = RAY_ERR_IO; + break; + } + } else { + memcpy(wb + wn, &len32, 4); + if (slen) memcpy(wb + wn + 4, ray_str_ptr(s), slen); + wn += 4 + slen; } written_size += 4 + slen; } + if (err == RAY_OK && wn && fwrite(wb, 1, wn, f) != wn) err = RAY_ERR_IO; + ray_sys_free(wb); if (fclose(f) != 0 && err == RAY_OK) err = RAY_ERR_IO; } if (err != RAY_OK) goto fail_tmp; diff --git a/src/table/domain.h b/src/table/domain.h index e5c0f9e6..b9ce6610 100644 --- a/src/table/domain.h +++ b/src/table/domain.h @@ -164,6 +164,17 @@ const int64_t* ray_sym_domain_runtime_lut(ray_sym_domain_t* dom); * the shared object immediately; ray_sym_domain_flush persists them. */ int64_t ray_sym_domain_intern(ray_sym_domain_t* dom, const char* str, size_t len); +/* Batch find-or-append of n strings (hashes[i] = (uint32_t)ray_hash_bytes + * of strs[i]); out_pos[i] receives the position. Repeats inside the + * batch resolve to one position. FILE: the lookup of the existing + * vocabulary runs in parallel on the pool; new strings are appended in + * batch order (a string's first occurrence), whatever the worker count. RUNTIME: one ray_sym_intern per entry. + * Returns false on OOM (out_pos then holds -1 for the entries not + * resolved). */ +bool ray_sym_domain_intern_batch(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos); + /* Number of entries in the domain. */ int64_t ray_sym_domain_count(ray_sym_domain_t* dom); diff --git a/test/rfl/io/csv_splayed_dedupe_overflow.rfl b/test/rfl/io/csv_splayed_dedupe_overflow.rfl new file mode 100644 index 00000000..5364d72d --- /dev/null +++ b/test/rfl/io/csv_splayed_dedupe_overflow.rfl @@ -0,0 +1,27 @@ +;; Splayed CSV load of a SYM column whose per-partition dictionary +;; overflows (src/io/csv.c: csv_dedup_task caps a partition at +;; CSV_DEDUP_MAX_ENTS distinct strings — 8192 in debug builds — and the +;; column then interns row by row into the symfile domain, +;; csv_intern_dicts_domain's fallback). 140k distinct strings exceed the +;; cap in at least one partition however the column is split (at most 16 +;; partitions). Release builds keep the cap at 1M and take the batched +;; path; the answers are the same either way. + +(.sys.exec "rm -rf rf_test_splayed_ovf rf_test_splayed_ovf.csv") -- 0 +(.sys.exec "seq 0 139999 | awk '{print $1\",k\"$1}' > rf_test_splayed_ovf.csv") -- 0 +(set Tovf (.csv.splayed [id s] [I64 SYM] "rf_test_splayed_ovf.csv" "rf_test_splayed_ovf/")) +(count Tovf) -- 140000 +(sum (at Tovf 'id)) -- 9799930000 +(count (distinct (at Tovf 's))) -- 140000 +(at (at Tovf 's) 0) -- 'k0 +(at (at Tovf 's) 8192) -- 'k8192 +(at (at Tovf 's) 139999) -- 'k139999 +;; every row's symbol is the one written for its id +(count (select {from: Tovf where: (== s (as 'SYM "k77777"))})) -- 1 +(at (at (select {id: id from: Tovf where: (== s (as 'SYM "k77777"))}) 'id) 0) -- 77777 +;; reload from disk agrees +(set Rovf (.db.splayed.get "rf_test_splayed_ovf/")) +(count Rovf) -- 140000 +(count (distinct (at Rovf 's))) -- 140000 +(at (at Rovf 's) 139999) -- 'k139999 +(.sys.exec "rm -rf rf_test_splayed_ovf rf_test_splayed_ovf.csv") -- 0 diff --git a/test/test_domain.c b/test/test_domain.c index 40dfca97..7da142c7 100644 --- a/test/test_domain.c +++ b/test/test_domain.c @@ -39,6 +39,9 @@ #include "table/domain.h" #include "store/col.h" #include "store/serde.h" +#include "core/pool.h" /* the batch intern probes on the pool */ +#include "ops/hash.h" /* ray_hash_bytes: batch intern takes prehashed entries */ +#include "mem/sys.h" #include "ops/ops.h" /* RAY_PARTED_BASE (parted-flatten adoption test) */ #include #include @@ -401,6 +404,108 @@ static test_result_t test_domain_open_basic(void) { PASS(); } +/* ray_sym_domain_intern_batch: one batch with repeats spread over the hash + * partitions, half the vocabulary already interned one by one. Every + * entry resolves to the position find() reports, repeats agree, the + * pre-interned half keeps its positions, the count grows by exactly the + * new distinct strings, a second identical batch changes nothing and "" + * is position 0. */ +static test_result_t test_domain_intern_batch(void) { + unlink(TMP_DOM_SYM_PATH); + unlink(TMP_DOM_SYM_PATH ".lk"); + TEST_ASSERT_NOT_NULL(ray_pool_get()); /* parallel probe path */ + + ray_sym_domain_t* dom = ray_sym_domain_open_or_create(TMP_DOM_SYM_PATH); + TEST_ASSERT_NOT_NULL(dom); + + enum { NV = 20000, NB = 60000, SL = 16 }; + char* vocab = (char*)ray_sys_alloc((size_t)NV * SL); + int64_t* pre = (int64_t*)ray_sys_alloc((size_t)NV * sizeof(int64_t)); + int64_t* seen = (int64_t*)ray_sys_alloc((size_t)NV * sizeof(int64_t)); + const char** strs = (const char**)ray_sys_alloc((size_t)NB * sizeof(char*)); + size_t* lens = (size_t*)ray_sys_alloc((size_t)NB * sizeof(size_t)); + uint32_t* hashes = (uint32_t*)ray_sys_alloc((size_t)NB * sizeof(uint32_t)); + int64_t* pos = (int64_t*)ray_sys_alloc((size_t)NB * sizeof(int64_t)); + int64_t* pos2 = (int64_t*)ray_sys_alloc((size_t)NB * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(vocab); TEST_ASSERT_NOT_NULL(pre); TEST_ASSERT_NOT_NULL(seen); + TEST_ASSERT_NOT_NULL(strs); TEST_ASSERT_NOT_NULL(lens); TEST_ASSERT_NOT_NULL(hashes); + TEST_ASSERT_NOT_NULL(pos); TEST_ASSERT_NOT_NULL(pos2); + + for (int v = 0; v < NV; v++) { + snprintf(vocab + (size_t)v * SL, SL, "v%05d_%c", v, 'a' + v % 26); + seen[v] = -1; + pre[v] = -1; + } + for (int v = 0; v < NV / 2; v++) { + const char* sv = vocab + (size_t)v * SL; + pre[v] = ray_sym_domain_intern(dom, sv, strlen(sv)); + TEST_ASSERT(pre[v] > 0, "pre-intern gets a position"); + } + int64_t count_before = ray_sym_domain_count(dom); + TEST_ASSERT_EQ_I(count_before, NV / 2 + 1); /* + reserved "" */ + + for (int j = 0; j < NB; j++) { + int v = (int)(((int64_t)j * 7919) % NV); /* every string ~3 times */ + const char* sv = vocab + (size_t)v * SL; + strs[j] = sv; + lens[j] = strlen(sv); + hashes[j] = (uint32_t)ray_hash_bytes(sv, lens[j]); + pos[j] = -7; + } + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, NB, strs, lens, hashes, pos)); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 1); + + for (int j = 0; j < NB; j++) { + int v = (int)(((int64_t)j * 7919) % NV); + TEST_ASSERT(pos[j] > 0 && pos[j] <= NV, "position in range, never 0"); + if (seen[v] < 0) seen[v] = pos[j]; + TEST_ASSERT_EQ_I(pos[j], seen[v]); /* repeats agree */ + if (pre[v] >= 0) TEST_ASSERT_EQ_I(pos[j], pre[v]); /* hits keep their position */ + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, strs[j], lens[j]), pos[j]); + ray_t* a = ray_sym_domain_str(dom, pos[j]); + TEST_ASSERT_NOT_NULL(a); + TEST_ASSERT_EQ_U(ray_str_len(a), lens[j]); + TEST_ASSERT_MEM_EQ(lens[j], ray_str_ptr(a), strs[j]); + } + /* distinct positions: every vocabulary entry got exactly one */ + for (int v = 0; v < NV; v++) TEST_ASSERT(seen[v] > 0, "every string appeared"); + + /* the same batch again: all hits, nothing appended */ + for (int j = 0; j < NB; j++) pos2[j] = -7; + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, NB, strs, lens, hashes, pos2)); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 1); + for (int j = 0; j < NB; j++) TEST_ASSERT_EQ_I(pos2[j], pos[j]); + + /* "" resolves to the reserved position 0; a fresh string still appends */ + { + const char* two[2] = { "", "brand_new_entry" }; + size_t tl[2] = { 0, strlen("brand_new_entry") }; + uint32_t th[2] = { (uint32_t)ray_hash_bytes("", 0), (uint32_t)ray_hash_bytes(two[1], tl[1]) }; + int64_t tp[2] = { -7, -7 }; + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, 2, two, tl, th, tp)); + TEST_ASSERT_EQ_I(tp[0], 0); + TEST_ASSERT_EQ_I(tp[1], NV + 1); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 2); + } + + /* the flushed file reopens with the same vocabulary */ + TEST_ASSERT_EQ_I(ray_sym_domain_flush(dom, false), RAY_OK); + ray_sym_domain_release(dom); + ray_sym_domain_t* re = ray_sym_domain_open(TMP_DOM_SYM_PATH); + TEST_ASSERT_NOT_NULL(re); + TEST_ASSERT_EQ_I(ray_sym_domain_count(re), NV + 2); + for (int j = 0; j < NB; j += 997) + TEST_ASSERT_EQ_I(ray_sym_domain_find(re, strs[j], lens[j]), pos[j]); + ray_sym_domain_release(re); + + ray_sys_free(vocab); ray_sys_free(pre); ray_sys_free(seen); + ray_sys_free(strs); ray_sys_free(lens); ray_sys_free(hashes); + ray_sys_free(pos); ray_sys_free(pos2); + unlink(TMP_DOM_SYM_PATH); + unlink(TMP_DOM_SYM_PATH ".lk"); + PASS(); +} + /* Task 7b: open_or_create on a missing file yields an empty writable * domain; "" is seeded at position 0 by the first intern; flush creates * the file; verify-base-unchanged makes a racing writer LOUD. */ @@ -2030,6 +2135,7 @@ const test_entry_t domain_entries[] = { { "domain/concat_text_nulls", test_domain_concat_text_nulls, domain_rt_setup, domain_rt_teardown }, { "domain/runtime_lut", test_domain_runtime_lut, domain_rt_setup, domain_rt_teardown }, { "domain/open_position0_validation", test_domain_open_position0_validation, domain_setup, domain_teardown }, + { "domain/intern_batch", test_domain_intern_batch, domain_setup, domain_teardown }, { "domain/dict_upsert_file_keys", test_domain_dict_upsert_file_keys, domain_rt_setup, domain_rt_teardown }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_index.c b/test/test_index.c index af3a0213..d79a0e3d 100644 --- a/test/test_index.c +++ b/test/test_index.c @@ -26,6 +26,8 @@ #include "test.h" #include #include "mem/heap.h" +#include "mem/sys.h" +#include "core/pool.h" #include "mem/cow.h" #include "vec/vec.h" #include "table/sym.h" @@ -327,6 +329,83 @@ static test_result_t test_index_hash_with_nulls_preserved(void) { PASS(); } +/* Large column: the build runs partition-parallel above 64k rows and must + * produce the serial walk's layout — groups in first-occurrence order, rows + * ascending inside a group, nulls excluded — checked against a reference + * computed the obvious way. */ +static test_result_t test_index_hash_large_parallel(void) { + ray_heap_init(); + /* The parallel build needs the pool; create it before the attach so + * the test does not silently take the serial fallback. */ + ray_pool_t* pool = ray_pool_get(); + TEST_ASSERT_NOT_NULL(pool); + const int64_t n = 300000, kmax = 5003; + ray_t* v = ray_vec_new(RAY_I64, n); + TEST_ASSERT_NOT_NULL(v); + int64_t* xs = (int64_t*)ray_data(v); + for (int64_t i = 0; i < n; i++) + xs[i] = (int64_t)(((uint64_t)i * 2654435761ull) % (uint64_t)kmax) - 17; + v->len = n; + /* every 977th row null */ + for (int64_t i = 0; i < n; i += 977) + TEST_ASSERT_EQ_I(ray_vec_set_null_checked(v, i, true), RAY_OK); + + /* reference: first-occurrence group ids and counts */ + int64_t* gid_of_key = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + int64_t* ref_key = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + int64_t* ref_cnt = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(gid_of_key); TEST_ASSERT_NOT_NULL(ref_key); TEST_ASSERT_NOT_NULL(ref_cnt); + for (int64_t k = 0; k < kmax; k++) { gid_of_key[k] = -1; ref_cnt[k] = 0; } + int64_t ref_groups = 0, ref_keys = 0; + for (int64_t i = 0; i < n; i++) { + if (ray_vec_is_null(v, i)) continue; + int64_t k = xs[i] + 17; + if (gid_of_key[k] < 0) { gid_of_key[k] = ref_groups; ref_key[ref_groups++] = xs[i]; } + ref_cnt[gid_of_key[k]]++; + ref_keys++; + } + + ray_t* w = v; + ray_t* r = ray_index_attach_hash(&w); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + ray_index_t* ix = ray_index_payload(w->index); + TEST_ASSERT_EQ_I((int)ix->kind, RAY_IDX_HASH); + TEST_ASSERT_EQ_I(ix->u.hash.n_keys, ref_keys); + TEST_ASSERT_EQ_I(ix->u.hash.n_groups, ref_groups); + const int64_t* gk = (const int64_t*)ray_data(ix->u.hash.gkeys); + const int64_t* of = (const int64_t*)ray_data(ix->u.hash.offs); + const int64_t* rw = (const int64_t*)ray_data(ix->u.hash.rows); + TEST_ASSERT_EQ_I(of[0], 0); + TEST_ASSERT_EQ_I(of[ref_groups], ref_keys); + /* Serial layout: groups in first-occurrence order, each with its count, + * rows ascending and all storing the group's key. */ + for (int64_t g = 0; g < ref_groups; g++) { + TEST_ASSERT_EQ_I(gk[g], ref_key[g]); + TEST_ASSERT_EQ_I(of[g + 1] - of[g], ref_cnt[g]); + for (int64_t j = of[g]; j < of[g + 1]; j++) { + TEST_ASSERT_EQ_I(xs[rw[j]], ref_key[g]); + if (j > of[g]) TEST_ASSERT_TRUE(rw[j] > rw[j - 1]); + } + } + /* table probes: every key resolves to its group, an absent key misses */ + for (int64_t k = 0; k < kmax; k += 61) { + const int64_t* grows = NULL; + int64_t gn = 0; + TEST_ASSERT_EQ_I(ray_index_hash_group(w, k - 17, &grows, &gn), 1); + TEST_ASSERT_EQ_I(gn, ref_cnt[gid_of_key[k]]); + } + { + const int64_t* grows = NULL; + int64_t gn = 0; + TEST_ASSERT_EQ_I(ray_index_hash_group(w, kmax + 1000, &grows, &gn), 0); + } + + ray_sys_free(gid_of_key); ray_sys_free(ref_key); ray_sys_free(ref_cnt); + ray_release(w); + ray_heap_destroy(); + PASS(); +} + /* ─── Sort index ──────────────────────────────────────────────────── */ static test_result_t test_index_sort_attach_drop(void) { @@ -3697,6 +3776,7 @@ const test_entry_t index_entries[] = { { "index/unsupported_type", test_index_unsupported_type, NULL, NULL }, { "index/hash_attach_drop", test_index_hash_attach_drop, NULL, NULL }, { "index/hash_with_nulls_preserved", test_index_hash_with_nulls_preserved, NULL, NULL }, + { "index/hash_large_parallel", test_index_hash_large_parallel, NULL, NULL }, { "index/sort_attach_drop", test_index_sort_attach_drop, NULL, NULL }, { "index/bloom_attach_drop", test_index_bloom_attach_drop, NULL, NULL }, { "index/replace_cross_kind", test_index_replace_cross_kind, NULL, NULL }, diff --git a/test/test_splay.c b/test/test_splay.c index 34188b46..d037cf5e 100644 --- a/test/test_splay.c +++ b/test/test_splay.c @@ -40,6 +40,10 @@ #include "lang/internal.h" /* ray_set/get_splayed_fn (surface resolver) */ #include "mem/heap.h" #include "table/sym.h" +#include "table/domain.h" /* symfile positions of a streamed CSV load */ +#include "io/csv.h" /* ray_csv_save_splayed_named_opts */ +#include "mem/sys.h" +#include "core/pool.h" #include #include #include @@ -68,6 +72,139 @@ static void rm_rf(const char* path) { (void)ray_test_rm_rf(path); } +/* Streamed CSV -> splayed load, several chunks. The symfile positions of + * the SYM columns are the ones the cell-by-cell writer gives: per chunk, + * columns in order, each column's strings by first occurrence — whatever + * the worker count or how the dictionaries were split into partitions. + * Chunks of 20k rows with >4k new strings each take the parallel batch. */ +static test_result_t test_csv_splayed_symfile_order(void) { + TEST_ASSERT_NOT_NULL(ray_pool_get()); + const char* dir = TMP_SPLAY_BASE "/csvorder"; + const char* csv = TMP_SPLAY_BASE "/csvorder.csv"; + rm_rf(dir); + mkdir(TMP_SPLAY_BASE, 0755); + + enum { NROWS = 60000, CHUNK = 20000, NA = 13000, NB = 9000 }; + FILE* f = fopen(csv, "wb"); + TEST_ASSERT_NOT_NULL(f); + fputs("a,b,v\n", f); + for (int r = 0; r < NROWS; r++) { + int ai = (int)(((int64_t)r * 7919) % NA); + if (r % 5 == 0) fprintf(f, "a%d,a%d,%d\n", ai, (ai + 11) % NA, r); /* b reuses a's strings */ + else fprintf(f, "a%d,b%d,%d\n", ai, (int)(((int64_t)r * 104729) % NB), r); + } + fclose(f); + + int8_t types[] = { RAY_SYM, RAY_SYM, RAY_I64 }; + ray_err_t err = ray_csv_save_splayed_named_opts(csv, ',', true, types, 3, NULL, 0, dir, CHUNK); + TEST_ASSERT_EQ_I(err, RAY_OK); + + /* expected positions: walk as the writer does */ + int64_t* pa = (int64_t*)ray_sys_alloc(NA * sizeof(int64_t)); + int64_t* pb = (int64_t*)ray_sys_alloc(NB * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(pa); TEST_ASSERT_NOT_NULL(pb); + for (int i = 0; i < NA; i++) pa[i] = -1; + for (int i = 0; i < NB; i++) pb[i] = -1; + int64_t next = 1; /* 0 is "" */ + for (int c0 = 0; c0 < NROWS; c0 += CHUNK) { + for (int r = c0; r < c0 + CHUNK; r++) { /* column a */ + int ai = (int)(((int64_t)r * 7919) % NA); + if (pa[ai] < 0) pa[ai] = next++; + } + for (int r = c0; r < c0 + CHUNK; r++) { /* column b */ + int ai = (int)(((int64_t)r * 7919) % NA); + if (r % 5 == 0) { int x = (ai + 11) % NA; if (pa[x] < 0) pa[x] = next++; } + else { int bi = (int)(((int64_t)r * 104729) % NB); if (pb[bi] < 0) pb[bi] = next++; } + } + } + + char sym_path[256]; + snprintf(sym_path, sizeof(sym_path), "%s/.sym", dir); + ray_sym_domain_t* dom = ray_sym_domain_open(sym_path); + TEST_ASSERT_NOT_NULL(dom); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), next); + char buf[32]; + for (int i = 0; i < NA; i++) { + if (pa[i] < 0) continue; + int n = snprintf(buf, sizeof(buf), "a%d", i); + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, buf, (size_t)n), pa[i]); + } + for (int i = 0; i < NB; i++) { + if (pb[i] < 0) continue; + int n = snprintf(buf, sizeof(buf), "b%d", i); + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, buf, (size_t)n), pb[i]); + } + ray_sym_domain_release(dom); + + /* and the cells read back the strings written */ + ray_t* t = ray_read_splayed(dir, sym_path); + TEST_ASSERT_FALSE(RAY_IS_ERR(t)); + TEST_ASSERT_EQ_I(ray_table_nrows(t), NROWS); + ray_t* ca = ray_table_get_col_idx(t, 0); + for (int r = 0; r < NROWS; r += 997) { + ray_t* cell = ray_sym_vec_cell(ca, r); + TEST_ASSERT_NOT_NULL(cell); + int n = snprintf(buf, sizeof(buf), "a%d", (int)(((int64_t)r * 7919) % NA)); + TEST_ASSERT_EQ_U(ray_str_len(cell), (size_t)n); + TEST_ASSERT_MEM_EQ((size_t)n, ray_str_ptr(cell), buf); + } + ray_release(t); + ray_sys_free(pa); ray_sys_free(pb); + rm_rf(dir); + unlink(csv); + PASS(); +} + +/* A chunk whose byte window holds no quote, in a file that has quotes in + * another chunk, is split into rows exactly as the whole file is: a lone + * '\r' ends a row. (The quote-free fast path of the parallel scanner did + * not treat it so, and merged two rows, losing a value.) */ +static test_result_t test_csv_splayed_quote_mode_per_file(void) { + TEST_ASSERT_NOT_NULL(ray_pool_get()); + const char* dir = TMP_SPLAY_BASE "/csvquote"; + const char* csv = TMP_SPLAY_BASE "/csvquote.csv"; + rm_rf(dir); + mkdir(TMP_SPLAY_BASE, 0755); + + enum { NROWS = 60000, CHUNK = 20000, LONE = 45007 }; + FILE* f = fopen(csv, "wb"); + TEST_ASSERT_NOT_NULL(f); + fputs("s,v\n", f); + fputs("\"q,1\",0\n", f); /* quotes only in chunk 0 */ + for (int r = 1; r < NROWS; r++) { + if (r == LONE) fprintf(f, "left\rright,%d\n", r); /* chunk 2 */ + else fprintf(f, "r%d,%d\n", r, r); + } + fclose(f); + + int8_t types[] = { RAY_SYM, RAY_I64 }; + ray_t* mem = ray_read_csv_named_opts(csv, ',', true, types, 2, NULL, 0); + TEST_ASSERT_FALSE(RAY_IS_ERR(mem)); + int64_t want = ray_table_nrows(mem); + TEST_ASSERT_EQ_I(want, NROWS + 1); /* the lone \r splits a row */ + + ray_err_t err = ray_csv_save_splayed_named_opts(csv, ',', true, types, 2, NULL, 0, dir, CHUNK); + TEST_ASSERT_EQ_I(err, RAY_OK); + char sym_path[256]; + snprintf(sym_path, sizeof(sym_path), "%s/.sym", dir); + ray_t* t = ray_read_splayed(dir, sym_path); + TEST_ASSERT_FALSE(RAY_IS_ERR(t)); + TEST_ASSERT_EQ_I(ray_table_nrows(t), want); + /* every row agrees with the in-memory read */ + ray_t* vm = ray_table_get_col_idx(mem, 1); + ray_t* vt = ray_table_get_col_idx(t, 1); + for (int64_t r = 0; r < want; r++) { + TEST_ASSERT_EQ_I(ray_vec_is_null(vt, r), ray_vec_is_null(vm, r)); + if (!ray_vec_is_null(vm, r)) + TEST_ASSERT_EQ_I(((int64_t*)ray_data(vt))[r], ((int64_t*)ray_data(vm))[r]); + } + ray_release(t); + ray_release(mem); + rm_rf(dir); + unlink(csv); + PASS(); +} + /* ========================================================================= * 1. ray_splay_save: NULL dir → RAY_ERR_IO * ========================================================================= */ @@ -2277,5 +2414,7 @@ const test_entry_t splay_entries[] = { { "splay/empty_sym_table_roundtrip", test_empty_sym_table_roundtrip, splay_setup, splay_teardown }, { "splay/resolution_order_independence", test_resolution_order_independence, splay_setup, splay_teardown }, { "splay/resolution_explicit_wins", test_resolution_explicit_wins, splay_setup, splay_teardown }, + { "splay/csv_symfile_order", test_csv_splayed_symfile_order, splay_setup, splay_teardown }, + { "splay/csv_quote_mode_per_file", test_csv_splayed_quote_mode_per_file, splay_setup, splay_teardown }, { NULL, NULL, NULL, NULL }, }; From f82dc86687bfa0bf009369d11c70e4202f1706d6 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 13:31:52 +0300 Subject: [PATCH 47/51] perf(select): fused top-k prunes chunks by zone extrema and visits the best chunks first (#643) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(select): fused top-k skips chunks whose zone cannot beat the K-th key The fused top-k evaluated the predicate over every row even when the first sort key is a stored integer/temporal column whose chunk-zone index shows most chunks cannot hold a top-K row (a table clustered by another key, sorted within it, puts a time column's early values in a few chunks). Each worker whose heap is full publishes its worst first-key value; the best of those bounds the final K-th key. A chunk whose zone minimum (asc) is above the bound, or maximum (desc) below it, is skipped before the predicate runs — its pages are never read. Chunks with nulls are kept when nulls sort first. On a 100M-row table ordering by a time column clustered under a counter (52 of 1526 chunks can hold the top 10): cold 1205 ms -> 157 ms, warm 30 ms -> 9 ms; with a second sort key 24 ms -> 5.5 ms. Test: fused/topk_zone_prune (asc, desc, two keys, K near the match count, all equal to an unindexed in-memory copy). Co-Authored-By: Claude Opus 5.5 * perf(select): publish the top-k pruning bound once per morsel Publishing on every heap replacement made the bound a contended cache line when rows arrive in the order being sought (a descending top-k over a column stored ascending replaces the root on nearly every row, and nothing is pruned there): 25-70% slower than without pruning. The bound is now published once per morsel, on its own cache line. Those shapes are back to the unpruned time; the prunable ones keep their gain. Test: topk_zone_prune gains a nullable key with an all-null chunk, asc/desc and two keys. Co-Authored-By: Claude Opus 5.5 * perf(select): fused top-k visits the most promising chunks first With a chunk-zone index on the first sort key the K-th key only tightens as fast as the physical order lets it: an `asc` over a descending column scans everything before the bound becomes useful, and a key whose best values sit in the middle prunes nothing until the workers get there. Order the chunks by their zone extremum (smallest minimum for asc, largest maximum for desc, null chunks first when nulls sort ahead) and dispatch the workers over that virtual row space, mapping each slot back to its chunk. The K-heaps then fill from the best chunks, the bound is tight after the first few, and the rest are skipped whatever the layout. Results are unchanged: the heap keeps the K best rows with a source-row tie break, so the visiting order cannot alter them. Local A/B on a 4M-row splayed table (release, 8 cores): asc over a descending key K=100 43 -> 2.7 ms, K=8000 212 -> 104 ms, with a filter 22 -> 0.2-1.7 ms; already-favourable and unindexed shapes unchanged. Co-Authored-By: Claude Fable 5.1 * fix(select): `if` over two STR columns under a sparse filter stays on the selected path Two cheap STR branches route `if` to the eager arm, which picks descriptors for every row of the table in one pass. Under a `where:` that keeps few rows that is a bad trade: the filtered result keeps the whole intermediate — both parents' bytes — alive for its few rows (`select {x: (if c S S2) where: (== st 1)}` over 500k rows held 20 MB for 10k rows; the selected path holds nothing beyond them). Take the eager arm only when the outer selection keeps more than a quarter of the rows (the projection's pre-compaction threshold); a sparse selection evaluates the branches over the selected rows only. No filter and a filter keeping most rows are unchanged. Co-Authored-By: Claude Fable 5.1 * perf(select): fused top-k re-tests the pruning bound before every morsel The chunk test ran once at chunk entry, so a bound that tightened while the chunk was being scanned (another worker's heap filled, or this one's) only took effect from the next chunk. Re-test it before every morsel — one relaxed load and a compare, no writes — and abandon the rest of the chunk once it cannot beat the bound. Local A/B on the 4M-row table (release, 8 cores): asc over a descending key K=100 2.6-3.0 -> 0.4 ms, the same with a filter 0.8-1.4 -> 0.25 ms; the other shapes unchanged. Co-Authored-By: Claude Fable 5.1 --------- Co-authored-by: Claude Opus 5.5 --- src/ops/fused_topk.c | 208 +++++++++++++++++++++-- src/ops/pivot.c | 27 +-- test/rfl/fused/topk_zone_prune.rfl | 68 ++++++++ test/rfl/strop/str_view_pool_compact.rfl | 12 +- 4 files changed, 290 insertions(+), 25 deletions(-) create mode 100644 test/rfl/fused/topk_zone_prune.rfl diff --git a/src/ops/fused_topk.c b/src/ops/fused_topk.c index 1c216006..6553e7dd 100644 --- a/src/ops/fused_topk.c +++ b/src/ops/fused_topk.c @@ -47,6 +47,7 @@ #include "ops/internal.h" #include "lang/internal.h" #include "core/pool.h" +#include "ops/idxop.h" /* chunk-zone extrema: skip chunks that cannot enter the top-K */ #include #include @@ -128,8 +129,118 @@ typedef struct { ray_t** sym_strings; uint32_t sym_count; _Atomic(uint32_t) oom; + /* Chunk pruning on the first sort key (integer / temporal column with a + * chunk-zone index): zmin/zmax per 1<esz) { + case 1: return (int64_t)((const uint8_t*)ks->base)[row]; + case 2: return (int64_t)((const int16_t*)ks->base)[row]; + case 4: return (int64_t)((const int32_t*)ks->base)[row]; + default: return ((const int64_t*)ks->base)[row]; + } +} + +/* A full heap's worst row publishes its first key as a pruning bound. */ +static void fpk_publish_bound(fpk_par_ctx_t* c, int64_t worst_row) { + if (!c->zmin) return; + const fpk_keyspec_t* ks = &c->keys[0]; + if (ks->has_nulls && ray_vec_is_null(ks->col, worst_row)) return; + int64_t v = fpk_key_i64(ks, worst_row); + int64_t cur = atomic_load_explicit(&c->bound, memory_order_relaxed); + bool set = atomic_load_explicit(&c->bound_set, memory_order_relaxed); + for (;;) { + bool better = !set || (c->zdesc ? v > cur : v < cur); + if (!better) return; + if (atomic_compare_exchange_weak_explicit(&c->bound, &cur, v, + memory_order_relaxed, memory_order_relaxed)) { + atomic_store_explicit(&c->bound_set, 1, memory_order_release); + return; + } + set = true; + } +} + +/* Chunk `g` cannot hold a row that beats the bound. */ +static inline bool fpk_chunk_pruned(const fpk_par_ctx_t* c, int64_t g) { + if (!atomic_load_explicit(&c->bound_set, memory_order_acquire)) return false; + if (g < 0 || g >= (int64_t)c->zn) return false; + if (c->znulls_better && c->znull && (c->znull[g >> 3] & (1u << (g & 7)))) return false; + int64_t b = atomic_load_explicit(&c->bound, memory_order_relaxed); + return c->zdesc ? c->zmax[g] < b : c->zmin[g] > b; +} + +/* Chunk `a` is more promising than chunk `b` for the first key's + * direction: null chunks lead when nulls sort ahead, then the smaller + * minimum (asc) or the larger maximum (desc); ties keep chunk order. */ +static inline bool fpk_chunk_better(const fpk_par_ctx_t* c, uint32_t a, uint32_t b) { + if (c->znulls_better && c->znull) { + bool na = (c->znull[a >> 3] >> (a & 7)) & 1; + bool nb = (c->znull[b >> 3] >> (b & 7)) & 1; + if (na != nb) return na; + } + int64_t ka = c->zdesc ? c->zmax[a] : c->zmin[a]; + int64_t kb = c->zdesc ? c->zmax[b] : c->zmin[b]; + if (ka != kb) return c->zdesc ? ka > kb : ka < kb; + return a < b; +} + +/* Heap-sort the chunk ids in `ord` (initially 0..n-1) best first. */ +static void fpk_order_chunks(const fpk_par_ctx_t* c, uint32_t* ord, uint32_t n) { + /* max-heap on "worse": the root is the least promising chunk, so + * popping it to the tail leaves the best chunk at ord[0]. */ + for (uint32_t i = n; i-- > 0;) { + /* sift ord[i] down */ + uint32_t k = i; + for (;;) { + uint32_t l = 2 * k + 1, r = l + 1, w = k; + if (l < n && fpk_chunk_better(c, ord[w], ord[l])) w = l; + if (r < n && fpk_chunk_better(c, ord[w], ord[r])) w = r; + if (w == k) break; + uint32_t t = ord[k]; ord[k] = ord[w]; ord[w] = t; + k = w; + } + if (i == 0) break; + } + for (uint32_t end = n; end > 1;) { + end--; + uint32_t t = ord[0]; ord[0] = ord[end]; ord[end] = t; + uint32_t k = 0; + for (;;) { + uint32_t l = 2 * k + 1, r = l + 1, w = k; + if (l < end && fpk_chunk_better(c, ord[w], ord[l])) w = l; + if (r < end && fpk_chunk_better(c, ord[w], ord[r])) w = r; + if (w == k) break; + uint32_t u = ord[k]; ord[k] = ord[w]; ord[w] = u; + k = w; + } + } +} + /* Compare two source rows by the multi-key sort spec. Returns * "a is worse than b" sense: positive means evict-a-first in the * max-heap of K-best entries. Short-circuits on first non-equal key. @@ -272,17 +383,17 @@ static inline void fpk_heapify(const fpk_par_ctx_t* c, int64_t* heap, int32_t n) fpk_sift_down(c, heap, n, i); } -/* Worker fn: scan rows [start, end), eval predicate per morsel, do - * heap inserts for passing rows. */ -static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { - fpk_par_ctx_t* c = (fpk_par_ctx_t*)raw; - if (atomic_load_explicit(&c->oom, memory_order_relaxed)) return; - int32_t k = (int32_t)c->k; - int64_t* hidx = &c->heap_idx[(size_t)worker_id * (size_t)k]; - int32_t hn = c->heap_n[worker_id]; - - int64_t row = start; +/* Scan physical rows [row, end): eval predicate per morsel, heap-insert + * the passing rows. `*hnp` is the worker's heap fill on entry and exit. + * `chunk` >= 0 is the zone chunk the rows belong to: the bound may tighten + * while the chunk is being scanned (another worker's heap filled, or this + * one's), so it is re-tested before every morsel — one relaxed load and a + * compare — and the rest of the chunk is abandoned once it cannot beat it. */ +static inline void fpk_scan_rows(fpk_par_ctx_t* c, int64_t* hidx, int32_t* hnp, + int32_t k, int64_t row, int64_t end, int64_t chunk) { + int32_t hn = *hnp; while (row < end) { + if (chunk >= 0 && fpk_chunk_pruned(c, chunk)) break; int64_t mend = row + RAY_MORSEL_ELEMS; if (mend > end) mend = end; int64_t mlen = mend - row; @@ -302,8 +413,43 @@ static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end fpk_sift_down(c, hidx, k, 0); } } + /* Once per morsel: publishing on every heap replacement contends + * on the bound when the rows arrive in the order being sought. */ + if (hn == k) fpk_publish_bound(c, hidx[0]); row = mend; } + *hnp = hn; +} + +/* Worker fn: [start, end) is a range of physical rows, or — with a zone + * order — of virtual rows whose chunk slots map to physical chunks. */ +static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { + fpk_par_ctx_t* c = (fpk_par_ctx_t*)raw; + if (atomic_load_explicit(&c->oom, memory_order_relaxed)) return; + int32_t k = (int32_t)c->k; + int64_t* hidx = &c->heap_idx[(size_t)worker_id * (size_t)k]; + int32_t hn = c->heap_n[worker_id]; + + if (!c->zmin) { + fpk_scan_rows(c, hidx, &hn, k, start, end, -1); + } else { + int64_t span = (int64_t)1 << c->zlog2; + int64_t v = start; + while (v < end) { + int64_t slot = v >> c->zlog2; + int64_t sbeg = slot << c->zlog2; + int64_t send = sbeg + span; + if (send > end) send = end; + int64_t g = c->zorder ? (int64_t)c->zorder[slot] : slot; + if (!fpk_chunk_pruned(c, g)) { + int64_t p0 = (g << c->zlog2) + (v - sbeg); + int64_t p1 = (g << c->zlog2) + (send - sbeg); + if (p1 > c->nrows) p1 = c->nrows; + if (p0 < p1) fpk_scan_rows(c, hidx, &hn, k, p0, p1, g); + } + v = send; + } + } c->heap_n[worker_id] = hn; } @@ -379,6 +525,28 @@ ray_t* ray_fused_topk_select(ray_t* tbl, ctx.n_keys = n_sort_keys; ctx.k = k; ctx.tbl = tbl; + ctx.nrows = nrows; + { + ray_t* kc = ctx.keys[0].col; + int8_t kt = ctx.keys[0].type; + if ((kt == RAY_I16 || kt == RAY_I32 || kt == RAY_I64 || kt == RAY_DATE || + kt == RAY_TIME || kt == RAY_TIMESTAMP) && + ray_index_kind(kc) == RAY_IDX_CHUNK_ZONE) { + ray_index_t* zx = ray_index_payload(kc->index); + if (zx->built_for_len == kc->len && !zx->u.chunk_zone.is_f64 && + zx->u.chunk_zone.mins && zx->u.chunk_zone.maxs && zx->u.chunk_zone.null_bits) { + ctx.zmin = (const int64_t*)ray_data(zx->u.chunk_zone.mins); + ctx.zmax = (const int64_t*)ray_data(zx->u.chunk_zone.maxs); + ctx.znull = (const uint8_t*)ray_data(zx->u.chunk_zone.null_bits); + ctx.zn = zx->u.chunk_zone.n_chunks; + ctx.zlog2 = zx->u.chunk_zone.chunk_log2; + ctx.zdesc = ctx.keys[0].desc; + ctx.znulls_better = ctx.keys[0].nulls_first; + if (((int64_t)ctx.zn << ctx.zlog2) < nrows) + ctx.zmin = NULL; /* index shorter than the column: no pruning */ + } + } + } /* Compile the predicate via a temp graph just for the WHERE clause. No * where: is a predicate with no children — every row passes — and the @@ -420,8 +588,24 @@ ray_t* ray_fused_topk_select(ray_t* tbl, return NULL; } - if (pool) ray_pool_dispatch(pool, fpk_par_fn, &ctx, nrows); - else fpk_par_fn(&ctx, 0, 0, nrows); + /* With a zone index the workers walk the chunks best-first over the + * virtual row space (the order is optional: without it the slots map + * to the physical chunks and pruning still applies). */ + int64_t span_rows = nrows; + ray_t* zord_hdr = NULL; + if (ctx.zmin) { + span_rows = (int64_t)ctx.zn << ctx.zlog2; + uint32_t* ord = (uint32_t*)scratch_alloc(&zord_hdr, + (size_t)ctx.zn * sizeof(uint32_t)); + if (ord) { + for (uint32_t g = 0; g < ctx.zn; g++) ord[g] = g; + fpk_order_chunks(&ctx, ord, ctx.zn); + ctx.zorder = ord; + } + } + if (pool) ray_pool_dispatch(pool, fpk_par_fn, &ctx, span_rows); + else fpk_par_fn(&ctx, 0, 0, span_rows); + if (zord_hdr) { scratch_free(zord_hdr); ctx.zorder = NULL; } if (atomic_load_explicit(&ctx.oom, memory_order_relaxed)) { scratch_free(idx_hdr); scratch_free(hn_hdr); diff --git a/src/ops/pivot.c b/src/ops/pivot.c index 2084ec31..1851634f 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -925,6 +925,22 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { * answer exactly as it would, or the same expression gets two types * depending on the worker count. Where it is not, this arm is the only * one there is and may report the type the rows actually have. */ + ray_t* outer_sel = g->selection; + if (outer_sel) { + ray_rowsel_t* sm = ray_rowsel_meta(outer_sel); + if (!sm || sm->nrows != nrows) return NULL; + } + + int64_t selected = outer_sel ? ray_rowsel_meta(outer_sel)->total_pass : nrows; + if (selected < 0 || selected > nrows) return NULL; + /* Under a sparse outer selection (a `where:` keeping a quarter of the + * rows or fewer) the STR eager arm is a bad trade: it builds a + * descriptor for every row of the table, and the filtered result then + * keeps that whole intermediate — both parents' bytes — alive for its + * few rows. Touching only the selected rows costs less and holds only + * their bytes. Same threshold as the projection's pre-compaction. */ + bool sparse_sel = outer_sel && selected * 4 <= nrows; + bool eager_possible = (op->out_type != RAY_STR && if_branch_trivial(g, then_op) && if_branch_trivial(g, else_op) && @@ -932,7 +948,7 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { if_type_eager_ok(else_op->out_type, op->out_type)) || /* two STR vector branches that are cheap and total: * the eager arm picks descriptors over one pass */ - (op->out_type == RAY_STR && + (op->out_type == RAY_STR && !sparse_sel && then_op->out_type == RAY_STR && else_op->out_type == RAY_STR && if_branch_cheap(g, then_op, 0) && if_branch_cheap(g, else_op, 0)); { @@ -946,15 +962,6 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { !if_make_branch_plan(g, else_op, &else_plan)) return NULL; - ray_t* outer_sel = g->selection; - if (outer_sel) { - ray_rowsel_t* sm = ray_rowsel_meta(outer_sel); - if (!sm || sm->nrows != nrows) return NULL; - } - - int64_t selected = outer_sel ? ray_rowsel_meta(outer_sel)->total_pass : nrows; - if (selected < 0 || selected > nrows) return NULL; - uint8_t* cond = (uint8_t*)ray_data(cond_v); int64_t true_count = if_selected_true_count(cond, nrows, outer_sel); int64_t false_count = selected - true_count; diff --git a/test/rfl/fused/topk_zone_prune.rfl b/test/rfl/fused/topk_zone_prune.rfl new file mode 100644 index 00000000..77609331 --- /dev/null +++ b/test/rfl/fused/topk_zone_prune.rfl @@ -0,0 +1,68 @@ +;; Fused top-k over a stored column with a chunk-zone index skips chunks +;; whose extremum cannot beat the current K-th key. The time column here +;; runs in two long ascending blocks (like a table sorted by (counter, +;; time)), so most chunks are prunable; every answer must equal the same +;; query over an unindexed in-memory copy, for asc/desc, two keys, K near +;; the number of matches, and a nullable key (below). +(.sys.exec "rm -rf rf_test_zprune rf_test_zprune.csv") -- 0 +(set n 600000) +(set i (til n)) +(set t (+ 1000000 (* 7 (% i 300000)))) +(set s (as 'SYMBOL (map (fn [j] (if (== 0 (% j 3)) "" (format "p%" (% j 101)))) i))) +(set v (% (* i 13) 1000)) +(.csv.write (table [t s v] (list t s v)) "rf_test_zprune.csv") -- 0 +(set T (.csv.splayed [t s v] [I64 SYM I64] "rf_test_zprune.csv" "rf_test_zprune/")) +(set Tm (.csv.read [t s v] [I64 SYM I64] "rf_test_zprune.csv")) +(at (.idx.info (at T 't)) 'kind) -- 'chunk_zone +(set eqt (fn [a b] (all (map (fn [c] (all (== (as 'STR (at a c)) (as 'STR (at b c))))) (cols a))))) +(set q1 (fn [x] (select {s: s t: t v: v from: x where: (!= s "") asc: t take: 10}))) +(set q2 (fn [x] (select {s: s t: t from: x where: (!= s "") asc: [t s] take: 10}))) +(set q3 (fn [x] (select {v: v t: t from: x where: (> v 500) desc: t take: 7}))) +(set q4 (fn [x] (select {v: v t: t from: x where: (> v 990) asc: t take: 50}))) +(set q5 (fn [x] (select {v: v t: t from: x where: (== v 3) asc: t take: 700}))) +(eqt (q1 T) (q1 Tm)) -- true +(eqt (q2 T) (q2 Tm)) -- true +(eqt (q3 T) (q3 Tm)) -- true +(eqt (q4 T) (q4 Tm)) -- true +(eqt (q5 T) (q5 Tm)) -- true +(at (q1 T) 't) -- [1000007 1000007 1000014 1000014 1000028 1000028 1000035 1000035 1000049 1000049] +;; a nullable key: one chunk entirely null and every 1000th row null — +;; nulls sort first under asc (their chunks are never skipped) and last +;; under desc +(.sys.exec "rm -rf rf_test_zprune_n") -- 0 +(set kn (as 'I64 (map (fn [j] (if (or (== 0 (% j 1000)) (and (>= j 131072) (< j 196608))) 0N (+ 5 (% j 300000)))) i))) +(.db.splayed.set "rf_test_zprune_n/" (table [kn v] (list kn v))) +(set TN (.db.splayed.get "rf_test_zprune_n/")) +(set TNm (table [kn v] (list kn v))) +(set qn1 (fn [x] (select {kn: kn v: v from: x where: (> v 100) asc: kn take: 20}))) +(set qn2 (fn [x] (select {kn: kn v: v from: x where: (> v 100) desc: kn take: 20}))) +(set qn3 (fn [x] (select {kn: kn v: v from: x where: (== v 7) asc: [kn v] take: 5}))) +(eqt (qn1 TN) (qn1 TNm)) -- true +(eqt (qn2 TN) (qn2 TNm)) -- true +(eqt (qn3 TN) (qn3 TNm)) -- true +(.sys.exec "rm -rf rf_test_zprune_n") -- 0 +(.sys.exec "rm -rf rf_test_zprune rf_test_zprune.csv") -- 0 +;; best-chunk-first: the chunks are visited in the order of their zone +;; extremum, so an `asc` over a descending column (its best chunk is the +;; last one) and a key whose best chunk sits in the middle prune the rest +;; whatever the physical order; answers equal the in-memory copy and the +;; planner's full sort +(.sys.exec "rm -rf rf_test_zprune_o") -- 0 +(set kd (- n i)) +(set km (abs (- i 300000))) +(.db.splayed.set "rf_test_zprune_o/" (table [kd km v] (list kd km v))) +(set TO (.db.splayed.get "rf_test_zprune_o/")) +(set TOm (table [kd km v] (list kd km v))) +(set qo1 (fn [x] (select {kd: kd v: v from: x asc: kd take: 8000}))) +(set qo2 (fn [x] (select {kd: kd v: v from: x where: (> v 500) asc: kd take: 100}))) +(set qo3 (fn [x] (select {km: km v: v from: x where: (< v 900) asc: km take: 300}))) +(set qo4 (fn [x] (select {km: km v: v from: x desc: km take: 5}))) +(set qo5 (fn [x] (select {km: km kd: kd from: x where: (== v 7) asc: [km kd] take: 40}))) +(eqt (qo1 TO) (qo1 TOm)) -- true +(eqt (qo2 TO) (qo2 TOm)) -- true +(eqt (qo3 TO) (qo3 TOm)) -- true +(eqt (qo4 TO) (qo4 TOm)) -- true +(eqt (qo5 TO) (qo5 TOm)) -- true +(at (qo1 TO) 'kd) -- (+ 1 (til 8000)) +(at (qo4 TO) 'km) -- [300000 299999 299999 299998 299998] +(.sys.exec "rm -rf rf_test_zprune_o") -- 0 diff --git a/test/rfl/strop/str_view_pool_compact.rfl b/test/rfl/strop/str_view_pool_compact.rfl index 5ed5eda9..a6e2e7b8 100644 --- a/test/rfl/strop/str_view_pool_compact.rfl +++ b/test/rfl/strop/str_view_pool_compact.rfl @@ -19,6 +19,12 @@ (count R) -- 10000 (< (- (direct) base) 1000000) -- true (all (== (at R 'x) (at (select {from: T x: (if c S S2) where: (== st 1)}) 'x))) -- true +(all (map (fn [j] (== (at (at R 'x) j) (if (at c (* j 50)) (at S (* j 50)) (at S2 (* j 50))))) (til 200))) -- true +(set R 0) +;; a filter that keeps most rows: the same values whichever arm ran +(set R (select {from: T x: (if c S S2) where: (> st 10)})) +(count R) -- 400000 +(all (== (at R 'x) (at (at (select {from: T x: (if c S S2)}) 'x) (where (> st 10))))) -- true (set R 0) ;; the same over two substring views (set R (select {from: T x: (if c (substr S 2 30) (substr S2 2 30)) where: (== st 1)})) @@ -29,9 +35,9 @@ (set R (select {from: T x: (if c S "a-rather-long-literal-string") where: (== st 1)})) (< (- (direct) base) 1000000) -- true (set R 0) -;; a per-row condition over two whole columns (the selected path — STR is -;; never routed to the eager arm when the condition is a column) copies only -;; the chosen rows' bytes: less than the two parent pools together +;; a per-row condition over two whole columns with no filter (two cheap STR +;; branches take the eager arm, which picks descriptors over one pass): the +;; result holds less than the two parent pools together (set R (select {from: T x: (if c S S2)})) (set grow (- (direct) base)) (< grow 40000000) -- true From 67861ce172c798ddb8ba9684001bfcb6740660a5 Mon Sep 17 00:00:00 2001 From: Anton Kundenko Date: Tue, 29 Sep 2026 16:42:20 +0200 Subject: [PATCH 48/51] fix(collection): hash int cells through f64 when a set meets a float probe (#646) * fix(collection): hash int cells through f64 when a set meets a float probe The row hashset behind find/except/union/sect/in hashed int cells with ray_hash_i64 and float cells with ray_hash_f64. Numeric equality crosses the two (atom_eq compares any two numerics through f64, and hs_eq_rows already defers to it), but equal values landed in different buckets, so the probe never reached the compare. A correct answer was a collision: (find [1.0 2.0 3.0] [3 1]) [0Nl 0] -> [2 0] (except [1 2 3] [1.0 3.0]) [1 2 3] -> [2] (sect [1 2 3] [2.0 3.0]) [] -> [2 3] (union [1 2] [2.0 3.0]) [1 2 2 3] -> [1 2 3] (in [1 0Nl 3] [1.0 0Nf]) [false true false] -> [true true false] The set gains a num_f64 mode in which int-family cells hash through f64. The first probe of a new type goes to a cold helper; if it is the other numeric class (or a list, whose numeric atoms already hash through f64), the set rehashes once and keeps the mode. A cached last-probe type keeps the per-probe cost to one compare, so same-type callers are unchanged. hs_eq_rows compares mixed numeric typed cells as f64 directly instead of boxing two atoms per probe. If the one-off rehash hits OOM, that probe is answered by a full scan and the next probe retries. 1M x 500k, -c 8, release, 3 alternating runs: except/sect i64-i64, f64-f64 within ~1% of dev except i64-f64 153 ms -> 80 ms (and now correct) sect i64-f64 155 ms -> 80 ms (and now correct) find i64 hay / f64 needles 34.5 ms -> 0.1 ms (and now correct) collection_branch_cov.rfl pinned the old miss -- its comment says the probe "may miss" because the hashes differ -- and now expects 2. New mixed_numeric.rfl covers find/except/union/sect/in in both directions, narrow ints, fractional misses, both-sides-null in, and sizes that grow the table before and after the switch. Closes #645 * fix(collection): skip the no-op rehash of a float-built set; pin ints past 2^53 Review follow-ups on #646. Only int cells change hash in num_f64 mode, so a float- or list-built set is already in the f64 layout: set the mode and skip hashset_rehash unless the source is int-class (except with an i64 probe against a 500k f64 set, 79 -> 73 ms). mixed_numeric.rfl gains the float-set/int-probe order and a note, with cases, that ints past 2^53 compare through f64 -- consistent with == -- while same-type ints sharing a double image stay distinct. --- src/ops/collection.c | 119 ++++++++++++++++-- test/rfl/collection/collection_branch_cov.rfl | 11 +- test/rfl/collection/mixed_numeric.rfl | 59 +++++++++ 3 files changed, 174 insertions(+), 15 deletions(-) create mode 100644 test/rfl/collection/mixed_numeric.rfl diff --git a/src/ops/collection.c b/src/ops/collection.c index 6b20c4b2..b61bc3f1 100644 --- a/src/ops/collection.c +++ b/src/ops/collection.c @@ -65,13 +65,60 @@ typedef struct hashset_t { int8_t src_type; /* ray_t.type */ bool src_has_nulls; void* src_data; /* pointer to typed data (or RAY_LIST elements) */ + /* Int-family cells hash through f64 (hs_hash_row). Off until a probe + * crosses numeric classes — see hs_needs_num_f64. */ + bool num_f64; + int8_t probe_type; /* last probe type vetted by hashset_adopt_probe */ } hashset_t; +/* Numeric class of a typed-vec type, matching is_numeric/atom_eq: every + * int width compares to every float width through f64. Temporal types are + * not numeric there (a DATE never equals an I32), so they stay class 0. */ +enum { HS_NUM_NONE = 0, HS_NUM_INT = 1, HS_NUM_FLT = 2 }; +static inline int hs_num_class(int8_t t) { + switch (t) { + case RAY_BOOL: case RAY_U8: case RAY_I16: case RAY_I32: case RAY_I64: + return HS_NUM_INT; + case RAY_F32: case RAY_F64: + return HS_NUM_FLT; + default: + return HS_NUM_NONE; + } +} + +/* Cell i of a numeric typed vec as f64 — as_f64 on the boxed atom. */ +static inline double hs_cell_f64(int8_t t, const void* data, int64_t i) { + switch (t) { + case RAY_BOOL: return (double)((const bool*)data)[i]; + case RAY_U8: return (double)((const uint8_t*)data)[i]; + case RAY_I16: return (double)((const int16_t*)data)[i]; + case RAY_I32: return (double)((const int32_t*)data)[i]; + case RAY_I64: return (double)((const int64_t*)data)[i]; + case RAY_F32: return (double)((const float*)data)[i]; + default: return ((const double*)data)[i]; + } +} + +/* Int cells hash with ray_hash_i64 and float cells (typed, or numeric atoms + * in a RAY_LIST) with ray_hash_f64, so 3 and 3.0 land in different buckets + * and hs_eq_rows never sees them (#645). An int side meeting a float or + * list side must hash its ints through f64 as well. */ +static inline bool hs_needs_num_f64(int8_t a, int8_t b) { + int ca = hs_num_class(a), cb = hs_num_class(b); + if (ca == HS_NUM_INT) return cb == HS_NUM_FLT || b == RAY_LIST; + if (cb == HS_NUM_INT) return ca == HS_NUM_FLT || a == RAY_LIST; + return false; +} + /* Hash a single row at index i in src. Mirrors atom_eq's coercion * rules: numeric types normalize through f64 so an I64 atom and an - * F64 atom holding the same value collide (boxed-list path only — a - * typed vec is homogeneous, so the dispatch picks one branch). */ -static uint64_t hs_hash_row(ray_t* src, int64_t i, int8_t t, void* data) { + * F64 atom holding the same value collide. A typed int vec hashes + * through i64 unless num_f64 is set — the set is being probed from the + * other numeric class (hs_needs_num_f64). */ +static uint64_t hs_hash_row(ray_t* src, int64_t i, int8_t t, void* data, + bool num_f64) { + if (num_f64 && hs_num_class(t) == HS_NUM_INT) + return ray_hash_f64(hs_cell_f64(t, data, i)); switch (t) { case RAY_I64: return ray_hash_i64(((const int64_t*)data)[i]); case RAY_I32: return ray_hash_i64((int64_t)((const int32_t*)data)[i]); @@ -169,6 +216,10 @@ static int hs_eq_rows(ray_t* a_src, int64_t ai, int8_t at, void* a_data, } } } + /* Mixed numeric typed vecs: atom_eq compares any two numerics + * through f64 — do the same without boxing a pair per probe. */ + if (hs_num_class(at) != HS_NUM_NONE && hs_num_class(bt) != HS_NUM_NONE) + return hs_cell_f64(at, a_data, ai) == hs_cell_f64(bt, b_data, bi); /* Fall back to atom_eq via boxed values. Used for cross-type * comparisons (e.g. except over typed I64 vs F64 vec) and the * RAY_LIST path. collection_elem allocates a temporary atom for @@ -208,6 +259,8 @@ static bool hashset_init(hashset_t* hs, ray_t* src, int64_t hint) { hs->src_type = src ? src->type : 0; hs->src_has_nulls = src ? ray_vec_may_have_nulls(src) : false; hs->src_data = src ? ray_data(src) : NULL; + hs->num_f64 = false; + hs->probe_type = hs->src_type; return true; } @@ -216,11 +269,11 @@ static void hashset_destroy(hashset_t* hs) { hs->slots = NULL; } -static bool hashset_grow(hashset_t* hs) { +/* Re-slot every stored row into a fresh table of new_cap, hashing with + * the set's current num_f64 mode. */ +static bool hashset_rehash(hashset_t* hs, int64_t new_cap) { int64_t old_cap = hs->cap; int64_t* old_slots = hs->slots; - int64_t new_cap = old_cap * 2; - if (new_cap < old_cap) return false; ray_t* nb = ray_alloc((size_t)new_cap * sizeof(int64_t)); if (!nb || RAY_IS_ERR(nb)) return false; int64_t* ns = (int64_t*)ray_data(nb); @@ -229,7 +282,8 @@ static bool hashset_grow(hashset_t* hs) { for (int64_t i = 0; i < old_cap; i++) { int64_t ridx = old_slots[i]; if (ridx == HS_EMPTY) continue; - uint64_t h = hs_hash_row(hs->src, ridx, hs->src_type, hs->src_data); + uint64_t h = hs_hash_row(hs->src, ridx, hs->src_type, hs->src_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)mask); while (ns[s] != HS_EMPTY) s = (s + 1) & mask; ns[s] = ridx; @@ -242,8 +296,50 @@ static bool hashset_grow(hashset_t* hs) { return true; } +static bool hashset_grow(hashset_t* hs) { + int64_t new_cap = hs->cap * 2; + if (new_cap < hs->cap) return false; + return hashset_rehash(hs, new_cap); +} + /* Probe the set for the row (probe_src, probe_i). Returns the stored * row index from the build-side vec on hit, HS_EMPTY on miss. */ +/* A probe of a type the set has not seen yet. From the other numeric + * class, rehash once so int cells hash through f64 — later probes and + * inserts keep the mode, and same-type callers never get here. Returns + * false on OOM: the slots keep their old layout, the type stays unvetted + * so the next probe retries, and the caller answers by hashset_scan. */ +static __attribute__((noinline, cold)) bool +hashset_adopt_probe(hashset_t* hs, int8_t probe_type) { + if (!hs->num_f64 && hs_needs_num_f64(hs->src_type, probe_type)) { + hs->num_f64 = true; + /* Only int cells change hash in this mode: a float- or list-built + * set already sits in the f64 layout, so it needs no rehash. */ + if (hs_num_class(hs->src_type) == HS_NUM_INT && + !hashset_rehash(hs, hs->cap)) { + hs->num_f64 = false; + return false; + } + } + hs->probe_type = probe_type; + return true; +} + +/* Hash-free probe for when the table could not be rehashed: compare the + * row against every stored one. */ +static __attribute__((noinline, cold)) int64_t +hashset_scan(hashset_t* hs, ray_t* probe_src, int64_t probe_i, + int8_t probe_type, void* probe_data) { + for (int64_t k = 0; k < hs->cap; k++) { + int64_t stored = hs->slots[k]; + if (stored != HS_EMPTY && + hs_eq_rows(probe_src, probe_i, probe_type, probe_data, + hs->src, stored, hs->src_type, hs->src_data)) + return stored; + } + return HS_EMPTY; +} + static int64_t hashset_find_xrow(hashset_t* hs, ray_t* probe_src, int64_t probe_i, int8_t probe_type, void* probe_data) { if (hs_row_is_null(probe_src, probe_i, probe_data)) @@ -277,7 +373,11 @@ static int64_t hashset_find_xrow(hashset_t* hs, ray_t* probe_src, int64_t probe_ return HS_EMPTY; } } - uint64_t h = hs_hash_row(probe_src, probe_i, probe_type, probe_data); + if (probe_type != hs->probe_type && + !hashset_adopt_probe(hs, probe_type)) + return hashset_scan(hs, probe_src, probe_i, probe_type, probe_data); + uint64_t h = hs_hash_row(probe_src, probe_i, probe_type, probe_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)hs->mask); while (hs->slots[s] != HS_EMPTY) { int64_t stored = hs->slots[s]; @@ -366,7 +466,8 @@ static bool hashset_insert(hashset_t* hs, int64_t i) { if (hs->count * 2 >= hs->cap) { if (!hashset_grow(hs)) { /* fall through, may degrade */ } } - uint64_t h = hs_hash_row(hs->src, i, hs->src_type, hs->src_data); + uint64_t h = hs_hash_row(hs->src, i, hs->src_type, hs->src_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)hs->mask); while (hs->slots[s] != HS_EMPTY) { int64_t stored = hs->slots[s]; diff --git a/test/rfl/collection/collection_branch_cov.rfl b/test/rfl/collection/collection_branch_cov.rfl index dc0e28bb..24708f78 100644 --- a/test/rfl/collection/collection_branch_cov.rfl +++ b/test/rfl/collection/collection_branch_cov.rfl @@ -218,12 +218,11 @@ ;; STR equality (lines 145-151) (count (union ["aa" "bb"] ["bb" "cc"])) -- 3 -;; Cross-type comparison via atom_eq fallback (lines 159-165) -;; except where vec1 is I64 and vec2 is F64 → cross-type hs_eq_rows -;; NOTE: hash values differ across types (I64 vs F64), so the hashset -;; probe may miss. The typed-vec path uses type-specific hashing, so -;; cross-type except returns all of vec1 (no matches found). -(count (except [1 2 3 4] (as 'F64 [2 3]))) -- 4 +;; Cross-type comparison: except where vec1 is I64 and vec2 is F64 → +;; cross-type hs_eq_rows. The first I64 probe switches the F64-built set +;; into num_f64 mode so 2 and 2.0 share a bucket; this used to return all +;; of vec1 because the two sides hashed through different functions (#645). +(count (except [1 2 3 4] (as 'F64 [2 3]))) -- 2 ;; ══════════════════════════════════════════════════════════════════════ ;; Section 10: hashset_grow — trigger by exceeding load factor (lines 202-226) diff --git a/test/rfl/collection/mixed_numeric.rfl b/test/rfl/collection/mixed_numeric.rfl new file mode 100644 index 00000000..028d45da --- /dev/null +++ b/test/rfl/collection/mixed_numeric.rfl @@ -0,0 +1,59 @@ +;; Set operations over mixed int/float typed vectors (#645). +;; +;; Numeric equality crosses widths and int/float: (== 3 3.0) is true, and +;; atom_eq compares any two numerics through f64. The row hashset used by +;; find/except/union/sect/in hashed int cells with ray_hash_i64 and float +;; cells with ray_hash_f64, so equal values landed in different buckets and +;; the probe missed; a correct answer was a hash collision. The set now +;; switches to hashing int cells through f64 when it meets a float probe. + +;; the contract these pin +(== [1 2 3] [1.0 2.0 3.0]) -- [true true true] + +;; find — both directions, and a haystack larger than the needles +(find [1.0 2.0 3.0] [3 1]) -- [2 0] +(find [1 2 3] [3.0 1.0]) -- [2 0] +(find (as 'F64 (til 100)) [3 1]) -- [3 1] +(find (til 100) [3.0 1.0 2.5]) -- [3 1 0Nl] + +;; except +(except [1 2 3] [1.0 3.0]) -- [2] +(except [1.0 2.0 3.0] [1 3]) -- [2.0] +(except [1i 2i 3i] [2.0]) -- [1i 3i] +(except [1h 2h] [2.0]) -- [1h] + +;; union — the shared value appears once +(count (union [1 2] [2.0 3.0])) -- 3 + +;; sect +(sect [1 2 3] [2.0 3.0]) -- [2 3] +(sect [1.0 2.0 3.0] [3 9]) -- [3.0] + +;; in with a null on BOTH sides still reaches the hashset +(in [1 0Nl 3] [1.0 0Nf]) -- [true true false] +(in [1.0 0Nf 3.0] [0Nl 3]) -- [false true true] + +;; a fractional float matches no int +(find [1 2 3] [2.5]) -- [0Nl] +(except [1 2 3] [2.5]) -- [1 2 3] + +;; same-type sets are unaffected +(find [1 2 3] [3 1]) -- [2 0] +(except [1 2 3] [1 3]) -- [2] +(find [1.0 2.0] [2.0]) -- [1] + +;; a float-built set probed by ints (no rehash needed), then by floats +(set _fs (as 'F64 (til 1000))) +(count (except (til 2000) _fs)) -- 1000 +(count (except (as 'F64 (til 2000)) _fs)) -- 1000 + +;; beyond 2^53 an int and a double compare through f64, so an int matches +;; the double its neighbour rounds to — (== 9007199254740993 +;; 9007199254740992.0) is true too. Same-type ints that share a double +;; image are still told apart. +(find [9007199254740993 9007199254740992 9007199254740994] [9007199254740992.0 9007199254740994.0]) -- [0 2] +(except [9007199254740993 9007199254740992 9007199254740994] [9007199254740992]) -- [9007199254740993 9007199254740994] + +;; enough rows to force hashset growth before and after the switch +(count (except (til 1000) (as 'F64 (til 500)))) -- 500 +(sum (find (as 'F64 (til 1000)) (til 1000))) -- 499500 From a581a77fd8e6a17e175a6efdde891ff56d584081 Mon Sep 17 00:00:00 2001 From: Anton Kundenko Date: Tue, 29 Sep 2026 19:06:45 +0200 Subject: [PATCH 49/51] test(agg): pin the LLC in dense_cache_bound so the route is runner-independent (#647) agg_contract/dense_cache_bound failed intermittently on the macOS debug CI job (got strategy 2, expected 1) across unrelated PRs. The group route bounds replicated task-local slabs by the probed last-level cache, and switches to partition ownership when fewer than three slabs fit. A macOS CI VM reports a few MB of L2, which fits fewer than three of the test's 2.5 MB slabs, so the route correctly chose partitioning while the test asserted the task-local strategy. ray_cache_llc_bytes gains a DEBUG-only pin, ray_cache_llc_set_for_test, following ray_ipc_set_auto_journal_eval_for_test. The three platform probes become static cache_llc_probe and one wrapper honours the pin. The test pins 32 MB (the size the route assumes when none is reported) around the routed query and restores it before any assert can return. It also asserts the pinned bound actually bites (3..19 slabs). Pinning 4 MB instead reproduces the CI failure exactly: test_agg_contract.c:1759 got 2, expected 1. make test 3953/3953. --- src/core/platform.c | 24 +++++++++++++++++++++--- src/core/platform.h | 4 ++++ test/test_agg_contract.c | 17 +++++++++++++---- 3 files changed, 38 insertions(+), 7 deletions(-) diff --git a/src/core/platform.c b/src/core/platform.c index bad99c91..5251a671 100644 --- a/src/core/platform.c +++ b/src/core/platform.c @@ -369,7 +369,7 @@ static uint64_t cache_sysfs_llc_bytes(void) { } #endif -uint64_t ray_cache_llc_bytes(void) { +static uint64_t cache_llc_probe(void) { static uint64_t cached = UINT64_MAX; if (cached != UINT64_MAX) return cached; uint64_t bytes = 0; @@ -642,7 +642,7 @@ uint32_t ray_physical_core_count(void) { /* Sum of every level-3 cache instance reported by the processor topology * (each SYSTEM_LOGICAL_PROCESSOR_INFORMATION cache record is one instance). * 0 when the query fails. */ -uint64_t ray_cache_llc_bytes(void) { +static uint64_t cache_llc_probe(void) { static uint64_t cached = UINT64_MAX; if (cached != UINT64_MAX) return cached; uint64_t bytes = 0; @@ -800,7 +800,7 @@ ray_err_t ray_thread_join(ray_thread_t t) { } uint32_t ray_thread_count(void) { return 1; } -uint64_t ray_cache_llc_bytes(void) { return 0; } +static uint64_t cache_llc_probe(void) { return 0; } /* Semaphore — counter-only. Single-threaded so wait never blocks (the * counter must already be positive when wait fires). */ @@ -818,3 +818,21 @@ void ray_sem_wait(ray_sem_t* s) { void ray_sem_signal(ray_sem_t* s) { (*s)++; } #endif /* RAY_OS_WASM */ + +#ifdef DEBUG +/* Test pin for the probed LLC size: routing that bounds replicated state by + * the cache (group dense slabs) otherwise picks a different strategy on + * every CI runner. 0 restores the platform probe. */ +static uint64_t g_llc_for_test = 0; + +void ray_cache_llc_set_for_test(uint64_t bytes) { + g_llc_for_test = bytes; +} +#endif + +uint64_t ray_cache_llc_bytes(void) { +#ifdef DEBUG + if (g_llc_for_test) return g_llc_for_test; +#endif + return cache_llc_probe(); +} diff --git a/src/core/platform.h b/src/core/platform.h index ea3633c5..e2d13785 100644 --- a/src/core/platform.h +++ b/src/core/platform.h @@ -180,6 +180,10 @@ uint32_t ray_physical_core_count(void); * 0 when the platform cannot report it. Bounds replicated per-task state * whose random-access working set must stay cache-resident to scale. */ uint64_t ray_cache_llc_bytes(void); +#ifdef DEBUG +/* Pin ray_cache_llc_bytes to `bytes` (0 = probe again). */ +void ray_cache_llc_set_for_test(uint64_t bytes); +#endif void ray_parallel_begin(void); void ray_parallel_end(void); diff --git a/test/test_agg_contract.c b/test/test_agg_contract.c index 2134755c..3f61880c 100644 --- a/test/test_agg_contract.c +++ b/test/test_agg_contract.c @@ -1734,7 +1734,13 @@ static test_result_t test_cancelled_group(void) { * 20-worker pool and a 100k-slot slab the raw replication (20 slabs) leaves * most caches, and the run must use at most floor(0.75 * LLC / slab) task * slabs (never fewer than the pool when everything fits). The result is - * identical either way. */ + * identical either way. + * + * The LLC is pinned to 32 MB (the size the route assumes when none is + * reported) for the routed query: ~10 slabs fit, so the bound bites but the + * run is not cache-starved. Unpinned, a small-cache runner (a macOS CI VM + * reports a few MB of L2) fits fewer than three slabs and the route rightly + * switches to partition ownership, failing the strategy assertion. */ static test_result_t test_dense_cache_bound(void) { ray_pool_destroy(); TEST_ASSERT_EQ_I(ray_pool_init_total(20), RAY_OK); @@ -1743,21 +1749,24 @@ static test_result_t test_dense_cache_bound(void) { "(set cb_t (table [k v] (list (as 'I32 (% (* cb_i 7919) 100000)) (% cb_i 13))))"); TEST_ASSERT_NOT_NULL(setup); TEST_ASSERT_FALSE(RAY_IS_ERR(setup)); ray_release(setup); agg_route_reset(); + ray_cache_llc_set_for_test(32ull << 20); ray_t* r = ray_eval_str("(select {from:cb_t by:k s:(sum v)})"); - TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_FALSE(RAY_IS_ERR(r)); agg_route_stats_t stats = agg_route_stats(); + uint64_t llc = ray_cache_llc_bytes(); + ray_cache_llc_set_for_test(0); /* before any assert can return */ + TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_FALSE(RAY_IS_ERR(r)); TEST_ASSERT_EQ_I(stats.routes[AGG_ROUTE_V2_DENSE], 1); TEST_ASSERT_EQ_I(stats.dense_strategy, AGG_DENSE_TASK_LOCAL); TEST_ASSERT_TRUE(stats.dense_tasks >= 2 && stats.dense_tasks <= 20); TEST_ASSERT_EQ_I(ray_table_nrows(r), 100000); - uint64_t llc = ray_cache_llc_bytes(); - if (llc > 0) { + { size_t block = agg_resolve(OP_SUM, RAY_I64)->state_size; double slots = (double)stats.dense_local_slots / stats.dense_tasks; double slab = slots * (block + sizeof(int64_t) + 1); double budget = (double)llc * 0.75; uint32_t cap = slab * 20 > budget ? (uint32_t)(budget / slab) : 20; if (cap < 2) cap = 2; + TEST_ASSERT_TRUE(cap >= 3 && cap < 20); /* the pin makes the bound bite */ TEST_ASSERT_EQ_I(stats.dense_tasks, cap); } /* The bounded run computes the same sums as the serial engine. */ From 4a7276d9eccf5705c8e535e988c77a99861613b2 Mon Sep 17 00:00:00 2001 From: Anton Kundenko Date: Tue, 29 Sep 2026 19:58:08 +0200 Subject: [PATCH 50/51] fix(in): admit the typed kernel unless both sides carry a null (#644) * fix(in): admit the typed kernel unless both sides carry a null #607 opened the typed membership kernel to text columns but kept it behind "both operands exactly null-free". That is stricter than the semantics require: the kernel (null matches nothing) and the hashset fallback (null equals null) disagree only on a null row against a null set element, which needs a null on each side. One null row in the column therefore sent the whole column to the per-row hashset probe. 351,393-row column, -c 8, one null row, null-free needles: SYM (in col one) 4,820 us -> 60 us I64 (in icol [0 1]) 5,730 us -> 80 us SYM null-free (in col one) 145 us -> 55 us (set tested first: a null-free set skips the column's null scan) It also fixes wrong answers: the hashset hashes i64 and f64 cells with different functions, so mixed int/float membership with a null on one side missed matches -- (in [1.0 0Nf 3.0] [1 3]) was [true false false]. Those shapes now take the kernel's double promotion. The both-sides-null mixed-type case still reaches the hashset and is still wrong; it is a hashset bug, not a gate one, and is left for a separate fix. A 23-case null edge matrix is byte-identical before and after apart from that corrected row. The in.rfl comment claiming a needle-only null "must still reject" is corrected, and one-sided nulls across every null-bearing width are pinned. * fix(in): temporal equals only its own type; hash large needle sets in the kernel Review follow-ups on widening the `in` kernel gate. Temporal types. The typed kernel classified DATE/TIME/TIMESTAMP as plain int-family and compared raw payloads, so (in [2000.01.01 0Nd] [0]) was [true false] while find, except and the hashset `in` -- all atom_eq, which never matches a DATE to an int or to a TIMESTAMP -- said no. It also compared a DATE's day count to a TIMESTAMP's nanoseconds: (in [2000.01.02] [2000.01.01D00:00:00.000000001]) was [true]. A temporal operand against a different type is now an empty probe, as SYM vs non-SYM already was. Bare `in`, the fused where: and the hashset agree whichever side carries a null. Large needle sets. Past the SIMD small-set size the kernel scanned every needle per row, O(rows x needles). The widened gate sent nullable int/float columns there, and null-free ones already went: a 351k-row column against 100k needles took ~1.2 s. Sets larger than IN_SIMD_SET (8) now get an open-addressing table over the live needles -- int64 values, or f64 bit patterns with -0.0 folded to +0.0 and NaN never inserted -- built once per call and shared by all workers. An allocation failure keeps the scan. 351,393 rows, -c 8, release, dev before #644 -> this commit: I64 with a null x 3 needles 6,500 us -> 50 us I64 with a null x 1k 10,600 us -> 1,400 us I64 with a null x 100k 10,000 us -> 2,500 us I64 null-free x 100k 1,155,000 us -> 1,500 us F64 with a null x 100k 10,500 us -> 2,500 us select where: in, null-free x 100k 1,155,500 us -> 2,000 us in.rfl pins temporal vs int and distinct temporal types with nulls on neither, either and both sides, the fused where:, and the 8/9-needle boundary, -0.0, NaN, duplicates and narrow ints on the hash path. make test 3952/3952; reverting exec.c fails in.rfl:169. --- src/ops/collection.c | 16 +++-- src/ops/exec.c | 133 +++++++++++++++++++++++++++++++------ test/rfl/collection/in.rfl | 104 +++++++++++++++++++++++++++-- 3 files changed, 225 insertions(+), 28 deletions(-) diff --git a/src/ops/collection.c b/src/ops/collection.c index b61bc3f1..d0618624 100644 --- a/src/ops/collection.c +++ b/src/ops/collection.c @@ -1486,10 +1486,18 @@ ray_t* ray_in_fn(ray_t* val, ray_t* vec) { * on a 351k-row column (#593). vec.h says as much where it * defines the two — "paths requiring null-free data use has_nulls * below". The exact check costs one pass and is not on a per-row - * path; a null-bearing operand still falls through, because the - * kernel's null-matches-nothing semantics differ from this path's - * null-equals-null. */ - if (!ray_vec_has_nulls(val) && !ray_vec_has_nulls(vec)) { + * path. + * + * Only BOTH sides null-bearing must fall through. The kernel's + * null-matches-nothing and this path's null-equals-null disagree + * on exactly one question — does a null row match a null set + * element — and that needs a null on each side. A null row + * against a null-free set, or a null set element against a + * null-free column, is false under both. Requiring both sides + * null-free sent any nullable column to the hashset: one null + * row cost ~25x. The set is tested first: it is usually the + * small side, and a null-free set skips the column scan. */ + if (!(ray_vec_has_nulls(vec) && ray_vec_has_nulls(val))) { ray_t* fast = ray_in_vec_exec(val, vec, false); if (fast) return fast; } diff --git a/src/ops/exec.c b/src/ops/exec.c index fe5dd93b..daa2fe30 100644 --- a/src/ops/exec.c +++ b/src/ops/exec.c @@ -27,6 +27,7 @@ #include "ops/rowsel.h" #include "ops/fused_group.h" #include "ops/idxop.h" +#include "ops/hash.h" #include "mem/heap.h" #include "mem/sys.h" #include "core/qstats.h" /* per-worker parallelism stats for profile spans */ @@ -611,6 +612,16 @@ typedef struct { * per-row linear set scan into one byte load — any set size, width- * specialized, vectorizable. NULL when not applicable. */ const uint8_t* symlut; + /* Needle hash, built when the set outgrows the SIMD small-set path + * (IN_SIMD_SET): open addressing over the live probe values, so a + * row costs one probe instead of a scan of the whole set. hti holds + * int64 values (empty = INT64_MIN; a real INT64_MIN needle sets + * ht_has_min), htf holds f64 bit patterns (empty = IN_HTF_EMPTY, a + * NaN, which is never inserted). Both NULL = linear scan. */ + const int64_t* hti; + const uint64_t* htf; + int64_t ht_mask; + bool ht_has_min; bool col_has_nulls; bool col_atom_null; bool col_is_atom; @@ -618,6 +629,51 @@ typedef struct { bool negate; } in_worker_ctx_t; +/* Sets up to this many live elements take the unrolled SIMD compare in + * exec_in_worker; larger ones get the needle hash (in_build_worker_ctx). */ +#define IN_SIMD_SET 8 +#define IN_HTF_EMPTY UINT64_C(0x7ff8000000000000) + +/* -0.0 and +0.0 compare equal, so they must share a slot. */ +static inline uint64_t in_f64_key(double v) { + uint64_t b; + memcpy(&b, &v, sizeof(b)); + return b == UINT64_C(0x8000000000000000) ? 0 : b; +} + +static inline int in_set_has_i(const in_worker_ctx_t* c, int64_t v) { + if (c->hti) { + if (v == INT64_MIN) return c->ht_has_min; + int64_t s = (int64_t)(ray_hash_i64(v) & (uint64_t)c->ht_mask); + for (;;) { + int64_t k = c->hti[s]; + if (k == v) return 1; + if (k == INT64_MIN) return 0; + s = (s + 1) & c->ht_mask; + } + } + for (int64_t j = 0; j < c->sv_len; j++) + if (v == c->svi[j]) return 1; + return 0; +} + +static inline int in_set_has_f(const in_worker_ctx_t* c, double v) { + if (c->htf) { + if (v != v) return 0; /* NaN equals nothing, as in the scan */ + uint64_t b = in_f64_key(v); + int64_t s = (int64_t)(ray_hash_i64((int64_t)b) & (uint64_t)c->ht_mask); + for (;;) { + uint64_t k = c->htf[s]; + if (k == b) return 1; + if (k == IN_HTF_EMPTY) return 0; + s = (s + 1) & c->ht_mask; + } + } + for (int64_t j = 0; j < c->sv_len; j++) + if (v == c->svf[j]) return 1; + return 0; +} + static void exec_in_worker(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { (void)worker_id; @@ -705,7 +761,7 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, return; } - if (!c->col_is_atom && !c->use_double && sv_len >= 1 && sv_len <= 8) { + if (!c->col_is_atom && !c->use_double && sv_len >= 1 && sv_len <= IN_SIMD_SET) { const int64_t* svi = c->svi; uint8_t neg = (uint8_t)negate; #define IN_FAST(CTYPE, FITS, SENT, HASNULL) do { \ @@ -765,7 +821,6 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, } if (c->use_double) { - const double* svf = c->svf; if (c->col_atom_null) { /* All elements are null — fill zeros */ for (int64_t i = start; i < end; i++) ob[i - ob_base] = 0; @@ -775,24 +830,17 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, double cv; if (c->col_is_atom) cv = (ct == RAY_F64) ? col->f64 : (double)col->i64; else IN_READ_F64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svf[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_f(c, cv) ^ negate); } } else { for (int64_t i = start; i < end; i++) { double cv; if (c->col_is_atom) cv = (ct == RAY_F64) ? col->f64 : (double)col->i64; else IN_READ_F64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svf[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_f(c, cv) ^ negate); } } } else { - const int64_t* svi = c->svi; if (c->col_atom_null) { for (int64_t i = start; i < end; i++) ob[i - ob_base] = 0; } else if (vec_has_nulls) { @@ -801,20 +849,14 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, int64_t cv; if (c->col_is_atom) cv = col->i64; else IN_READ_I64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svi[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_i(c, cv) ^ negate); } } else { for (int64_t i = start; i < end; i++) { int64_t cv; if (c->col_is_atom) cv = col->i64; else IN_READ_I64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svi[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_i(c, cv) ^ negate); } } } @@ -885,6 +927,17 @@ static in_ctx_status_t in_build_worker_ctx(ray_t* col, ray_t* set, bool negate, int col_class = CLASSIFY(ct); int set_class = CLASSIFY(st); + /* A temporal value equals only its own type: atom_eq — and so the + * hashset behind find/except and the null-bearing `in` — never matches + * a DATE to an int or to a TIMESTAMP. Comparing raw payloads here + * matched day 0 to 0i, and a DATE's day count to a TIMESTAMP's + * nanoseconds. Empty probe, exactly like SYM vs non-SYM below. */ + #define IS_TEMPORAL(t) \ + ((t) == RAY_DATE || (t) == RAY_TIME || (t) == RAY_TIMESTAMP) + if ((IS_TEMPORAL(ct) || IS_TEMPORAL(st)) && ct != st) + set_len = 0; + #undef IS_TEMPORAL + /* Mixed SYM vs non-SYM → treat as an empty probe. A SYM set * containing resolved sym IDs has no meaning when compared to a * raw integer column, so nothing can match — but we still drop @@ -1043,10 +1096,52 @@ static in_ctx_status_t in_build_worker_ctx(ray_t* col, ray_t* set, bool negate, } } + /* Large set on a non-LUT column: hash the live needles once so each row + * costs one probe. The linear scan was O(rows × needles) — a nullable + * I64 column against 100k needles took over a second. An allocation + * failure keeps the scan: slower, same answer. */ + const int64_t* hti = NULL; + const uint64_t* htf = NULL; + int64_t ht_mask = 0; + bool ht_has_min = false; + if (!ray_is_atom(col) && !symlut && sv_len > IN_SIMD_SET) { + int64_t cap = 16; + while (cap < sv_len * 2) cap <<= 1; + ray_t* hh = ray_alloc((size_t)cap * sizeof(int64_t)); + if (hh) { + ht_mask = cap - 1; + if (use_double) { + uint64_t* t = (uint64_t*)ray_data(hh); + for (int64_t k = 0; k < cap; k++) t[k] = IN_HTF_EMPTY; + for (int64_t j = 0; j < sv_len; j++) { + if (svf[j] != svf[j]) continue; /* NaN matches nothing */ + uint64_t b = in_f64_key(svf[j]); + int64_t sl = (int64_t)(ray_hash_i64((int64_t)b) & (uint64_t)ht_mask); + while (t[sl] != IN_HTF_EMPTY && t[sl] != b) sl = (sl + 1) & ht_mask; + t[sl] = b; + } + htf = t; + } else { + int64_t* t = (int64_t*)ray_data(hh); + for (int64_t k = 0; k < cap; k++) t[k] = INT64_MIN; + for (int64_t j = 0; j < sv_len; j++) { + int64_t v = svi[j]; + if (v == INT64_MIN) { ht_has_min = true; continue; } + int64_t sl = (int64_t)(ray_hash_i64(v) & (uint64_t)ht_mask); + while (t[sl] != INT64_MIN && t[sl] != v) sl = (sl + 1) & ht_mask; + t[sl] = v; + } + hti = t; + } + *lut_hdr_out = hh; /* freed by every caller with the LUT */ + } + } + *out_ctx = (in_worker_ctx_t){ .col = col, .svf = svf, .svi = svi, .sv_len = sv_len, .symlut = symlut, + .hti = hti, .htf = htf, .ht_mask = ht_mask, .ht_has_min = ht_has_min, .ob = NULL, .ob_base = 0, .ct = ct, .col_has_nulls = col_has_nulls, .col_atom_null = col_atom_null, diff --git a/test/rfl/collection/in.rfl b/test/rfl/collection/in.rfl index 026417e4..9f25838c 100644 --- a/test/rfl/collection/in.rfl +++ b/test/rfl/collection/in.rfl @@ -86,9 +86,10 @@ (in (list) [1h 2h]) -- (list) ;; ========== TEXT (SYM/STR) NULL SEMANTICS (#593) ========== -;; `in` admits the typed verdict-LUT kernel only when BOTH sides are exactly -;; null-free, because that kernel uses null-matches-nothing semantics while -;; this path is null-equals-null. The gate used to ask +;; `in` admits the typed verdict-LUT kernel unless BOTH sides carry a null: +;; the kernel uses null-matches-nothing semantics while the hashset path is +;; null-equals-null, and the two disagree only on a null row against a null +;; set element. The gate used to ask ;; ray_vec_may_have_nulls, which is unconditionally true for SYM and STR ;; (their null is a payload value, not an attribute bit) — so the kernel was ;; unreachable for text columns and every symbol `in` fell through to the @@ -102,8 +103,8 @@ ;; null in the column, non-null needle — the null matches nothing else (in (as 'SYM ["a" "" "b"]) (as 'SYM ["a"])) -- [true false false] -;; null only in the NEEDLES: the column is null-free but the gate must still -;; reject, since a null needle needs null-equals-null against nothing +;; null only in the NEEDLES: a null-free column has no row for the null +;; needle to equal, so both semantics agree and the kernel is admitted (in (as 'SYM ["a" "b"]) (as 'SYM ["" "a"])) -- [true false] ;; nulls on both sides @@ -113,6 +114,38 @@ (in (as 'SYM ["a" "b" "c"]) (as 'SYM ["b" "c"])) -- [false true true] (in ["a" "b" "c"] ["b" "c"]) -- [false true true] +;; ========== ONE-SIDED NULLS REACH THE KERNEL (#593) ========== +;; A nullable column against null-free needles (or the reverse) is admitted: +;; a null row can only match a null needle. Before, one null row sent the +;; whole column to the hashset, whose i64/f64 hashes disagree across numeric +;; types — so the mixed-type rows below also answered wrongly. + +;; nullable column, null-free needles, every null-bearing width +(in [1 0Nl 3] [1 3]) -- [true false true] +(in [1i 0Ni 3i] [1 3]) -- [true false true] +(in [1h 0Nh 3h] [1h]) -- [true false false] +(in [1.0 0Nf 3.0] [1.0 3.0]) -- [true false true] +(in [2024.01.01 0Nd] [2024.01.01]) -- [true false] +(in ["a" "" "b"] ["a"]) -- [true false false] + +;; the i32 null sentinel is INT32_MIN: a non-null needle of that value must +;; not match the null row +(in [1i 0Ni 3i] [-2147483648 3]) -- [false false true] + +;; null-free column, null needles: the null needle matches nothing +(in [1 2 3] [0Nl 3]) -- [false false true] +(in [1.0 2.0 3.0] [0Nf 3.0]) -- [false false true] +(in [2024.01.01 2024.01.02] [0Nd 2024.01.02]) -- [false true] +(in ["a" "b"] ["" "a"]) -- [true false] + +;; mixed int/float with a null on one side — was [true false false] and +;; [false false false] on the hashset path +(in [1.0 0Nf 3.0] [1 3]) -- [true false true] +(in [1 0Nl 3] [1.0 3.0]) -- [true false true] + +;; nulls on both sides still take the null-equals-null path +(in [1 0Nl 3] [0Nl 3]) -- [false true true] + ;; empty needle set over a null-free symbol column (in (as 'SYM ["a" "b"]) (as 'SYM [])) -- [false false] @@ -123,3 +156,64 @@ ;; duplicate needles must not double-count or change the verdict (in (as 'SYM ["a" "b"]) (as 'SYM ["a" "a" "a"])) -- [true false] + +;; ========== TEMPORAL EQUALS ONLY ITS OWN TYPE (#593 review) ========== +;; atom_eq never matches a DATE to an int or to a TIMESTAMP, and neither do +;; find / except or the hashset `in`. The typed kernel compared raw +;; payloads — day 0 matched 0i, a DATE's day count matched a TIMESTAMP's +;; nanoseconds — so widening its gate made the answer depend on where the +;; nulls were. Every shape below answers the same with nulls on either +;; side, none, or (for the hashset) both. + +;; temporal column, int needles: nothing matches +(in [2000.01.01 2000.01.03] [0i 2i]) -- [false false] +(in [2000.01.01 0Nd 2000.01.03] [0i 2i]) -- [false false false] +(in [2000.01.01 2000.01.03] [0Ni 2i]) -- [false false] +(in [2000.01.01 0Nd] [0]) -- [false false] +(in [00:00:01.000 0Nt] [1000i]) -- [false false] +(in [2000.01.01D00:00:00.000000000 0Np] [0]) -- [false false] + +;; int column, temporal needles: nothing matches +(in [0i 2i] [2000.01.01 2000.01.03]) -- [false false] +(in [0 0Nl] [2000.01.01]) -- [false false] +(in [0 1] [2000.01.01 0Nd]) -- [false false] + +;; distinct temporal types never match — not even by raw payload +(in [2000.01.02] [2000.01.01D00:00:00.000000001]) -- [false] +(in [2000.01.02 0Nd] [2000.01.02D00:00:00.000000000]) -- [false false] +(in [2000.01.01D00:00:00.000000000 0Np] [2000.01.01]) -- [false false] +(in [00:00:00.001] [2000.01.01D00:00:00.000000001]) -- [false] + +;; same with nulls on both sides: only null equals null +(in [2000.01.01 0Nd] [0Ni 0i]) -- [false true] + +;; same-type temporal still matches, with or without nulls +(in [2000.01.01 0Nd 2000.01.03] [2000.01.03]) -- [false false true] +(in [2000.01.01 2000.01.03] [2000.01.01 0Nd]) -- [true false] + +;; the fused where: and find agree with the bare in +(count (select {from: (table [d] (list [2000.01.01 2000.01.03 0Nd])) where: (in d [0i 2i])})) -- 0 +(find [2000.01.01 2000.01.03] [0i]) -- [0Nl] + +;; ========== LARGE NEEDLE SETS HASH INSIDE THE KERNEL ========== +;; Past the SIMD small-set size (8) the kernel hashes the needles instead of +;; scanning them per row, so a nullable column against thousands of needles +;; stays O(rows + needles). Pins the boundary and the table's edge values. + +;; 8 needles (SIMD path) and 9 (hash path) answer alike +(in [1 2 3 0Nl 20] [1 3 5 7 9 11 13 15]) -- [true false true false false] +(in [1 2 3 0Nl 20] [1 3 5 7 9 11 13 15 20]) -- [true false true false true] +(in [1.0 2.0 0Nf 20.0] [1.0 3.0 5.0 7.0 9.0 11.0 13.0 15.0 20.0]) -- [true false false true] + +;; -0.0 and +0.0 share a slot; NaN / null never match +(in [-0.0 0.0 1.5] (as 'F64 (til 20))) -- [true true false] +(in [0Nf 2.5] (concat (as 'F64 (til 20)) [2.5])) -- [false true] + +;; duplicates in a large set, and a large set against a nullable column +(sum (in (concat (til 1000) [0Nl]) (take (til 100) 5000))) -- 100 +(sum (in (concat (til 1000) [0Nl]) (* 2 (til 1000)))) -- 500 +(sum (in (concat (as 'F64 (til 1000)) [0Nf]) (as 'F64 (* 2 (til 1000))))) -- 500 + +;; a large set of narrow ints against a narrow column +(sum (in (as 'I32 (til 1000)) (as 'I32 (* 3 (til 100))))) -- 100 +(sum (in (as 'I16 (til 1000)) (as 'I16 (* 3 (til 100))))) -- 100 From 7b7acdff8acf6f31dd367e440f5b1a5b9af79000 Mon Sep 17 00:00:00 2001 From: Serhii Savchuk Date: Tue, 29 Sep 2026 21:39:23 +0300 Subject: [PATCH 51/51] fix(agg): integer mean divides the exact 128-bit sum; mapped index tail unmapped after drop (#648) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(agg): integer mean divides the exact 128-bit sum The mean of an integer column depended on where it was computed. The group engines divided a WRAPPED int64 sum — a whole-table `select {a: (avg id)}` over signed 64-bit ids answered -5.59e10 for a true mean of 2.53e18, and any per-group total past 2^63 was garbage — while the vector path accumulated in double and lost low bits past 2^53, so its last bits moved with the accumulation order. The chunk-zone metadata could only answer a mean when every partial sum stayed below 2^53. Every integer mean now divides the exact 128-bit sum of its rows, converted to double by one formula, so the vector reduction, the parted column path, the columnar, hash-row, scatter and slice group engines, the streaming engine, pivot and the metadata agree bit for bit: - the chunk-zone aggregates keep the high word of each chunk's sum in a third block (older two-block indexes still load; without high words the metadata answers only when the extrema prove the total fits); - the keyless reduction and the group accumulators carry a high word beside the wrapped sum, allocated only when an integer avg over an I64 / TIMESTAMP column (or a table of 2^31 rows or more) is present — narrower inputs cannot leave int64 and keep their plain sum; - the streaming engine keeps its 16-byte state: a signed running sum plus the wrap count packed with the row count, exact while a group has fewer than 2^32 rows (larger tables stay on the legacy engines); narrow inputs use an exact int64 state; - a bare I64 column sums in blocks whose mode (values below 2^53, below 2^53 in magnitude, or general) is escalated once. `sum` keeps its int64 wraparound contract. The grouped combine over a parted table (per-partition sums joined afterwards) is not changed here and still divides the wrapped total. Tests: group/avg_exact_i128 (exact means over values that overflow int64 within every group, across the keyless, columnar, hash, scatter, slice, streaming and pivot paths, nulls, narrow types, a splayed copy), test_agg_engine avg_exact_i128_engines (v2 on and off), store/splayed_zone_aggs (metadata means past 2^53 and past 2^63). Co-Authored-By: Claude Fable 5.1 * fix(store): a mapped column that drops its index still unmaps the index tail A column loaded by mmap with an inline index is longer than its payload, and ray_free sized the unmap from the attached index. Once the loaded column itself dropped that index — an in-place edit of its only reference detaches a mapped index without releasing it — or when ray_index_inline_map had discarded a child that is still in the file, the unmap covered the payload only and the index tail stayed mapped for the life of the process. A mapped column that carries an index now registers its region in the file-map registry under its own address with the true mapped length, the way a string column's pool already does, and ray_free consults the registry for every mapped column before falling back to the size it can derive. Test: index/mapped_drop_unmaps_tail (the tail page must be unmapped after the sole reference drops its index and is freed). Co-Authored-By: Claude Fable 5.1 * test(index): lay the unmap-tail fixture out by the platform page size The index region must reach a page of its own for the test to observe the short unmap; with 16 KiB pages (Apple silicon) the 4 KiB layout put the whole file in one page and msync on a misaligned address failed. The payload now ends 64 bytes short of the second page whatever the page size, and msync is issued on a page-aligned range. Co-Authored-By: Claude Fable 5.1 * fix(group): slice-group fast branches read TIME as 4 bytes; heap mirror consults the registry The slice-group fast branches took RAY_TIME through the int64 path (`t == RAY_I64 || t == RAY_TIME`, and the F64 x int product's `tb == RAY_I64 || tb == RAY_TIME`), although a TIME cell is a 4-byte millisecond count: a contiguous chunk of n rows would read 2n int32 cells, pair neighbouring times into one value and run past the column. TIME and DATE now take the 4-byte branch with I32; the int64 branch serves I64 and TIMESTAMP. mapped_block_bytes, the read-only mirror of the size ray_free hands to the unmap, consulted the file-map registry for string columns only; since an indexed mapped column registers its region too, it consults the registry for every mapped column, as ray_free does. Co-Authored-By: Claude Fable 5.1 * fix(group): the shared-stream pairing gate admits the int sides the product branches handle The F64 x int product pairing still admitted a TIME int side after TIME moved to the 4-byte branch, while no typed branch of sg_prod_range handles it — the generic branch zeroes the sibling sum under a comment that relies on the gate. Admit I64 and I32 only, in step with the branches (not reachable from the language: a temporal factor is rejected before a product forms). Co-Authored-By: Claude Fable 5.1 --------- Co-authored-by: Claude Fable 5.1 Co-authored-by: Anton Kundenko --- src/mem/heap.c | 14 +- src/ops/agg.c | 91 +++-- src/ops/agg_engine.c | 8 + src/ops/agg_stream.c | 147 ++++++-- src/ops/group.c | 535 +++++++++++++++++++++++---- src/ops/idxop.c | 21 +- src/ops/idxop.h | 35 +- src/ops/internal.h | 9 + src/ops/pivot.c | 11 +- src/ops/query.c | 6 +- src/store/col.c | 24 +- test/rfl/group/avg_exact_i128.rfl | 139 +++++++ test/rfl/store/splayed_zone_aggs.rfl | 27 +- test/test_agg_engine.c | 96 +++++ test/test_index.c | 65 ++++ 15 files changed, 1057 insertions(+), 171 deletions(-) create mode 100644 test/rfl/group/avg_exact_i128.rfl diff --git a/src/mem/heap.c b/src/mem/heap.c index 45244056..02e41fe5 100644 --- a/src/mem/heap.c +++ b/src/mem/heap.c @@ -1683,8 +1683,11 @@ void ray_free(ray_t* v) { * pool before appending — has to ask the registry, and only a * string column was ever registered. So the lock stays off the * ordinary free entirely, and a mutated column pays it once. */ + /* A column loaded with an inline index registered its region too + * (col.c): the index may have been detached since, and only the + * descriptor still knows the mapped length. */ ray_file_map_t* m = col_map; - if (!m && v->type == RAY_STR) m = ray_file_map_lookup(v); + if (!m) m = ray_file_map_lookup(v); if (m) { ray_file_map_release(m); if (h) RAY_STAT(h->stats.free_count++); @@ -2942,9 +2945,14 @@ void ray_parallel_end(void) { * else is the page-rounded payload plus an inline passenger index. */ static size_t mapped_block_bytes(const ray_t* v) { if (v->type == RAY_TABLE || v->type == RAY_DICT || v->type == RAY_LIST) return 0; - if (v->type == RAY_STR) { + { + /* Same route as ray_free: the pool's descriptor for a string + * column, else the registry — which also holds every indexed + * mapped column (col.c), whether or not its index is still + * attached. */ ray_file_map_t* m = NULL; - if (v->str_pool && !RAY_IS_ERR(v->str_pool) && v->str_pool->mmod == 3) + if (v->type == RAY_STR && v->str_pool && !RAY_IS_ERR(v->str_pool) && + v->str_pool->mmod == 3) m = v->str_pool->file_map; if (!m) m = ray_file_map_lookup(v); if (m) return m->len; diff --git a/src/ops/agg.c b/src/ops/agg.c index 138afb65..fe7d7f9d 100644 --- a/src/ops/agg.c +++ b/src/ops/agg.c @@ -169,12 +169,14 @@ static ray_t* agg_parted_avg(ray_t* x) { if (!agg_parted_numeric_base(base)) return ray_error("type", "avg expects a numeric or temporal parted column, got %s", ray_type_name(base)); ray_t** segs = (ray_t**)ray_data(x); double sum = 0.0; + int64_t hi = 0; uint64_t lo = 0; /* integer segments: exact 128-bit sum */ int64_t cnt = 0; + bool fp = base == RAY_F64 || base == RAY_F32; for (int64_t s = 0; s < x->len; s++) { ray_t* seg = segs[s]; if (!seg) continue; int has_nulls = ray_vec_may_have_nulls(seg); - if (base == RAY_F64 || base == RAY_F32) { + if (fp) { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; if (base == RAY_F64) sum += ((double*)ray_data(seg))[i]; @@ -184,12 +186,12 @@ static ray_t* agg_parted_avg(ray_t* x) { } else { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; - sum += (double)agg_read_i64(seg, i); cnt++; + ray_i128_add(&hi, &lo, agg_read_i64(seg, i)); cnt++; } } } if (cnt == 0) return ray_typed_null(-RAY_F64); - return make_f64(sum / (double)cnt); + return make_f64((fp ? sum : ray_i128_to_f64(hi, lo)) / (double)cnt); } static ray_t* agg_parted_prod(ray_t* x) { @@ -404,36 +406,63 @@ static ray_t* agg_pair_vec(ray_t* x, ray_t* y, uint16_t op) { return ray_f64(ray_f64_fin(num / sqrt(dx * dy))); } -/* Whole-column sum / non-null count of an integer column from its - * chunk-zone index (per-chunk int64 sums with wraparound and non-null - * counts), in O(n_chunks). `exact_f64` reports whether every partial sum - * an accumulation in double could meet stays below 2^53 in magnitude, i.e. - * the sum converted to double equals any double accumulation of the rows. - * Returns false when the column has no such index for its current length. */ -bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64) { - if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return false; +/* Per-chunk aggregates of an integer column's chunk-zone index, or NULL: + * [sum low words | non-null counts | sum high words] (the last block only + * in the current layout). */ +static const int64_t* zone_aggs(ray_t* x, uint32_t* n_out, bool* have_hi) { + if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return NULL; ray_index_t* ix = ray_index_payload(x->index); if (ix->built_for_len != x->len || ix->u.chunk_zone.is_f64 || !ix->u.chunk_zone.aggs) - return false; + return NULL; uint32_t n = ix->u.chunk_zone.n_chunks; - if (ix->u.chunk_zone.aggs->len != 2 * (int64_t)n) return false; - const int64_t* ag = (const int64_t*)ray_data(ix->u.chunk_zone.aggs); - const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); - const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + int64_t len = ix->u.chunk_zone.aggs->len; + if (len != 2 * (int64_t)n && len != 3 * (int64_t)n) return NULL; + *n_out = n; + *have_hi = len == 3 * (int64_t)n; + return (const int64_t*)ray_data(ix->u.chunk_zone.aggs); +} + +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; uint64_t sum = 0; int64_t nn = 0; - double bound = 0.0; for (uint32_t g = 0; g < n; g++) { sum += (uint64_t)ag[g]; nn += ag[n + g]; - if (ag[n + g] > 0) { - double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); - bound += (a > b ? a : b) * (double)ag[n + g]; - } } *sum_out = (int64_t)sum; *nn_out = nn; - if (exact_f64) *exact_f64 = bound < 9007199254740992.0; /* 2^53 */ + return true; +} + +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; + ray_index_t* ix = ray_index_payload(x->index); + const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); + const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + int64_t hi = 0; uint64_t lo = 0; int64_t nn = 0; + double bound = 0.0; + for (uint32_t g = 0; g < n; g++) { + int64_t cnt = ag[n + g]; + nn += cnt; + if (have_hi) { + ray_i128_add128(&hi, &lo, ag[2 * (int64_t)n + g], (uint64_t)ag[g]); + } else { + /* no high words stored: the wrapped low word is the chunk's + * exact sum only when no partial sum could leave int64 */ + if (cnt > 0) { + double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); + bound += (a > b ? a : b) * (double)cnt; + } + ray_i128_add(&hi, &lo, ag[g]); + } + } + if (!have_hi && !(bound < 9223372036854775808.0)) return false; /* 2^63 */ + *hi_out = hi; *lo_out = lo; *nn_out = nn; return true; } @@ -453,7 +482,7 @@ ray_t* ray_sum_fn(ray_t* x) { /* Integer columns with per-chunk sums in their zone index. */ if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { int64_t zs, zn; - if (ray_zone_int_sum(x, &zs, &zn, NULL)) return make_i64(zs); + if (ray_zone_int_sum(x, &zs, &zn)) return make_i64(zs); } /* Narrow/temporal types need specific return constructors that the * DAG executor doesn't provide — use scalar path for these. TIMESTAMP @@ -608,14 +637,14 @@ ray_t* ray_avg_fn(ray_t* x) { /* Canonical admission: numeric + temporal (→ F64); SYM/STR/GUID are * non-numeric → type error (the DAG path otherwise averaged raw ids). */ if (!agg_type_admitted(OP_AVG, x->type)) return ray_error("type", "avg expects a numeric or temporal vector, got %s", ray_type_name(x->type)); - /* Integer columns with per-chunk sums: exact when no partial sum can - * leave double's integer range (then every accumulation order in - * double gives the same value). */ + /* Integer columns with per-chunk sums: the exact 128-bit total is + * what the row-wise reduction computes too, so the answers agree + * bit for bit. */ if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { - int64_t zs, zn; bool exact = false; - if (ray_zone_int_sum(x, &zs, &zn, &exact) && exact) { + int64_t zh, zn; uint64_t zl; + if (ray_zone_int_sum128(x, &zh, &zl, &zn)) { if (zn == 0) return ray_typed_null(-RAY_F64); - return make_f64((double)zs / (double)zn); + return make_f64(ray_i128_to_f64(zh, zl) / (double)zn); } } AGG_VEC_VIA_DAG(x, ray_avg); @@ -663,7 +692,7 @@ ray_t* ray_min_fn(ray_t* x) { * zone has them; the sentinel alone cannot tell a * column of INT64_MAX values from an empty one. */ int64_t zs_, zn_; - bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); if (have_nn ? zn_ == 0 : mn == INT64_MAX) return ray_typed_null(-x->type); /* Preserve the column's storage width on the result. */ switch (x->type) { @@ -723,7 +752,7 @@ ray_t* ray_max_fn(ray_t* x) { * zone has them; the sentinel alone cannot tell a * column of INT64_MIN values from an empty one. */ int64_t zs_, zn_; - bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); if (have_nn ? zn_ == 0 : mx == INT64_MIN) return ray_typed_null(-x->type); switch (x->type) { case RAY_BOOL: return ray_bool((bool)mx); diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index 81ffe665..984faff6 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -184,6 +184,14 @@ agg_v2_reason_t agg_v2_admission(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { const agg_vtable_t* vt = agg_resolve(ext->agg_ops[a], ic->type); if (!vt) return AGG_V2_AGG_TYPE; if (vt->kind != ACC_STREAMING) return AGG_V2_BUFFERED; + /* The streaming 64-bit integer avg packs its 128-bit high word with + * the group count in one word (agg_stream.c): exact while every + * group has fewer than 2^32 rows; the narrower kernels keep a plain + * int64 sum, exact below 2^31 rows. A table that large keeps the + * legacy engines, whose accumulators carry a full high word. */ + if (ext->agg_ops[a] == OP_AVG && ic->type != RAY_F64 && ic->type != RAY_F32 && + ray_table_nrows(tbl) >= ((int64_t)1 << ((ic->type == RAY_I64 || ic->type == RAY_TIMESTAMP) ? 32 : 31))) + return AGG_V2_AGG_TYPE; } return AGG_V2_ADMITTED; } diff --git a/src/ops/agg_stream.c b/src/ops/agg_stream.c index 8912d186..083b5a3c 100644 --- a/src/ops/agg_stream.c +++ b/src/ops/agg_stream.c @@ -5,6 +5,7 @@ #include "ops/ops.h" #include "ops/internal.h" /* ray_f64_fin (single-null float model) */ #include "lang/internal.h" /* ray_median_dbl_inplace */ +#include "ops/idxop.h" /* ray_i128_add / ray_i128_to_f64: exact integer avg */ #include #include #include /* realloc/free for the buffered median accumulator */ @@ -311,6 +312,80 @@ static const agg_vtable_t AVG_F64 = { .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, }; +/* ---- avg, integer inputs: exact 128-bit sum --------------------------- + * Every integer/temporal avg kernel below keeps the exact sum and divides + * ray_i128_to_f64(hi, lo) by the count — the same bits as the keyless + * reduction, the legacy group engines and the chunk-zone metadata, for any + * column (a double running sum loses low bits past 2^53 and an int64 one + * wraps). + * + * The state stays 16 bytes (the dense plans budget slot traffic by state + * size): `lo` is a signed int64 running sum and `hc` packs, as hi:32 | + * cnt:32, the number of times that sum wrapped past +/-2^63 with the row + * count — the total is wraps * 2^64 + lo. A signed-overflow test per row + * is a never-taken branch on ordinary data, cheaper than a carry chain. + * Exact for any int64 values while a group holds fewer than 2^32 rows + * (|sum| < 2^95, so the wrap count fits 32 signed bits); agg_v2_admission + * keeps tables of 2^32 rows or more off the v2 engine when an integer avg + * is present. */ +typedef struct { int64_t lo; uint64_t hc; } avg_i128_state; +#define AVG_I128_WRAPS(hc) ((int64_t)(int32_t)(uint32_t)((hc) >> 32)) +#define AVG_I128_CNT(hc) ((int64_t)((hc) & 0xffffffffu)) +#define AVG_I128_ONE_WRAP ((uint64_t)1 << 32) +static void avg_i128_init(void* s) { + avg_i128_state* st = (avg_i128_state*)s; st->lo = 0; st->hc = 0; +} +static inline void avg_i128_add(avg_i128_state* st, int64_t v) { + int64_t o = st->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + st->lo = nw; + st->hc += 1u; + /* signed overflow: o and v share a sign the result lacks */ + if (RAY_UNLIKELY(((o ^ nw) & (v ^ nw)) < 0)) + st->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static void avg_i128_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; avg_i128_state* dd = (avg_i128_state*)d; const avg_i128_state* ss = (const avg_i128_state*)s; + int64_t o = dd->lo, v = ss->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + dd->lo = nw; + dd->hc += ss->hc; + if (((o ^ nw) & (v ^ nw)) < 0) + dd->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static double avg_i128_final_result(const void* s) { + const avg_i128_state* st = s; + int64_t cnt = AVG_I128_CNT(st->hc); + if (!cnt) return NULL_F64; + /* two's-complement (hi, lo): the signed low word borrows one from the + * wrap count when negative */ + int64_t hi = AVG_I128_WRAPS(st->hc) + (st->lo < 0 ? -1 : 0); + return ray_f64_fin(ray_i128_to_f64(hi, (uint64_t)st->lo) / (double)cnt); +} +AGG_SCALAR_FINAL(avg_i128_final, double, ray_f64, value != value) +#define AVG_I128_UPDATE_BODY \ + avg_i128_add((avg_i128_state*)((char*)base + (size_t)gids[i]*stride), (int64_t)d[i]) + +/* ---- avg, inputs of at most 32 bits: exact int64 sum ------------------- + * Fewer than 2^31 values of at most 32 bits sum to less than 2^63, so a + * plain int64 running sum is exact (agg_v2_admission keeps larger tables + * off v2 for these kernels) and (double)sum / cnt is bit for bit what the + * 128-bit form would give. */ +typedef struct { int64_t sum; int64_t cnt; } avg_i64_state; +static void avg_i64_init(void* s) { ((avg_i64_state*)s)->sum = 0; ((avg_i64_state*)s)->cnt = 0; } +static void avg_i64_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; ((avg_i64_state*)d)->sum += ((const avg_i64_state*)s)->sum; + ((avg_i64_state*)d)->cnt += ((const avg_i64_state*)s)->cnt; +} +static double avg_i64_final_result(const void* s) { + const avg_i64_state* st = s; + return st->cnt ? ray_f64_fin((double)st->sum / (double)st->cnt) : NULL_F64; +} +AGG_SCALAR_FINAL(avg_i64_final, double, ray_f64, value != value) +#define AVG_I64_UPDATE_BODY \ + avg_i64_state* st = (avg_i64_state*)((char*)base + (size_t)gids[i]*stride); \ + st->sum += (int64_t)d[i]; st->cnt++ + /* ---- variance family, I64 (sumsq as int64 unsigned-wrap; formula group.c:2190) -- */ /* Shifted-data accumulator: sums are of (v - k), where k is the first * value this state sees. The textbook one-pass form sumsq/n - mean^2 @@ -797,15 +872,14 @@ static void avg_bool_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_BOOL_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_bool_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_bool_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_bool_native_update(void* base, size_t stride, const uint32_t* gids, @@ -910,15 +984,14 @@ static void avg_u8_native_update(void* base, size_t stride, const uint32_t* gids int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_U8_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_u8_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_u8_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_u8_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1023,15 +1096,14 @@ static void avg_i16_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int16_t* d = (const int16_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I16_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i16_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i16_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i16_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1136,15 +1208,14 @@ static void avg_i32_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I32_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i32_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i32_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1203,15 +1274,14 @@ static void avg_i64_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_I64_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i64_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_i64_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void min_f32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1363,15 +1433,14 @@ static void avg_date_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_DATE_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_date_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_date_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_date_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1459,15 +1528,14 @@ static void avg_time_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIME_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_time_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_time_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_time_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1575,15 +1643,14 @@ static void avg_timestamp_native_update(void* base, size_t stride, const uint32_ int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIMESTAMP_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_timestamp_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_timestamp_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void var_timestamp_native_update(void* base, size_t stride, const uint32_t* gids, diff --git a/src/ops/group.c b/src/ops/group.c index bb737c18..5226285f 100644 --- a/src/ops/group.c +++ b/src/ops/group.c @@ -43,6 +43,19 @@ static inline bool group_fp_type(int8_t t) { return t == RAY_F32 || t == RAY_F64; } +/* Does an integer AVG over this input need the 128-bit high word next to + * its int64 sum? Only a 64-bit input can push a group's total past int64 + * in fewer than 2^31 rows; a narrower input (and a strlen fusion, whose + * lengths are smaller still) stays exact in the wrapped int64 sum, so it + * skips the carry and the per-slot high word entirely — the dense + * direct-array path in particular carves that word per worker. An + * unknown type (0: a linear plan without a source vector) is treated as + * 64-bit. */ +static inline bool group_avg_needs_hi(int8_t t, int64_t nrows) { + return t == RAY_I64 || t == RAY_TIMESTAMP || t == RAY_SYM || t == 0 || + nrows >= ((int64_t)1 << 31); +} + /* * group_key_f64_bits -- read an F64 GROUP BY key's bits, canonicalised. * @@ -131,10 +144,13 @@ int64_t ray_group_perpart_runs(void) { typedef struct { double sum_f, min_f, max_f, prod_f, first_f, last_f, sum_sq_f; int64_t sum_i, min_i, max_i, prod_i, first_i, last_i, sum_sq_i; - /* Parallel f64 sum of the integer stream — used by AVG so the - * mean of an i64 column whose sum exceeds 2^63 stays accurate - * instead of being whatever (uint64) wrap left in sum_i. */ + /* f64 sum of the integer stream (variance / stddev of integers). */ double sum_d; + /* High word of the exact 128-bit sum of the integer stream; sum_i is + * its low word. AVG of an integer column divides this exact total, + * so the mean is the same bits whatever the morsel split and equals + * the value the chunk-zone metadata computes from its per-chunk sums. */ + int64_t sum_hi; int64_t cnt; int64_t zero_count; bool has_first; @@ -145,7 +161,7 @@ static void reduce_acc_init(reduce_acc_t* acc) { acc->prod_f = 1.0; acc->first_f = 0; acc->last_f = 0; acc->sum_sq_f = 0; acc->sum_i = 0; acc->min_i = INT64_MAX; acc->max_i = INT64_MIN; acc->prod_i = 1; acc->first_i = 0; acc->last_i = 0; acc->sum_sq_i = 0; - acc->sum_d = 0; + acc->sum_d = 0; acc->sum_hi = 0; acc->cnt = 0; acc->zero_count = 0; acc->has_first = false; } @@ -329,13 +345,14 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, #define RED_NEED_MAX (1u << 7) #define RED_NEED_FIRST (1u << 8) #define RED_NEED_LAST (1u << 9) +#define RED_NEED_SUM_128 (1u << 10) /* integer stream: exact 128-bit sum in (sum_hi, sum_i) */ #define RED_MASK_ANY (RED_NEED_COUNT | RED_NEED_ZERO) #define RED_MASK_MIN (RED_NEED_COUNT | RED_NEED_MIN) #define RED_MASK_MAX (RED_NEED_COUNT | RED_NEED_MAX) #define RED_MASK_FIRST (RED_NEED_COUNT | RED_NEED_FIRST) #define RED_MASK_LAST (RED_NEED_COUNT | RED_NEED_LAST) -#define RED_MASK_AVG_I (RED_NEED_SUM_D | RED_NEED_COUNT) +#define RED_MASK_AVG_I (RED_NEED_SUM_128 | RED_NEED_COUNT) #define RED_MASK_AVG_F (RED_NEED_SUM | RED_NEED_COUNT) #define RED_MASK_STATS_I (RED_NEED_SUM_D | RED_NEED_SUM_SQ | RED_NEED_COUNT) #define RED_MASK_STATS_F (RED_NEED_SUM | RED_NEED_SUM_SQ | RED_NEED_COUNT) @@ -368,6 +385,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -396,6 +418,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -662,7 +689,11 @@ static void reduce_merge(reduce_acc_t* dst, const reduce_acc_t* src, int8_t in_t break; case OP_AVG: if (fp) dst->sum_f += src->sum_f; - else dst->sum_d += src->sum_d; + else { + uint64_t lo = (uint64_t)dst->sum_i + (uint64_t)src->sum_i; + dst->sum_hi += src->sum_hi + (lo < (uint64_t)src->sum_i ? 1 : 0); + dst->sum_i = (int64_t)lo; + } dst->cnt += src->cnt; break; case OP_VAR: case OP_VAR_POP: case OP_STDDEV: case OP_STDDEV_POP: @@ -4268,7 +4299,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: result = ray_i64(scan_n); break; - case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : merged.sum_d / merged.cnt)) : ray_typed_null(-RAY_F64); break; + case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : ray_i128_to_f64(merged.sum_hi, (uint64_t)merged.sum_i) / merged.cnt)) : ray_typed_null(-RAY_F64); break; case OP_FIRST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.first_f) : reduction_i64_result(merged.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_LAST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.last_f) : reduction_i64_result(merged.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_VAR: case OP_VAR_POP: @@ -4309,7 +4340,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: return ray_i64(scan_n); - case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : acc.sum_d / acc.cnt)) : ray_typed_null(-RAY_F64); + case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : ray_i128_to_f64(acc.sum_hi, (uint64_t)acc.sum_i) / acc.cnt)) : ray_typed_null(-RAY_F64); case OP_FIRST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.first_f) : reduction_i64_result(acc.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_LAST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.last_f) : reduction_i64_result(acc.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_VAR: case OP_VAR_POP: @@ -4683,6 +4714,8 @@ bool ght_compute_layout(ght_layout_t* out, uint32_t n_keys, uint32_t n_aggs, if (need_flags & GHT_NEED_MIN) { out->off_min = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_MAX) { out->off_max = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_SUMSQ) { out->off_sumsq = (uint16_t)off; off += block; } + /* High words of the exact 128-bit integer sums (integer AVG). */ + if (need_flags & GHT_NEED_SUM128) { out->off_sum_hi = (uint16_t)off; off += block; } /* Per-slot row-index bounds for FIRST/LAST. Two int64 blocks of * n_agg_vals slots each, allocated only when needed. */ if (has_first_last) { @@ -5310,6 +5343,14 @@ static inline void init_accum_from_entry(char* row, const char* entry, } else { memcpy(row + ly->off_sum + s * 8, agg_data + s * 8, 8); } + /* 128-bit sum: the high word sign-extends the first integer + * value (the row was zeroed above, so 0 is right for F64, + * truthy and non-negative slots). */ + if ((nf & GHT_NEED_SUM128) && !(af & GHT_AF_F64) && + !(aflags2[a] & GHT_AF2_TRUTHY)) { + int64_t v; memcpy(&v, agg_data + s * 8, 8); + if (v < 0) { int64_t hi = -1; memcpy(row + ly->off_sum_hi + s * 8, &hi, 8); } + } } if (nf & GHT_NEED_MIN) memcpy(row + ly->off_min + s * 8, agg_data + s * 8, 8); if (nf & GHT_NEED_MAX) memcpy(row + ly->off_max + s * 8, agg_data + s * 8, 8); @@ -5432,6 +5473,7 @@ static inline void accum_from_entry(char* row, const char* entry, else if (af & GHT_AF_LAST) { if (take_last) memcpy(row + ly->off_sum + s * 8, val, 8); } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -5560,6 +5602,7 @@ static void accum_from_entry_nullable(char* row, const char* entry, } } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -6713,7 +6756,10 @@ static void radix_phase3_fn(void* ctx, uint32_t worker_id, int64_t start, int64_ case OP_AVG: if (nn == 0) { v = NULL_F64; grp_set_null(ao->vec, di); break; } v = sf ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (ao->affine) v += ao->bias_f64; break; case OP_MIN: @@ -6926,7 +6972,9 @@ static inline uint32_t group_merge_row(group_ht_t* ht, const uint8_t* const aflags = ly->agg_flags; const int8_t* const vslot = ly->agg_val_slot; uint16_t off_sum = ly->off_sum; + uint16_t off_sum_hi = ly->off_sum_hi; bool need_sum = (ly->need_flags & GHT_NEED_SUM) != 0; + bool need_sum128 = (ly->need_flags & GHT_NEED_SUM128) != 0; for (;;) { uint32_t sv = ht->slots[slot]; if (sv == HT_EMPTY) { @@ -6965,10 +7013,16 @@ static inline uint32_t group_merge_row(group_ht_t* ht, memcpy(&sv_f, src_row + off, 8); // cppcheck-suppress invalidPointerCast // GHT rows are 8-byte aligned and off_sum/s*8 are multiples of 8 *(double*)(row + off) += sv_f; + } else if (need_sum128) { + int64_t sv_i, sv_hi; + memcpy(&sv_i, src_row + off, 8); + memcpy(&sv_hi, src_row + off_sum_hi + (size_t)s * 8, 8); + ray_i128_add128((int64_t*)(row + off_sum_hi + (size_t)s * 8), + (uint64_t*)(row + off), sv_hi, (uint64_t)sv_i); } else { int64_t sv_i; memcpy(&sv_i, src_row + off, 8); - *(int64_t*)(row + off) += sv_i; + *(int64_t*)(row + off) = wrap_add_i64(*(int64_t*)(row + off), sv_i); } } } @@ -7107,7 +7161,7 @@ static void radix_v2_phase1_fn(void* ctx, uint32_t worker_id, * NOTE: "hashing is identical" is load-bearing and is enforced by the * v2_packed branch in the staging loop — see the comment there. */ if (!wide_any && !inline_str && !nullable && ly->null_words == 0 && - (ly->need_flags & ~(uint32_t)GHT_NEED_SUM) == 0 && + (ly->need_flags & ~(uint32_t)(GHT_NEED_SUM | GHT_NEED_SUM128)) == 0 && !(ly->agg_flags_any & (GHT_AF_FIRST | GHT_AF_LAST | GHT_AF_BINARY | GHT_AF_HOLISTIC)) && nk >= 1 && nk <= 2 && ly->entry_stride <= 64) { @@ -7515,6 +7569,11 @@ typedef union { double f; int64_t i; } da_val_t; typedef struct { da_val_t* sum; /* SUM/AVG/FIRST/LAST [n_slots * n_aggs] */ + int64_t* sum_hi; /* high words of the exact 128-bit integer sums + * [n_slots * n_aggs]; allocated only when an + * integer AVG is present (sum[].i is then the low + * word — still the wrapped int64 SUM), NULL + * otherwise so SUM-only queries pay nothing. */ da_val_t* min_val; /* MIN [n_slots * n_aggs] */ da_val_t* max_val; /* MAX [n_slots * n_aggs] */ double* sumsq_f64; /* sum-of-squares for STDDEV/VAR */ @@ -7535,6 +7594,7 @@ typedef struct { double* sumxy; /* Σxy */ /* Arena headers */ ray_t* _h_sum; + ray_t* _h_sum_hi; ray_t* _h_min; ray_t* _h_max; ray_t* _h_sumsq; @@ -7549,6 +7609,7 @@ typedef struct { static inline void da_accum_free(da_accum_t* a) { scratch_free(a->_h_sum); + scratch_free(a->_h_sum_hi); scratch_free(a->_h_min); scratch_free(a->_h_max); scratch_free(a->_h_sumsq); @@ -7566,11 +7627,18 @@ static inline void da_accum_free(da_accum_t* a) { * non-NULL) carries the per-(group, agg) non-null row count: AVG/VAR/ * STDDEV use it as the divisor and MIN/MAX/PROD/FIRST/LAST emit a typed * null when it is zero. Pass NULL to keep the legacy count[gid]-divisor - * behaviour (callers without HAS_NULLS aggs need not allocate it). */ + * behaviour (callers without HAS_NULLS aggs need not allocate it). + * sum_hi64 (if non-NULL) carries the high words of the exact 128-bit + * integer sums: an integer AVG then divides ray_i128_to_f64(hi, lo), the + * same bits as the keyless reduction and the chunk-zone metadata. A + * caller passing NULL must not emit an integer AVG. The strlen fusion + * keeps the plain wrapped add (a sum of lengths never leaves int64), so a + * high word of 0 is exact there. */ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* ext, ray_t* const* agg_vecs, uint32_t grp_count, uint32_t n_aggs, const double* sum_f64, const int64_t* sum_i64, + const int64_t* sum_hi64, const double* min_f64, const double* max_f64, const int64_t* min_i64, const int64_t* max_i64, const int64_t* counts, @@ -7657,7 +7725,9 @@ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* break; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } - v = is_f64 ? sum_f64[idx] / nn : (double)sum_i64[idx] / nn; + v = is_f64 ? sum_f64[idx] / nn + : sum_hi64 ? ray_i128_to_f64(sum_hi64[idx], (uint64_t)sum_i64[idx]) / nn + : (double)sum_i64[idx] / nn; if (affine && affine[a].enabled) v += affine[a].bias_f64; break; @@ -7872,12 +7942,14 @@ typedef struct { int64_t* keys; int64_t* counts; da_val_t* sums; + int64_t* sums_hi; /* high words of the 128-bit integer sums (integer AVG) */ uint32_t cap; uint32_t size; ray_t* _h_used; ray_t* _h_keys; ray_t* _h_counts; ray_t* _h_sums; + ray_t* _h_sums_hi; } sparse_i64_ht_t; static inline uint64_t sparse_i64_mix(uint64_t x) { @@ -7906,6 +7978,7 @@ static void sparse_i64_free(sparse_i64_ht_t* ht) { scratch_free(ht->_h_keys); scratch_free(ht->_h_counts); scratch_free(ht->_h_sums); + scratch_free(ht->_h_sums_hi); memset(ht, 0, sizeof(*ht)); } @@ -8025,7 +8098,7 @@ static int64_t da_count_emit_keep_min_u32(const uint32_t* counts, } static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, - bool need_sum) { + bool need_sum, bool need_sum128) { memset(ht, 0, sizeof(*ht)); if (cap < 1024) cap = 1024; cap = sparse_i64_pow2(cap); @@ -8037,7 +8110,12 @@ static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, ht->sums = (da_val_t*)scratch_calloc(&ht->_h_sums, (size_t)cap * n_aggs * sizeof(da_val_t)); } - if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums)) { + if (need_sum && need_sum128) { + ht->sums_hi = (int64_t*)scratch_calloc(&ht->_h_sums_hi, + (size_t)cap * n_aggs * sizeof(int64_t)); + } + if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums) || + (need_sum && need_sum128 && !ht->sums_hi)) { sparse_i64_free(ht); return false; } @@ -8057,7 +8135,7 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, bool need_sum) { sparse_i64_ht_t old = *ht; sparse_i64_ht_t nw; - if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum)) + if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum, old.sums_hi != NULL)) return false; for (uint32_t i = 0; i < old.cap; i++) { if (!old.used[i]) continue; @@ -8068,6 +8146,9 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, if (need_sum) memcpy(&nw.sums[(size_t)s * n_aggs], &old.sums[(size_t)i * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (nw.sums_hi) + memcpy(&nw.sums_hi[(size_t)s * n_aggs], &old.sums_hi[(size_t)i * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); nw.size++; } sparse_i64_free(&old); @@ -8089,6 +8170,9 @@ static bool sparse_i64_touch(sparse_i64_ht_t* ht, int64_t key, uint8_t n_aggs, if (need_sum) memset(&ht->sums[(size_t)s * n_aggs], 0, (size_t)n_aggs * sizeof(da_val_t)); + if (ht->sums_hi) + memset(&ht->sums_hi[(size_t)s * n_aggs], 0, + (size_t)n_aggs * sizeof(int64_t)); ht->size++; } *out_slot = s; @@ -8392,16 +8476,67 @@ static inline int64_t scalar_i64_at(const void* ptr, int8_t type, int64_t r) { return read_col_i64(ptr, r, type, 0); /* attrs=0: agg columns are numeric, never SYM */ } +/* Exact 128-bit sum of a contiguous int64 run, kept vectorizable. + * Blocks of 1024 rows, three forms, escalating once for the rest of the + * run when a block fails a check (a column's values are alike): + * 0. non-negative values below 2^53 (ids, counts, timestamps): one OR of + * the raw values next to the wrapped sum proves every value is in + * [0, 2^53), so the block sum cannot have overflowed int64; + * 1. values in [-2^53, 2^53): the same with the values biased by 2^53 + * (one extra add); + * 2. anything: two plain running sums, the wrapped total and the sum of + * the (arithmetically shifted) high 32-bit halves; with n < 2^32 the + * low halves' sum is below 2^64, so it is recovered exactly from the + * wrapped total, and the block's value is sh * 2^32 + low. + * A failed check re-reads only that block (8 KiB, L1-hot). Same bits as + * ray_i128_add per element (idxop.h) at a fraction of its cost. */ +static inline void group_i128_sum_i64_range(const int64_t* restrict x, int64_t n, + int64_t* hi, uint64_t* lo) { + const int64_t blk = 1024; + const uint64_t bias = (uint64_t)1 << 53; + int mode = 0; + for (int64_t b = 0; b < n; b += blk) { + int64_t e = (n - b > blk) ? b + blk : n; + if (mode == 0) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u; } + if ((o >> 53) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 1; + } + if (mode == 1) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u + bias; } + if ((o >> 54) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 2; + } + uint64_t sl = 0; int64_t sh = 0; + for (int64_t r = b; r < e; r++) { + int64_t v = x[r]; + sl += (uint64_t)v; + sh += v >> 32; + } + uint64_t low = sl - ((uint64_t)sh << 32); /* sum of the low halves */ + ray_i128_add128(hi, lo, sh >> 32, (uint64_t)sh << 32); + ray_i128_add128(hi, lo, 0, low); + } +} + /* Tight SIMD-friendly loop for single SUM/AVG on i64 (no mask). - * Note: int64 sum can overflow; caller responsibility to use appropriate types. */ + * Note: the int64 sum wraps (SUM's contract); an integer AVG also carries + * the high word (acc->sum_hi) so it divides the exact 128-bit total. */ static void scalar_sum_i64_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { scalar_ctx_t* c = (scalar_ctx_t*)ctx; da_accum_t* acc = &c->accums[worker_id]; const int64_t* restrict data = (const int64_t*)c->agg_ptrs[0]; - int64_t sum = 0; - for (int64_t r = start; r < end; r++) - sum = wrap_add_i64(sum, data[r]); - acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + if (acc->sum_hi) { + group_i128_sum_i64_range(data + start, end - start, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else { + int64_t sum = 0; + for (int64_t r = start; r < end; r++) + sum = wrap_add_i64(sum, data[r]); + acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + } acc->count[0] += end - start; } @@ -8424,6 +8559,41 @@ static void scalar_sum_linear_i64_fn(void* ctx, uint32_t worker_id, int64_t star const agg_linear_t* lin = &c->agg_linear[0]; int64_t n = end - start; + if (acc->sum_hi) { + /* Integer AVG needs the exact 128-bit total. A bare column (one + * term, coefficient 1, no bias — the common keyless avg) streams + * the vectorized block sum; a narrow column's block total cannot + * leave int64 at all. Any other linear shape walks the rows: the + * per-row value wraps like the materialized expression would and + * the 128-bit total of those rows is exact. */ + bool bare = lin->n_terms == 1 && lin->coeff_i64[0] == 1 && + lin->bias_i64 == 0 && lin->term_ptrs[0]; + int8_t t0 = lin->term_types[0]; + if (bare && (t0 == RAY_I64 || t0 == RAY_TIMESTAMP)) { + group_i128_sum_i64_range((const int64_t*)lin->term_ptrs[0] + start, n, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else if (bare && t0 != RAY_SYM) { + /* <= 32-bit values: 2^30 of them stay far inside int64 */ + const int64_t blk = (int64_t)1 << 30; + for (int64_t b = start; b < end; b += blk) { + int64_t e = (end - b > blk) ? b + blk : end; + int64_t sum = 0; + for (int64_t r = b; r < e; r++) + sum += scalar_i64_at(lin->term_ptrs[0], t0, r); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, sum); + } + } else { + for (int64_t r = start; r < end; r++) { + int64_t iv = lin->bias_i64; + for (uint8_t t = 0; t < lin->n_terms; t++) + iv = wrap_add_i64(iv, wrap_mul_i64(lin->coeff_i64[t], + scalar_i64_at(lin->term_ptrs[t], lin->term_types[t], r))); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, iv); + } + } + acc->count[0] += n; + return; + } int64_t sum = wrap_mul_i64(lin->bias_i64, n); /* n_terms is bounded by AGG_LINEAR_MAX_TERMS (8, internal.h) — an * unrelated fixed cap on linear-expression arity, not a GROUP n_keys/ @@ -8500,7 +8670,8 @@ static inline void scalar_accum_row(scalar_ctx_t* c, da_accum_t* acc, int64_t r) if (nn) nn[a]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[a], (uint64_t*)&acc->sum[a].i, iv); + else acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); if (acc->sumsq_f64) acc->sumsq_f64[a] += fv * fv; if (nn) nn[a]++; } @@ -8654,7 +8825,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 uint8_t v_attrs = c->agg_cols[a] ? c->agg_cols[a]->attrs : 0; int64_t v = read_col_i64(c->agg_ptrs[a], r, c->agg_types[a], v_attrs); if (RAY_LIKELY(!((inm >> a) & 1) || v != c->agg_int_null_sentinel[a])) { - acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, v); + else acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); if (nn) nn[idx]++; } } @@ -8718,7 +8890,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 if (nn) nn[idx]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, iv); + else acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); if (acc->sumsq_f64) acc->sumsq_f64[idx] += fv * fv; if (nn) nn[idx]++; } @@ -9109,6 +9282,9 @@ static void da_merge_fn(void* ctx, uint32_t wid, int64_t start, int64_t end) { } } else if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -10252,6 +10428,8 @@ typedef struct { const uint16_t* agg_ops; uint8_t n_aggs; da_val_t* partials; /* [n_tasks * n_aggs], zeroed */ + int64_t* partials_hi; /* high words of the integer partials + * (NULL unless an integer AVG) */ double* partial_sumsq; double* partial_sum_y; double* partial_sumsq_y; @@ -10291,7 +10469,7 @@ static inline double sg_prod_range(const agg_prod_t* p, int64_t r0, int64_t n, a2 += x[j + 2] * y[j + 2]; a3 += x[j + 3] * y[j + 3]; } for (; j < n; j++) a0 += x[j] * y[j]; - } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIME)) { + } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIMESTAMP)) { const double* restrict x = (const double*)pa + r0; const int64_t* restrict y = (const int64_t*)pb + r0; uint64_t s0 = 0; @@ -10458,32 +10636,45 @@ static void sg_accum_fn(void* raw, uint32_t wid, int64_t tstart, int64_t tend) { int8_t t = av->type; uint8_t at = av->attrs; uint64_t acc = 0; + int64_t hi = 0; /* 128-bit high word (integer AVG) */ double ssq = 0.0; bool need_sq = c->partial_sumsq && (op == OP_STDDEV || op == OP_STDDEV_POP || op == OP_VAR || op == OP_VAR_POP); - if (contig && (t == RAY_I64 || t == RAY_TIME)) { + if (contig && (t == RAY_I64 || t == RAY_TIMESTAMP)) { const int64_t* restrict x = (const int64_t*)p + r0; - for (int64_t j = 0; j < n; j++) { - int64_t v = x[j]; - acc += (uint64_t)v; - if (need_sq) { double d = (double)v; ssq += d * d; } + if (c->partials_hi) { + /* exact 128-bit total, still a vectorized stream */ + group_i128_sum_i64_range(x, n, &hi, &acc); + if (need_sq) + for (int64_t j = 0; j < n; j++) { double d = (double)x[j]; ssq += d * d; } + } else { + for (int64_t j = 0; j < n; j++) { + int64_t v = x[j]; + acc += (uint64_t)v; + if (need_sq) { double d = (double)v; ssq += d * d; } + } } - } else if (contig && t == RAY_I32) { + } else if (contig && (t == RAY_I32 || t == RAY_DATE || t == RAY_TIME)) { + /* the 4-byte family: I32 and the day / millisecond temporals */ const int32_t* restrict x = (const int32_t*)p + r0; for (int64_t j = 0; j < n; j++) { int64_t v = (int64_t)x[j]; acc += (uint64_t)v; if (need_sq) { double d = (double)v; ssq += d * d; } } + /* n <= SG_CHUNK_ROWS 32-bit values: the int64 sum is + * exact, its high word is the sign. */ + hi = (int64_t)acc < 0 ? -1 : 0; } else { for (int64_t j = 0; j < n; j++) { int64_t v = read_col_i64(p, rows[j], t, at); - acc += (uint64_t)v; + ray_i128_add(&hi, &acc, v); if (need_sq) { double d = (double)v; ssq += d * d; } } } c->partials[idx].i = (int64_t)acc; + if (c->partials_hi) c->partials_hi[idx] = hi; if (need_sq) c->partial_sumsq[idx] = ssq; } } @@ -10652,7 +10843,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const ray_idx_slice_t* slices = (K > 0) ? (const ray_idx_slice_t*)ray_data(g->sg_slices_hdr) : NULL; - bool need_sumsq = false, need_pair = false; + bool need_sumsq = false, need_pair = false, need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { uint16_t aop = ext->agg_ops[a]; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || @@ -10661,12 +10852,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, need_sumsq = true; if (agg_is_binary_agg(aop)) need_pair = true; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + if (aop == OP_AVG && !prod[a].enabled && agg_vecs[a] && + !group_fp_type(agg_vecs[a]->type) && + group_avg_needs_hi(agg_vecs[a]->type, ray_table_nrows(tbl))) + need_sum128 = true; } ray_t *sum_hdr = NULL, *cnt_hdr = NULL, *task_hdr = NULL, *part_hdr = NULL; ray_t *sumsq_hdr = NULL, *sum_y_hdr = NULL, *sumsq_y_hdr = NULL, *sumxy_hdr = NULL; ray_t *part_sumsq_hdr = NULL, *part_sum_y_hdr = NULL, *part_sumsq_y_hdr = NULL, *part_sumxy_hdr = NULL; + ray_t *sum_hi_hdr = NULL, *part_hi_hdr = NULL; da_val_t* sums = NULL; + int64_t* sums_hi = NULL; double *sumsq = NULL, *sum_y = NULL, *sumsq_y = NULL, *sumxy = NULL; int64_t* counts = NULL; if (K > 0) { @@ -10674,6 +10872,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, (size_t)K * n_aggs * sizeof(da_val_t)); counts = (int64_t*)scratch_calloc(&cnt_hdr, (size_t)K * sizeof(int64_t)); + if (need_sum128) + sums_hi = (int64_t*)scratch_calloc(&sum_hi_hdr, + (size_t)K * n_aggs * sizeof(int64_t)); if (need_sumsq) sumsq = (double*)scratch_calloc(&sumsq_hdr, (size_t)K * n_aggs * sizeof(double)); @@ -10692,12 +10893,16 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, n_tasks += (slices[i].n + SG_CHUNK_ROWS - 1) / SG_CHUNK_ROWS; sg_task_t* tasks = NULL; da_val_t* partials = NULL; + int64_t* part_hi = NULL; double *part_sumsq = NULL, *part_sum_y = NULL, *part_sumsq_y = NULL, *part_sumxy = NULL; if (sums && counts) { task_hdr = ray_alloc((size_t)n_tasks * (int64_t)sizeof(sg_task_t)); tasks = task_hdr ? (sg_task_t*)ray_data(task_hdr) : NULL; partials = (da_val_t*)scratch_calloc(&part_hdr, (size_t)n_tasks * n_aggs * sizeof(da_val_t)); + if (need_sum128) + part_hi = (int64_t*)scratch_calloc(&part_hi_hdr, + (size_t)n_tasks * n_aggs * sizeof(int64_t)); if (need_sumsq) part_sumsq = (double*)scratch_calloc(&part_sumsq_hdr, (size_t)n_tasks * n_aggs * sizeof(double)); @@ -10711,12 +10916,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } if (!sums || !counts || !tasks || !partials || + (need_sum128 && (!sums_hi || !part_hi)) || (need_sumsq && (!sumsq || !part_sumsq)) || (need_pair && (!sum_y || !sumsq_y || !sumxy || !part_sum_y || !part_sumsq_y || !part_sumxy))) { scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); scratch_free(part_hi_hdr); if (task_hdr) ray_free(task_hdr); scratch_free(part_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); @@ -10733,7 +10940,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } sg_ctx_t ctx = { slices, tasks, agg_vecs, agg_vecs2, prod, ext->agg_ops, - n_aggs, partials, part_sumsq, part_sum_y, + n_aggs, partials, part_hi, part_sumsq, part_sum_y, part_sumsq_y, part_sumxy, {0}, {0} }; /* Shared-stream pairing: a bare-scan SUM/AVG over the same column * a product's int side already streams rides the product loop — @@ -10748,10 +10955,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (prod[a].ta != RAY_F64) { ip = prod[a].pa; it = prod[a].ta; } else if (prod[a].tb != RAY_F64) { ip = prod[a].pb; it = prod[a].tb; } else continue; /* F64×F64 — no int side */ - if (it != RAY_I64 && it != RAY_TIME && it != RAY_I32) continue; + if (it != RAY_I64 && it != RAY_I32) continue; for (uint32_t b = 0; b < n_aggs; b++) { if (b == a || !agg_vecs[b] || ctx.fused_by[b] >= 0) continue; if (ext->agg_ops[b] != OP_SUM && ext->agg_ops[b] != OP_AVG) continue; + /* The product loop yields only the wrapped int64 side sum; + * an integer AVG needs the 128-bit total, so it streams + * its own column. */ + if (ext->agg_ops[b] == OP_AVG && part_hi) continue; if (ray_data(agg_vecs[b]) != ip || agg_vecs[b]->type != it) continue; ctx.pair_sum[a] = (int8_t)b; ctx.fused_by[b] = (int8_t)a; @@ -10778,6 +10989,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (pair || prod[a].enabled || ext->agg_ops[a] == OP_COUNT || (agg_vecs[a] && group_fp_type(agg_vecs[a]->type))) sums[di].f += partials[si].f; + else if (sums_hi) + ray_i128_add128(&sums_hi[di], (uint64_t*)&sums[di].i, + part_hi[si], (uint64_t)partials[si].i); else sums[di].i = (int64_t)((uint64_t)sums[di].i + (uint64_t)partials[si].i); @@ -10790,7 +11004,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } ray_free(task_hdr); - scratch_free(part_hdr); + scratch_free(part_hdr); scratch_free(part_hi_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); scratch_free(part_sumsq_y_hdr); scratch_free(part_sumxy_hdr); } @@ -10803,6 +11017,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } ray_t* kc = col_vec_new(key_col, K > 0 ? K : 1); @@ -10812,6 +11027,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } if (kc->type == RAY_SYM) @@ -10826,17 +11042,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } emit_agg_columns(&result, g, ext, agg_vecs, (uint32_t)K, n_aggs, - (double*)sums, (int64_t*)sums, + (double*)sums, (int64_t*)sums, sums_hi, NULL, NULL, NULL, NULL, counts, NULL, prod, sumsq, NULL, sum_y, sumsq_y, sumxy); scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } @@ -10996,6 +11214,7 @@ typedef struct { int64_t n_scan; uint8_t key_esz; bool sp_need_sum; + bool sp_need_sum128; /* an integer AVG: carry high words */ const int64_t* match_idx; ray_t* rowsel; ray_t* match_idx_block; @@ -11066,6 +11285,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { int64_t n_scan = c->n_scan; uint8_t key_esz = c->key_esz; bool sp_need_sum = c->sp_need_sum; + bool sp_need_sum128 = c->sp_need_sum128; const int64_t* match_idx = c->match_idx; ray_t* rowsel = c->rowsel; ray_t* match_idx_block = c->match_idx_block; @@ -11077,16 +11297,23 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { : (1u << 20); const uint64_t max_dense_cap = 1u << 24; bool count_only_first = (key_types[0] == RAY_SYM); - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)cap * sizeof(uint32_t)); da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ bool dyn_ok = range_count != NULL; if (dyn_ok && sp_need_sum && !count_only_first) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, (size_t)cap * n_aggs * sizeof(da_val_t)); dyn_ok = range_sum != NULL; + if (dyn_ok && sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)cap * n_aggs * sizeof(int64_t)); + dyn_ok = range_sum_hi != NULL; + } } uint64_t max_seen = 0; @@ -11203,6 +11430,19 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memset(range_sum + (size_t)old_cap * n_aggs, 0, \ (size_t)(cap - old_cap) * n_aggs * sizeof(da_val_t)); \ } \ + if (range_sum_hi) { \ + int64_t* new_hi = (int64_t*)scratch_realloc( \ + &range_hi_hdr, \ + (size_t)old_cap * n_aggs * sizeof(int64_t), \ + (size_t)cap * n_aggs * sizeof(int64_t)); \ + if (!new_hi) { \ + dyn_ok = false; \ + goto dyn_dense_done; \ + } \ + range_sum_hi = new_hi; \ + memset(range_sum_hi + (size_t)old_cap * n_aggs, 0, \ + (size_t)(cap - old_cap) * n_aggs * sizeof(int64_t)); \ + } \ } \ have_dyn_key = true; \ if (off > max_seen) max_seen = off; \ @@ -11218,6 +11458,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (range_sum_hi) \ + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11272,7 +11516,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11287,7 +11531,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11298,16 +11542,21 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11316,9 +11565,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (sp_need_sum && !range_sum) + if (sp_need_sum && !range_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, + (size_t)grp_count * n_aggs * sizeof(int64_t)); + } uint32_t gi = 0; for (uint64_t off = 0; off <= max_seen; off++) { @@ -11334,6 +11587,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } if (!range_sum) range_count[off] = gi + 1u; gi++; @@ -11357,6 +11614,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (dense_sum_hi) \ + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11402,13 +11663,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { * row is non-null and the legacy count-based divisor is * correct. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11417,7 +11678,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { return result; } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); /* Dynamic-dense probe bailed (unbounded key or no surviving row): shared @@ -11966,6 +12227,8 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, * once n_aggs exceeded its bit width. */ bool *sc_int_null_has = (bool*)(agg_types + vla_aggs); bool sc_any_nullable = false; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool sc_need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { if (agg_prod[a].enabled) { /* Fused product: F64 accumulate, no source vec. */ @@ -11996,6 +12259,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, sc_int_null_sentinel[a] = 0; sc_int_null_has[a] = false; } + if (ext->agg_ops[a] == OP_AVG && !group_fp_type(agg_types[a]) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + sc_need_sum128 = true; } if (!match_idx && !rowsel && !sc_any_nullable && n_aggs > 1) { @@ -12051,8 +12317,21 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } + /* An integer AVG needs the exact 128-bit total: take it from + * the chunk-zone metadata when the column carries it (the + * same (hi, lo) ray_avg_fn reads), otherwise skip the shortcut + * and let the parallel scan below carry the high word. */ + int64_t zone_hi = 0, zone_nn = 0; uint64_t zone_lo = 0; + bool zone_128 = false; + if (one_base_input && base_col && sc_need_sum128 && + !group_fp_type(base_type)) { + zone_128 = ray_zone_int_sum128(base_col, &zone_hi, &zone_lo, &zone_nn) + && zone_nn == nrows; + if (!zone_128) one_base_input = false; + } if (one_base_input && base_col) { - ray_t* base_sum_obj = ray_sum_fn(base_col); + ray_t* base_sum_obj = zone_128 ? ray_i64((int64_t)zone_lo) + : ray_sum_fn(base_col); if (!base_sum_obj || RAY_IS_ERR(base_sum_obj)) { scratch_free(sc_vla_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -12084,12 +12363,14 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, ray_t *sum_hdr = NULL, *cnt_hdr = NULL; double* sums_f64 = NULL; int64_t* sums_i64 = NULL; + int64_t* sums_hi = NULL; if (base_is_f64) sums_f64 = (double*)scratch_alloc(&sum_hdr, (size_t)n_aggs * sizeof(double)); else + /* [sums | high words] in one carve */ sums_i64 = (int64_t*)scratch_alloc(&sum_hdr, - (size_t)n_aggs * sizeof(int64_t)); + (size_t)2 * n_aggs * sizeof(int64_t)); int64_t* counts = (int64_t*)scratch_alloc(&cnt_hdr, sizeof(int64_t)); if ((base_is_f64 ? (sums_f64 != NULL) : (sums_i64 != NULL)) && @@ -12100,6 +12381,11 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { for (uint32_t a = 0; a < n_aggs; a++) sums_i64[a] = base_sum_i64; + if (zone_128) { + sums_hi = sums_i64 + n_aggs; + for (uint32_t a = 0; a < n_aggs; a++) + sums_hi[a] = zone_hi; + } } counts[0] = nrows; @@ -12119,7 +12405,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - sums_f64, sums_i64, + sums_f64, sums_i64, sums_hi, NULL, NULL, NULL, NULL, counts, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); @@ -12172,6 +12458,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const size_t sc_line = 64; size_t sc_words = 1; /* count[1] */ if (need_flags & DA_NEED_SUM) sc_words += n_aggs; /* sum */ + if (sc_need_sum128) sc_words += n_aggs; /* sum_hi */ if (need_flags & DA_NEED_MIN) sc_words += n_aggs; /* min_val */ if (need_flags & DA_NEED_MAX) sc_words += n_aggs; /* max_val */ if (need_flags & DA_NEED_SUMSQ) sc_words += n_aggs; /* sumsq_f64 */ @@ -12188,6 +12475,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (need_flags & DA_NEED_SUM) { sc_acc[w].sum = (da_val_t*)(void*)(blk + off); off += n_aggs; } + if (sc_need_sum128) { + sc_acc[w].sum_hi = blk + off; off += n_aggs; + } if (need_flags & DA_NEED_MIN) { sc_acc[w].min_val = (da_val_t*)(void*)(blk + off); off += n_aggs; for (uint32_t a = 0; a < n_aggs; a++) { @@ -12298,6 +12588,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { if (group_fp_type(agg_types[a])) m->sum[a].f += wa->sum[a].f; + else if (m->sum_hi) + ray_i128_add128(&m->sum_hi[a], (uint64_t*)&m->sum[a].i, + wa->sum_hi[a], (uint64_t)wa->sum[a].i); else m->sum[a].i = wrap_add_i64(m->sum[a].i, wa->sum[a].i); } @@ -12363,7 +12656,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - (double*)m->sum, (int64_t*)m->sum, + (double*)m->sum, (int64_t*)m->sum, m->sum_hi, (double*)m->min_val, (double*)m->max_val, (int64_t*)m->min_val, (int64_t*)m->max_val, m->count, agg_affine, agg_prod, m->sumsq_f64, m->nn_count, @@ -12727,6 +13020,8 @@ da_path:; int64_t da_int_null_sentinel[vla_aggs]; uint64_t agg_f64_mask = 0; uint64_t da_int_null_mask = 0; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool da_need_sum128 = false; /* Track whether any agg column can produce a null so we can * allocate per-(group, agg) non-null counts only when required. * F64 with HAS_NULLS uses NaN-skip; sentinel-typed integers @@ -12778,6 +13073,9 @@ da_path:; agg_types[a] = 0; da_int_null_sentinel[a] = 0; } + if (ext->agg_ops[a] == OP_AVG && !(agg_f64_mask & ((uint64_t)1 << a)) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + da_need_sum128 = true; } ray_pool_t* da_pool = ray_pool_get(); @@ -12788,6 +13086,7 @@ da_path:; * cells must remain O(contributing rows). */ uint32_t arrays_per_agg = 0; if (need_flags & DA_NEED_SUM) arrays_per_agg += 1; + if (da_need_sum128) arrays_per_agg += 1; /* sum_hi */ if (need_flags & DA_NEED_MIN) arrays_per_agg += 1; if (need_flags & DA_NEED_MAX) arrays_per_agg += 1; if (need_flags & DA_NEED_SUMSQ) arrays_per_agg += 1; @@ -12839,6 +13138,11 @@ da_path:; total * sizeof(da_val_t)); if (!accums[w].sum) { alloc_ok = false; break; } } + if (da_need_sum128) { + accums[w].sum_hi = (int64_t*)scratch_calloc(&accums[w]._h_sum_hi, + total * sizeof(int64_t)); + if (!accums[w].sum_hi) { alloc_ok = false; break; } + } if (need_flags & DA_NEED_SUMSQ) { accums[w].sumsq_f64 = (double*)scratch_calloc(&accums[w]._h_sumsq, total * sizeof(double)); @@ -13020,6 +13324,7 @@ da_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP || agg_is_binary_agg(aop)) { if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } else if (aop == OP_PROD) { /* Use per-(group, agg) non-null counts when @@ -13146,6 +13451,9 @@ da_path:; /* binary aggs accumulate Σx as double even * for integer x-columns — merge as float. */ merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -13206,6 +13514,7 @@ da_path:; da_accum_free(&accums[w]); da_val_t* da_sum = merged->sum; /* may be NULL if !DA_NEED_SUM */ + int64_t* da_sum_hi = merged->sum_hi; /* NULL unless an integer AVG */ da_val_t* da_min_val = merged->min_val; /* may be NULL if !DA_NEED_MIN */ da_val_t* da_max_val = merged->max_val; /* may be NULL if !DA_NEED_MAX */ double* da_sumsq = merged->sumsq_f64; /* may be NULL if !DA_NEED_SUMSQ */ @@ -13271,8 +13580,9 @@ da_path:; size_t dense_total = (size_t)grp_count * n_aggs; ray_t *_h_dsum = NULL, *_h_dmin = NULL, *_h_dmax = NULL; ray_t *_h_dsq = NULL, *_h_dcnt = NULL, *_h_dnn = NULL; - ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL; + ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL, *_h_dshi = NULL; da_val_t* dense_sum = da_sum ? (da_val_t*)scratch_alloc(&_h_dsum, dense_total * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = da_sum_hi ? (int64_t*)scratch_alloc(&_h_dshi, dense_total * sizeof(int64_t)) : NULL; da_val_t* dense_min_val = da_min_val ? (da_val_t*)scratch_alloc(&_h_dmin, dense_total * sizeof(da_val_t)) : NULL; da_val_t* dense_max_val = da_max_val ? (da_val_t*)scratch_alloc(&_h_dmax, dense_total * sizeof(da_val_t)) : NULL; double* dense_sumsq = da_sumsq ? (double*)scratch_alloc(&_h_dsq, dense_total * sizeof(double)) : NULL; @@ -13294,6 +13604,7 @@ da_path:; size_t si = (size_t)s * n_aggs + a; size_t di = (size_t)gi * n_aggs + a; if (dense_sum) dense_sum[di] = da_sum[si]; + if (dense_sum_hi) dense_sum_hi[di] = da_sum_hi[si]; if (dense_min_val) dense_min_val[di] = da_min_val[si]; if (dense_max_val) dense_max_val[di] = da_max_val[si]; if (dense_sumsq) dense_sumsq[di] = da_sumsq[si]; @@ -13306,7 +13617,7 @@ da_path:; } emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, (double*)dense_min_val, (double*)dense_max_val, (int64_t*)dense_min_val, (int64_t*)dense_max_val, dense_counts, agg_affine, agg_prod, dense_sumsq, @@ -13317,6 +13628,7 @@ da_path:; scratch_free(_h_dsq); scratch_free(_h_dcnt); scratch_free(_h_dnn); scratch_free(_h_dsy); scratch_free(_h_dsqy); scratch_free(_h_dxy); + scratch_free(_h_dshi); da_accum_free(&accums[0]); scratch_free(accums_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -13346,6 +13658,9 @@ da_path:; sp_eligible = false; } bool sp_need_sum = false; + /* An integer AVG divides the exact 128-bit sum: the scatter arrays + * and the sparse table then carry a high word per slot. */ + bool sp_need_sum128 = false; for (uint32_t a = 0; a < n_aggs && sp_eligible; a++) { uint16_t op = ext->agg_ops[a]; if (op == OP_COUNT) continue; @@ -13360,8 +13675,13 @@ da_path:; * accum_from_entry inherits the same nullable-agg gap.) */ if (agg_vecs[a] && ray_vec_may_have_nulls(agg_vecs[a])) sp_eligible = false; - else + else { sp_need_sum = true; + if (op == OP_AVG && !(agg_vecs[a] && group_fp_type(agg_vecs[a]->type)) && + !agg_strlen[a] && + group_avg_needs_hi(agg_vecs[a] ? agg_vecs[a]->type : 0, nrows)) + sp_need_sum128 = true; + } } } @@ -13441,7 +13761,8 @@ da_path:; .strlen_sym_count = strlen_sym_count, .agg_f64_mask = agg_f64_mask, .n_aggs = n_aggs, .n_keys = n_keys, .n_scan = n_scan, .key_esz = key_esz, - .sp_need_sum = sp_need_sum, .match_idx = match_idx, + .sp_need_sum = sp_need_sum, .sp_need_sum128 = sp_need_sum128, + .match_idx = match_idx, .rowsel = rowsel, .match_idx_block = match_idx_block, .vla_hdr = vla_hdr, .emit_filter = emit_filter, }; @@ -13473,13 +13794,14 @@ da_path:; ? (uint64_t)((uint64_t)max_key - (uint64_t)min_key + 1u) : 0u; if (have_key && key_range > 0 && key_range <= (1u << 26)) { - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)key_range * sizeof(uint32_t)); if (!range_count) goto ht_path; da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ if (sp_need_sum && key_range <= (1u << 24)) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, @@ -13488,6 +13810,16 @@ da_path:; scratch_free(cnt_hdr); goto ht_path; } + if (sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)key_range * n_aggs * sizeof(int64_t)); + if (!range_sum_hi) { + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); + scratch_free(cnt_hdr); + goto ht_path; + } + } } for (int64_t i = 0; i < n_scan; i++) { @@ -13511,6 +13843,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (range_sum_hi) + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13534,7 +13870,7 @@ da_path:; ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13550,7 +13886,7 @@ da_path:; /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13566,11 +13902,16 @@ da_path:; ? (da_val_t*)scratch_calloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_calloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13596,6 +13937,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } range_count[off] = gi + 1u; gi++; @@ -13622,6 +13967,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13630,7 +13979,7 @@ da_path:; } } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_op_ext_t* key_ext = find_ext(g, ext->keys[0]); int64_t name_id = key_ext ? key_ext->sym : 0; @@ -13641,12 +13990,12 @@ da_path:; * emit-filter range path only runs when sp_eligible was * true. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13668,7 +14017,7 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false, false)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13687,7 +14036,8 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum, + sp_need_sum128)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13714,6 +14064,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (sp_ht.sums_hi) + ray_i128_add(&sp_ht.sums_hi[(size_t)slot * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13769,15 +14123,20 @@ da_path:; } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc(&_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13787,14 +14146,17 @@ da_path:; if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (use_emit_filter && sp_need_sum) + if (use_emit_filter && sp_need_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, (size_t)grp_count * n_aggs * sizeof(int64_t)); + } sparse_i64_ht_t heavy_ht; memset(&heavy_ht, 0, sizeof(heavy_ht)); if (use_emit_filter && grp_count > 0) { - if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false, false)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13816,7 +14178,7 @@ da_path:; if (use_emit_filter) { int32_t hslot; if (!sparse_i64_touch(&heavy_ht, sp_ht.keys[s], n_aggs, false, &hslot)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&heavy_ht); sparse_i64_free(&sp_ht); @@ -13832,6 +14194,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &sp_ht.sums[(size_t)s * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &sp_ht.sums_hi[(size_t)s * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } gi++; } @@ -13859,6 +14225,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)out_gi * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13876,12 +14246,12 @@ da_path:; * and is gated to null-free agg columns (sp_eligible guard at * ~line 5737), so counts[gi] is the correct divisor. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13919,6 +14289,10 @@ ht_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_PROD || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_FIRST || aop == OP_LAST) ght_need |= GHT_NEED_SUM; + /* Integer avg divides the exact 128-bit sum: carry its high words. */ + if (aop == OP_AVG && agg_vecs[a] && !group_fp_type(agg_vecs[a]->type) && + !agg_strlen[a] && group_avg_needs_hi(agg_vecs[a]->type, nrows)) + ght_need |= GHT_NEED_SUM128; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP) { ght_need |= GHT_NEED_SUM; ght_need |= GHT_NEED_SUMSQ; } if (agg_is_binary_agg(aop)) @@ -15581,7 +15955,10 @@ sequential_fallback:; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } v = is_f64 ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (agg_affine[a].enabled) v += agg_affine[a].bias_f64; break; case OP_MIN: diff --git a/src/ops/idxop.c b/src/ops/idxop.c index 873984f5..cc9f9a36 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -605,7 +605,8 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, int64_t s = (int64_t)g * csz; int64_t e = s + csz; if (e > n) e = n; int64_t mn = INT64_MAX, mx = INT64_MIN; - uint64_t sum = 0; /* wraps like the engine's int64 sum */ + uint64_t sum = 0; /* low word: wraps like the engine's int64 sum */ + int64_t hi = 0; /* high word of the exact 128-bit sum */ int64_t nn = 0; bool any_null = false; for (int64_t i = s; i < e; i++) { @@ -620,13 +621,15 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, } if (val < mn) mn = val; if (val > mx) mx = val; - sum += (uint64_t)val; + ray_i128_add(&hi, &sum, val); nn++; } if (ix->u.chunk_zone.aggs) { int64_t* ag = (int64_t*)ray_data(ix->u.chunk_zone.aggs); ag[g] = (int64_t)sum; ag[n_chunks + g] = nn; + if (ix->u.chunk_zone.aggs->len >= 3 * (int64_t)n_chunks) + ag[2 * (int64_t)n_chunks + g] = hi; } /* Empty (all-null) chunks keep mn=INT64_MAX / mx=INT64_MIN so * the reduce path's min(mins[*]) / max(maxs[*]) ignores them. */ @@ -824,9 +827,10 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2) { ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; if (!ix->u.chunk_zone.is_f64) { - ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } - aggs->len = 2 * (int64_t)n_chunks; + aggs->len = 3 * (int64_t)n_chunks; ix->u.chunk_zone.aggs = aggs; } @@ -880,9 +884,10 @@ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2) { ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; if (!ix->u.chunk_zone.is_f64) { - ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } - aggs->len = 2 * (int64_t)n_chunks; + aggs->len = 3 * (int64_t)n_chunks; ix->u.chunk_zone.aggs = aggs; } @@ -1118,9 +1123,11 @@ ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size) { } *slots[i] = c; } + /* two layouts: [lo | nn] (2 per chunk) and [lo | nn | hi] (3 per chunk) */ if (ix->kind == RAY_IDX_CHUNK_ZONE && ix->u.chunk_zone.aggs && (ix->u.chunk_zone.aggs->type != RAY_I64 || ix->u.chunk_zone.is_f64 || - ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks)) + (ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks && + ix->u.chunk_zone.aggs->len != 3 * (int64_t)ix->u.chunk_zone.n_chunks))) ix->u.chunk_zone.aggs = NULL; ix->markers |= RAY_MARK_MMAP; idx->mmod = 1; diff --git a/src/ops/idxop.h b/src/ops/idxop.h index 076a7747..609895a9 100644 --- a/src/ops/idxop.h +++ b/src/ops/idxop.h @@ -277,11 +277,40 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2); * compute an index for persistence without COWing a shared column. */ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2); +/* 128-bit two's-complement accumulation of int64 values: (hi, lo) += v. + * The engine's integer avg sums this way — exact for any column, and the + * same bits whatever the morsel split — and the chunk-zone metadata keeps + * the per-chunk (hi, lo) so it can answer the same value. */ +static inline void ray_i128_add(int64_t* hi, uint64_t* lo, int64_t v) { + uint64_t l = *lo + (uint64_t)v; + *hi += (v < 0 ? -1 : 0) + (l < (uint64_t)v ? 1 : 0); + *lo = l; +} +static inline void ray_i128_add128(int64_t* hi, uint64_t* lo, int64_t vhi, uint64_t vlo) { + uint64_t l = *lo + vlo; + *hi += vhi + (l < vlo ? 1 : 0); + *lo = l; +} +/* Double nearest the 128-bit value (hi, lo): the magnitude is converted + * (high word scaled by 2^64 plus the low word) and the sign reapplied, so + * a small negative total is not lost in 2^64 - |s|. One formula + * everywhere so every path agrees bit for bit. */ +static inline double ray_i128_to_f64(int64_t hi, uint64_t lo) { + bool neg = hi < 0; + uint64_t h = (uint64_t)hi, l = lo; + if (neg) { l = ~l + 1u; h = ~h + (l == 0 ? 1u : 0u); } + double d = (double)h * 18446744073709551616.0 + (double)l; + return neg ? -d : d; +} + /* Whole-column sum (int64 wraparound) and non-null count of an integer * column from its chunk-zone per-chunk aggregates; false when the column - * carries none for its current length. exact_f64 (optional): whether the - * sum is exactly what a double accumulation of the rows gives. */ -bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64); + * carries none for its current length. */ +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out); +/* The exact 128-bit sum (hi, lo) and non-null count from the same + * metadata: true when the per-chunk high words are stored, or when no + * partial sum can wrap int64 (then the wrapped sum is the exact one). */ +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out); /* Build a RAY_IDX_DICT (codes + distinct values) for STR vector `v` WITHOUT * attaching it — standalone RAY_INDEX object (caller releases / attaches). diff --git a/src/ops/internal.h b/src/ops/internal.h index 3a271d27..f4381f88 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -1228,6 +1228,12 @@ ray_t* desc_vec_eager(ray_t* x); /* OP_PEARSON_CORR per-group accumulators: x-side piggybacks on SUM and * SUMSQ blocks; this flag enables the y-side blocks (Σy, Σy², Σxy). */ #define GHT_NEED_PEARSON 0x10 +/* Exact integer AVG: an extra int64 block (off_sum_hi) carries the high + * word of a 128-bit two's-complement sum next to each off_sum slot, so an + * integer mean never divides a wrapped int64 (ray_i128_add / ray_i128_to_f64 + * in idxop.h). Set whenever an OP_AVG agg has a non-float input; SUM keeps + * reading off_sum alone (int64 wraparound is its contract). */ +#define GHT_NEED_SUM128 0x20 /* ── ght_layout_t — inline-or-spill, fixed-size, by-value embeddable ── * @@ -1300,6 +1306,9 @@ typedef struct { uint16_t off_sum_y; uint16_t off_sumsq_y; uint16_t off_sumxy; + /* High words of the 128-bit integer sums (GHT_NEED_SUM128); 0 when the + * layout carries none. */ + uint16_t off_sum_hi; /* Earliest contributing source row for this group. Every packed entry * carries its source row in the tail slot; partition merges retain the * minimum so output order is independent of radix partition count. */ diff --git a/src/ops/pivot.c b/src/ops/pivot.c index 1851634f..9d25bcc3 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -1607,6 +1607,12 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { uint8_t need_flags = GHT_NEED_SUM; /* always need sum (used for FIRST/LAST too) */ if (agg_op == OP_MIN) need_flags |= GHT_NEED_MIN; if (agg_op == OP_MAX) need_flags |= GHT_NEED_MAX; + /* Integer avg divides the exact 128-bit sum (high words in off_sum_hi), + * like every group engine — never a wrapped int64. */ + if (agg_op == OP_AVG && (vcol->type == RAY_I64 || vcol->type == RAY_TIMESTAMP || + (vcol->type != RAY_F64 && vcol->type != RAY_F32 && + nrows >= ((int64_t)1 << 31)))) + need_flags |= GHT_NEED_SUM128; /* n_keys/n_aggs are no longer capped: ght_compute_layout spills to an * owned heap block (ly.spill_hdr) whenever n_keys exceeds GHT_INLINE @@ -2143,7 +2149,10 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } v = val_is_f64 ? ROW_RD_F64(row, ly.off_sum, s) / nn - : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; + : (ly.need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly.off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly.off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; break; case OP_MIN: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } diff --git a/src/ops/query.c b/src/ops/query.c index f00da25a..028c53bc 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -549,7 +549,7 @@ static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t uint16_t op = resolve_agg_opcode(el[0]->i64); bool int_col = col->type == RAY_I64 || col->type == RAY_I32 || col->type == RAY_I16 || col->type == RAY_U8; - int64_t zs, zn; bool exact = false; + int64_t zs, zn, zh; uint64_t zl; switch (op) { case OP_COUNT: break; case OP_MIN: case OP_MAX: { @@ -560,10 +560,10 @@ static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t break; } case OP_SUM: - if (!int_col || !ray_zone_int_sum(col, &zs, &zn, NULL)) return NULL; + if (!int_col || !ray_zone_int_sum(col, &zs, &zn)) return NULL; break; case OP_AVG: - if (!int_col || !ray_zone_int_sum(col, &zs, &zn, &exact) || !exact) return NULL; + if (!int_col || !ray_zone_int_sum128(col, &zh, &zl, &zn)) return NULL; break; default: return NULL; } diff --git a/src/store/col.c b/src/store/col.c index c38a523e..295a1736 100644 --- a/src/store/col.c +++ b/src/store/col.c @@ -1754,9 +1754,29 @@ static ray_t* col_mmap_impl(const char* path, struct ray_sym_domain_s* dom, ray_t* r = ray_index_attach_built(&vec, idx); if (r && !RAY_IS_ERR(r)) vec = r; } - /* The munmap size is derived from the column + index at free time - * (ray_free), so no aux slot is needed — leaving str_pool intact on STR. */ /* Attach failure → column loads unindexed; correctness unaffected. */ + /* The mapping is longer than the payload by the index region, and + * ray_free can only size it from the attached index. Once the + * column drops that index — an in-place edit of the loaded column's + * only reference detaches a mapped index without releasing it — or + * when ray_index_inline_map discarded a child that is still in the + * file, the tail would stay mapped for the life of the process. So + * a mapped column that carries an index registers its region under + * its own address with the true length, exactly as str_pool_cow does + * for a string column whose pool pointer stops leading to it; a + * string column already reaches the region through its pool. */ + if (vec->type != RAY_STR && (vec->attrs & RAY_ATTR_HAS_INDEX)) { + ray_file_map_t* m = (ray_file_map_t*)ray_sys_alloc(sizeof(*m)); + if (m) { + m->base = cm.mapped; + m->len = cm.mapped_size; + m->rc = 1; + m->next = NULL; + ray_file_map_register(vec, m); + } + /* No descriptor (oom): the free falls back to sizing from the + * index, the behaviour before this registration existed. */ + } } return vec; diff --git a/test/rfl/group/avg_exact_i128.rfl b/test/rfl/group/avg_exact_i128.rfl new file mode 100644 index 00000000..0217ee91 --- /dev/null +++ b/test/rfl/group/avg_exact_i128.rfl @@ -0,0 +1,139 @@ +;; avg of an integer column divides the EXACT 128-bit sum in every group +;; engine: keyless, dense/hash/radix keyed (v2), the float-key legacy hash +;; path, the small-table serial finishes, the slice-indexed path, pivot, +;; nullable inputs and a splayed copy — all the same bits as (avg vec). +;; SUM keeps its int64 wraparound contract next to the exact mean. +;; A column of 2^62 averages to exactly 2^62; alternating +(2^63-1-i) / +;; -(2^63-1-i) pairs sum to exactly -1 per pair, so their mean is -0.5 +;; whenever a group holds whole pairs (keys derive from (div d 2)). +(set n 70000) +(set d (til n)) +(set hc (take [4611686018427387904] n)) +(set hv (* (- (* 2 (% d 2)) 1) (- 9223372036854775807 d))) +(set p (div d 2)) +(set k3 (% p 3)) +(set f3 (as 'F64 k3)) +(set k350 (% p 350)) +(set ksp (* (% p 350) 1000003)) +(set k35k (% p 35000)) +(set T (table [c v d k3 f3 k350 ksp k35k] (list hc hv d k3 f3 k350 ksp k35k))) +(set E 4611686018427387904.0) +(set allE (fn [x] (== (count x) (sum (== x E))))) +(set allH (fn [x] (== (count x) (sum (== x -0.5))))) +;; the bare column +(avg hc) -- 4611686018427387904.0 +(avg hv) -- -0.5 +;; keyless select: one agg, a where, two aggs over one column (sum stays wrapped) +(at (select {from: T b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v)}) 'b) -- [-0.5] +(== (first (at (select {from: T b: (avg v)}) 'b)) (avg hv)) -- true +(== (first (at (select {from: T b: (avg c)}) 'b)) (avg hc)) -- true +(at (select {from: T where: (>= d 2) b: (avg v)}) 'b) -- [-0.5] +(at (select {from: T where: (>= d 2) b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg c) s: (sum c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v) s: (sum v)}) 'b) -- [-0.5] +(at (select {from: T b: (avg v) s: (sum v)}) 's) -- [-35000] +(at (select {from: T b: (avg c) m: (max c)}) 'b) -- [4611686018427387904.0] +;; few groups (dense), 350 groups, 35000 groups of two rows (each group's +;; own total overflows int64), sparse keys, a where, two keys +(allE (at (select {from: T by: k3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: k350 b: (avg v)}) 'b)) -- true +(count (select {from: T by: k35k b: (avg c)})) -- 35000 +(allE (at (select {from: T by: k35k b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k35k b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: ksp b: (avg v)}) 'b)) -- true +(allE (at (select {from: T by: ksp b: (avg c) s: (sum c)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: k3 b: k350} m: (avg v)}) 'm)) -- true +;; avg exact while sum of the same column wraps (23334, 23334, 23332 rows of 2^62) +(at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 's) -- [0Nl 0Nl 0] +(allE (at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 'b)) -- true +;; float keys take the legacy hash engine (radix merge + row emit) +(allE (at (select {from: T by: f3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v) s: (sum v)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: f3 b: (as 'F64 k350)} m: (avg v)}) 'm)) -- true +;; small tables: the serial finishes +(allH (at (select {from: T where: (< d 60) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T where: (< d 60) by: f3 b: (avg v)}) 'b)) -- true +(at (select {from: T where: (< d 60) b: (avg v)}) 'b) -- [-0.5] +;; narrow and temporal inputs agree with the bare vector avg +(set i32 (as 'I32 (* (- (* 2 (% d 2)) 1) (- 2147483647 (% d 1000))))) +(set i16 (as 'I16 (* (- (* 2 (% d 2)) 1) (- 32767 (% d 100))))) +(set u8 (as 'U8 (% d 256))) +(set bo (== (% d 3) 0)) +(set dt (as 'DATE (- 2147483647 (% d 1000)))) +(set tm (as 'TIME (- 2147483647 (% d 1000)))) +(set ts (as 'TIMESTAMP hv)) +(set TN (table [k3 f3 i32 i16 u8 bo dt tm ts] (list k3 f3 i32 i16 u8 bo dt tm ts))) +(at (select {from: TN a: (avg i32)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg i16)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg ts)}) 'a) -- [-0.5] +(== (first (at (select {from: TN a: (avg u8)}) 'a)) (avg u8)) -- true +(== (first (at (select {from: TN a: (avg bo)}) 'a)) (avg bo)) -- true +(== (first (at (select {from: TN a: (avg dt)}) 'a)) (avg dt)) -- true +(== (first (at (select {from: TN a: (avg tm)}) 'a)) (avg tm)) -- true +(at (select {from: TN by: k3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(== (at (select {from: TN by: k3 asc: k3 a: (avg u8)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg u8)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg bo)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg bo)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg dt)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg dt)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg tm)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg tm)}) 'a)) -- [true true true] +;; nulls are skipped; an all-null group is null +(set hcn (* hc (at [1 0N] (as 'I64 (== (% d 7) 0))))) +(set TNu (table [c d k3 f3] (list hcn d k3 f3))) +(at (select {from: TNu a: (avg c)}) 'a) -- [4611686018427387904.0] +(== (first (at (select {from: TNu a: (avg c)}) 'a)) (avg hcn)) -- true +(at (select {from: TNu a: (avg c) s: (sum c)}) 'a) -- [4611686018427387904.0] +(at (select {from: TNu where: (>= d 2) a: (avg c)}) 'a) -- [4611686018427387904.0] +(allE (at (select {from: TNu by: k3 a: (avg c)}) 'a)) -- true +(allE (at (select {from: TNu by: f3 a: (avg c)}) 'a)) -- true +(set TA (table [c k3 f3] (list (* hc (at [1 0N] (as 'I64 (== k3 2)))) k3 f3))) +(at (select {from: TA by: k3 asc: k3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: TA by: f3 asc: f3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: (select {from: TA where: (== k3 2)}) a: (avg c)}) 'a) -- [0Nf] +;; pivot +(set TP (table [r pc c v] (list (% p 3) (% (div d 6) 2) hc hv))) +(set P (pivot TP 'r 'pc 'v avg)) +(at P (at (cols P) 1)) -- [-0.5 -0.5 -0.5] +(at P (at (cols P) 2)) -- [-0.5 -0.5 -0.5] +(set Q (pivot TP 'r 'pc 'c avg)) +(allE (at Q (at (cols Q) 1))) -- true +(allE (at Q (at (cols Q) 2))) -- true +;; slice-indexed group (hash index on the SYM key + in-filter) +(set TS (update {from: (table [s v c] (list (as 'SYM (% p 5)) hv hc)) s: (.idx.hash s)})) +(at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg v)}) 'a) -- [-0.5 -0.5 -0.5] +(allE (at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg c) s2: (sum c)}) 'a)) -- true +;; single-key sparse scatter path (wide I32 key, avg over strlen of a STR +;; column, a count, a where): the mean of the lengths, bit for bit +(set cid (as 'I32 (* 35000 (% (* d 2654435761) 6000)))) +(set url (concat "http://example.com/some/path/" (as 'STR (* d (% d 977))))) +(set H (table [cid url c] (list cid url hc))) +(set R (select {from: H by: cid asc: cid l: (avg (strlen url)) n: (count url) where: (!= url "")})) +(count R) -- 6000 +(== (first (at R 'l)) (avg (strlen (at (select {from: H where: (== cid 0)}) 'url)))) -- true +(== (at R 'l) (at (select {from: H by: cid asc: cid l: (avg (strlen url)) where: (!= url "")}) 'l)) -- (take [true] 6000) +(allE (at (select {from: H by: cid l: (avg c) n: (count url)}) 'l)) -- true +(allE (at (select {from: H by: cid l: (avg c) s: (sum c) where: (!= url "")}) 'l)) -- true +;; a splayed copy answers exactly what the in-memory table answers +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 +(set Tm (table [c v d k3 f3 k350] (list hc hv d k3 f3 k350))) +(.db.splayed.set "rf_test_avg_i128/" Tm) +(set Ts (.db.splayed.get "rf_test_avg_i128/")) +(at (select {from: Ts a: (avg v) b: (avg c)}) 'a) -- [-0.5] +(at (select {from: Ts a: (avg v) b: (avg c)}) 'b) -- [4611686018427387904.0] +(== (at (select {from: Ts a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm a: (avg v) b: (avg c)}) 'b)) -- [true] +(== (at (select {from: Ts a: (avg v)}) 'a) (at (select {from: Tm a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts where: (>= d 2) a: (avg v)}) 'a) (at (select {from: Tm where: (>= d 2) a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a)) -- [true true true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(allH (at (select {from: Ts by: k350 a: (avg v)}) 'a)) -- true +(== (at (select {from: Ts by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(== (avg (at Ts 'c)) (first (at (select {from: Ts a: (avg c)}) 'a))) -- true +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl index 4716dc30..e340e9c3 100644 --- a/test/rfl/store/splayed_zone_aggs.rfl +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -22,9 +22,11 @@ (at (at R 's16) 0) -- (sum (at Tm 'a16)) (at (at R 'mn) 0) -- 2000.01.01 (at (at R 'c) 0) -- 200000 -;; sums past double's integer range: avg is left to the row-wise path and -;; still agrees; a float column and a filter take the usual path too +;; sums past double's integer range: the metadata keeps the exact 128-bit +;; per-chunk sums and the row-wise reduction accumulates the same way, so +;; avg agrees bit for bit; a float column and a filter take the usual path (at (select {from: T a: (avg big)}) 'a) -- (at (select {from: Tm a: (avg big)}) 'a) +(avg (at T 'big)) -- (avg (at Tm 'big)) (at (select {from: T a: (sum a16) b: (sum f)}) 'b) -- (at (select {from: Tm a: (sum a16) b: (sum f)}) 'b) (at (select {from: T a: (sum a16) where: (> a32 0)}) 'a) -- (at (select {from: Tm a: (sum a16) where: (> a32 0)}) 'a) ;; scalar forms read the same metadata @@ -44,6 +46,27 @@ (at (select {from: TX a: (min m) b: (max m)}) 'a) -- [9223372036854775807] (min (at TX 'm)) -- 9223372036854775807 (.sys.exec "rm -rf rf_test_zone_max") -- 0 +;; the exact mean of values that overflow int64 within every chunk: +;; alternating +(2^63-1-i) / -(2^63-1-i) pairs sum to exactly one per pair, +;; and a column of 2^62 averages to exactly 2^62 — from the metadata, from +;; the row-wise reduction, and from a parted copy +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 +(set hn 150000) +(set hi (til hn)) +(set hv (* (- (* 2 (% hi 2)) 1) (- 9223372036854775807 hi))) +(set hc (take [4611686018427387904] hn)) +(.db.splayed.set "rf_test_zone_huge/" (table [hv hc] (list hv hc))) +(set TH (.db.splayed.get "rf_test_zone_huge/")) +(set THm (table [hv hc] (list hv hc))) +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(avg hv) -- -0.5 +(avg (at TH 'hv)) -- -0.5 +(avg (at TH 'hc)) -- 4611686018427387904.0 +(at (select {from: TH a: (avg hv) where: (>= hi 0)}) 'a) -- (at (select {from: THm a: (avg hv) where: (>= hi 0)}) 'a) +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 ;; an empty table keeps the planner's answers (at (select {from: (select {from: T where: (< a64 -1000000)}) c: (count a16)}) 'c) -- (at (select {from: (select {from: Tm where: (< a64 -1000000)}) c: (count a16)}) 'c) (.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 diff --git a/test/test_agg_engine.c b/test/test_agg_engine.c index ebd94475..3adb1a8b 100644 --- a/test/test_agg_engine.c +++ b/test/test_agg_engine.c @@ -2103,6 +2103,101 @@ static test_result_t test_group_values_f64(void) { PASS(); } + +/* ══════════════════════════════════════════════════════════════════════ + * Exact integer AVG on the legacy engines (v2 OFF) and on v2 (ON). + * A column of 2^62 averages to exactly 2^62 and alternating + * +(2^63-1-i) / -(2^63-1-i) pairs to exactly -0.5 whenever a group holds + * whole pairs — only the 128-bit sum gives those bits (the wrapped int64 + * total is garbage, a double running sum loses the low bits). The key + * type and the row count steer the legacy ladder: an I64 key over 70000 + * rows takes the direct-array path, the same key over 60 rows the serial + * finish, an F64 key the hash/radix row layout, and a sparse I64 key + * (range far above the row count) the single-key sparse paths, which now + * hand an integer AVG to the row layout. */ +static ray_t* avg128_make(int64_t n, bool f64_key, bool sparse_key) { + ray_t* kvec = ray_vec_new(f64_key ? RAY_F64 : RAY_I64, n); kvec->len = n; + ray_t* vvec = ray_vec_new(RAY_I64, n); vvec->len = n; + ray_t* cvec = ray_vec_new(RAY_I64, n); cvec->len = n; + int64_t* vd = (int64_t*)ray_data(vvec); + int64_t* cd = (int64_t*)ray_data(cvec); + for (int64_t i = 0; i < n; i++) { + int64_t k = (i / 2) % 3; + if (sparse_key) k *= 1000000007LL; + if (f64_key) ((double*)ray_data(kvec))[i] = (double)k; + else ((int64_t*)ray_data(kvec))[i] = k; + int64_t mag = INT64_MAX - i; + vd[i] = (i & 1) ? mag : -mag; + cd[i] = (int64_t)1 << 62; + } + ray_t* tbl = ray_table_new(3); + tbl = ray_table_add_col(tbl, ray_sym_intern("k", 1), kvec); ray_release(kvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("v", 1), vvec); ray_release(vvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("c", 1), cvec); ray_release(cvec); + return tbl; +} +static ray_op_t* gb_avg128(ray_graph_t* g) { + ray_op_t* k = ray_scan(g, "k"); ray_op_t* v = ray_scan(g, "v"); ray_op_t* c = ray_scan(g, "c"); + uint16_t ops[] = { OP_AVG, OP_AVG, OP_SUM, OP_COUNT }; + ray_op_t* ins[] = { v, c, c, c }; ray_op_t* keys[] = { k }; + return ray_group(g, keys, 1, ops, ins, 4); +} +/* Run gb_avg128 with the v2 flag as given; the 2nd and 3rd result columns + * are the two means. Fails unless every group is exactly -0.5 / 2^62. */ +static test_result_t avg128_check(ray_t* tbl, bool v2, const char* what) { + ray_agg_engine_v2 = v2; + ray_graph_t* g = ray_graph_new(tbl); + ray_t* r = ray_execute(g, gb_avg128(g)); + if (r && ray_is_lazy(r)) r = ray_lazy_materialize(r); + ray_agg_engine_v2 = true; /* restore default */ + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + if (!r || RAY_IS_ERR(r) || r->type != RAY_TABLE || ray_table_ncols(r) != 5) { + res = (test_result_t){ TEST_FAIL, "avg128: bad result shape" }; + } else { + ray_t* mv = ray_table_get_col_idx(r, 1); + ray_t* mc = ray_table_get_col_idx(r, 2); + int64_t ng = ray_table_nrows(r); + if (ng != 3 || !mv || !mc || mv->type != RAY_F64 || mc->type != RAY_F64) { + res = (test_result_t){ TEST_FAIL, "avg128: expected 3 groups of F64 means" }; + } else { + for (int64_t i = 0; i < ng; i++) { + double a = ((const double*)ray_data(mv))[i]; + double b = ((const double*)ray_data(mc))[i]; + if (a != -0.5 || b != 4611686018427387904.0) { + snprintf(ray_test_fail_buf, sizeof ray_test_fail_buf, + "%s: group %lld mean(v)=%.17g mean(c)=%.17g (want -0.5, 2^62)", + what, (long long)i, a, b); + res = (test_result_t){ TEST_FAIL, ray_test_fail_buf }; + break; + } + } + } + } + if (r && !RAY_IS_ERR(r)) ray_release(r); + ray_graph_free(g); + return res; +} +static test_result_t test_avg_exact_i128_engines(void) { + ray_heap_init(); (void)ray_sym_init(); + struct { int64_t n; bool f64; bool sparse; const char* name; } shapes[] = { + { HC_N, false, false, "i64 key, 70000 rows" }, + { 60, false, false, "i64 key, 60 rows" }, + { HC_N, true, false, "f64 key, 70000 rows" }, + { 60, true, false, "f64 key, 60 rows" }, + { HC_N, false, true, "sparse i64 key, 70000 rows" }, + { 60, false, true, "sparse i64 key, 60 rows" }, + }; + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + for (size_t i = 0; i < sizeof(shapes) / sizeof(shapes[0]) && res.status == TEST_PASS; i++) { + ray_t* tbl = avg128_make(shapes[i].n, shapes[i].f64, shapes[i].sparse); + res = avg128_check(tbl, false, shapes[i].name); + if (res.status == TEST_PASS) res = avg128_check(tbl, true, shapes[i].name); + ray_release(tbl); + } + ray_sym_destroy(); ray_heap_destroy(); + return res; +} + const test_entry_t agg_engine_entries[] = { { "pearson_old_engine_r_vs_r2", test_pearson_old_engine_r_vs_r2, NULL, NULL }, { "diff_group_pearson_1k", test_diff_group_pearson_1k, NULL, NULL }, @@ -2179,5 +2274,6 @@ const test_entry_t agg_engine_entries[] = { { "group_keys_i_i32", test_group_keys_i_i32, NULL, NULL }, { "group_keys_multi", test_group_keys_multi, NULL, NULL }, { "agg_run_one_i64", test_agg_run_one_i64, NULL, NULL }, + { "avg_exact_i128_engines", test_avg_exact_i128_engines, NULL, NULL }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_index.c b/test/test_index.c index d79a0e3d..bcba719f 100644 --- a/test/test_index.c +++ b/test/test_index.c @@ -35,6 +35,9 @@ #include "ops/rowsel.h" #include "store/col.h" #include +#include +#include +#include #include #include #include @@ -581,6 +584,67 @@ static test_result_t test_index_persistence_roundtrip(void) { PASS(); } +/* ─── Mapped column drops its own index: the whole mapping is unmapped ── + * + * A column loaded by mmap with an inline index region is longer than its + * payload. Dropping the index from the loaded column itself (the sole + * reference: an in-place edit does exactly this) used to leave the index + * tail mapped for the life of the process, because ray_free sized the + * unmap from the index it no longer had. The file is laid out so the + * region crosses into a page of its own; after the free that page must + * be gone (msync reports ENOMEM on an unmapped range). */ +static test_result_t test_index_mapped_drop_unmaps_tail(void) { + ray_heap_init(); + /* Lay the file out so the inline index region crosses into a page of + * its own whatever the page size (4 KiB on Linux, 16 KiB on Apple + * silicon): the payload ends 64 bytes short of the second page. */ + long pg = sysconf(_SC_PAGESIZE); + TEST_ASSERT_TRUE(pg >= 4096); + int64_t n = (2 * (int64_t)pg - 96) / 8; + ray_t* v = ray_vec_new(RAY_I64, n); + for (int64_t i = 0; i < n; i++) { int64_t x = i * 3; v = ray_vec_append(v, &x); } + TEST_ASSERT_FALSE(RAY_IS_ERR(v)); + ray_t* w = v; + TEST_ASSERT_FALSE(RAY_IS_ERR(ray_index_attach_chunk_zone(&w, 8))); + + char path[] = "/tmp/idx_drop_unmap_XXXXXX"; + int fd = mkstemp(path); + TEST_ASSERT_TRUE(fd >= 0); + close(fd); + TEST_ASSERT_EQ_I(ray_col_save(w, path), RAY_OK); /* writes the inline index region too */ + ray_release(w); + + struct stat st; + TEST_ASSERT_EQ_I(stat(path, &st), 0); + TEST_ASSERT_TRUE(st.st_size > 2 * pg); /* the region reaches a further page */ + size_t mapped = ((size_t)st.st_size + (size_t)pg - 1) & ~((size_t)pg - 1); + + ray_t* m = ray_col_mmap(path); + TEST_ASSERT_FALSE(RAY_IS_ERR(m)); + TEST_ASSERT_EQ_U(m->mmod, 1); + TEST_ASSERT_TRUE(m->attrs & RAY_ATTR_HAS_INDEX); + TEST_ASSERT_EQ_I((int)ray_index_payload(m->index)->kind, RAY_IDX_CHUNK_ZONE); + char* last_page = (char*)m + mapped - (size_t)pg; + TEST_ASSERT_EQ_I(msync(last_page, (size_t)pg, MS_ASYNC), 0); /* mapped while loaded */ + + /* Sole reference: the drop detaches the mapped index in place. */ + ray_t* d = m; + ray_t* r = ray_index_drop(&d); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + TEST_ASSERT_TRUE(d == m); + TEST_ASSERT_FALSE(d->attrs & RAY_ATTR_HAS_INDEX); + int64_t* data = (int64_t*)ray_data(d); + TEST_ASSERT_EQ_I(data[n - 1], (n - 1) * 3); + + ray_release(d); + errno = 0; + int rc = msync(last_page, (size_t)pg, MS_ASYNC); + TEST_ASSERT_TRUE(rc == -1 && errno == ENOMEM); /* the tail page is unmapped */ + unlink(path); + ray_heap_destroy(); + PASS(); +} + /* ─── Slice null detection on indexed/parent vec ───────────────────── */ static test_result_t test_index_aux_helper_slice(void) { @@ -3784,6 +3848,7 @@ const test_entry_t index_entries[] = { { "index/null_readers_through_helper", test_index_null_readers_through_helper, NULL, NULL }, { "index/aux_helper_slice", test_index_aux_helper_slice, NULL, NULL }, { "index/drop_under_shared_cow", test_index_drop_under_shared_cow, NULL, NULL }, + { "index/mapped_drop_unmaps_tail", test_index_mapped_drop_unmaps_tail, NULL, NULL }, { "index/persistence_roundtrip", test_index_persistence_roundtrip, NULL, NULL }, { "index/bool_zone_and_hash", test_index_bool_zone_and_hash, NULL, NULL }, { "index/i16_zone_and_hash", test_index_i16_zone_and_hash, NULL, NULL },