diff --git a/RELEASE.md b/RELEASE.md index 76421ebe8..248133049 100644 --- a/RELEASE.md +++ b/RELEASE.md @@ -29,6 +29,26 @@ truth for the version** — no version literal is ever hand-edited in source. - packages `rayforce-X.Y.Z--.tar.gz` + a `.sha256` checksum, - publishes a GitHub Release with a feature-oriented changelog and the artifacts. +5. **Merge `master` back into `dev`** right after the release PR merges: + + ```sh + git fetch origin + git switch dev && git pull --ff-only + git merge --no-ff -s ours origin/master \ + -m "chore: merge master back into dev after the vX.Y.Z release (#N)" + git diff origin/dev --stat # must print nothing + git push origin dev + ``` + + Merging the release PR leaves a squash or merge commit on `master` that `dev` + doesn't contain. `master` requires branches to be up to date before merging, + so without this step the next release PR is blocked as "out of date with the + base". After a squash, it also shows conflicts that aren't real. Copying the + release commit onto `dev` doesn't help: only a merge makes `master`'s commit + an ancestor of `dev`. `-s ours` is correct because `dev` already contains + everything the release shipped, so the empty `git diff` proves the merge + changed no files. If `master` ever carries a hotfix that isn't on `dev`, + drop `-s ours` and do a normal merge instead. That's the whole ritual. **Never edit the version in source to make a release** — the tag is authoritative. diff --git a/bench/concat_text_nulls.c b/bench/concat_text_nulls.c new file mode 100644 index 000000000..6ce308c99 --- /dev/null +++ b/bench/concat_text_nulls.c @@ -0,0 +1,45 @@ +/* Focused concat benchmark: build with release library, run before/after. */ +#define _POSIX_C_SOURCE 200809L +#include +#include "mem/heap.h" +#include "vec/vec.h" +#include +#include + +static double now(void) { + struct timespec t; + clock_gettime(CLOCK_MONOTONIC, &t); + return t.tv_sec + t.tv_nsec * 1e-9; +} + +int main(void) { + ray_heap_init(); + const int64_t n = 1000000; + const int reps = 40; + for (int kind = 0; kind < 2; kind++) { + for (int mode = 0; mode < 3; mode++) { + ray_t* a = ray_vec_new(kind ? RAY_SYM : RAY_STR, n); + if (!a || RAY_IS_ERR(a)) return 1; + if (kind) { + a->len = n; + for (int64_t i = 0; i < n; i++) ((int64_t*)ray_data(a))[i] = 1; + } else { + for (int64_t i = 0; i < n; i++) a = ray_str_vec_append(a, "abc", 3); + } + if (mode == 1) ray_vec_set_null(a, 0, true); + if (mode == 2) ray_vec_set_null(a, n - 1, true); + double start = now(); + for (int r = 0; r < reps; r++) { + ray_t* out = ray_vec_concat(a, a); + if (!out || RAY_IS_ERR(out) || out->len != 2 * n) return 2; + ray_release(out); + } + printf("%s %s rows=%lld reps=%d ms_per_concat=%.6f\n", + kind ? "SYM64" : "STR-inline", mode == 0 ? "no-null" : mode == 1 ? "first-null" : "last-null", + (long long)n, reps, (now() - start) * 1000 / reps); + ray_release(a); + } + } + ray_heap_destroy(); + return 0; +} diff --git a/docs/docs/guides/memory.md b/docs/docs/guides/memory.md index a0050195e..ba1780ea8 100644 --- a/docs/docs/guides/memory.md +++ b/docs/docs/guides/memory.md @@ -385,7 +385,7 @@ This tells you the join needed about 1 GB of temporary memory beyond what was al | Tool | What It Does | When to Use | |---|---|---| | `(.sys.mem 0)` | Returns heap allocation statistics | Monitor memory usage, detect leaks | -| `(.sys.gc 0)` | Flushes caches, releases pages | Between heavy queries, before benchmarks | +| `(.sys.gc)` | Flushes caches, releases pages | Between heavy queries, before benchmarks | | `(.sys.info 0)` | Shows system and runtime info | Check total RAM, CPU count, OS details | | `(timeit expr)` | Measures execution time of one expression | Benchmark a specific operation | | `:timeit` | Toggles profiling for all REPL expressions | Interactive performance exploration | diff --git a/docs/docs/namespaces/log.md b/docs/docs/namespaces/log.md index fbde54b66..7a366e971 100644 --- a/docs/docs/namespaces/log.md +++ b/docs/docs/namespaces/log.md @@ -15,11 +15,11 @@ The journal is intentionally minimal: there's no per-entry timestamp or transact | [`.log.write`](#log-write) | unary | — | Append a serialised expression to the open journal. | | [`.log.sync`](#log-sync) | variadic | — | `fsync` the journal. | | [`.log.snapshot`](#log-snapshot) | variadic | restricted | Write a snapshot of current state and roll to a fresh segment. | -| [`.log.roll`](#log-roll) | variadic | restricted | Close the active segment and start a new one. | +| [`.log.roll`](#log-roll) | nullary | restricted | Close the active segment and start a new one. | | [`.log.replay`](#log-replay) | unary | restricted | Replay a journal file; return entry count. | | [`.log.validate`](#log-validate) | unary | — | Scan a journal file; return `(chunks valid_bytes)`. | | [`.log.close`](#log-close) | variadic | restricted | Flush and close the active journal. | -| [`.log.purge`](#log-purge) | variadic | restricted | Close the active journal and delete all its files. | +| [`.log.purge`](#log-purge) | nullary | restricted | Close the active journal and delete all its files. | ## `.log.open` { #log-open } diff --git a/src/core/platform.c b/src/core/platform.c index bad99c91d..5251a6712 100644 --- a/src/core/platform.c +++ b/src/core/platform.c @@ -369,7 +369,7 @@ static uint64_t cache_sysfs_llc_bytes(void) { } #endif -uint64_t ray_cache_llc_bytes(void) { +static uint64_t cache_llc_probe(void) { static uint64_t cached = UINT64_MAX; if (cached != UINT64_MAX) return cached; uint64_t bytes = 0; @@ -642,7 +642,7 @@ uint32_t ray_physical_core_count(void) { /* Sum of every level-3 cache instance reported by the processor topology * (each SYSTEM_LOGICAL_PROCESSOR_INFORMATION cache record is one instance). * 0 when the query fails. */ -uint64_t ray_cache_llc_bytes(void) { +static uint64_t cache_llc_probe(void) { static uint64_t cached = UINT64_MAX; if (cached != UINT64_MAX) return cached; uint64_t bytes = 0; @@ -800,7 +800,7 @@ ray_err_t ray_thread_join(ray_thread_t t) { } uint32_t ray_thread_count(void) { return 1; } -uint64_t ray_cache_llc_bytes(void) { return 0; } +static uint64_t cache_llc_probe(void) { return 0; } /* Semaphore — counter-only. Single-threaded so wait never blocks (the * counter must already be positive when wait fires). */ @@ -818,3 +818,21 @@ void ray_sem_wait(ray_sem_t* s) { void ray_sem_signal(ray_sem_t* s) { (*s)++; } #endif /* RAY_OS_WASM */ + +#ifdef DEBUG +/* Test pin for the probed LLC size: routing that bounds replicated state by + * the cache (group dense slabs) otherwise picks a different strategy on + * every CI runner. 0 restores the platform probe. */ +static uint64_t g_llc_for_test = 0; + +void ray_cache_llc_set_for_test(uint64_t bytes) { + g_llc_for_test = bytes; +} +#endif + +uint64_t ray_cache_llc_bytes(void) { +#ifdef DEBUG + if (g_llc_for_test) return g_llc_for_test; +#endif + return cache_llc_probe(); +} diff --git a/src/core/platform.h b/src/core/platform.h index ea3633c56..e2d137856 100644 --- a/src/core/platform.h +++ b/src/core/platform.h @@ -180,6 +180,10 @@ uint32_t ray_physical_core_count(void); * 0 when the platform cannot report it. Bounds replicated per-task state * whose random-access working set must stay cache-resident to scale. */ uint64_t ray_cache_llc_bytes(void); +#ifdef DEBUG +/* Pin ray_cache_llc_bytes to `bytes` (0 = probe again). */ +void ray_cache_llc_set_for_test(uint64_t bytes); +#endif void ray_parallel_begin(void); void ray_parallel_end(void); diff --git a/src/io/csv.c b/src/io/csv.c index 25302c1f9..f4d583ef7 100644 --- a/src/io/csv.c +++ b/src/io/csv.c @@ -151,6 +151,7 @@ static inline void scratch_free(ray_t* hdr) { typedef struct { const char* ptr; uint32_t len; + uint32_t hash; /* ray_hash_bytes of the field; filled for SYM columns */ } csv_strref_t; RAY_INLINE const char* scan_field(const char* p, const char* buf_end, @@ -952,9 +953,15 @@ static size_t csv_scan_split_at(const char* buf, size_t file_size, size_t s, /* Returns the row count (>=0), -1 if interrupted, or -2 when the parallel * path does not apply and the caller should run the serial scan. */ +/* force_quotes: the caller scans a byte window of a file and knows quotes + * occur somewhere in the file's data. The quote-aware state machine is + * then used even when this window holds none, so a window is split into + * rows exactly as the whole file would be (the quote-free fast path treats + * a lone '\r' and "\n\r" differently). */ static int64_t build_row_offsets_par(const char* buf, size_t buf_size, size_t data_offset, uint64_t prog_base, uint64_t prog_len, + bool force_quotes, int64_t** offsets_out, ray_t** hdr_out) { *offsets_out = NULL; *hdr_out = NULL; @@ -1023,7 +1030,7 @@ static int64_t build_row_offsets_par(const char* buf, size_t buf_size, total_q += quote_cnt[i]; total_term += term_cnt[i]; } - ctx.has_quotes = total_q != 0; + ctx.has_quotes = total_q != 0 || force_quotes; /* Now nudge the boundaries for the state machine pass 1 just chose. The * byte skipped is always '\n' or '\r', never '"', so the parities @@ -1201,7 +1208,7 @@ static int64_t build_row_offsets(const char* buf, size_t buf_size, * -2 means "not applicable" (small file, no pool, allocation refused) and * falls back to the serial scan, which defines the semantics. */ int64_t par = build_row_offsets_par(buf, buf_size, data_offset, - prog_base, prog_len, + prog_base, prog_len, false, offsets_out, hdr_out); if (par != -2) return par; @@ -1211,6 +1218,59 @@ static int64_t build_row_offsets(const char* buf, size_t buf_size, offsets_out, hdr_out); } +/* Row starts of the next `max_rows` rows from `data_offset`, found by the + * parallel scanner over a byte window instead of the serial walk: the + * window is sized from `avg_row` (bytes per row seen so far) with slack, and + * doubled when it holds fewer than max_rows + 1 row starts before the end + * of the file (the extra start proves the max_rows-th row is complete). + * Returns the row count, -1 if interrupted, or -2 when the parallel path + * does not apply (small window, no pool) — the caller then runs the serial + * limited scan. */ +static int64_t build_row_offsets_window(const char* buf, size_t buf_size, + size_t data_offset, int64_t max_rows, + size_t avg_row, bool data_has_quotes, + int64_t** offsets_out, ray_t** hdr_out, + size_t* next_offset_out) { + *offsets_out = NULL; *hdr_out = NULL; + if (next_offset_out) *next_offset_out = data_offset; + if (max_rows <= 0 || data_offset >= buf_size) return 0; + if (avg_row < 8) avg_row = 8; + size_t window = (size_t)max_rows * avg_row + (size_t)max_rows * avg_row / 4 + (64u << 10); + for (;;) { + size_t end = data_offset + window; + if (end > buf_size || end < data_offset) end = buf_size; + /* Readahead hint for the window about to be scanned: the scanner's + * tasks fault the pages in parallel, which a cold file serves best + * when the kernel already streams the range. */ + { + size_t ps = (size_t)sysconf(_SC_PAGESIZE); + size_t a = data_offset & ~(ps - 1); + madvise((void*)(buf + a), end - a, MADV_WILLNEED); + } + int64_t* offs = NULL; ray_t* hdr = NULL; + int64_t n = build_row_offsets_par(buf, end, data_offset, 0, 0, + data_has_quotes, &offs, &hdr); + if (n < 0) return n; /* -1 interrupted, -2 not applicable */ + if (n == 0) { scratch_free(hdr); return -2; } + if (n > max_rows) { + /* row max_rows - 1 ends before start max_rows: complete */ + if (next_offset_out) *next_offset_out = (size_t)offs[max_rows]; + *offsets_out = offs; *hdr_out = hdr; + return max_rows; + } + if (end == buf_size) { + /* every remaining row, the last one ended by the file */ + if (next_offset_out) *next_offset_out = buf_size; + *offsets_out = offs; *hdr_out = hdr; + return n; + } + /* the window held at most max_rows starts: widen and rescan */ + scratch_free(hdr); + if (window > SIZE_MAX / 2) return -2; + window *= 2; + } +} + static int64_t build_row_offsets_limited(const char* buf, size_t buf_size, size_t data_offset, int64_t max_rows, bool data_has_quotes, @@ -1381,6 +1441,7 @@ typedef struct { uint32_t len; const char* ptr; int64_t gid; /* global sym id, filled in step B */ + int64_t row; /* first row the string occurs on (domain batch order) */ } csv_dedup_ent_t; typedef struct { @@ -1438,8 +1499,32 @@ typedef struct { /* [n_cols] "this column wrote a canonical null", recorded where the value * is written. See csv_note_empty. */ bool* empties; + /* Target FILE domain (splayed save): the distinct strings are + * interned straight into the table's symfile domain, batched over + * every SYM column of the chunk, and the codes are positions in it. + * NULL: runtime symbol table, id-order-preserving serial walk. */ + struct ray_sym_domain_s* dom; + /* Hash partitions per SYM column (domain target only; 1 otherwise): + * dicts[i * n_part + p] holds the distinct strings of column cols[i] + * whose hash >> part_shift == p. Partitions of one column are + * deduplicated by independent tasks. */ + int n_part; + int part_shift; } csv_dedup_ctx_t; +static inline int csv_dedup_part(const csv_dedup_ctx_t* dd, uint32_t hash) { + return dd->n_part > 1 ? (int)(hash >> dd->part_shift) : 0; +} + +/* Every partition of column i deduplicated without overflow. */ +static inline bool csv_col_dict_ok(const csv_dedup_ctx_t* dd, int i) { + for (int p = 0; p < dd->n_part; p++) { + const csv_dedup_t* d = &dd->dicts[i * dd->n_part + p]; + if (!d->done || d->overflow) return false; + } + return true; +} + /* HAS_NULLS accounting for SYM/STR columns (step 9c). * * bfb5b380 made an empty SYM/STR cell a canonical null and, where the parse @@ -1460,10 +1545,14 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, (void)worker_id; (void)end_i; csv_dedup_ctx_t* ctx = (csv_dedup_ctx_t*)arg; csv_dedup_t* d = &ctx->dicts[start]; - const csv_strref_t* refs = ctx->str_refs[ctx->cols[start]]; + int col_i = (int)(start / ctx->n_part); + int part = (int)(start % ctx->n_part); + const csv_strref_t* refs = ctx->str_refs[ctx->cols[col_i]]; /* Local codes land in the destination id array and are replaced in place - * by step C, so the dedupe needs no per-row scratch of its own. */ - uint32_t* codes = (uint32_t*)ctx->col_data[ctx->cols[start]]; + * by step C, so the dedupe needs no per-row scratch of its own. With + * partitions each task owns the rows whose hash falls in its partition; + * partition 0 also writes the null codes. */ + uint32_t* codes = (uint32_t*)ctx->col_data[ctx->cols[col_i]]; int64_t n_rows = ctx->n_rows; if (!csv_dedup_grow(d) || !csv_dedup_grow_ents(d)) { @@ -1475,8 +1564,9 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, for (int64_t r = 0; r < n_rows; r++) { if (RAY_UNLIKELY((r & 1023) == 0 && ray_interrupted())) return; - if (refs[r].ptr == NULL) { codes[r] = 0; continue; } - uint32_t h = (uint32_t)ray_hash_bytes(refs[r].ptr, refs[r].len); + if (refs[r].ptr == NULL) { if (part == 0) codes[r] = 0; continue; } + uint32_t h = refs[r].hash; /* computed by the parse */ + if (csv_dedup_part(ctx, h) != part) continue; uint32_t mask = d->n_slots - 1; uint32_t j = h & mask; uint32_t found = 0; @@ -1507,6 +1597,7 @@ static void csv_dedup_task(void* arg, uint32_t worker_id, e->len = refs[r].len; e->ptr = refs[r].ptr; e->gid = 0; + e->row = r; d->n_ents++; codes[r] = d->n_ents; /* code = entry index + 1 */ d->slots[j] = d->n_ents; @@ -1523,8 +1614,9 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, int64_t start, int64_t end_i) { (void)worker_id; (void)end_i; csv_dedup_ctx_t* ctx = (csv_dedup_ctx_t*)arg; - const csv_dedup_t* d = &ctx->dicts[start]; - if (d->overflow) return; /* step B already wrote real ids */ + if (!csv_col_dict_ok(ctx, (int)start)) return; /* step B already wrote real ids */ + const csv_dedup_t* dcol = &ctx->dicts[start * ctx->n_part]; + const csv_strref_t* refs = ctx->str_refs[ctx->cols[start]]; uint32_t* ids = (uint32_t*)ctx->col_data[ctx->cols[start]]; int64_t n_rows = ctx->n_rows; uint32_t empty = (uint32_t)ctx->empty_gid; @@ -1532,7 +1624,8 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, for (int64_t r = 0; r < n_rows; r++) { if (RAY_UNLIKELY((r & 1023) == 0 && ray_interrupted())) return; uint32_t code = ids[r]; - uint32_t id = code ? (uint32_t)d->ents[code - 1].gid : empty; + uint32_t id = code ? (uint32_t)dcol[csv_dedup_part(ctx, refs[r].hash)].ents[code - 1].gid + : empty; ids[r] = id; saw_null |= (id == 0); } @@ -1551,9 +1644,135 @@ static void csv_dedup_map_task(void* arg, uint32_t worker_id, * empty string anyway, so collapsing them is the only deterministic answer the * parser can give. Empty fields are local code 0 out of the dedupe and are * mapped to that id by step C. */ +/* Step B for a FILE domain target: one batch for the whole chunk, laid out + * the way the cell-by-cell writer met the strings — columns in order, each + * column's strings by first occurrence — so the symfile gets the same + * positions whatever the worker count and however a column's dictionary + * was split into hash partitions. The domain probes the existing + * vocabulary in parallel and appends the new strings in batch order. A + * column whose dictionary overflowed contributes its rows directly (the + * batch dedupes them) and gets its ids written here. */ +/* Append column i's dictionary entries to out[] by first row. Each hash + * partition lists its strings in row order already, so this is a merge of + * n_part sorted runs (n_part is small). */ +static int64_t csv_col_ents_by_row(csv_dedup_ctx_t* dd, int i, csv_dedup_ent_t** out) { + int np = dd->n_part; + csv_dedup_t* d0 = &dd->dicts[i * np]; + if (np == 1) { + for (uint32_t e = 0; e < d0->n_ents; e++) out[e] = &d0->ents[e]; + return d0->n_ents; + } + uint32_t cur[64]; + for (int p = 0; p < np; p++) cur[p] = 0; + int64_t n = 0; + for (;;) { + int best = -1; + int64_t br = 0; + for (int p = 0; p < np; p++) { + if (cur[p] >= d0[p].n_ents) continue; + int64_t r = d0[p].ents[cur[p]].row; + if (best < 0 || r < br) { best = p; br = r; } + } + if (best < 0) break; + out[n++] = &d0[best].ents[cur[best]++]; + } + return n; +} + +static bool csv_intern_dicts_domain(csv_dedup_ctx_t* dd, int n_sym, + int64_t* col_max_ids, + uint64_t prog_base, uint64_t prog_len) { + struct ray_sym_domain_s* dom = dd->dom; + /* Position 0 of a symfile domain is "" (reserved on creation). */ + if (ray_sym_domain_intern(dom, "", 0) != 0) return false; + dd->empty_gid = 0; + + int64_t total = 0; + for (int i = 0; i < n_sym; i++) { + if (csv_col_dict_ok(dd, i)) { + for (int p = 0; p < dd->n_part; p++) total += dd->dicts[i * dd->n_part + p].n_ents; + } else { + total += dd->n_rows; + } + } + if (total == 0) { + if (prog_len) ray_progress_span_set(prog_base + prog_len); + return true; + } + + ray_t *hs = NULL, *hl = NULL, *hh = NULL, *hp = NULL, *ho = NULL; + const char** strs = (const char**)scratch_alloc(&hs, (size_t)total * sizeof(char*)); + size_t* lens = (size_t*)scratch_alloc(&hl, (size_t)total * sizeof(size_t)); + uint32_t* hashes = (uint32_t*)scratch_alloc(&hh, (size_t)total * sizeof(uint32_t)); + int64_t* pos = (int64_t*)scratch_alloc(&hp, (size_t)total * sizeof(int64_t)); + /* order[k]: the dictionary entry behind batch slot k (dict columns) */ + csv_dedup_ent_t** order = (csv_dedup_ent_t**)scratch_alloc(&ho, + (size_t)total * sizeof(csv_dedup_ent_t*)); + bool ok = strs && lens && hashes && pos && order; + + int64_t k = 0; + for (int i = 0; ok && i < n_sym; i++) { + if (csv_col_dict_ok(dd, i)) { + int64_t ne = csv_col_ents_by_row(dd, i, order + k); + for (int64_t e = 0; e < ne; e++, k++) { + strs[k] = order[k]->ptr; + lens[k] = order[k]->len; + hashes[k] = order[k]->hash; + } + } else { + const csv_strref_t* refs = dd->str_refs[dd->cols[i]]; + for (int64_t r = 0; r < dd->n_rows; r++) { + if (refs[r].ptr == NULL) continue; + strs[k] = refs[r].ptr; + lens[k] = refs[r].len; + hashes[k] = refs[r].hash; + k++; + } + } + } + if (ok) ok = ray_sym_domain_intern_batch(dom, k, strs, lens, hashes, pos); + + /* Hand the positions back in the same walk. */ + k = 0; + for (int i = 0; ok && i < n_sym; i++) { + int c = dd->cols[i]; + int64_t max_id = 0; + if (csv_col_dict_ok(dd, i)) { + int64_t ne = 0; + for (int p = 0; p < dd->n_part; p++) ne += dd->dicts[i * dd->n_part + p].n_ents; + for (int64_t e = 0; e < ne; e++, k++) { + int64_t id = pos[k]; + if (id < 0) { ok = false; id = 0; } + order[k]->gid = id; + if (id > max_id) max_id = id; + } + } else { + const csv_strref_t* refs = dd->str_refs[c]; + uint32_t* ids = (uint32_t*)dd->col_data[c]; + bool saw_null = false; + for (int64_t r = 0; r < dd->n_rows; r++) { + if (refs[r].ptr == NULL) { ids[r] = 0; saw_null = true; continue; } + int64_t id = pos[k++]; + if (id < 0) { ok = false; id = 0; } + ids[r] = (uint32_t)id; + saw_null |= (id == 0); + if (id > max_id) max_id = id; + } + csv_note_empty(dd->empties, c, saw_null); + } + if (col_max_ids) col_max_ids[c] = max_id; + } + + scratch_free(hs); scratch_free(hl); scratch_free(hh); scratch_free(hp); scratch_free(ho); + if (prog_len) ray_progress_span_set(prog_base + prog_len); + return ok; +} + static bool csv_intern_dicts(csv_dedup_ctx_t* dd, int n_sym, int64_t* col_max_ids, uint64_t prog_base, uint64_t prog_len) { + if (dd->dom) + return csv_intern_dicts_domain(dd, n_sym, col_max_ids, prog_base, prog_len); bool ok = true; /* Same first call, same reason, as the serial walk: sym 0 is reserved by @@ -1739,6 +1958,7 @@ typedef struct { int n_sym; bool* empties; /* [n_cols] SYM/STR column wrote a null */ bool intern_ok; + struct ray_sym_domain_s* dom; /* SYM target domain, NULL = runtime */ } csv_finalize_ctx_t; /* dispatch 1: [0, n_fill) fill a RAY_STR column, [n_fill, n_fill+n_sym) dedupe @@ -1768,7 +1988,7 @@ static void csv_finalize_task(void* arg, uint32_t worker_id, * prog_base/prog_len describe this phase's slice of the load's byte axis; pass * len 0 when no progress span is active (the streaming conversion path). */ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, - bool* fill_ok, int* sym_cols, csv_dedup_t* dicts, + bool* fill_ok, int* sym_cols, bool* empties, uint64_t prog_base, uint64_t prog_len) { int n_fill = 0, n_sym = 0; @@ -1778,7 +1998,31 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, else sym_cols[n_sym++] = c; } for (int i = 0; i < n_fill; i++) fill_ok[i] = true; - memset(dicts, 0, (size_t)n_sym * sizeof(csv_dedup_t)); + + ray_pool_t* pool = ray_pool_get(); + bool par = pool && ray_pool_total_workers(pool) >= 2; + + /* Partitions per SYM column. The runtime path keeps one dictionary per + * column (its id order is the serial walk's); a domain target has no + * order to keep, so the heavy columns are split by hash until the + * dedupe tasks cover the pool about twice over. */ + int n_part = 1; + if (ctx->dom && par && n_sym > 0) { + int64_t want = (int64_t)ray_pool_total_workers(pool) * 2 / n_sym; + while (n_part * 2 <= want && n_part < 16) n_part *= 2; + while (n_part > 1 && (int64_t)n_fill + (int64_t)n_sym * n_part > (int64_t)RAY_POOL_INIT_TASKS) + n_part /= 2; + } + int part_shift = 32; + for (int q = n_part; q > 1; q >>= 1) part_shift--; + + ray_t* dicts_hdr = NULL; + csv_dedup_t* dicts = NULL; + if (n_sym > 0) { + dicts = (csv_dedup_t*)scratch_calloc(&dicts_hdr, + (size_t)n_sym * (size_t)n_part * sizeof(csv_dedup_t)); + if (!dicts) return false; + } ctx->fill_cols = fill_cols; ctx->n_fill = n_fill; @@ -1793,6 +2037,9 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, ctx->dd.n_rows = ctx->n_rows; ctx->dd.empty_gid = 0; ctx->dd.empties = ctx->empties; + ctx->dd.dom = ctx->dom; + ctx->dd.n_part = n_part; + ctx->dd.part_shift = part_shift; /* Slice the phase: dedupe+fill is the bulk, the serial intern touches only * distinct strings, the remap is one linear pass per SYM column. Measured @@ -1800,10 +2047,8 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, uint64_t w1 = prog_len * 6 / 10; uint64_t w2 = prog_len / 10; - int64_t n_tasks = (int64_t)n_fill + (int64_t)n_sym; - ray_pool_t* pool = ray_pool_get(); - bool par = pool && ray_pool_total_workers(pool) >= 2 && - n_tasks > 0 && n_tasks <= (int64_t)RAY_POOL_INIT_TASKS; + int64_t n_tasks = (int64_t)n_fill + (int64_t)n_sym * n_part; + par = par && n_tasks > 0 && n_tasks <= (int64_t)RAY_POOL_INIT_TASKS; if (prog_len) ray_progress_span_phase("finalize", prog_base, w1); if (par) ray_pool_dispatch_n(pool, csv_finalize_task, ctx, (uint32_t)n_tasks); @@ -1830,12 +2075,14 @@ static bool csv_finalize_run(csv_finalize_ctx_t* ctx, int* fill_cols, if (ray_interrupted()) goto fail; } - for (int i = 0; i < n_sym; i++) csv_dedup_release(&dicts[i]); + for (int i = 0; i < n_sym * n_part; i++) csv_dedup_release(&dicts[i]); + scratch_free(dicts_hdr); for (int i = 0; i < n_fill; i++) if (!fill_ok[i]) return false; return true; fail: - for (int i = 0; i < n_sym; i++) csv_dedup_release(&dicts[i]); + for (int i = 0; i < n_sym * n_part; i++) csv_dedup_release(&dicts[i]); + scratch_free(dicts_hdr); return false; } @@ -2021,6 +2268,10 @@ static void csv_parse_fn(void* arg, uint32_t worker_id, } ctx->str_refs[c][row].ptr = fld; ctx->str_refs[c][row].len = (uint32_t)flen; + /* SYM columns are deduplicated by hash right after + * the parse; hash here while the bytes are hot. */ + if (ctx->resolved_types[c] == RAY_SYM) + ctx->str_refs[c][row].hash = (uint32_t)ray_hash_bytes(fld, flen); } break; } @@ -2202,6 +2453,8 @@ static bool csv_parse_serial(const char* buf, size_t buf_size, } str_refs[c][row].ptr = fld; str_refs[c][row].len = (uint32_t)flen; + if (resolved_types[c] == RAY_SYM) + str_refs[c][row].hash = (uint32_t)ray_hash_bytes(fld, flen); } break; } @@ -2479,7 +2732,8 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, const int64_t* row_offsets, int64_t n_rows, int ncols, char delimiter, const int64_t* col_name_ids, - const int8_t* resolved_types) { + const int8_t* resolved_types, + struct ray_sym_domain_s* sym_dom) { /* Defensive guard: RAY_CSV_AUTO_TAG must be resolved to a concrete width * before reaching this point (by csv_resolve_auto_in_place or * csv_resolve_auto_streamed). If a marker slips through, the resolution @@ -2505,6 +2759,11 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, for (int j = 0; j < c; j++) ray_release(col_vecs[j]); return NULL; } + if (type == RAY_SYM && sym_dom) { + /* Cells are positions in the target symfile domain. */ + ray_sym_domain_retain(sym_dom); + col_vecs[c]->sym_domain = sym_dom; + } col_vecs[c]->len = n_rows; col_data[c] = ray_data(col_vecs[c]); } @@ -2645,14 +2904,14 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, .col_vecs = col_vecs, .n_rows = n_rows, .sym_max_ids = sym_max_ids, + .dom = sym_dom, }; int fill_cols[CSV_MAX_COLS]; int sym_cols[CSV_MAX_COLS]; bool fill_ok[CSV_MAX_COLS]; - csv_dedup_t dicts[CSV_MAX_COLS]; /* No progress span on the conversion path — pass a zero-length slice. */ bool fin_ok = csv_finalize_run(&fctx, fill_cols, fill_ok, - sym_cols, dicts, col_wrote_null, 0, 0); + sym_cols, col_wrote_null, 0, 0); if (!fin_ok || ray_interrupted()) { csv_free_escaped_strrefs(str_ref_bufs, ncols, parse_types, n_rows, buf, file_size, row_done, col_had_escaped); @@ -2693,6 +2952,7 @@ static ray_t* csv_materialize_rows(const char* buf, size_t file_size, if (new_w >= RAY_SYM_W32) continue; ray_t* narrow = ray_sym_vec_new(new_w, n_rows); if (!narrow || RAY_IS_ERR(narrow)) continue; + ray_sym_vec_adopt_domain(narrow, col_vecs[c]); narrow->len = n_rows; const uint32_t* src = (const uint32_t*)col_data[c]; void* dst = ray_data(narrow); @@ -3123,9 +3383,8 @@ static ray_t* csv_read_named_opts_inner(const char* path, char delimiter, bool h int fill_cols[CSV_MAX_COLS]; int sym_cols[CSV_MAX_COLS]; bool fill_ok[CSV_MAX_COLS]; - csv_dedup_t dicts[CSV_MAX_COLS]; bool fin_ok = csv_finalize_run(&fctx, fill_cols, fill_ok, - sym_cols, dicts, col_wrote_null, + sym_cols, col_wrote_null, prog_parse_end, (uint64_t)file_size - prog_parse_end); if (!fin_ok || ray_interrupted()) { @@ -3283,7 +3542,15 @@ typedef struct { * fixed at W32: a streaming writer can't know the final vocabulary * before the last chunk, and W32 covers any STRL count. */ struct ray_sym_domain_s* dom; + /* runtime id -> domain position, direct-mapped: the chunk vecs are + * runtime-domain and a column's values repeat across rows and chunks, + * so a value is interned into the symfile's domain (a locked probe) the + * first time it is met and looked up here after. */ + int64_t* lut_id; /* [CSV_SPLAYED_LUT] runtime id per slot, -1 empty */ + uint32_t* lut_pos; /* [CSV_SPLAYED_LUT] position per slot */ } csv_splayed_col_writer_t; +#define CSV_SPLAYED_LUT_BITS 19 +#define CSV_SPLAYED_LUT (1u << CSV_SPLAYED_LUT_BITS) static ray_err_t csv_splayed_writer_open(csv_splayed_col_writer_t* w, const char* dir, int64_t name_id, @@ -3314,9 +3581,25 @@ static ray_err_t csv_splayed_writer_open(csv_splayed_col_writer_t* w, if (!w->fp) return RAY_ERR_IO; ray_t zero = {0}; if (fwrite(&zero, 1, 32, w->fp) != 32) return RAY_ERR_IO; + if (type == RAY_SYM) { + /* best effort: without the cache every cell probes the domain */ + w->lut_id = (int64_t*)ray_alloc_raw((size_t)CSV_SPLAYED_LUT * sizeof(int64_t)); + w->lut_pos = (uint32_t*)ray_alloc_raw((size_t)CSV_SPLAYED_LUT * sizeof(uint32_t)); + if (!w->lut_id || !w->lut_pos) { + ray_free_raw(w->lut_id); ray_free_raw(w->lut_pos); + w->lut_id = NULL; w->lut_pos = NULL; + } else { + memset(w->lut_id, 0xff, (size_t)CSV_SPLAYED_LUT * sizeof(int64_t)); + } + } return RAY_OK; } +static void csv_splayed_writer_drop_lut(csv_splayed_col_writer_t* w) { + ray_free_raw(w->lut_id); ray_free_raw(w->lut_pos); + w->lut_id = NULL; w->lut_pos = NULL; +} + static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, ray_t* col) { if (!w->fp || !col || RAY_IS_ERR(col)) return RAY_ERR_TYPE; @@ -3328,18 +3611,40 @@ static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, * resolve each cell through the chunk vec's own domain and * find-or-append into the target (distinct work rides the * write). The domain is flushed before the column files are - * committed (close), preserving the sym-first crash ordering. */ + * committed (close), preserving the sym-first crash ordering. + * A runtime-domain chunk vec goes through the id -> position + * cache: only a value's first encounter pays the domain probe. */ + bool direct = ray_sym_vec_domain(col) == w->dom; + bool cached = ray_sym_vec_domain(col) == ray_sym_runtime_domain() && w->lut_id; + const void* cd = ray_data(col); uint32_t buf[8192]; for (int64_t off = 0; off < n; ) { int64_t cnt = n - off; if (cnt > (int64_t)(sizeof(buf) / sizeof(buf[0]))) cnt = (int64_t)(sizeof(buf) / sizeof(buf[0])); for (int64_t i = 0; i < cnt; i++) { - ray_t* s = ray_sym_vec_cell(col, off + i); - if (!s) return RAY_ERR_CORRUPT; - int64_t pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), - ray_str_len(s)); - if (pos < 0) return RAY_ERR_OOM; + int64_t pos; + if (direct) { + /* Already encoded over the target domain. */ + pos = ray_read_sym(cd, off + i, RAY_SYM, col->attrs); + } else if (cached) { + int64_t id = ray_read_sym(cd, off + i, RAY_SYM, col->attrs); + uint32_t slot = (uint32_t)(((uint64_t)id * 0x9E3779B97F4A7C15ull) >> (64 - CSV_SPLAYED_LUT_BITS)); + if (w->lut_id[slot] == id) { + pos = w->lut_pos[slot]; + } else { + ray_t* s = ray_sym_str(id); + if (!s) return RAY_ERR_CORRUPT; + pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), ray_str_len(s)); + if (pos < 0) return RAY_ERR_OOM; + w->lut_id[slot] = id; w->lut_pos[slot] = (uint32_t)pos; + } + } else { + ray_t* s = ray_sym_vec_cell(col, off + i); + if (!s) return RAY_ERR_CORRUPT; + pos = ray_sym_domain_intern(w->dom, ray_str_ptr(s), ray_str_len(s)); + if (pos < 0) return RAY_ERR_OOM; + } buf[i] = (uint32_t)pos; /* Position 0 of any symfile is the empty string (domain.c * enforces that reservation on open), so a re-encoded cell is @@ -3365,7 +3670,29 @@ static ray_err_t csv_splayed_writer_append(csv_splayed_col_writer_t* w, return RAY_OK; } +/* Append task: column `start` of the chunk table into its writer. The + * first failure is kept (a later task cannot clear it). */ +typedef struct { + csv_splayed_col_writer_t* writers; + ray_t* tbl; + int ncols; + _Atomic(ray_err_t) err; +} csv_splayed_append_ctx_t; + +static void csv_splayed_append_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + csv_splayed_append_ctx_t* a = (csv_splayed_append_ctx_t*)raw; + if (atomic_load_explicit(&a->err, memory_order_relaxed) != RAY_OK) return; + ray_t* col = ray_table_get_col_idx(a->tbl, (int64_t)start); + ray_err_t e = csv_splayed_writer_append(&a->writers[start], col); + if (e != RAY_OK) { + ray_err_t ok = RAY_OK; + atomic_compare_exchange_strong_explicit(&a->err, &ok, e, memory_order_relaxed, memory_order_relaxed); + } +} + static ray_err_t csv_splayed_writer_close(csv_splayed_col_writer_t* w) { + csv_splayed_writer_drop_lut(w); if (!w->fp) return RAY_OK; ray_err_t err = RAY_OK; @@ -3397,6 +3724,7 @@ static ray_err_t csv_splayed_writer_close(csv_splayed_col_writer_t* w) { } static void csv_splayed_writer_abort(csv_splayed_col_writer_t* w) { + csv_splayed_writer_drop_lut(w); if (w->fp) fclose(w->fp); w->fp = NULL; remove(w->tmp_path); @@ -3630,16 +3958,27 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool size_t chunk_offset = data_offset; bool wrote_any = false; + size_t avg_row_bytes = 64; /* refined from every chunk scanned */ while (chunk_offset < file_size || !wrote_any) { ray_t* row_offsets_hdr = NULL; int64_t* row_offsets = NULL; size_t next_offset = chunk_offset; int64_t cnt = 0; if (chunk_offset < file_size) { - cnt = build_row_offsets_limited(buf, file_size, chunk_offset, - rows_per_chunk, data_has_quotes, - &row_offsets, - &row_offsets_hdr, &next_offset); + /* Parallel scan over a byte window sized from the rows seen so + * far; the serial walk remains the fallback and the semantics. */ + cnt = build_row_offsets_window(buf, file_size, chunk_offset, + rows_per_chunk, avg_row_bytes, + data_has_quotes, + &row_offsets, &row_offsets_hdr, + &next_offset); + if (cnt == -2) + cnt = build_row_offsets_limited(buf, file_size, chunk_offset, + rows_per_chunk, data_has_quotes, + &row_offsets, + &row_offsets_hdr, &next_offset); + if (cnt > 0 && next_offset > chunk_offset) + avg_row_bytes = (next_offset - chunk_offset) / (size_t)cnt; if (cnt <= 0) { scratch_free(row_offsets_hdr); err = (cnt < 0) ? RAY_ERR_CANCEL : RAY_ERR_IO; @@ -3649,7 +3988,7 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool ray_t* tbl = csv_materialize_rows(buf, file_size, row_offsets, cnt, ncols, delimiter, col_name_ids, - resolved_types); + resolved_types, sym_dom); scratch_free(row_offsets_hdr); if (!tbl || RAY_IS_ERR(tbl)) { err = (tbl && RAY_IS_ERR(tbl)) ? ray_err_from_obj(tbl) @@ -3658,14 +3997,31 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool break; } - for (int c = 0; c < ncols; c++) { - ray_t* col = ray_table_get_col_idx(tbl, c); - err = csv_splayed_writer_append(&writers[c], col); - if (err != RAY_OK) break; + /* One task per column: each writer owns its file, its cache and + * its symfile domain (the domain probe takes the domain lock, the + * symbol table read its own), so the columns of a chunk are + * encoded and written side by side. */ + { + csv_splayed_append_ctx_t actx = { .writers = writers, .tbl = tbl, + .ncols = ncols, .err = RAY_OK }; + ray_pool_t* wpool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(wpool, ncols, 2)) + ray_pool_dispatch_n(wpool, csv_splayed_append_task, &actx, (uint32_t)ncols); + else + for (int c = 0; c < ncols; c++) csv_splayed_append_task(&actx, 0, c, c + 1); + err = actx.err; } ray_release(tbl); if (err != RAY_OK) break; wrote_any = true; + /* The chunk's bytes are done with: drop them from the mapping so a + * long file does not pin its whole length in resident memory. */ + if (next_offset > chunk_offset) { + size_t ps = (size_t)sysconf(_SC_PAGESIZE); + size_t a = (chunk_offset + ps - 1) & ~(ps - 1); + size_t b = next_offset & ~(ps - 1); + if (b > a) madvise((void*)(buf + a), b - a, MADV_DONTNEED); + } if (cnt == 0) break; chunk_offset = next_offset; } @@ -3673,8 +4029,9 @@ ray_err_t ray_csv_save_splayed_named_opts(const char* path, char delimiter, bool /* Flush the symfile BEFORE committing column files (writer_close * renames tmp → final): columns must never reference positions the * symfile doesn't persist (sym-first crash ordering). */ - if (err == RAY_OK && sym_dom) + if (err == RAY_OK && sym_dom) { err = ray_sym_domain_flush(sym_dom, false); + } for (int c = 0; c < ncols; c++) { ray_err_t cerr = (err == RAY_OK) ? csv_splayed_writer_close(&writers[c]) @@ -3886,7 +4243,8 @@ ray_err_t ray_csv_save_parted_named_opts(const char* path, char delimiter, bool } ray_t* tbl = csv_materialize_rows(buf, file_size, row_offsets, - cnt, ncols, delimiter, col_name_ids, resolved_types); + cnt, ncols, delimiter, col_name_ids, resolved_types, + NULL); if (!tbl || RAY_IS_ERR(tbl)) { err = (tbl && RAY_IS_ERR(tbl)) ? ray_err_from_obj(tbl) : RAY_ERR_OOM; diff --git a/src/mem/arena.c b/src/mem/arena.c index df44caf85..cc2e6e0b1 100644 --- a/src/mem/arena.c +++ b/src/mem/arena.c @@ -109,26 +109,54 @@ ray_t* ray_arena_alloc(ray_arena_t* arena, size_t nbytes) { return v; } -ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { +size_t ray_arena_str_bytes(size_t len) { + if (len < 7) return 32; + /* [U8 header (32) | data (len+1) | pad to 32 | STR header (32)] */ + return (((32 + len + 1) + 31) & ~(size_t)31) + 32; +} + +void* ray_arena_alloc_raw(ray_arena_t* arena, size_t nbytes) { + if (!arena) return NULL; + if (nbytes > SIZE_MAX - (ARENA_ALIGN - 1)) return NULL; + size_t block_size = ARENA_ALIGN_UP(nbytes); + ray_arena_chunk_t* c = arena->chunks; + if (c->used + block_size > c->cap) { + size_t new_cap = arena->chunk_size; + if (block_size > new_cap) new_cap = ARENA_ALIGN_UP(block_size); + ray_arena_chunk_t* nc = arena_new_chunk(new_cap); + if (!nc) return NULL; + nc->next = arena->chunks; + arena->chunks = nc; + c = nc; + } + void* p = chunk_data(c) + c->used; + c->used += block_size; + return p; +} + +ray_t* ray_arena_str_at(void* at, const char* s, size_t len) { if (len < 7) { - /* SSO: bytes inline in the header (ray_arena_alloc zeroes it and sets - * RAY_ATTR_ARENA + rc=1). */ - ray_t* v = ray_arena_alloc(arena, 0); - if (!v) return NULL; + /* SSO: bytes inline in the header. */ + ray_t* v = (ray_t*)at; + memset(v, 0, 32); + v->attrs = RAY_ATTR_ARENA; + ray_atomic_store(&v->rc, 1); v->type = -RAY_STR; v->slen = (uint8_t)len; if (len > 0) memcpy(v->sdata, s, len); v->sdata[len] = '\0'; return v; } - /* Long string: fused single allocation for the U8 data vec + the STR atom. + /* Long string: fused single block for the U8 data vec + the STR atom. * Layout: [U8 ray_t header (32) | data (len+1) | pad to 32 | STR header (32)]. - * One arena_alloc instead of two. 32-byte arena alignment keeps the atom's - * obj pointer low byte out of is_sso()'s 1..7 SSO range. */ + * 32-byte arena alignment keeps the atom's obj pointer low byte out of + * is_sso()'s 1..7 SSO range. */ size_t data_size = len + 1; size_t chars_block = ((32 + data_size) + 31) & ~(size_t)31; /* align up to 32 */ - ray_t* chars = ray_arena_alloc(arena, chars_block); - if (!chars) return NULL; + ray_t* chars = (ray_t*)at; + memset(chars, 0, 32); + chars->attrs = RAY_ATTR_ARENA; + ray_atomic_store(&chars->rc, 1); chars->type = RAY_U8; chars->len = (int64_t)len; memcpy(ray_data(chars), s, len); @@ -143,6 +171,12 @@ ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { return v; } +ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len) { + void* at = ray_arena_alloc_raw(arena, ray_arena_str_bytes(len)); + if (!at) return NULL; + return ray_arena_str_at(at, s, len); +} + bool ray_arena_reserve(ray_arena_t* arena, size_t bytes) { if (!arena) return false; if (bytes == 0) return true; diff --git a/src/mem/arena.h b/src/mem/arena.h index 1cce80eb8..d7861fd5d 100644 --- a/src/mem/arena.h +++ b/src/mem/arena.h @@ -45,6 +45,14 @@ ray_t* ray_arena_alloc(ray_arena_t* arena, size_t nbytes); * heap (used by the global sym table and FILE sym domains). NULL on OOM. */ ray_t* ray_arena_str(ray_arena_t* arena, const char* s, size_t len); +/* Bulk string construction: reserve one raw region for many atoms, then + * build each atom in place (ray_arena_str == alloc_raw + str_at). The + * region is arena memory: 32-byte aligned, released with the arena. Lets + * a caller carve one region per worker and build atoms in parallel. */ +size_t ray_arena_str_bytes(size_t len); /* bytes one atom needs */ +void* ray_arena_alloc_raw(ray_arena_t* arena, size_t nbytes); /* NULL on OOM */ +ray_t* ray_arena_str_at(void* at, const char* s, size_t len); /* at: ray_arena_str_bytes(len) */ + /* Ensure the arena can serve subsequent allocations totalling at least * `bytes` without the head chunk needing to grow. If the head chunk has * enough free space already, this is a no-op; otherwise a new chunk with diff --git a/src/mem/heap.c b/src/mem/heap.c index 452440569..02e41fe55 100644 --- a/src/mem/heap.c +++ b/src/mem/heap.c @@ -1683,8 +1683,11 @@ void ray_free(ray_t* v) { * pool before appending — has to ask the registry, and only a * string column was ever registered. So the lock stays off the * ordinary free entirely, and a mutated column pays it once. */ + /* A column loaded with an inline index registered its region too + * (col.c): the index may have been detached since, and only the + * descriptor still knows the mapped length. */ ray_file_map_t* m = col_map; - if (!m && v->type == RAY_STR) m = ray_file_map_lookup(v); + if (!m) m = ray_file_map_lookup(v); if (m) { ray_file_map_release(m); if (h) RAY_STAT(h->stats.free_count++); @@ -2942,9 +2945,14 @@ void ray_parallel_end(void) { * else is the page-rounded payload plus an inline passenger index. */ static size_t mapped_block_bytes(const ray_t* v) { if (v->type == RAY_TABLE || v->type == RAY_DICT || v->type == RAY_LIST) return 0; - if (v->type == RAY_STR) { + { + /* Same route as ray_free: the pool's descriptor for a string + * column, else the registry — which also holds every indexed + * mapped column (col.c), whether or not its index is still + * attached. */ ray_file_map_t* m = NULL; - if (v->str_pool && !RAY_IS_ERR(v->str_pool) && v->str_pool->mmod == 3) + if (v->type == RAY_STR && v->str_pool && !RAY_IS_ERR(v->str_pool) && + v->str_pool->mmod == 3) m = v->str_pool->file_map; if (!m) m = ray_file_map_lookup(v); if (m) return m->len; diff --git a/src/ops/agg.c b/src/ops/agg.c index 51b0138d1..fe7d7f9d4 100644 --- a/src/ops/agg.c +++ b/src/ops/agg.c @@ -169,12 +169,14 @@ static ray_t* agg_parted_avg(ray_t* x) { if (!agg_parted_numeric_base(base)) return ray_error("type", "avg expects a numeric or temporal parted column, got %s", ray_type_name(base)); ray_t** segs = (ray_t**)ray_data(x); double sum = 0.0; + int64_t hi = 0; uint64_t lo = 0; /* integer segments: exact 128-bit sum */ int64_t cnt = 0; + bool fp = base == RAY_F64 || base == RAY_F32; for (int64_t s = 0; s < x->len; s++) { ray_t* seg = segs[s]; if (!seg) continue; int has_nulls = ray_vec_may_have_nulls(seg); - if (base == RAY_F64 || base == RAY_F32) { + if (fp) { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; if (base == RAY_F64) sum += ((double*)ray_data(seg))[i]; @@ -184,12 +186,12 @@ static ray_t* agg_parted_avg(ray_t* x) { } else { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; - sum += (double)agg_read_i64(seg, i); cnt++; + ray_i128_add(&hi, &lo, agg_read_i64(seg, i)); cnt++; } } } if (cnt == 0) return ray_typed_null(-RAY_F64); - return make_f64(sum / (double)cnt); + return make_f64((fp ? sum : ray_i128_to_f64(hi, lo)) / (double)cnt); } static ray_t* agg_parted_prod(ray_t* x) { @@ -404,6 +406,66 @@ static ray_t* agg_pair_vec(ray_t* x, ray_t* y, uint16_t op) { return ray_f64(ray_f64_fin(num / sqrt(dx * dy))); } +/* Per-chunk aggregates of an integer column's chunk-zone index, or NULL: + * [sum low words | non-null counts | sum high words] (the last block only + * in the current layout). */ +static const int64_t* zone_aggs(ray_t* x, uint32_t* n_out, bool* have_hi) { + if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return NULL; + ray_index_t* ix = ray_index_payload(x->index); + if (ix->built_for_len != x->len || ix->u.chunk_zone.is_f64 || !ix->u.chunk_zone.aggs) + return NULL; + uint32_t n = ix->u.chunk_zone.n_chunks; + int64_t len = ix->u.chunk_zone.aggs->len; + if (len != 2 * (int64_t)n && len != 3 * (int64_t)n) return NULL; + *n_out = n; + *have_hi = len == 3 * (int64_t)n; + return (const int64_t*)ray_data(ix->u.chunk_zone.aggs); +} + +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; + uint64_t sum = 0; + int64_t nn = 0; + for (uint32_t g = 0; g < n; g++) { + sum += (uint64_t)ag[g]; + nn += ag[n + g]; + } + *sum_out = (int64_t)sum; + *nn_out = nn; + return true; +} + +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; + ray_index_t* ix = ray_index_payload(x->index); + const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); + const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + int64_t hi = 0; uint64_t lo = 0; int64_t nn = 0; + double bound = 0.0; + for (uint32_t g = 0; g < n; g++) { + int64_t cnt = ag[n + g]; + nn += cnt; + if (have_hi) { + ray_i128_add128(&hi, &lo, ag[2 * (int64_t)n + g], (uint64_t)ag[g]); + } else { + /* no high words stored: the wrapped low word is the chunk's + * exact sum only when no partial sum could leave int64 */ + if (cnt > 0) { + double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); + bound += (a > b ? a : b) * (double)cnt; + } + ray_i128_add(&hi, &lo, ag[g]); + } + } + if (!have_hi && !(bound < 9223372036854775808.0)) return false; /* 2^63 */ + *hi_out = hi; *lo_out = lo; *nn_out = nn; + return true; +} + ray_t* ray_sum_fn(ray_t* x) { if (ray_is_lazy(x)) return ray_lazy_append(x, OP_SUM); if (RAY_IS_PARTED(x->type)) return agg_parted_sum(x); @@ -417,6 +479,11 @@ ray_t* ray_sum_fn(ray_t* x) { /* Canonical admission: numeric + TIME (duration); DATE/TIMESTAMP are * absolute points and SYM/STR/GUID are non-numeric → type error. */ if (!agg_type_admitted(OP_SUM, x->type)) return ray_error("type", "sum expects a numeric or time-duration vector, got %s", ray_type_name(x->type)); + /* Integer columns with per-chunk sums in their zone index. */ + if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { + int64_t zs, zn; + if (ray_zone_int_sum(x, &zs, &zn)) return make_i64(zs); + } /* Narrow/temporal types need specific return constructors that the * DAG executor doesn't provide — use scalar path for these. TIMESTAMP * is rejected by agg_type_admitted() above, so only the duration-like @@ -570,6 +637,16 @@ ray_t* ray_avg_fn(ray_t* x) { /* Canonical admission: numeric + temporal (→ F64); SYM/STR/GUID are * non-numeric → type error (the DAG path otherwise averaged raw ids). */ if (!agg_type_admitted(OP_AVG, x->type)) return ray_error("type", "avg expects a numeric or temporal vector, got %s", ray_type_name(x->type)); + /* Integer columns with per-chunk sums: the exact 128-bit total is + * what the row-wise reduction computes too, so the answers agree + * bit for bit. */ + if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { + int64_t zh, zn; uint64_t zl; + if (ray_zone_int_sum128(x, &zh, &zl, &zn)) { + if (zn == 0) return ray_typed_null(-RAY_F64); + return make_f64(ray_i128_to_f64(zh, zl) / (double)zn); + } + } AGG_VEC_VIA_DAG(x, ray_avg); } if (!is_list(x)) return ray_error("type", "avg expects a numeric vector, atom, or list, got %s", ray_type_name(x->type)); @@ -597,7 +674,7 @@ ray_t* ray_min_fn(ray_t* x) { * (mutation paths call ray_index_drop). */ if (ray_index_kind(x) == RAY_IDX_CHUNK_ZONE) { ray_index_t* ix = ray_index_payload(x->index); - if (ix->built_for_len == x->len) { + if (ix->built_for_len == x->len && ix->u.chunk_zone.mins) { uint32_t n_chunks = ix->u.chunk_zone.n_chunks; if (ix->u.chunk_zone.is_f64) { const double* mins = (const double*)ray_data(ix->u.chunk_zone.mins); @@ -611,7 +688,12 @@ ray_t* ray_min_fn(ray_t* x) { int64_t mn = INT64_MAX; for (uint32_t g = 0; g < n_chunks; g++) if (mins[g] < mn) mn = mins[g]; - if (mn == INT64_MAX) return ray_typed_null(-x->type); + /* All-null is what the non-null counts say when the + * zone has them; the sentinel alone cannot tell a + * column of INT64_MAX values from an empty one. */ + int64_t zs_, zn_; + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); + if (have_nn ? zn_ == 0 : mn == INT64_MAX) return ray_typed_null(-x->type); /* Preserve the column's storage width on the result. */ switch (x->type) { case RAY_BOOL: return ray_bool((bool)mn); @@ -652,7 +734,7 @@ ray_t* ray_max_fn(ray_t* x) { if (ray_is_vec(x)) { if (ray_index_kind(x) == RAY_IDX_CHUNK_ZONE) { ray_index_t* ix = ray_index_payload(x->index); - if (ix->built_for_len == x->len) { + if (ix->built_for_len == x->len && ix->u.chunk_zone.maxs) { uint32_t n_chunks = ix->u.chunk_zone.n_chunks; if (ix->u.chunk_zone.is_f64) { const double* maxs = (const double*)ray_data(ix->u.chunk_zone.maxs); @@ -666,7 +748,12 @@ ray_t* ray_max_fn(ray_t* x) { int64_t mx = INT64_MIN; for (uint32_t g = 0; g < n_chunks; g++) if (maxs[g] > mx) mx = maxs[g]; - if (mx == INT64_MIN) return ray_typed_null(-x->type); + /* All-null is what the non-null counts say when the + * zone has them; the sentinel alone cannot tell a + * column of INT64_MIN values from an empty one. */ + int64_t zs_, zn_; + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); + if (have_nn ? zn_ == 0 : mx == INT64_MIN) return ray_typed_null(-x->type); switch (x->type) { case RAY_BOOL: return ray_bool((bool)mx); case RAY_U8: return ray_u8((uint8_t)mx); diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index 9f19eb1ab..984faff60 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -184,6 +184,14 @@ agg_v2_reason_t agg_v2_admission(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { const agg_vtable_t* vt = agg_resolve(ext->agg_ops[a], ic->type); if (!vt) return AGG_V2_AGG_TYPE; if (vt->kind != ACC_STREAMING) return AGG_V2_BUFFERED; + /* The streaming 64-bit integer avg packs its 128-bit high word with + * the group count in one word (agg_stream.c): exact while every + * group has fewer than 2^32 rows; the narrower kernels keep a plain + * int64 sum, exact below 2^31 rows. A table that large keeps the + * legacy engines, whose accumulators carry a full high word. */ + if (ext->agg_ops[a] == OP_AVG && ic->type != RAY_F64 && ic->type != RAY_F32 && + ray_table_nrows(tbl) >= ((int64_t)1 << ((ic->type == RAY_I64 || ic->type == RAY_TIMESTAMP) ? 32 : 31))) + return AGG_V2_AGG_TYPE; } return AGG_V2_ADMITTED; } @@ -4645,15 +4653,34 @@ static int64_t agg_index_winner(void* raw, const int64_t* rows, int64_t count) { #undef INDEX_FIRST_VALID } int64_t best = -1; + /* A FILE domain's entries are compared off the mapped vocabulary: no + * atom per row — the lazily built atoms of a wide vocabulary cost far + * more than the compare and stay for the domain's lifetime. Positions + * past the file prefix, and the runtime domain, go through the atoms. */ + struct ray_sym_domain_s* dom = ray_sym_vec_domain(c->src); + ray_sym_domain_raw_t vocab; + bool vocab_ok = ray_sym_domain_raw_pin(dom, &vocab); for (int64_t j = 0; j < count; j++) { if ((j & 65535) == 0 && ray_interrupted()) return -1; int64_t r = rows[c->kind == OP_LAST ? count - 1 - j : j]; if (ray_vec_is_null(c->src, r)) continue; if (best < 0) best = r; if (c->kind == OP_FIRST || c->kind == OP_LAST) break; - ray_t* x = ray_group_sym_read(&c->symbols, ray_sym_vec_domain(c->src), ray_read_sym(data, r, c->src->type, c->src->attrs)); - ray_t* y = ray_group_sym_read(&c->symbols, ray_sym_vec_domain(c->src), ray_read_sym(data, best, c->src->type, c->src->attrs)); - int cmp = ray_str_cmp(x, y); + int64_t ir = ray_read_sym(data, r, c->src->type, c->src->attrs); + int64_t ib = ray_read_sym(data, best, c->src->type, c->src->attrs); + int cmp; + if (vocab_ok && ir >= 0 && ib >= 0 && ir < vocab.count && ib < vocab.count) { + size_t lr, lb; + const char* pr = ray_sym_domain_raw_str(&vocab, ir, &lr); + const char* pb = ray_sym_domain_raw_str(&vocab, ib, &lb); + size_t m = lr < lb ? lr : lb; + cmp = m ? memcmp(pr, pb, m) : 0; + if (cmp == 0) cmp = (lr > lb) - (lr < lb); + } else { + ray_t* x = ray_group_sym_read(&c->symbols, dom, ir); + ray_t* y = ray_group_sym_read(&c->symbols, dom, ib); + cmp = ray_str_cmp(x, y); + } if (c->kind == OP_MIN ? cmp < 0 : cmp > 0) best = r; } return best; diff --git a/src/ops/agg_stream.c b/src/ops/agg_stream.c index 8912d1860..083b5a3c9 100644 --- a/src/ops/agg_stream.c +++ b/src/ops/agg_stream.c @@ -5,6 +5,7 @@ #include "ops/ops.h" #include "ops/internal.h" /* ray_f64_fin (single-null float model) */ #include "lang/internal.h" /* ray_median_dbl_inplace */ +#include "ops/idxop.h" /* ray_i128_add / ray_i128_to_f64: exact integer avg */ #include #include #include /* realloc/free for the buffered median accumulator */ @@ -311,6 +312,80 @@ static const agg_vtable_t AVG_F64 = { .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, }; +/* ---- avg, integer inputs: exact 128-bit sum --------------------------- + * Every integer/temporal avg kernel below keeps the exact sum and divides + * ray_i128_to_f64(hi, lo) by the count — the same bits as the keyless + * reduction, the legacy group engines and the chunk-zone metadata, for any + * column (a double running sum loses low bits past 2^53 and an int64 one + * wraps). + * + * The state stays 16 bytes (the dense plans budget slot traffic by state + * size): `lo` is a signed int64 running sum and `hc` packs, as hi:32 | + * cnt:32, the number of times that sum wrapped past +/-2^63 with the row + * count — the total is wraps * 2^64 + lo. A signed-overflow test per row + * is a never-taken branch on ordinary data, cheaper than a carry chain. + * Exact for any int64 values while a group holds fewer than 2^32 rows + * (|sum| < 2^95, so the wrap count fits 32 signed bits); agg_v2_admission + * keeps tables of 2^32 rows or more off the v2 engine when an integer avg + * is present. */ +typedef struct { int64_t lo; uint64_t hc; } avg_i128_state; +#define AVG_I128_WRAPS(hc) ((int64_t)(int32_t)(uint32_t)((hc) >> 32)) +#define AVG_I128_CNT(hc) ((int64_t)((hc) & 0xffffffffu)) +#define AVG_I128_ONE_WRAP ((uint64_t)1 << 32) +static void avg_i128_init(void* s) { + avg_i128_state* st = (avg_i128_state*)s; st->lo = 0; st->hc = 0; +} +static inline void avg_i128_add(avg_i128_state* st, int64_t v) { + int64_t o = st->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + st->lo = nw; + st->hc += 1u; + /* signed overflow: o and v share a sign the result lacks */ + if (RAY_UNLIKELY(((o ^ nw) & (v ^ nw)) < 0)) + st->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static void avg_i128_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; avg_i128_state* dd = (avg_i128_state*)d; const avg_i128_state* ss = (const avg_i128_state*)s; + int64_t o = dd->lo, v = ss->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + dd->lo = nw; + dd->hc += ss->hc; + if (((o ^ nw) & (v ^ nw)) < 0) + dd->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static double avg_i128_final_result(const void* s) { + const avg_i128_state* st = s; + int64_t cnt = AVG_I128_CNT(st->hc); + if (!cnt) return NULL_F64; + /* two's-complement (hi, lo): the signed low word borrows one from the + * wrap count when negative */ + int64_t hi = AVG_I128_WRAPS(st->hc) + (st->lo < 0 ? -1 : 0); + return ray_f64_fin(ray_i128_to_f64(hi, (uint64_t)st->lo) / (double)cnt); +} +AGG_SCALAR_FINAL(avg_i128_final, double, ray_f64, value != value) +#define AVG_I128_UPDATE_BODY \ + avg_i128_add((avg_i128_state*)((char*)base + (size_t)gids[i]*stride), (int64_t)d[i]) + +/* ---- avg, inputs of at most 32 bits: exact int64 sum ------------------- + * Fewer than 2^31 values of at most 32 bits sum to less than 2^63, so a + * plain int64 running sum is exact (agg_v2_admission keeps larger tables + * off v2 for these kernels) and (double)sum / cnt is bit for bit what the + * 128-bit form would give. */ +typedef struct { int64_t sum; int64_t cnt; } avg_i64_state; +static void avg_i64_init(void* s) { ((avg_i64_state*)s)->sum = 0; ((avg_i64_state*)s)->cnt = 0; } +static void avg_i64_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; ((avg_i64_state*)d)->sum += ((const avg_i64_state*)s)->sum; + ((avg_i64_state*)d)->cnt += ((const avg_i64_state*)s)->cnt; +} +static double avg_i64_final_result(const void* s) { + const avg_i64_state* st = s; + return st->cnt ? ray_f64_fin((double)st->sum / (double)st->cnt) : NULL_F64; +} +AGG_SCALAR_FINAL(avg_i64_final, double, ray_f64, value != value) +#define AVG_I64_UPDATE_BODY \ + avg_i64_state* st = (avg_i64_state*)((char*)base + (size_t)gids[i]*stride); \ + st->sum += (int64_t)d[i]; st->cnt++ + /* ---- variance family, I64 (sumsq as int64 unsigned-wrap; formula group.c:2190) -- */ /* Shifted-data accumulator: sums are of (v - k), where k is the first * value this state sees. The textbook one-pass form sumsq/n - mean^2 @@ -797,15 +872,14 @@ static void avg_bool_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_BOOL_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_bool_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_bool_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_bool_native_update(void* base, size_t stride, const uint32_t* gids, @@ -910,15 +984,14 @@ static void avg_u8_native_update(void* base, size_t stride, const uint32_t* gids int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_U8_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_u8_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_u8_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_u8_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1023,15 +1096,14 @@ static void avg_i16_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int16_t* d = (const int16_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I16_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i16_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i16_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i16_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1136,15 +1208,14 @@ static void avg_i32_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I32_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i32_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i32_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1203,15 +1274,14 @@ static void avg_i64_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_I64_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i64_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_i64_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void min_f32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1363,15 +1433,14 @@ static void avg_date_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_DATE_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_date_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_date_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_date_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1459,15 +1528,14 @@ static void avg_time_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIME_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_time_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_time_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_time_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1575,15 +1643,14 @@ static void avg_timestamp_native_update(void* base, size_t stride, const uint32_ int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIMESTAMP_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_timestamp_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_timestamp_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void var_timestamp_native_update(void* base, size_t stride, const uint32_t* gids, diff --git a/src/ops/collection.c b/src/ops/collection.c index 6b20c4b24..d06186240 100644 --- a/src/ops/collection.c +++ b/src/ops/collection.c @@ -65,13 +65,60 @@ typedef struct hashset_t { int8_t src_type; /* ray_t.type */ bool src_has_nulls; void* src_data; /* pointer to typed data (or RAY_LIST elements) */ + /* Int-family cells hash through f64 (hs_hash_row). Off until a probe + * crosses numeric classes — see hs_needs_num_f64. */ + bool num_f64; + int8_t probe_type; /* last probe type vetted by hashset_adopt_probe */ } hashset_t; +/* Numeric class of a typed-vec type, matching is_numeric/atom_eq: every + * int width compares to every float width through f64. Temporal types are + * not numeric there (a DATE never equals an I32), so they stay class 0. */ +enum { HS_NUM_NONE = 0, HS_NUM_INT = 1, HS_NUM_FLT = 2 }; +static inline int hs_num_class(int8_t t) { + switch (t) { + case RAY_BOOL: case RAY_U8: case RAY_I16: case RAY_I32: case RAY_I64: + return HS_NUM_INT; + case RAY_F32: case RAY_F64: + return HS_NUM_FLT; + default: + return HS_NUM_NONE; + } +} + +/* Cell i of a numeric typed vec as f64 — as_f64 on the boxed atom. */ +static inline double hs_cell_f64(int8_t t, const void* data, int64_t i) { + switch (t) { + case RAY_BOOL: return (double)((const bool*)data)[i]; + case RAY_U8: return (double)((const uint8_t*)data)[i]; + case RAY_I16: return (double)((const int16_t*)data)[i]; + case RAY_I32: return (double)((const int32_t*)data)[i]; + case RAY_I64: return (double)((const int64_t*)data)[i]; + case RAY_F32: return (double)((const float*)data)[i]; + default: return ((const double*)data)[i]; + } +} + +/* Int cells hash with ray_hash_i64 and float cells (typed, or numeric atoms + * in a RAY_LIST) with ray_hash_f64, so 3 and 3.0 land in different buckets + * and hs_eq_rows never sees them (#645). An int side meeting a float or + * list side must hash its ints through f64 as well. */ +static inline bool hs_needs_num_f64(int8_t a, int8_t b) { + int ca = hs_num_class(a), cb = hs_num_class(b); + if (ca == HS_NUM_INT) return cb == HS_NUM_FLT || b == RAY_LIST; + if (cb == HS_NUM_INT) return ca == HS_NUM_FLT || a == RAY_LIST; + return false; +} + /* Hash a single row at index i in src. Mirrors atom_eq's coercion * rules: numeric types normalize through f64 so an I64 atom and an - * F64 atom holding the same value collide (boxed-list path only — a - * typed vec is homogeneous, so the dispatch picks one branch). */ -static uint64_t hs_hash_row(ray_t* src, int64_t i, int8_t t, void* data) { + * F64 atom holding the same value collide. A typed int vec hashes + * through i64 unless num_f64 is set — the set is being probed from the + * other numeric class (hs_needs_num_f64). */ +static uint64_t hs_hash_row(ray_t* src, int64_t i, int8_t t, void* data, + bool num_f64) { + if (num_f64 && hs_num_class(t) == HS_NUM_INT) + return ray_hash_f64(hs_cell_f64(t, data, i)); switch (t) { case RAY_I64: return ray_hash_i64(((const int64_t*)data)[i]); case RAY_I32: return ray_hash_i64((int64_t)((const int32_t*)data)[i]); @@ -169,6 +216,10 @@ static int hs_eq_rows(ray_t* a_src, int64_t ai, int8_t at, void* a_data, } } } + /* Mixed numeric typed vecs: atom_eq compares any two numerics + * through f64 — do the same without boxing a pair per probe. */ + if (hs_num_class(at) != HS_NUM_NONE && hs_num_class(bt) != HS_NUM_NONE) + return hs_cell_f64(at, a_data, ai) == hs_cell_f64(bt, b_data, bi); /* Fall back to atom_eq via boxed values. Used for cross-type * comparisons (e.g. except over typed I64 vs F64 vec) and the * RAY_LIST path. collection_elem allocates a temporary atom for @@ -208,6 +259,8 @@ static bool hashset_init(hashset_t* hs, ray_t* src, int64_t hint) { hs->src_type = src ? src->type : 0; hs->src_has_nulls = src ? ray_vec_may_have_nulls(src) : false; hs->src_data = src ? ray_data(src) : NULL; + hs->num_f64 = false; + hs->probe_type = hs->src_type; return true; } @@ -216,11 +269,11 @@ static void hashset_destroy(hashset_t* hs) { hs->slots = NULL; } -static bool hashset_grow(hashset_t* hs) { +/* Re-slot every stored row into a fresh table of new_cap, hashing with + * the set's current num_f64 mode. */ +static bool hashset_rehash(hashset_t* hs, int64_t new_cap) { int64_t old_cap = hs->cap; int64_t* old_slots = hs->slots; - int64_t new_cap = old_cap * 2; - if (new_cap < old_cap) return false; ray_t* nb = ray_alloc((size_t)new_cap * sizeof(int64_t)); if (!nb || RAY_IS_ERR(nb)) return false; int64_t* ns = (int64_t*)ray_data(nb); @@ -229,7 +282,8 @@ static bool hashset_grow(hashset_t* hs) { for (int64_t i = 0; i < old_cap; i++) { int64_t ridx = old_slots[i]; if (ridx == HS_EMPTY) continue; - uint64_t h = hs_hash_row(hs->src, ridx, hs->src_type, hs->src_data); + uint64_t h = hs_hash_row(hs->src, ridx, hs->src_type, hs->src_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)mask); while (ns[s] != HS_EMPTY) s = (s + 1) & mask; ns[s] = ridx; @@ -242,8 +296,50 @@ static bool hashset_grow(hashset_t* hs) { return true; } +static bool hashset_grow(hashset_t* hs) { + int64_t new_cap = hs->cap * 2; + if (new_cap < hs->cap) return false; + return hashset_rehash(hs, new_cap); +} + /* Probe the set for the row (probe_src, probe_i). Returns the stored * row index from the build-side vec on hit, HS_EMPTY on miss. */ +/* A probe of a type the set has not seen yet. From the other numeric + * class, rehash once so int cells hash through f64 — later probes and + * inserts keep the mode, and same-type callers never get here. Returns + * false on OOM: the slots keep their old layout, the type stays unvetted + * so the next probe retries, and the caller answers by hashset_scan. */ +static __attribute__((noinline, cold)) bool +hashset_adopt_probe(hashset_t* hs, int8_t probe_type) { + if (!hs->num_f64 && hs_needs_num_f64(hs->src_type, probe_type)) { + hs->num_f64 = true; + /* Only int cells change hash in this mode: a float- or list-built + * set already sits in the f64 layout, so it needs no rehash. */ + if (hs_num_class(hs->src_type) == HS_NUM_INT && + !hashset_rehash(hs, hs->cap)) { + hs->num_f64 = false; + return false; + } + } + hs->probe_type = probe_type; + return true; +} + +/* Hash-free probe for when the table could not be rehashed: compare the + * row against every stored one. */ +static __attribute__((noinline, cold)) int64_t +hashset_scan(hashset_t* hs, ray_t* probe_src, int64_t probe_i, + int8_t probe_type, void* probe_data) { + for (int64_t k = 0; k < hs->cap; k++) { + int64_t stored = hs->slots[k]; + if (stored != HS_EMPTY && + hs_eq_rows(probe_src, probe_i, probe_type, probe_data, + hs->src, stored, hs->src_type, hs->src_data)) + return stored; + } + return HS_EMPTY; +} + static int64_t hashset_find_xrow(hashset_t* hs, ray_t* probe_src, int64_t probe_i, int8_t probe_type, void* probe_data) { if (hs_row_is_null(probe_src, probe_i, probe_data)) @@ -277,7 +373,11 @@ static int64_t hashset_find_xrow(hashset_t* hs, ray_t* probe_src, int64_t probe_ return HS_EMPTY; } } - uint64_t h = hs_hash_row(probe_src, probe_i, probe_type, probe_data); + if (probe_type != hs->probe_type && + !hashset_adopt_probe(hs, probe_type)) + return hashset_scan(hs, probe_src, probe_i, probe_type, probe_data); + uint64_t h = hs_hash_row(probe_src, probe_i, probe_type, probe_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)hs->mask); while (hs->slots[s] != HS_EMPTY) { int64_t stored = hs->slots[s]; @@ -366,7 +466,8 @@ static bool hashset_insert(hashset_t* hs, int64_t i) { if (hs->count * 2 >= hs->cap) { if (!hashset_grow(hs)) { /* fall through, may degrade */ } } - uint64_t h = hs_hash_row(hs->src, i, hs->src_type, hs->src_data); + uint64_t h = hs_hash_row(hs->src, i, hs->src_type, hs->src_data, + hs->num_f64); int64_t s = (int64_t)(h & (uint64_t)hs->mask); while (hs->slots[s] != HS_EMPTY) { int64_t stored = hs->slots[s]; @@ -1385,10 +1486,18 @@ ray_t* ray_in_fn(ray_t* val, ray_t* vec) { * on a 351k-row column (#593). vec.h says as much where it * defines the two — "paths requiring null-free data use has_nulls * below". The exact check costs one pass and is not on a per-row - * path; a null-bearing operand still falls through, because the - * kernel's null-matches-nothing semantics differ from this path's - * null-equals-null. */ - if (!ray_vec_has_nulls(val) && !ray_vec_has_nulls(vec)) { + * path. + * + * Only BOTH sides null-bearing must fall through. The kernel's + * null-matches-nothing and this path's null-equals-null disagree + * on exactly one question — does a null row match a null set + * element — and that needs a null on each side. A null row + * against a null-free set, or a null set element against a + * null-free column, is false under both. Requiring both sides + * null-free sent any nullable column to the hashset: one null + * row cost ~25x. The set is tested first: it is usually the + * small side, and a null-free set skips the column scan. */ + if (!(ray_vec_has_nulls(vec) && ray_vec_has_nulls(val))) { ray_t* fast = ray_in_vec_exec(val, vec, false); if (fast) return fast; } diff --git a/src/ops/exec.c b/src/ops/exec.c index de07dbbe4..daa2fe30c 100644 --- a/src/ops/exec.c +++ b/src/ops/exec.c @@ -27,6 +27,7 @@ #include "ops/rowsel.h" #include "ops/fused_group.h" #include "ops/idxop.h" +#include "ops/hash.h" #include "mem/heap.h" #include "mem/sys.h" #include "core/qstats.h" /* per-worker parallelism stats for profile spans */ @@ -611,6 +612,16 @@ typedef struct { * per-row linear set scan into one byte load — any set size, width- * specialized, vectorizable. NULL when not applicable. */ const uint8_t* symlut; + /* Needle hash, built when the set outgrows the SIMD small-set path + * (IN_SIMD_SET): open addressing over the live probe values, so a + * row costs one probe instead of a scan of the whole set. hti holds + * int64 values (empty = INT64_MIN; a real INT64_MIN needle sets + * ht_has_min), htf holds f64 bit patterns (empty = IN_HTF_EMPTY, a + * NaN, which is never inserted). Both NULL = linear scan. */ + const int64_t* hti; + const uint64_t* htf; + int64_t ht_mask; + bool ht_has_min; bool col_has_nulls; bool col_atom_null; bool col_is_atom; @@ -618,6 +629,51 @@ typedef struct { bool negate; } in_worker_ctx_t; +/* Sets up to this many live elements take the unrolled SIMD compare in + * exec_in_worker; larger ones get the needle hash (in_build_worker_ctx). */ +#define IN_SIMD_SET 8 +#define IN_HTF_EMPTY UINT64_C(0x7ff8000000000000) + +/* -0.0 and +0.0 compare equal, so they must share a slot. */ +static inline uint64_t in_f64_key(double v) { + uint64_t b; + memcpy(&b, &v, sizeof(b)); + return b == UINT64_C(0x8000000000000000) ? 0 : b; +} + +static inline int in_set_has_i(const in_worker_ctx_t* c, int64_t v) { + if (c->hti) { + if (v == INT64_MIN) return c->ht_has_min; + int64_t s = (int64_t)(ray_hash_i64(v) & (uint64_t)c->ht_mask); + for (;;) { + int64_t k = c->hti[s]; + if (k == v) return 1; + if (k == INT64_MIN) return 0; + s = (s + 1) & c->ht_mask; + } + } + for (int64_t j = 0; j < c->sv_len; j++) + if (v == c->svi[j]) return 1; + return 0; +} + +static inline int in_set_has_f(const in_worker_ctx_t* c, double v) { + if (c->htf) { + if (v != v) return 0; /* NaN equals nothing, as in the scan */ + uint64_t b = in_f64_key(v); + int64_t s = (int64_t)(ray_hash_i64((int64_t)b) & (uint64_t)c->ht_mask); + for (;;) { + uint64_t k = c->htf[s]; + if (k == b) return 1; + if (k == IN_HTF_EMPTY) return 0; + s = (s + 1) & c->ht_mask; + } + } + for (int64_t j = 0; j < c->sv_len; j++) + if (v == c->svf[j]) return 1; + return 0; +} + static void exec_in_worker(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { (void)worker_id; @@ -705,7 +761,7 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, return; } - if (!c->col_is_atom && !c->use_double && sv_len >= 1 && sv_len <= 8) { + if (!c->col_is_atom && !c->use_double && sv_len >= 1 && sv_len <= IN_SIMD_SET) { const int64_t* svi = c->svi; uint8_t neg = (uint8_t)negate; #define IN_FAST(CTYPE, FITS, SENT, HASNULL) do { \ @@ -765,7 +821,6 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, } if (c->use_double) { - const double* svf = c->svf; if (c->col_atom_null) { /* All elements are null — fill zeros */ for (int64_t i = start; i < end; i++) ob[i - ob_base] = 0; @@ -775,24 +830,17 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, double cv; if (c->col_is_atom) cv = (ct == RAY_F64) ? col->f64 : (double)col->i64; else IN_READ_F64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svf[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_f(c, cv) ^ negate); } } else { for (int64_t i = start; i < end; i++) { double cv; if (c->col_is_atom) cv = (ct == RAY_F64) ? col->f64 : (double)col->i64; else IN_READ_F64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svf[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_f(c, cv) ^ negate); } } } else { - const int64_t* svi = c->svi; if (c->col_atom_null) { for (int64_t i = start; i < end; i++) ob[i - ob_base] = 0; } else if (vec_has_nulls) { @@ -801,20 +849,14 @@ static void exec_in_worker(void* vctx, uint32_t worker_id, int64_t cv; if (c->col_is_atom) cv = col->i64; else IN_READ_I64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svi[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_i(c, cv) ^ negate); } } else { for (int64_t i = start; i < end; i++) { int64_t cv; if (c->col_is_atom) cv = col->i64; else IN_READ_I64(cv, i); - int found = 0; - for (int64_t j = 0; j < sv_len; j++) - if (cv == svi[j]) { found = 1; break; } - ob[i - ob_base] = (uint8_t)(found ^ negate); + ob[i - ob_base] = (uint8_t)(in_set_has_i(c, cv) ^ negate); } } } @@ -885,6 +927,17 @@ static in_ctx_status_t in_build_worker_ctx(ray_t* col, ray_t* set, bool negate, int col_class = CLASSIFY(ct); int set_class = CLASSIFY(st); + /* A temporal value equals only its own type: atom_eq — and so the + * hashset behind find/except and the null-bearing `in` — never matches + * a DATE to an int or to a TIMESTAMP. Comparing raw payloads here + * matched day 0 to 0i, and a DATE's day count to a TIMESTAMP's + * nanoseconds. Empty probe, exactly like SYM vs non-SYM below. */ + #define IS_TEMPORAL(t) \ + ((t) == RAY_DATE || (t) == RAY_TIME || (t) == RAY_TIMESTAMP) + if ((IS_TEMPORAL(ct) || IS_TEMPORAL(st)) && ct != st) + set_len = 0; + #undef IS_TEMPORAL + /* Mixed SYM vs non-SYM → treat as an empty probe. A SYM set * containing resolved sym IDs has no meaning when compared to a * raw integer column, so nothing can match — but we still drop @@ -1043,10 +1096,52 @@ static in_ctx_status_t in_build_worker_ctx(ray_t* col, ray_t* set, bool negate, } } + /* Large set on a non-LUT column: hash the live needles once so each row + * costs one probe. The linear scan was O(rows × needles) — a nullable + * I64 column against 100k needles took over a second. An allocation + * failure keeps the scan: slower, same answer. */ + const int64_t* hti = NULL; + const uint64_t* htf = NULL; + int64_t ht_mask = 0; + bool ht_has_min = false; + if (!ray_is_atom(col) && !symlut && sv_len > IN_SIMD_SET) { + int64_t cap = 16; + while (cap < sv_len * 2) cap <<= 1; + ray_t* hh = ray_alloc((size_t)cap * sizeof(int64_t)); + if (hh) { + ht_mask = cap - 1; + if (use_double) { + uint64_t* t = (uint64_t*)ray_data(hh); + for (int64_t k = 0; k < cap; k++) t[k] = IN_HTF_EMPTY; + for (int64_t j = 0; j < sv_len; j++) { + if (svf[j] != svf[j]) continue; /* NaN matches nothing */ + uint64_t b = in_f64_key(svf[j]); + int64_t sl = (int64_t)(ray_hash_i64((int64_t)b) & (uint64_t)ht_mask); + while (t[sl] != IN_HTF_EMPTY && t[sl] != b) sl = (sl + 1) & ht_mask; + t[sl] = b; + } + htf = t; + } else { + int64_t* t = (int64_t*)ray_data(hh); + for (int64_t k = 0; k < cap; k++) t[k] = INT64_MIN; + for (int64_t j = 0; j < sv_len; j++) { + int64_t v = svi[j]; + if (v == INT64_MIN) { ht_has_min = true; continue; } + int64_t sl = (int64_t)(ray_hash_i64(v) & (uint64_t)ht_mask); + while (t[sl] != INT64_MIN && t[sl] != v) sl = (sl + 1) & ht_mask; + t[sl] = v; + } + hti = t; + } + *lut_hdr_out = hh; /* freed by every caller with the LUT */ + } + } + *out_ctx = (in_worker_ctx_t){ .col = col, .svf = svf, .svi = svi, .sv_len = sv_len, .symlut = symlut, + .hti = hti, .htf = htf, .ht_mask = ht_mask, .ht_has_min = ht_has_min, .ob = NULL, .ob_base = 0, .ct = ct, .col_has_nulls = col_has_nulls, .col_atom_null = col_atom_null, @@ -1839,9 +1934,117 @@ static ray_t* exec_elementwise_tree(ray_graph_t* g, ray_op_t* root) { return result; } +/* Nodes whose result may be shared between consumers: pure, vector-valued + * operators. Structural ops (scan, filter, group, sort, join …) either + * carry side state (g->selection) or return their input, and stay out. */ +static inline bool op_memoizable(uint16_t o) { + switch (o) { + case OP_IF: case OP_LIKE: case OP_ILIKE: case OP_UPPER: case OP_LOWER: + case OP_STRLEN: case OP_SUBSTR: case OP_REPLACE: case OP_TRIM: + case OP_CONCAT: case OP_STR_FIND: case OP_EXTRACT: case OP_DATE_TRUNC: + case OP_IN: case OP_NOT_IN: + return true; + default: + return op_is_elementwise(o); + } +} + +/* Count each node's consumers over the in_id edges (plus one for the + * root's return) and arm the memo when some memoizable node has more than + * one. The memo keeps its own ref on every stored value until + * exec_memo_end: a consumer that receives an input with rc == 1 may reuse + * the buffer in place, and a fused window or a sibling still holding a raw + * pointer into that buffer would then read the overwritten values — so no + * shared value is ever handed out as the sole reference. Shared + * intermediates therefore live to the end of the execution; before this a + * shared node was recomputed per consumer instead. Values belong to the + * table the memo was armed over (memo_table): while g->table is swapped + * for a sub-table the memo is inert. Returns whether it armed; not + * re-entrant on purpose — a nested execution of the same graph leaves the + * outer memo alone and must not tear it down. */ +static bool exec_memo_begin(ray_graph_t* g, ray_op_t* root) { + if (!g || g->memo_uses || g->node_count == 0) return false; + uint32_t nc = g->node_count; + ray_t* hdr = NULL; + char* mem = (char*)scratch_calloc(&hdr, (size_t)nc * (sizeof(ray_t*) + sizeof(uint32_t))); + if (!mem) return false; + ray_t** vals = (ray_t**)mem; + uint32_t* uses = (uint32_t*)(mem + (size_t)nc * sizeof(ray_t*)); + for (uint32_t i = 0; i < nc; i++) { + ray_op_t* n = &g->nodes[i]; + if (n->flags & OP_FLAG_DEAD) continue; + for (uint8_t k = 0; k < n->arity && k < 2; k++) + if (n->in_id[k] != RAY_OP_NONE && n->in_id[k] < nc) uses[n->in_id[k]]++; + /* Operands kept in the ext node: the third input of if / substr / + * replace, the trailing arguments of concat. */ + if (n->opcode == OP_IF || n->opcode == OP_SUBSTR || n->opcode == OP_REPLACE) { + ray_op_ext_t* e = find_ext(g, n->id); + if (e && e->third_in < nc) uses[e->third_in]++; + } else if (n->opcode == OP_CONCAT) { + ray_op_ext_t* e = find_ext(g, n->id); + if (e) { + int n_args = (int)e->sym; + const uint32_t* trail = (const uint32_t*)((const char*)(e + 1)); + for (int a = 2; a < n_args; a++) + if (trail[a - 2] < nc) uses[trail[a - 2]]++; + } + } + } + if (root && root->id < nc) uses[root->id]++; + bool any = false; + for (uint32_t i = 0; i < nc && !any; i++) + any = uses[i] > 1 && op_memoizable(g->nodes[i].opcode); + if (!any) { scratch_free(hdr); return false; } + g->memo_vals = vals; g->memo_uses = uses; g->memo_n = nc; g->memo_hdr = hdr; + g->memo_table = g->table; + return true; +} + +static void exec_memo_end(ray_graph_t* g) { + if (!g || !g->memo_uses) return; + for (uint32_t i = 0; i < g->memo_n; i++) + if (g->memo_vals[i]) ray_release(g->memo_vals[i]); + scratch_free(g->memo_hdr); + g->memo_vals = NULL; g->memo_uses = NULL; g->memo_n = 0; g->memo_hdr = NULL; + g->memo_table = NULL; +} + +/* A sub-evaluation over another table (an `if` branch over its compacted + * rows) gets a memo of its own: the outer memo is set aside — its values + * belong to the outer table — and a fresh one is armed over the current + * g->table with the sub-root's census. Pop tears the inner memo down and + * puts the outer one back. */ +void ray_exec_memo_push(ray_graph_t* g, ray_op_t* root, ray_exec_memo_save_t* save) { + save->vals = g->memo_vals; save->uses = g->memo_uses; save->n = g->memo_n; + save->hdr = g->memo_hdr; save->table = g->memo_table; + g->memo_vals = NULL; g->memo_uses = NULL; g->memo_n = 0; g->memo_hdr = NULL; + g->memo_table = NULL; + (void)exec_memo_begin(g, root); +} + +void ray_exec_memo_pop(ray_graph_t* g, const ray_exec_memo_save_t* save) { + exec_memo_end(g); + g->memo_vals = save->vals; g->memo_uses = save->uses; g->memo_n = save->n; + g->memo_hdr = save->hdr; g->memo_table = save->table; +} + ray_t* exec_node(ray_graph_t* g, ray_op_t* op) { if (!op) return ray_error("nyi", NULL); + /* Shared node already computed: hand out a ref (the memo's own ref goes + * at exec_memo_end). */ + bool memo_on = g->memo_uses && g->table == g->memo_table && + op->id < g->memo_n && op_memoizable(op->opcode); + if (memo_on && g->memo_vals[op->id]) { + ray_t* v = g->memo_vals[op->id]; + ray_retain(v); + /* The memo keeps its own ref until exec_memo_end: a consumer that + * reaches its input with rc == 1 may reuse the buffer in place, and + * another consumer may still hold a raw pointer into it. */ + return v; + } + bool memo_shared = memo_on && g->memo_uses[op->id] > 1; + /* Per-op cancellation checkpoint. Long fused pipelines iterate * exec_node many times; this catches Ctrl-C between operators * without adding cost to the per-row hot path. */ @@ -1879,6 +2082,15 @@ ray_t* exec_node(ray_graph_t* g, ray_op_t* op) { ray_t* _prof_result = exec_node_inner(g, op); tl_exec_depth--; + /* First consumer of a shared node: keep a ref for the others (a lazy + * value is not kept — materialising it is the consumer's business). */ + if (memo_shared && _prof_result && !RAY_IS_ERR(_prof_result) && + !ray_is_lazy(_prof_result) && g->memo_uses && g->table == g->memo_table && + !g->memo_vals[op->id]) { + ray_retain(_prof_result); + g->memo_vals[op->id] = _prof_result; + } + if (profiling) { ray_prof_span_t* ep = ray_profile_span_end(oname); if (ep) { @@ -3926,7 +4138,11 @@ static ray_t* ray_execute_inner(ray_graph_t* g, ray_op_t* root) { if (seg_count == 0 || !dag_can_stream(g, root)) { /* Non-parted table or DAG contains ops that need specialized merge: * use existing flat-materialization path. */ + /* Only the call that armed the memo tears it down: a nested + * execution of the same graph leaves the outer memo alone. */ + bool memo_armed = exec_memo_begin(g, root); ray_t* result = exec_node(g, root); + if (memo_armed) exec_memo_end(g); if (g->selection && result && !RAY_IS_ERR(result) && result->type == RAY_TABLE) { /* Projection-aware compaction: a select publishes the keys-only diff --git a/src/ops/expr.c b/src/ops/expr.c index 81a76cbcb..a35192083 100644 --- a/src/ops/expr.c +++ b/src/ops/expr.c @@ -584,6 +584,7 @@ static uint8_t expr_ensure_type(ray_expr_t* out, uint8_t src, int8_t target) { out->regs[r].kind = REG_SCRATCH; out->regs[r].type = target; out->regs[r].nullable = out->regs[src].nullable; + out->regs[r].null_src = out->regs[src].null_src; out->n_regs++; out->n_scratch++; out->ins[out->n_ins++] = (expr_ins_t){ @@ -593,6 +594,34 @@ static uint8_t expr_ensure_type(ray_expr_t* out, uint8_t src, int8_t target) { return r; } +/* Can this instruction write a null sentinel into its destination even + * when every source lane is a real value? F64: any op that ray_f64_fin / + * a zero-divisor test canonicalizes to NULL_F64 (overflow, x/0, sqrt(<0), + * log(<=0), ...). I64: zero-divisor IDIV/MOD and the INT64_MIN overflow of + * NEG/ABS. Such a destination is nullable for every downstream instruction + * (so CAST F64->I64, I64 arithmetic and MIN2/MAX2 pick their null-aware + * kernels) and, when nothing upstream is a nullable column, the output flag + * comes from a precise sentinel scan rather than the conservative attr. */ +static bool expr_op_generates_null(uint16_t op, int8_t ot, int8_t t1, bool binary) { + if (ot == RAY_F64) { + switch (op) { + case OP_ADD: case OP_SUB: case OP_MUL: + case OP_DIV: case OP_IDIV: case OP_MOD: case OP_POW: + case OP_SQRT: case OP_LOG: case OP_EXP: + case OP_SIN: case OP_ASIN: case OP_COS: case OP_ACOS: + case OP_TAN: case OP_ATAN: case OP_RECIPROCAL: + return true; + default: + return false; /* NEG/ABS/CEIL/FLOOR/ROUND/CAST/MIN2/MAX2: finite -> finite */ + } + } + if (ot == RAY_I64) { + if (binary) return op == OP_DIV || op == OP_IDIV || op == OP_MOD; + return (op == OP_NEG || op == OP_ABS) && t1 == RAY_I64; + } + return false; +} + /* Which (opcode, dst-type, src1-type) shapes have null-aware kernel * variants? Landing per-family: * Task 5: CAST shapes + F64 arithmetic (IEEE-propagating, no variant needed) @@ -804,6 +833,7 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].type = (base == RAY_F64 || base == RAY_F32) ? RAY_F64 : RAY_I64; out->regs[r].nullable = col_nulls; + out->regs[r].null_src = col_nulls; out->has_parted = true; } else { out->regs[r].col_type = col->type; @@ -815,6 +845,7 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].type = (col->type == RAY_F64 || col->type == RAY_F32) ? RAY_F64 : RAY_I64; out->regs[r].nullable = col_nulls; + out->regs[r].null_src = col_nulls; } } else if (node->opcode == OP_CONST) { ray_op_ext_t* ext = find_ext(g, node->id); @@ -878,37 +909,57 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { else ot = RAY_I64; - /* Type promotion: ensure both sources match for the operation. - * Skip for OP_CAST — the instruction itself IS the conversion. */ + /* Type promotion: every source of a non-CAST instruction + * must sit in the lane type its kernel reads. Scratch + * registers may hold I32/I16 (narrowing CAST results) or + * BOOL (comparison results): an I64 or comparison kernel + * reading such a buffer as 8-byte lanes returns garbage, so + * widen them here (the CAST kernels map the narrow null + * sentinels). A promotion that cannot be placed (register + * or instruction budget) bails instead of running the + * mismatched kernel. Skip for OP_CAST — the instruction + * itself IS the conversion. */ +#define EXPR_PROMOTE(sreg, T) do { \ + (sreg) = expr_ensure_type(out, (sreg), (T)); \ + if (out->regs[(sreg)].type != (T)) \ + EXPR_BAIL(EXPR_BAIL_REGS); \ + } while (0) if (op == OP_CAST) { /* No promotion needed; CAST handles the conversion */ - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_F64 && s2 != 0xFF) { - /* Arithmetic with f64 output — promote i64 inputs to f64 */ - s1 = expr_ensure_type(out, s1, RAY_F64); - s2 = expr_ensure_type(out, s2, RAY_F64); - r = out->n_regs; /* re-read after possible CAST inserts */ - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_F64 && s2 == 0xFF) { - /* Unary f64 — promote input */ - s1 = expr_ensure_type(out, s1, RAY_F64); - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); - } else if (ot == RAY_BOOL && s2 != 0xFF && t1 != t2) { - /* Comparison with mixed types — promote both to f64 */ - int8_t pt = (t1 == RAY_F64 || t2 == RAY_F64) ? RAY_F64 : RAY_I64; - s1 = expr_ensure_type(out, s1, pt); - s2 = expr_ensure_type(out, s2, pt); - r = out->n_regs; - if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); + } else if (ot == RAY_F64) { + /* f64 arithmetic / unary math — promote i64 inputs to f64 */ + EXPR_PROMOTE(s1, RAY_F64); + if (s2 != 0xFF) EXPR_PROMOTE(s2, RAY_F64); + } else if (ot == RAY_I64) { + /* i64 arithmetic / NEG / ABS / SIGNUM — widen narrow and + * BOOL sources. An F64 source stays F64: only SIGNUM + * reaches here with one, and its kernel reads doubles. */ + if (out->regs[s1].type != RAY_F64) EXPR_PROMOTE(s1, RAY_I64); + if (s2 != 0xFF && out->regs[s2].type != RAY_F64) EXPR_PROMOTE(s2, RAY_I64); + } else if (ot == RAY_BOOL && s2 != 0xFF && + ((op >= OP_EQ && op <= OP_GE) || op == OP_AND || op == OP_OR)) { + /* Comparison / AND / OR — both sides in one lane type: + * BOOL stays BOOL only when both sides are BOOL. */ + int8_t pt = (t1 == RAY_F64 || t2 == RAY_F64) ? RAY_F64 + : (t1 == RAY_BOOL && t2 == RAY_BOOL) ? RAY_BOOL : RAY_I64; + if (pt != RAY_BOOL) { + EXPR_PROMOTE(s1, pt); + EXPR_PROMOTE(s2, pt); + } } +#undef EXPR_PROMOTE + r = out->n_regs; /* re-read after possible CAST inserts */ + if (r >= EXPR_MAX_REGS) EXPR_BAIL(EXPR_BAIL_REGS); /* Compute nullability from the FINAL (post-promotion) s1/s2. * Inserted CASTs inherit nullable from their source (Step 3), * so the promoted regs already carry the right flag. */ bool in_null = out->regs[s1].nullable || (s2 != 0xFF && out->regs[s2].nullable); + bool in_src = out->regs[s1].null_src || + (s2 != 0xFF && out->regs[s2].null_src); + bool gen_null = expr_op_generates_null(op, ot, out->regs[s1].type, + s2 != 0xFF); bool ins_null_aware = false; bool dst_nullable = false; if (in_null) { @@ -959,7 +1010,12 @@ bool expr_compile(ray_graph_t* g, ray_t* tbl, ray_op_t* root, ray_expr_t* out) { out->regs[r].kind = REG_SCRATCH; out->regs[r].type = ot; - out->regs[r].nullable = dst_nullable; + /* nullable: lanes may hold a sentinel, propagated from a + * source or produced by this op. null_src: that possibility + * traces to a nullable column (conservative output attr); + * without it the output flag comes from a precise scan. */ + out->regs[r].nullable = dst_nullable || gen_null; + out->regs[r].null_src = dst_nullable && in_src; out->n_scratch++; if (out->n_ins >= EXPR_MAX_INS) EXPR_BAIL(EXPR_BAIL_INS); @@ -1863,8 +1919,9 @@ static void expr_full_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t e /* Per-worker scratch buffers (heap-allocated via arena, morsel-sized) */ ray_t* scratch_hdr = NULL; + /* one morsel buffer per register the expression uses */ char* scratch_mem = (char*)scratch_alloc(&scratch_hdr, - (size_t)EXPR_MAX_REGS * EXPR_MORSEL * 8); + (size_t)(expr->n_regs ? expr->n_regs : 1) * EXPR_MORSEL * 8); if (!scratch_mem) return; void* scratch[EXPR_MAX_REGS]; for (uint8_t r = 0; r < expr->n_regs; r++) @@ -1941,57 +1998,27 @@ static void mark_i64_overflow_as_null(ray_t* result, int64_t off, int64_t len) { } } -/* The fused unary path may produce INT64_MIN via signed-overflow only for - * OP_NEG and OP_ABS over an i64 source (output type i64). Detect those - * shapes from the last instruction in the compiled expression. */ -static bool expr_last_op_overflows_i64(const ray_expr_t* expr) { - if (expr->out_type != RAY_I64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - if (last->opcode != OP_NEG && last->opcode != OP_ABS) return false; - if (last->src2 != 0xFF) return false; /* unary only */ - if (expr->regs[last->src1].type != RAY_I64) return false; - if (expr->regs[last->dst].type != RAY_I64) return false; - return true; -} - -/* The fused binary path writes NULL_I64 for an i64 DIV/IDIV/MOD whose - * divisor is zero (or the INT64_MIN/-1 overflow case) — null-model - * null handling requires HAS_NULLS set when that sentinel lands. When the - * output register is already marked `nullable` the conservative flag below - * covers it; this detector handles the case where it is not, reusing the - * mark_i64_overflow_as_null scan (which flips HAS_NULLS for any NULL_I64 - * lane). Detect the shape from the last instruction. */ -static bool expr_last_op_divmod_i64(const ray_expr_t* expr) { - if (expr->out_type != RAY_I64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - if (last->opcode != OP_DIV && last->opcode != OP_IDIV && - last->opcode != OP_MOD) return false; - if (last->src2 == 0xFF) return false; /* binary only */ - if (expr->regs[last->dst].type != RAY_I64) return false; - return true; -} - -/* Single-null float model: the fused F64 kernels canonicalize any non-finite - * result (overflow → ±Inf, div/mod-by-zero, sqrt(<0), log(≤0), exp(overflow)) - * to NULL_F64 in-buffer. Detect when the last instruction is such an F64 - * producer so the caller runs the cheap post-scan that flips HAS_NULLS for any - * 0Nf lane. Used to set HAS_NULLS conservatively from the op shape (no - * per-element scan — the scan is a full extra memory pass that regressed the - * hot float kernels ~50%); see the call site. Matches the fallback path's - * shape-based flagging so VM ≡ fallback. */ -static bool expr_last_op_produces_f64_null(const ray_expr_t* expr) { - if (expr->out_type != RAY_F64 || expr->n_ins == 0) return false; - const expr_ins_t* last = &expr->ins[expr->n_ins - 1]; - switch (last->opcode) { - case OP_ADD: case OP_SUB: case OP_MUL: - case OP_DIV: case OP_IDIV: case OP_MOD: case OP_POW: - case OP_SQRT: case OP_LOG: case OP_EXP: - case OP_SIN: case OP_ASIN: case OP_COS: case OP_ACOS: - case OP_TAN: case OP_ATAN: case OP_RECIPROCAL: - return true; - default: - return false; /* NEG/ABS/CEIL/FLOOR/ROUND/CAST/MIN2/MAX2: finite→finite */ - } +/* HAS_NULLS on a fused output. A register whose nullability traces to a + * nullable column gets the conservative attr — REQUIRED, not cosmetic: + * group.c feeds this vec to aggregates whose check-free fast path is gated + * on the attr; a missing attr with sentinel lanes = wrong aggregates. A + * register that is nullable only because some instruction in the program + * can produce a sentinel (x/0, overflow -> Inf, sqrt(<0), |INT64_MIN|, ...) + * is scanned instead, so a pure-finite result keeps HAS_NULLS unset — + * critical because this output is often an input to the NEXT op / + * aggregate, and a spurious HAS_NULLS would force that consumer onto the + * slow null-aware path (measured: conservative flagging regressed chained + * float kernels catastrophically by poisoning inputs). The scan looks at + * the whole program's generators, not just the last instruction: the + * sentinel survives every downstream null-aware kernel (abs, +, cast, ...). + * Produced once (post-join of all morsels), so the pass is not the per-op + * hot loop. Mirrors the fallback's shape-based flagging (VM == fallback). */ +static void expr_flag_output_nulls(const ray_expr_t* expr, ray_t* out, int64_t nrows) { + uint8_t o = expr->out_reg; + if (!expr->regs[o].nullable) return; + if (expr->regs[o].null_src) { out->attrs |= RAY_ATTR_HAS_NULLS; return; } + if (expr->out_type == RAY_I64) mark_i64_overflow_as_null(out, 0, nrows); + else if (expr->out_type == RAY_F64) mark_f64_nonfinite_as_null(out, 0, nrows); } /* Evaluate compiled expression over parted (segmented) columns. @@ -2063,24 +2090,7 @@ static ray_t* expr_eval_full_parted(const ray_expr_t* expr, int64_t nrows) { global_off += seg_len; } - if (expr_last_op_overflows_i64(expr) || expr_last_op_divmod_i64(expr)) - mark_i64_overflow_as_null(out, 0, nrows); - /* Single-null float model: flip HAS_NULLS PRECISELY if an F64 producer - * canonicalized a non-finite result to NULL_F64 in-buffer. Scan-based (not - * conservative-by-shape) so a pure-finite result keeps HAS_NULLS unset — - * critical because this fused output is often an input to the NEXT op / - * aggregate, and a spurious HAS_NULLS would force that consumer onto the - * slow null-aware path (measured: conservative flagging regressed chained - * float kernels catastrophically by poisoning inputs). This output is - * produced once (post-join of all morsels) so the single pass here is not - * the per-op hot loop; mirrors the i64-overflow mark above. */ - if (expr_last_op_produces_f64_null(expr)) - mark_f64_nonfinite_as_null(out, 0, nrows); - /* Conservative "may contain nulls" — REQUIRED, not cosmetic: group.c - * feeds this vec to aggregates whose check-free fast path is gated on - * the attr; a missing attr with sentinel lanes = wrong aggregates. */ - if (expr->regs[expr->out_reg].nullable) - out->attrs |= RAY_ATTR_HAS_NULLS; + expr_flag_output_nulls(expr, out, nrows); return out; } @@ -2104,24 +2114,7 @@ ray_t* expr_eval_full(const ray_expr_t* expr, int64_t nrows) { else expr_full_fn(&ctx, 0, 0, nrows); - if (expr_last_op_overflows_i64(expr) || expr_last_op_divmod_i64(expr)) - mark_i64_overflow_as_null(out, 0, nrows); - /* Single-null float model: flip HAS_NULLS PRECISELY if an F64 producer - * canonicalized a non-finite result to NULL_F64 in-buffer. Scan-based (not - * conservative-by-shape) so a pure-finite result keeps HAS_NULLS unset — - * critical because this fused output is often an input to the NEXT op / - * aggregate, and a spurious HAS_NULLS would force that consumer onto the - * slow null-aware path (measured: conservative flagging regressed chained - * float kernels catastrophically by poisoning inputs). This output is - * produced once (post-join of all morsels) so the single pass here is not - * the per-op hot loop; mirrors the i64-overflow mark above. */ - if (expr_last_op_produces_f64_null(expr)) - mark_f64_nonfinite_as_null(out, 0, nrows); - /* Conservative "may contain nulls" — REQUIRED, not cosmetic: group.c - * feeds this vec to aggregates whose check-free fast path is gated on - * the attr; a missing attr with sentinel lanes = wrong aggregates. */ - if (expr->regs[expr->out_reg].nullable) - out->attrs |= RAY_ATTR_HAS_NULLS; + expr_flag_output_nulls(expr, out, nrows); return out; } diff --git a/src/ops/fused_pred.c b/src/ops/fused_pred.c index 391d741a9..94d30219d 100644 --- a/src/ops/fused_pred.c +++ b/src/ops/fused_pred.c @@ -104,8 +104,8 @@ static int fp_atom_col_compatible(int8_t atom_type, int8_t col_type) { } /* Numeric, temporal, STR and GUID comparisons have an explicit typed leg - * with null-as-minimum ordering. SYM equality uses domain codes. LIKE/IN keep - * their stricter null-free admission because they use different evaluators. */ + * with null-as-minimum ordering. SYM equality uses domain codes. IN keeps + * its stricter null-free admission because it uses a different evaluator. */ static int fp_col_supported_op(const ray_t* col, int eq_or_ne) { if (!col) return 0; if (col->type >= RAY_BOOL && col->type <= RAY_TIMESTAMP) return 1; @@ -114,8 +114,8 @@ static int fp_col_supported_op(const ray_t* col, int eq_or_ne) { return !ray_vec_has_nulls(col); } -/* Strict form — no nullable column at all. Used by the shapes whose - * evaluator arm is not an equality compare (LIKE, IN). */ +/* Strict form — no nullable column at all. Used by the shape whose + * evaluator arm is not an equality compare (IN). */ static int fp_col_supported(const ray_t* col) { return col && !ray_vec_has_nulls(col); } @@ -208,7 +208,10 @@ static int fp_check_like(ray_t* expr, ray_t* tbl) { if (!fp_expr_const_str(elems[2])) return 0; if (tbl) { ray_t* col = ray_table_get_col(tbl, lhs->i64); - if (!col || !fp_col_supported(col)) return 0; + /* A null text cell is the empty string on every like path (the + * kernel and this evaluator both match the pattern against ""), + * so a nullable column is admitted. */ + if (!col) return 0; if (col->type != RAY_STR && col->type != RAY_SYM) return 0; } return 1; @@ -445,10 +448,13 @@ void fp_eval_cmp(const fp_cmp_t* p, int64_t start, int64_t end, for (int64_t r = 0; r < n; r++) { uint64_t sid = (uint64_t)read_by_esz(base, start + r, esz_l); if (sid >= lut_n || !lut) { - bits[r] = 0; + bits[r] = p->like_empty_match; continue; } - uint8_t state = lut[sid]; + /* The LUT is shared by the workers: relaxed atomics on its + * bytes (every writer stores the same answer for a cell, + * so a repeated resolve is the only cost of a race). */ + uint8_t state = __atomic_load_n(&lut[sid], __ATOMIC_RELAXED); if (!state) { const char* sp = NULL; size_t sl = 0; @@ -466,7 +472,7 @@ void fp_eval_cmp(const fp_cmp_t* p, int64_t start, int64_t end, : (uint8_t)ray_glob_match(sp, sl, p->pat_str, p->pat_len); } state = (uint8_t)(match ? 2 : 1); - lut[sid] = state; + __atomic_store_n(&lut[sid], state, __ATOMIC_RELAXED); } bits[r] = (uint8_t)(state == 2); } @@ -608,8 +614,8 @@ static inline uint8_t fp_eval_cmp_one(const fp_cmp_t* p, int64_t row) { if (p->col_type == RAY_SYM) { uint64_t sid = (uint64_t)read_by_esz(p->col_base, row, p->col_esz); if (sid >= p->like_lut_count || !p->like_lut) - return 0; - uint8_t state = p->like_lut[sid]; + return p->like_empty_match; + uint8_t state = __atomic_load_n(&p->like_lut[sid], __ATOMIC_RELAXED); if (!state) { /* NULL sym_strings ⇒ FILE-domain column (see fp_eval_cmp) */ const char* sp = NULL; @@ -629,7 +635,7 @@ static inline uint8_t fp_eval_cmp_one(const fp_cmp_t* p, int64_t row) { : (uint8_t)ray_glob_match(sp, sl, p->pat_str, p->pat_len); } state = (uint8_t)(match ? 2 : 1); - p->like_lut[sid] = state; + __atomic_store_n(&p->like_lut[sid], state, __ATOMIC_RELAXED); } return (uint8_t)(state == 2); } @@ -797,8 +803,10 @@ static int fp_compile_cmp(ray_graph_t* g, ray_op_t* pred_op, ray_t* tbl, return 0; } if (out->op == FP_LIKE) { + /* A nullable text column is admitted: a null cell is the empty + * string on every like path, matched against the pattern once + * here (like_empty_match). */ if (col->type != RAY_STR && col->type != RAY_SYM) return -1; - if (!fp_col_supported(col)) return -1; ray_t* cv_like = rext->literal; if (!cv_like || cv_like->type != -RAY_STR) return -1; out->col_type = col->type; @@ -810,6 +818,9 @@ static int fp_compile_cmp(ray_graph_t* g, ray_op_t* pred_op, ray_t* tbl, out->pat_str = ray_str_ptr(cv_like); out->pat_len = ray_str_len(cv_like); out->pat_compiled = ray_glob_compile(out->pat_str, out->pat_len); + out->like_empty_match = (out->pat_compiled.shape != RAY_GLOB_SHAPE_NONE) + ? (uint8_t)ray_glob_match_compiled(&out->pat_compiled, "", 0) + : (uint8_t)ray_glob_match("", 0, out->pat_str, out->pat_len); if (col->type == RAY_SYM) { /* Cell ids are positions in the COLUMN's domain. Runtime * domain: borrow the global string snapshot (lock-free per diff --git a/src/ops/fused_pred.h b/src/ops/fused_pred.h index d26ee6b24..be2fea7fc 100644 --- a/src/ops/fused_pred.h +++ b/src/ops/fused_pred.h @@ -82,6 +82,9 @@ typedef struct { * read straight from the mapping (see ray_sym_domain_raw_pin). */ ray_sym_domain_raw_t like_raw; uint8_t like_raw_ok; + /* pattern vs "": the answer for a null text cell and for a cell id the + * LUT does not cover (the same rule as the bare like kernel). */ + uint8_t like_empty_match; struct ray_sym_domain_s* like_dom; } fp_cmp_t; diff --git a/src/ops/fused_topk.c b/src/ops/fused_topk.c index ddf2ddfc3..6553e7ddc 100644 --- a/src/ops/fused_topk.c +++ b/src/ops/fused_topk.c @@ -47,8 +47,10 @@ #include "ops/internal.h" #include "lang/internal.h" #include "core/pool.h" +#include "ops/idxop.h" /* chunk-zone extrema: skip chunks that cannot enter the top-K */ #include +#include /* Use the same predicate-shape detector as fused_group. Single comparison * or AND of comparisons against literals on flat int/temporal/SYM columns. */ @@ -127,8 +129,118 @@ typedef struct { ray_t** sym_strings; uint32_t sym_count; _Atomic(uint32_t) oom; + /* Chunk pruning on the first sort key (integer / temporal column with a + * chunk-zone index): zmin/zmax per 1<esz) { + case 1: return (int64_t)((const uint8_t*)ks->base)[row]; + case 2: return (int64_t)((const int16_t*)ks->base)[row]; + case 4: return (int64_t)((const int32_t*)ks->base)[row]; + default: return ((const int64_t*)ks->base)[row]; + } +} + +/* A full heap's worst row publishes its first key as a pruning bound. */ +static void fpk_publish_bound(fpk_par_ctx_t* c, int64_t worst_row) { + if (!c->zmin) return; + const fpk_keyspec_t* ks = &c->keys[0]; + if (ks->has_nulls && ray_vec_is_null(ks->col, worst_row)) return; + int64_t v = fpk_key_i64(ks, worst_row); + int64_t cur = atomic_load_explicit(&c->bound, memory_order_relaxed); + bool set = atomic_load_explicit(&c->bound_set, memory_order_relaxed); + for (;;) { + bool better = !set || (c->zdesc ? v > cur : v < cur); + if (!better) return; + if (atomic_compare_exchange_weak_explicit(&c->bound, &cur, v, + memory_order_relaxed, memory_order_relaxed)) { + atomic_store_explicit(&c->bound_set, 1, memory_order_release); + return; + } + set = true; + } +} + +/* Chunk `g` cannot hold a row that beats the bound. */ +static inline bool fpk_chunk_pruned(const fpk_par_ctx_t* c, int64_t g) { + if (!atomic_load_explicit(&c->bound_set, memory_order_acquire)) return false; + if (g < 0 || g >= (int64_t)c->zn) return false; + if (c->znulls_better && c->znull && (c->znull[g >> 3] & (1u << (g & 7)))) return false; + int64_t b = atomic_load_explicit(&c->bound, memory_order_relaxed); + return c->zdesc ? c->zmax[g] < b : c->zmin[g] > b; +} + +/* Chunk `a` is more promising than chunk `b` for the first key's + * direction: null chunks lead when nulls sort ahead, then the smaller + * minimum (asc) or the larger maximum (desc); ties keep chunk order. */ +static inline bool fpk_chunk_better(const fpk_par_ctx_t* c, uint32_t a, uint32_t b) { + if (c->znulls_better && c->znull) { + bool na = (c->znull[a >> 3] >> (a & 7)) & 1; + bool nb = (c->znull[b >> 3] >> (b & 7)) & 1; + if (na != nb) return na; + } + int64_t ka = c->zdesc ? c->zmax[a] : c->zmin[a]; + int64_t kb = c->zdesc ? c->zmax[b] : c->zmin[b]; + if (ka != kb) return c->zdesc ? ka > kb : ka < kb; + return a < b; +} + +/* Heap-sort the chunk ids in `ord` (initially 0..n-1) best first. */ +static void fpk_order_chunks(const fpk_par_ctx_t* c, uint32_t* ord, uint32_t n) { + /* max-heap on "worse": the root is the least promising chunk, so + * popping it to the tail leaves the best chunk at ord[0]. */ + for (uint32_t i = n; i-- > 0;) { + /* sift ord[i] down */ + uint32_t k = i; + for (;;) { + uint32_t l = 2 * k + 1, r = l + 1, w = k; + if (l < n && fpk_chunk_better(c, ord[w], ord[l])) w = l; + if (r < n && fpk_chunk_better(c, ord[w], ord[r])) w = r; + if (w == k) break; + uint32_t t = ord[k]; ord[k] = ord[w]; ord[w] = t; + k = w; + } + if (i == 0) break; + } + for (uint32_t end = n; end > 1;) { + end--; + uint32_t t = ord[0]; ord[0] = ord[end]; ord[end] = t; + uint32_t k = 0; + for (;;) { + uint32_t l = 2 * k + 1, r = l + 1, w = k; + if (l < end && fpk_chunk_better(c, ord[w], ord[l])) w = l; + if (r < end && fpk_chunk_better(c, ord[w], ord[r])) w = r; + if (w == k) break; + uint32_t u = ord[k]; ord[k] = ord[w]; ord[w] = u; + k = w; + } + } +} + /* Compare two source rows by the multi-key sort spec. Returns * "a is worse than b" sense: positive means evict-a-first in the * max-heap of K-best entries. Short-circuits on first non-equal key. @@ -271,17 +383,17 @@ static inline void fpk_heapify(const fpk_par_ctx_t* c, int64_t* heap, int32_t n) fpk_sift_down(c, heap, n, i); } -/* Worker fn: scan rows [start, end), eval predicate per morsel, do - * heap inserts for passing rows. */ -static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { - fpk_par_ctx_t* c = (fpk_par_ctx_t*)raw; - if (atomic_load_explicit(&c->oom, memory_order_relaxed)) return; - int32_t k = (int32_t)c->k; - int64_t* hidx = &c->heap_idx[(size_t)worker_id * (size_t)k]; - int32_t hn = c->heap_n[worker_id]; - - int64_t row = start; +/* Scan physical rows [row, end): eval predicate per morsel, heap-insert + * the passing rows. `*hnp` is the worker's heap fill on entry and exit. + * `chunk` >= 0 is the zone chunk the rows belong to: the bound may tighten + * while the chunk is being scanned (another worker's heap filled, or this + * one's), so it is re-tested before every morsel — one relaxed load and a + * compare — and the rest of the chunk is abandoned once it cannot beat it. */ +static inline void fpk_scan_rows(fpk_par_ctx_t* c, int64_t* hidx, int32_t* hnp, + int32_t k, int64_t row, int64_t end, int64_t chunk) { + int32_t hn = *hnp; while (row < end) { + if (chunk >= 0 && fpk_chunk_pruned(c, chunk)) break; int64_t mend = row + RAY_MORSEL_ELEMS; if (mend > end) mend = end; int64_t mlen = mend - row; @@ -301,8 +413,43 @@ static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end fpk_sift_down(c, hidx, k, 0); } } + /* Once per morsel: publishing on every heap replacement contends + * on the bound when the rows arrive in the order being sought. */ + if (hn == k) fpk_publish_bound(c, hidx[0]); row = mend; } + *hnp = hn; +} + +/* Worker fn: [start, end) is a range of physical rows, or — with a zone + * order — of virtual rows whose chunk slots map to physical chunks. */ +static void fpk_par_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { + fpk_par_ctx_t* c = (fpk_par_ctx_t*)raw; + if (atomic_load_explicit(&c->oom, memory_order_relaxed)) return; + int32_t k = (int32_t)c->k; + int64_t* hidx = &c->heap_idx[(size_t)worker_id * (size_t)k]; + int32_t hn = c->heap_n[worker_id]; + + if (!c->zmin) { + fpk_scan_rows(c, hidx, &hn, k, start, end, -1); + } else { + int64_t span = (int64_t)1 << c->zlog2; + int64_t v = start; + while (v < end) { + int64_t slot = v >> c->zlog2; + int64_t sbeg = slot << c->zlog2; + int64_t send = sbeg + span; + if (send > end) send = end; + int64_t g = c->zorder ? (int64_t)c->zorder[slot] : slot; + if (!fpk_chunk_pruned(c, g)) { + int64_t p0 = (g << c->zlog2) + (v - sbeg); + int64_t p1 = (g << c->zlog2) + (send - sbeg); + if (p1 > c->nrows) p1 = c->nrows; + if (p0 < p1) fpk_scan_rows(c, hidx, &hn, k, p0, p1, g); + } + v = send; + } + } c->heap_n[worker_id] = hn; } @@ -378,17 +525,43 @@ ray_t* ray_fused_topk_select(ray_t* tbl, ctx.n_keys = n_sort_keys; ctx.k = k; ctx.tbl = tbl; + ctx.nrows = nrows; + { + ray_t* kc = ctx.keys[0].col; + int8_t kt = ctx.keys[0].type; + if ((kt == RAY_I16 || kt == RAY_I32 || kt == RAY_I64 || kt == RAY_DATE || + kt == RAY_TIME || kt == RAY_TIMESTAMP) && + ray_index_kind(kc) == RAY_IDX_CHUNK_ZONE) { + ray_index_t* zx = ray_index_payload(kc->index); + if (zx->built_for_len == kc->len && !zx->u.chunk_zone.is_f64 && + zx->u.chunk_zone.mins && zx->u.chunk_zone.maxs && zx->u.chunk_zone.null_bits) { + ctx.zmin = (const int64_t*)ray_data(zx->u.chunk_zone.mins); + ctx.zmax = (const int64_t*)ray_data(zx->u.chunk_zone.maxs); + ctx.znull = (const uint8_t*)ray_data(zx->u.chunk_zone.null_bits); + ctx.zn = zx->u.chunk_zone.n_chunks; + ctx.zlog2 = zx->u.chunk_zone.chunk_log2; + ctx.zdesc = ctx.keys[0].desc; + ctx.znulls_better = ctx.keys[0].nulls_first; + if (((int64_t)ctx.zn << ctx.zlog2) < nrows) + ctx.zmin = NULL; /* index shorter than the column: no pruning */ + } + } + } - /* Compile the predicate via a temp graph just for the WHERE clause. */ + /* Compile the predicate via a temp graph just for the WHERE clause. No + * where: is a predicate with no children — every row passes — and the + * heap does the whole sort-and-take in one pass. */ ray_graph_t* g = ray_graph_new(tbl); if (!g) { fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } - ray_op_t* pred_dag = compile_expr_dag(g, where_expr); - if (!pred_dag) { ray_graph_free(g); fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } - if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { - fp_pred_cleanup(&ctx.pred); - ray_graph_free(g); - fpk_unpin_keys(ctx.keys, n_sort_keys); - return NULL; + if (where_expr) { + ray_op_t* pred_dag = compile_expr_dag(g, where_expr); + if (!pred_dag) { ray_graph_free(g); fpk_unpin_keys(ctx.keys, n_sort_keys); return NULL; } + if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + fpk_unpin_keys(ctx.keys, n_sort_keys); + return NULL; + } } if (sym_needed) { @@ -415,8 +588,24 @@ ray_t* ray_fused_topk_select(ray_t* tbl, return NULL; } - if (pool) ray_pool_dispatch(pool, fpk_par_fn, &ctx, nrows); - else fpk_par_fn(&ctx, 0, 0, nrows); + /* With a zone index the workers walk the chunks best-first over the + * virtual row space (the order is optional: without it the slots map + * to the physical chunks and pruning still applies). */ + int64_t span_rows = nrows; + ray_t* zord_hdr = NULL; + if (ctx.zmin) { + span_rows = (int64_t)ctx.zn << ctx.zlog2; + uint32_t* ord = (uint32_t*)scratch_alloc(&zord_hdr, + (size_t)ctx.zn * sizeof(uint32_t)); + if (ord) { + for (uint32_t g = 0; g < ctx.zn; g++) ord[g] = g; + fpk_order_chunks(&ctx, ord, ctx.zn); + ctx.zorder = ord; + } + } + if (pool) ray_pool_dispatch(pool, fpk_par_fn, &ctx, span_rows); + else fpk_par_fn(&ctx, 0, 0, span_rows); + if (zord_hdr) { scratch_free(zord_hdr); ctx.zorder = NULL; } if (atomic_load_explicit(&ctx.oom, memory_order_relaxed)) { scratch_free(idx_hdr); scratch_free(hn_hdr); @@ -482,3 +671,209 @@ ray_t* ray_fused_topk_select(ray_t* tbl, } return result; } + +/* ───── Fused filter + positional take ──────────────────────────────── + * Chunks of FTK_CHUNK_ROWS rows, numbered from the end the answer comes + * from (row 0 for the first K, the last row for the last |K|), one pool + * task per chunk in that order — the workers sweep the table from that + * end together. Each worker appends passing rows to its own list until + * it holds |K|; those |K| bound the answer, so it publishes the |K|-th + * row as the cutoff: no row beyond it can be among the first |K| passing + * rows of the table, and every later chunk returns at once. The lists + * are merged by row id at the end and the |K| nearest the scanned end + * are gathered. + * ──────────────────────────────────────────────────────────────────── */ + +#define FTK_CHUNK_ROWS (64 * 1024) + +typedef struct { + fp_pred_t pred; + int64_t nrows; + int64_t k; /* |K| */ + bool from_end; + int64_t* rows; /* [nw * k] per-worker row ids, in scan order */ + int32_t* rows_n; /* [nw] */ + _Atomic(int64_t) cutoff; /* forward: rows >= cutoff are out; backward: rows <= cutoff */ +} ftk_ctx_t; + +static inline bool ftk_beyond(const ftk_ctx_t* c, int64_t row) { + int64_t cut = atomic_load_explicit(&c->cutoff, memory_order_relaxed); + return c->from_end ? row <= cut : row >= cut; +} + +static void ftk_publish(ftk_ctx_t* c, int64_t row) { + int64_t cur = atomic_load_explicit(&c->cutoff, memory_order_relaxed); + for (;;) { + bool tighter = c->from_end ? row > cur : row < cur; + if (!tighter) return; + if (atomic_compare_exchange_weak_explicit(&c->cutoff, &cur, row, + memory_order_relaxed, memory_order_relaxed)) + return; + } +} + +static void ftk_task_fn(void* raw, uint32_t worker_id, int64_t start, int64_t end) { + (void)end; + ftk_ctx_t* c = (ftk_ctx_t*)raw; + int32_t k = (int32_t)c->k; + int64_t* my = &c->rows[(size_t)worker_id * (size_t)k]; + int32_t n = c->rows_n[worker_id]; + if (n >= k) return; /* this worker's list is complete */ + + /* chunk `start` counted from the scanned end */ + int64_t lo, hi; + if (!c->from_end) { + lo = start * FTK_CHUNK_ROWS; + hi = lo + FTK_CHUNK_ROWS; + if (hi > c->nrows) hi = c->nrows; + } else { + hi = c->nrows - start * FTK_CHUNK_ROWS; + lo = hi - FTK_CHUNK_ROWS; + if (lo < 0) lo = 0; + } + if (ftk_beyond(c, c->from_end ? hi - 1 : lo)) return; + + uint8_t bits[RAY_MORSEL_ELEMS]; + if (!c->from_end) { + for (int64_t row = lo; row < hi && n < k; ) { + if (ftk_beyond(c, row)) break; + int64_t mend = row + RAY_MORSEL_ELEMS; + if (mend > hi) mend = hi; + fp_eval_pred(&c->pred, row, mend, bits); + for (int64_t r = 0; r < mend - row && n < k; r++) + if (bits[r]) my[n++] = row + r; + row = mend; + } + } else { + for (int64_t mend = hi; mend > lo && n < k; ) { + if (ftk_beyond(c, mend - 1)) break; + int64_t row = mend - RAY_MORSEL_ELEMS; + if (row < lo) row = lo; + fp_eval_pred(&c->pred, row, mend, bits); + for (int64_t r = mend - row - 1; r >= 0 && n < k; r--) + if (bits[r]) my[n++] = row + r; + mend = row; + } + } + c->rows_n[worker_id] = n; + if (n >= k) ftk_publish(c, my[k - 1]); +} + +ray_t* ray_fused_take_select(ray_t* tbl, + ray_t* where_expr, + int64_t k, + const int64_t* out_col_syms, + const int64_t* out_alias_syms, + uint32_t n_out) +{ + if (!tbl || tbl->type != RAY_TABLE || !where_expr || k == 0 || n_out == 0) return NULL; + if (k == INT64_MIN) return NULL; + bool from_end = k < 0; + if (from_end) k = -k; + if (k > FPK_MAX_K) return NULL; + int64_t nrows = ray_table_nrows(tbl); + if (nrows <= 0 || k >= nrows) return NULL; + + for (uint32_t c = 0; c < n_out; c++) { + ray_t* col = ray_table_get_col(tbl, out_col_syms[c]); + if (!col) return NULL; + int8_t ot = col->type; + if (RAY_IS_PARTED(ot) || ot == RAY_MAPCOMMON) return NULL; + if (!ray_is_vec(col)) return NULL; + } + + ftk_ctx_t ctx; + memset(&ctx, 0, sizeof(ctx)); + ctx.nrows = nrows; + ctx.k = k; + ctx.from_end = from_end; + atomic_store_explicit(&ctx.cutoff, from_end ? -1 : INT64_MAX, memory_order_relaxed); + + ray_graph_t* g = ray_graph_new(tbl); + if (!g) return NULL; + ray_op_t* pred_dag = compile_expr_dag(g, where_expr); + if (!pred_dag) { ray_graph_free(g); return NULL; } + if (fp_compile_pred(g, pred_dag, tbl, &ctx.pred) != 0) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + + ray_pool_t* pool = ray_pool_get(); + uint32_t nw = pool ? ray_pool_total_workers(pool) : 1; + ray_t* rows_hdr = NULL; + ray_t* n_hdr = NULL; + ctx.rows = (int64_t*)scratch_alloc(&rows_hdr, (size_t)nw * (size_t)k * sizeof(int64_t)); + ctx.rows_n = (int32_t*)scratch_calloc(&n_hdr, (size_t)nw * sizeof(int32_t)); + if (!ctx.rows || !ctx.rows_n) { + if (rows_hdr) scratch_free(rows_hdr); + if (n_hdr) scratch_free(n_hdr); + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + + int64_t n_chunks = (nrows + FTK_CHUNK_ROWS - 1) / FTK_CHUNK_ROWS; + if (ray_pool_par_dispatch_ok(pool, n_chunks, 2)) + ray_pool_dispatch_n(pool, ftk_task_fn, &ctx, (uint32_t)n_chunks); + else + for (int64_t t = 0; t < n_chunks; t++) ftk_task_fn(&ctx, 0, t, t + 1); + + /* Merge: each list is in scan order; pick the row nearest the scanned + * end across lists k times, then present in table order. */ + int64_t out[FPK_MAX_K]; + int32_t out_n = 0; + ray_t* pos_hdr = NULL; + int32_t* pos = (int32_t*)scratch_calloc(&pos_hdr, (size_t)nw * sizeof(int32_t)); + if (!pos) { + scratch_free(rows_hdr); scratch_free(n_hdr); + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return NULL; + } + while (out_n < (int32_t)k) { + int64_t best = -1; + uint32_t bw = 0; + for (uint32_t w = 0; w < nw; w++) { + if (pos[w] >= ctx.rows_n[w]) continue; + int64_t r = ctx.rows[(size_t)w * (size_t)k + (size_t)pos[w]]; + if (best < 0 || (from_end ? r > best : r < best)) { best = r; bw = w; } + } + if (best < 0) break; + pos[bw]++; + out[out_n++] = best; + } + scratch_free(pos_hdr); + scratch_free(rows_hdr); + scratch_free(n_hdr); + if (from_end) { + for (int32_t i = 0, j = out_n - 1; i < j; i++, j--) { + int64_t t = out[i]; out[i] = out[j]; out[j] = t; + } + } + + ray_t* result = ray_table_new(n_out); + if (!result || RAY_IS_ERR(result)) { + fp_pred_cleanup(&ctx.pred); + ray_graph_free(g); + return result ? result : ray_error("oom", NULL); + } + int build_ok = 1; + for (uint32_t c = 0; c < n_out; c++) { + int64_t cs = out_col_syms[c]; + int64_t alias = out_alias_syms ? out_alias_syms[c] : cs; + ray_t* src = ray_table_get_col(tbl, cs); + if (!src) { build_ok = 0; break; } + ray_t* col = gather_by_idx(src, out, out_n); + if (!col || RAY_IS_ERR(col)) { build_ok = 0; break; } + result = ray_table_add_col(result, alias, col); + ray_release(col); + } + ray_graph_free(g); + fp_pred_cleanup(&ctx.pred); + if (!build_ok) { + ray_release(result); + return ray_error("schema", NULL); + } + return result; +} diff --git a/src/ops/fused_topk.h b/src/ops/fused_topk.h index df3a26833..4b164ae4a 100644 --- a/src/ops/fused_topk.h +++ b/src/ops/fused_topk.h @@ -72,6 +72,23 @@ int ray_fused_topk_supported(ray_t* where_expr, ray_t* tbl); * * Returns NULL on shape miss (errors during predicate compile etc.) so * the caller can fall back to the unfused FILTER + SORT_TAKE path. */ +/* Fused filter + positional take: `(select {cols… from: T where: + * take: K})` with no ordering, K > 0 (the first K rows that pass, in + * table order) or K < 0 (the last |K|). The scan stops as soon as the + * answer is known: the table is walked in chunks from the near end, each + * worker keeps at most |K| row ids, and a worker that has |K| publishes + * the row beyond which nothing can still be part of the answer. Also + * serves `asc: key take: K` when the key column carries RAY_ATTR_SORTED + * and no nulls (the first K passing rows are then the K smallest, ties in + * table order — what the stable sort returns). + * Returns NULL on a shape the path does not take (caller falls back). */ +ray_t* ray_fused_take_select(ray_t* tbl, + ray_t* where_expr, + int64_t k, + const int64_t* out_col_syms, + const int64_t* out_alias_syms, + uint32_t n_out); + ray_t* ray_fused_topk_select(ray_t* tbl, ray_t* where_expr, const int64_t* sort_key_syms, diff --git a/src/ops/group.c b/src/ops/group.c index 8ebc04187..5226285f2 100644 --- a/src/ops/group.c +++ b/src/ops/group.c @@ -43,6 +43,19 @@ static inline bool group_fp_type(int8_t t) { return t == RAY_F32 || t == RAY_F64; } +/* Does an integer AVG over this input need the 128-bit high word next to + * its int64 sum? Only a 64-bit input can push a group's total past int64 + * in fewer than 2^31 rows; a narrower input (and a strlen fusion, whose + * lengths are smaller still) stays exact in the wrapped int64 sum, so it + * skips the carry and the per-slot high word entirely — the dense + * direct-array path in particular carves that word per worker. An + * unknown type (0: a linear plan without a source vector) is treated as + * 64-bit. */ +static inline bool group_avg_needs_hi(int8_t t, int64_t nrows) { + return t == RAY_I64 || t == RAY_TIMESTAMP || t == RAY_SYM || t == 0 || + nrows >= ((int64_t)1 << 31); +} + /* * group_key_f64_bits -- read an F64 GROUP BY key's bits, canonicalised. * @@ -131,10 +144,13 @@ int64_t ray_group_perpart_runs(void) { typedef struct { double sum_f, min_f, max_f, prod_f, first_f, last_f, sum_sq_f; int64_t sum_i, min_i, max_i, prod_i, first_i, last_i, sum_sq_i; - /* Parallel f64 sum of the integer stream — used by AVG so the - * mean of an i64 column whose sum exceeds 2^63 stays accurate - * instead of being whatever (uint64) wrap left in sum_i. */ + /* f64 sum of the integer stream (variance / stddev of integers). */ double sum_d; + /* High word of the exact 128-bit sum of the integer stream; sum_i is + * its low word. AVG of an integer column divides this exact total, + * so the mean is the same bits whatever the morsel split and equals + * the value the chunk-zone metadata computes from its per-chunk sums. */ + int64_t sum_hi; int64_t cnt; int64_t zero_count; bool has_first; @@ -145,7 +161,7 @@ static void reduce_acc_init(reduce_acc_t* acc) { acc->prod_f = 1.0; acc->first_f = 0; acc->last_f = 0; acc->sum_sq_f = 0; acc->sum_i = 0; acc->min_i = INT64_MAX; acc->max_i = INT64_MIN; acc->prod_i = 1; acc->first_i = 0; acc->last_i = 0; acc->sum_sq_i = 0; - acc->sum_d = 0; + acc->sum_d = 0; acc->sum_hi = 0; acc->cnt = 0; acc->zero_count = 0; acc->has_first = false; } @@ -162,13 +178,24 @@ static void reduce_acc_init(reduce_acc_t* acc) { static inline bool sym_lex_lt(struct ray_sym_domain_s* dom, int64_t a, int64_t b) { if (a == b) return false; - ray_t* sa = ray_sym_domain_str(dom, a); - ray_t* sb = ray_sym_domain_str(dom, b); - if (!sa || !sb) return a < b; - const char* pa = ray_str_ptr(sa); - const char* pb = ray_str_ptr(sb); - size_t la = ray_str_len(sa); - size_t lb = ray_str_len(sb); + const char* pa; const char* pb; + size_t la, lb; + /* A FILE domain's entries are read off the mapped vocabulary: no atom + * is materialised per compare (the lazily built atoms cost more than + * the compare itself on a wide vocabulary). Positions past the file + * prefix, and the runtime domain, resolve through the atoms as before. */ + ray_sym_domain_raw_t raw; + if (ray_sym_domain_raw_pin(dom, &raw) && a >= 0 && b >= 0 && + a < raw.count && b < raw.count) { + pa = ray_sym_domain_raw_str(&raw, a, &la); + pb = ray_sym_domain_raw_str(&raw, b, &lb); + } else { + ray_t* sa = ray_sym_domain_str(dom, a); + ray_t* sb = ray_sym_domain_str(dom, b); + if (!sa || !sb) return a < b; + pa = ray_str_ptr(sa); pb = ray_str_ptr(sb); + la = ray_str_len(sa); lb = ray_str_len(sb); + } size_t m = la < lb ? la : lb; int c = memcmp(pa, pb, m); if (c != 0) return c < 0; @@ -318,13 +345,14 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, #define RED_NEED_MAX (1u << 7) #define RED_NEED_FIRST (1u << 8) #define RED_NEED_LAST (1u << 9) +#define RED_NEED_SUM_128 (1u << 10) /* integer stream: exact 128-bit sum in (sum_hi, sum_i) */ #define RED_MASK_ANY (RED_NEED_COUNT | RED_NEED_ZERO) #define RED_MASK_MIN (RED_NEED_COUNT | RED_NEED_MIN) #define RED_MASK_MAX (RED_NEED_COUNT | RED_NEED_MAX) #define RED_MASK_FIRST (RED_NEED_COUNT | RED_NEED_FIRST) #define RED_MASK_LAST (RED_NEED_COUNT | RED_NEED_LAST) -#define RED_MASK_AVG_I (RED_NEED_SUM_D | RED_NEED_COUNT) +#define RED_MASK_AVG_I (RED_NEED_SUM_128 | RED_NEED_COUNT) #define RED_MASK_AVG_F (RED_NEED_SUM | RED_NEED_COUNT) #define RED_MASK_STATS_I (RED_NEED_SUM_D | RED_NEED_SUM_SQ | RED_NEED_COUNT) #define RED_MASK_STATS_F (RED_NEED_SUM | RED_NEED_SUM_SQ | RED_NEED_COUNT) @@ -357,6 +385,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -385,6 +418,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -651,7 +689,11 @@ static void reduce_merge(reduce_acc_t* dst, const reduce_acc_t* src, int8_t in_t break; case OP_AVG: if (fp) dst->sum_f += src->sum_f; - else dst->sum_d += src->sum_d; + else { + uint64_t lo = (uint64_t)dst->sum_i + (uint64_t)src->sum_i; + dst->sum_hi += src->sum_hi + (lo < (uint64_t)src->sum_i ? 1 : 0); + dst->sum_i = (int64_t)lo; + } dst->cnt += src->cnt; break; case OP_VAR: case OP_VAR_POP: case OP_STDDEV: case OP_STDDEV_POP: @@ -4257,7 +4299,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: result = ray_i64(scan_n); break; - case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : merged.sum_d / merged.cnt)) : ray_typed_null(-RAY_F64); break; + case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : ray_i128_to_f64(merged.sum_hi, (uint64_t)merged.sum_i) / merged.cnt)) : ray_typed_null(-RAY_F64); break; case OP_FIRST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.first_f) : reduction_i64_result(merged.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_LAST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.last_f) : reduction_i64_result(merged.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_VAR: case OP_VAR_POP: @@ -4298,7 +4340,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: return ray_i64(scan_n); - case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : acc.sum_d / acc.cnt)) : ray_typed_null(-RAY_F64); + case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : ray_i128_to_f64(acc.sum_hi, (uint64_t)acc.sum_i) / acc.cnt)) : ray_typed_null(-RAY_F64); case OP_FIRST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.first_f) : reduction_i64_result(acc.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_LAST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.last_f) : reduction_i64_result(acc.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_VAR: case OP_VAR_POP: @@ -4672,6 +4714,8 @@ bool ght_compute_layout(ght_layout_t* out, uint32_t n_keys, uint32_t n_aggs, if (need_flags & GHT_NEED_MIN) { out->off_min = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_MAX) { out->off_max = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_SUMSQ) { out->off_sumsq = (uint16_t)off; off += block; } + /* High words of the exact 128-bit integer sums (integer AVG). */ + if (need_flags & GHT_NEED_SUM128) { out->off_sum_hi = (uint16_t)off; off += block; } /* Per-slot row-index bounds for FIRST/LAST. Two int64 blocks of * n_agg_vals slots each, allocated only when needed. */ if (has_first_last) { @@ -5299,6 +5343,14 @@ static inline void init_accum_from_entry(char* row, const char* entry, } else { memcpy(row + ly->off_sum + s * 8, agg_data + s * 8, 8); } + /* 128-bit sum: the high word sign-extends the first integer + * value (the row was zeroed above, so 0 is right for F64, + * truthy and non-negative slots). */ + if ((nf & GHT_NEED_SUM128) && !(af & GHT_AF_F64) && + !(aflags2[a] & GHT_AF2_TRUTHY)) { + int64_t v; memcpy(&v, agg_data + s * 8, 8); + if (v < 0) { int64_t hi = -1; memcpy(row + ly->off_sum_hi + s * 8, &hi, 8); } + } } if (nf & GHT_NEED_MIN) memcpy(row + ly->off_min + s * 8, agg_data + s * 8, 8); if (nf & GHT_NEED_MAX) memcpy(row + ly->off_max + s * 8, agg_data + s * 8, 8); @@ -5421,6 +5473,7 @@ static inline void accum_from_entry(char* row, const char* entry, else if (af & GHT_AF_LAST) { if (take_last) memcpy(row + ly->off_sum + s * 8, val, 8); } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -5549,6 +5602,7 @@ static void accum_from_entry_nullable(char* row, const char* entry, } } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -6702,7 +6756,10 @@ static void radix_phase3_fn(void* ctx, uint32_t worker_id, int64_t start, int64_ case OP_AVG: if (nn == 0) { v = NULL_F64; grp_set_null(ao->vec, di); break; } v = sf ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (ao->affine) v += ao->bias_f64; break; case OP_MIN: @@ -6915,7 +6972,9 @@ static inline uint32_t group_merge_row(group_ht_t* ht, const uint8_t* const aflags = ly->agg_flags; const int8_t* const vslot = ly->agg_val_slot; uint16_t off_sum = ly->off_sum; + uint16_t off_sum_hi = ly->off_sum_hi; bool need_sum = (ly->need_flags & GHT_NEED_SUM) != 0; + bool need_sum128 = (ly->need_flags & GHT_NEED_SUM128) != 0; for (;;) { uint32_t sv = ht->slots[slot]; if (sv == HT_EMPTY) { @@ -6954,10 +7013,16 @@ static inline uint32_t group_merge_row(group_ht_t* ht, memcpy(&sv_f, src_row + off, 8); // cppcheck-suppress invalidPointerCast // GHT rows are 8-byte aligned and off_sum/s*8 are multiples of 8 *(double*)(row + off) += sv_f; + } else if (need_sum128) { + int64_t sv_i, sv_hi; + memcpy(&sv_i, src_row + off, 8); + memcpy(&sv_hi, src_row + off_sum_hi + (size_t)s * 8, 8); + ray_i128_add128((int64_t*)(row + off_sum_hi + (size_t)s * 8), + (uint64_t*)(row + off), sv_hi, (uint64_t)sv_i); } else { int64_t sv_i; memcpy(&sv_i, src_row + off, 8); - *(int64_t*)(row + off) += sv_i; + *(int64_t*)(row + off) = wrap_add_i64(*(int64_t*)(row + off), sv_i); } } } @@ -7096,7 +7161,7 @@ static void radix_v2_phase1_fn(void* ctx, uint32_t worker_id, * NOTE: "hashing is identical" is load-bearing and is enforced by the * v2_packed branch in the staging loop — see the comment there. */ if (!wide_any && !inline_str && !nullable && ly->null_words == 0 && - (ly->need_flags & ~(uint32_t)GHT_NEED_SUM) == 0 && + (ly->need_flags & ~(uint32_t)(GHT_NEED_SUM | GHT_NEED_SUM128)) == 0 && !(ly->agg_flags_any & (GHT_AF_FIRST | GHT_AF_LAST | GHT_AF_BINARY | GHT_AF_HOLISTIC)) && nk >= 1 && nk <= 2 && ly->entry_stride <= 64) { @@ -7504,6 +7569,11 @@ typedef union { double f; int64_t i; } da_val_t; typedef struct { da_val_t* sum; /* SUM/AVG/FIRST/LAST [n_slots * n_aggs] */ + int64_t* sum_hi; /* high words of the exact 128-bit integer sums + * [n_slots * n_aggs]; allocated only when an + * integer AVG is present (sum[].i is then the low + * word — still the wrapped int64 SUM), NULL + * otherwise so SUM-only queries pay nothing. */ da_val_t* min_val; /* MIN [n_slots * n_aggs] */ da_val_t* max_val; /* MAX [n_slots * n_aggs] */ double* sumsq_f64; /* sum-of-squares for STDDEV/VAR */ @@ -7524,6 +7594,7 @@ typedef struct { double* sumxy; /* Σxy */ /* Arena headers */ ray_t* _h_sum; + ray_t* _h_sum_hi; ray_t* _h_min; ray_t* _h_max; ray_t* _h_sumsq; @@ -7538,6 +7609,7 @@ typedef struct { static inline void da_accum_free(da_accum_t* a) { scratch_free(a->_h_sum); + scratch_free(a->_h_sum_hi); scratch_free(a->_h_min); scratch_free(a->_h_max); scratch_free(a->_h_sumsq); @@ -7555,11 +7627,18 @@ static inline void da_accum_free(da_accum_t* a) { * non-NULL) carries the per-(group, agg) non-null row count: AVG/VAR/ * STDDEV use it as the divisor and MIN/MAX/PROD/FIRST/LAST emit a typed * null when it is zero. Pass NULL to keep the legacy count[gid]-divisor - * behaviour (callers without HAS_NULLS aggs need not allocate it). */ + * behaviour (callers without HAS_NULLS aggs need not allocate it). + * sum_hi64 (if non-NULL) carries the high words of the exact 128-bit + * integer sums: an integer AVG then divides ray_i128_to_f64(hi, lo), the + * same bits as the keyless reduction and the chunk-zone metadata. A + * caller passing NULL must not emit an integer AVG. The strlen fusion + * keeps the plain wrapped add (a sum of lengths never leaves int64), so a + * high word of 0 is exact there. */ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* ext, ray_t* const* agg_vecs, uint32_t grp_count, uint32_t n_aggs, const double* sum_f64, const int64_t* sum_i64, + const int64_t* sum_hi64, const double* min_f64, const double* max_f64, const int64_t* min_i64, const int64_t* max_i64, const int64_t* counts, @@ -7646,7 +7725,9 @@ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* break; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } - v = is_f64 ? sum_f64[idx] / nn : (double)sum_i64[idx] / nn; + v = is_f64 ? sum_f64[idx] / nn + : sum_hi64 ? ray_i128_to_f64(sum_hi64[idx], (uint64_t)sum_i64[idx]) / nn + : (double)sum_i64[idx] / nn; if (affine && affine[a].enabled) v += affine[a].bias_f64; break; @@ -7848,6 +7929,12 @@ typedef struct { ray_t* rowsel; ray_t** sym_strings; /* borrowed sym snapshot for strlen-on-SYM aggs */ uint32_t sym_count; + int64_t da_n_scan; /* rows to scan (ranged tasks split it) */ + /* Per-agg raw vocabulary snapshot of a FILE-domain SYM column, pinned + * once for the whole accumulate: strlen and lexical min/max read the + * entry off the mapping instead of pinning (and checking null) per row. */ + ray_sym_domain_raw_t* agg_raw; + uint8_t* agg_raw_ok; } da_ctx_t; typedef struct { @@ -7855,12 +7942,14 @@ typedef struct { int64_t* keys; int64_t* counts; da_val_t* sums; + int64_t* sums_hi; /* high words of the 128-bit integer sums (integer AVG) */ uint32_t cap; uint32_t size; ray_t* _h_used; ray_t* _h_keys; ray_t* _h_counts; ray_t* _h_sums; + ray_t* _h_sums_hi; } sparse_i64_ht_t; static inline uint64_t sparse_i64_mix(uint64_t x) { @@ -7889,6 +7978,7 @@ static void sparse_i64_free(sparse_i64_ht_t* ht) { scratch_free(ht->_h_keys); scratch_free(ht->_h_counts); scratch_free(ht->_h_sums); + scratch_free(ht->_h_sums_hi); memset(ht, 0, sizeof(*ht)); } @@ -8008,7 +8098,7 @@ static int64_t da_count_emit_keep_min_u32(const uint32_t* counts, } static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, - bool need_sum) { + bool need_sum, bool need_sum128) { memset(ht, 0, sizeof(*ht)); if (cap < 1024) cap = 1024; cap = sparse_i64_pow2(cap); @@ -8020,7 +8110,12 @@ static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, ht->sums = (da_val_t*)scratch_calloc(&ht->_h_sums, (size_t)cap * n_aggs * sizeof(da_val_t)); } - if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums)) { + if (need_sum && need_sum128) { + ht->sums_hi = (int64_t*)scratch_calloc(&ht->_h_sums_hi, + (size_t)cap * n_aggs * sizeof(int64_t)); + } + if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums) || + (need_sum && need_sum128 && !ht->sums_hi)) { sparse_i64_free(ht); return false; } @@ -8040,7 +8135,7 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, bool need_sum) { sparse_i64_ht_t old = *ht; sparse_i64_ht_t nw; - if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum)) + if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum, old.sums_hi != NULL)) return false; for (uint32_t i = 0; i < old.cap; i++) { if (!old.used[i]) continue; @@ -8051,6 +8146,9 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, if (need_sum) memcpy(&nw.sums[(size_t)s * n_aggs], &old.sums[(size_t)i * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (nw.sums_hi) + memcpy(&nw.sums_hi[(size_t)s * n_aggs], &old.sums_hi[(size_t)i * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); nw.size++; } sparse_i64_free(&old); @@ -8072,6 +8170,9 @@ static bool sparse_i64_touch(sparse_i64_ht_t* ht, int64_t key, uint8_t n_aggs, if (need_sum) memset(&ht->sums[(size_t)s * n_aggs], 0, (size_t)n_aggs * sizeof(da_val_t)); + if (ht->sums_hi) + memset(&ht->sums_hi[(size_t)s * n_aggs], 0, + (size_t)n_aggs * sizeof(int64_t)); ht->size++; } *out_slot = s; @@ -8375,16 +8476,67 @@ static inline int64_t scalar_i64_at(const void* ptr, int8_t type, int64_t r) { return read_col_i64(ptr, r, type, 0); /* attrs=0: agg columns are numeric, never SYM */ } +/* Exact 128-bit sum of a contiguous int64 run, kept vectorizable. + * Blocks of 1024 rows, three forms, escalating once for the rest of the + * run when a block fails a check (a column's values are alike): + * 0. non-negative values below 2^53 (ids, counts, timestamps): one OR of + * the raw values next to the wrapped sum proves every value is in + * [0, 2^53), so the block sum cannot have overflowed int64; + * 1. values in [-2^53, 2^53): the same with the values biased by 2^53 + * (one extra add); + * 2. anything: two plain running sums, the wrapped total and the sum of + * the (arithmetically shifted) high 32-bit halves; with n < 2^32 the + * low halves' sum is below 2^64, so it is recovered exactly from the + * wrapped total, and the block's value is sh * 2^32 + low. + * A failed check re-reads only that block (8 KiB, L1-hot). Same bits as + * ray_i128_add per element (idxop.h) at a fraction of its cost. */ +static inline void group_i128_sum_i64_range(const int64_t* restrict x, int64_t n, + int64_t* hi, uint64_t* lo) { + const int64_t blk = 1024; + const uint64_t bias = (uint64_t)1 << 53; + int mode = 0; + for (int64_t b = 0; b < n; b += blk) { + int64_t e = (n - b > blk) ? b + blk : n; + if (mode == 0) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u; } + if ((o >> 53) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 1; + } + if (mode == 1) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u + bias; } + if ((o >> 54) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 2; + } + uint64_t sl = 0; int64_t sh = 0; + for (int64_t r = b; r < e; r++) { + int64_t v = x[r]; + sl += (uint64_t)v; + sh += v >> 32; + } + uint64_t low = sl - ((uint64_t)sh << 32); /* sum of the low halves */ + ray_i128_add128(hi, lo, sh >> 32, (uint64_t)sh << 32); + ray_i128_add128(hi, lo, 0, low); + } +} + /* Tight SIMD-friendly loop for single SUM/AVG on i64 (no mask). - * Note: int64 sum can overflow; caller responsibility to use appropriate types. */ + * Note: the int64 sum wraps (SUM's contract); an integer AVG also carries + * the high word (acc->sum_hi) so it divides the exact 128-bit total. */ static void scalar_sum_i64_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { scalar_ctx_t* c = (scalar_ctx_t*)ctx; da_accum_t* acc = &c->accums[worker_id]; const int64_t* restrict data = (const int64_t*)c->agg_ptrs[0]; - int64_t sum = 0; - for (int64_t r = start; r < end; r++) - sum = wrap_add_i64(sum, data[r]); - acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + if (acc->sum_hi) { + group_i128_sum_i64_range(data + start, end - start, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else { + int64_t sum = 0; + for (int64_t r = start; r < end; r++) + sum = wrap_add_i64(sum, data[r]); + acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + } acc->count[0] += end - start; } @@ -8407,6 +8559,41 @@ static void scalar_sum_linear_i64_fn(void* ctx, uint32_t worker_id, int64_t star const agg_linear_t* lin = &c->agg_linear[0]; int64_t n = end - start; + if (acc->sum_hi) { + /* Integer AVG needs the exact 128-bit total. A bare column (one + * term, coefficient 1, no bias — the common keyless avg) streams + * the vectorized block sum; a narrow column's block total cannot + * leave int64 at all. Any other linear shape walks the rows: the + * per-row value wraps like the materialized expression would and + * the 128-bit total of those rows is exact. */ + bool bare = lin->n_terms == 1 && lin->coeff_i64[0] == 1 && + lin->bias_i64 == 0 && lin->term_ptrs[0]; + int8_t t0 = lin->term_types[0]; + if (bare && (t0 == RAY_I64 || t0 == RAY_TIMESTAMP)) { + group_i128_sum_i64_range((const int64_t*)lin->term_ptrs[0] + start, n, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else if (bare && t0 != RAY_SYM) { + /* <= 32-bit values: 2^30 of them stay far inside int64 */ + const int64_t blk = (int64_t)1 << 30; + for (int64_t b = start; b < end; b += blk) { + int64_t e = (end - b > blk) ? b + blk : end; + int64_t sum = 0; + for (int64_t r = b; r < e; r++) + sum += scalar_i64_at(lin->term_ptrs[0], t0, r); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, sum); + } + } else { + for (int64_t r = start; r < end; r++) { + int64_t iv = lin->bias_i64; + for (uint8_t t = 0; t < lin->n_terms; t++) + iv = wrap_add_i64(iv, wrap_mul_i64(lin->coeff_i64[t], + scalar_i64_at(lin->term_ptrs[t], lin->term_types[t], r))); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, iv); + } + } + acc->count[0] += n; + return; + } int64_t sum = wrap_mul_i64(lin->bias_i64, n); /* n_terms is bounded by AGG_LINEAR_MAX_TERMS (8, internal.h) — an * unrelated fixed cap on linear-expression arity, not a GROUP n_keys/ @@ -8483,7 +8670,8 @@ static inline void scalar_accum_row(scalar_ctx_t* c, da_accum_t* acc, int64_t r) if (nn) nn[a]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[a], (uint64_t*)&acc->sum[a].i, iv); + else acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); if (acc->sumsq_f64) acc->sumsq_f64[a] += fv * fv; if (nn) nn[a]++; } @@ -8559,6 +8747,47 @@ static void scalar_accum_fn(void* ctx, uint32_t worker_id, int64_t start, int64_ * Fast path for SUM/AVG-only queries: eliminates op-code dispatch and da_read_val * dual-write overhead. The branch on c->all_sum is perfectly predicted (invariant * across all rows). */ +/* strlen of agg column a at row r. STR: the descriptor's length (a null + * is the empty string, length 0). SYM: null is id 0; a FILE-domain entry's + * length is the u32 prefix in the pinned mapping; anything else resolves + * as group_strlen_at_cached does. */ +static inline int64_t da_strlen_at(const da_ctx_t* c, uint32_t a, int64_t r) { + const ray_t* col = c->agg_cols[a]; + if (col->type == RAY_STR) { + const ray_str_t* elems; const char* pool; (void)pool; + str_resolve(col, &elems, &pool); + return (int64_t)elems[r].len; + } + if (col->type == RAY_SYM) { + int64_t sid = ray_read_sym(ray_data((ray_t*)col), r, RAY_SYM, col->attrs); + if (sid == 0) return 0; + if (c->agg_raw_ok && c->agg_raw_ok[a] && sid > 0 && sid < c->agg_raw[a].count) { + size_t sl; + (void)ray_sym_domain_raw_str(&c->agg_raw[a], sid, &sl); + return (int64_t)sl; + } + } + return group_strlen_at_cached(col, r, c->sym_strings, c->sym_count); +} + +/* Lexical x < y for two cells of agg column a_idx (a SYM column): the + * pinned mapping when both positions are in the file prefix, sym_lex_lt + * otherwise. */ +static inline bool da_sym_lex_lt(const da_ctx_t* c, uint32_t a_idx, int64_t x, int64_t y) { + if (x == y) return false; + if (c->agg_raw_ok && c->agg_raw_ok[a_idx] && x >= 0 && y >= 0 && + x < c->agg_raw[a_idx].count && y < c->agg_raw[a_idx].count) { + size_t lx, ly; + const char* px = ray_sym_domain_raw_str(&c->agg_raw[a_idx], x, &lx); + const char* py = ray_sym_domain_raw_str(&c->agg_raw[a_idx], y, &ly); + size_t m = lx < ly ? lx : ly; + int r = m ? memcmp(px, py, m) : 0; + if (r != 0) return r < 0; + return lx < ly; + } + return sym_lex_lt(ray_sym_vec_domain(c->agg_cols[a_idx]), x, y); +} + static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64_t r) { uint8_t n_aggs = c->n_aggs; acc->count[gid]++; @@ -8581,9 +8810,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 } if (!c->agg_ptrs[a]) continue; if (c->agg_strlen && c->agg_strlen[a]) { - acc->sum[idx].i = wrap_add_i64( - acc->sum[idx].i, - group_strlen_at_cached(c->agg_cols[a], r, c->sym_strings, c->sym_count)); + acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, da_strlen_at(c, a, r)); if (nn) nn[idx]++; } else if (f64m & ((uint64_t)1 << a)) { /* NaN payload = null, skip from sum. */ @@ -8598,7 +8825,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 uint8_t v_attrs = c->agg_cols[a] ? c->agg_cols[a]->attrs : 0; int64_t v = read_col_i64(c->agg_ptrs[a], r, c->agg_types[a], v_attrs); if (RAY_LIKELY(!((inm >> a) & 1) || v != c->agg_int_null_sentinel[a])) { - acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, v); + else acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); if (nn) nn[idx]++; } } @@ -8634,8 +8862,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 fv = prod_val_f64(&c->agg_prod[a], r); iv = (int64_t)fv; } else if (c->agg_strlen && c->agg_strlen[a]) { - iv = group_strlen_at_cached(c->agg_cols[a], r, - c->sym_strings, c->sym_count); + iv = da_strlen_at(c, a, r); fv = (double)iv; } else { uint8_t attrs = c->agg_cols[a] ? c->agg_cols[a]->attrs : 0; @@ -8663,7 +8890,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 if (nn) nn[idx]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, iv); + else acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); if (acc->sumsq_f64) acc->sumsq_f64[idx] += fv * fv; if (nn) nn[idx]++; } @@ -8727,7 +8955,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 } else if (c->agg_types[a] == RAY_SYM && !int_null) { /* Lex compare for SYM; INT64_MAX = "not seen yet". */ if (acc->min_val[idx].i == INT64_MAX || - sym_lex_lt(ray_sym_vec_domain(c->agg_cols[a]), iv, acc->min_val[idx].i)) + da_sym_lex_lt(c, a, iv, acc->min_val[idx].i)) acc->min_val[idx].i = iv; } else if (!int_null) { if (iv < acc->min_val[idx].i) acc->min_val[idx].i = iv; @@ -8738,7 +8966,7 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 if (fv == fv && fv > acc->max_val[idx].f) acc->max_val[idx].f = fv; } else if (c->agg_types[a] == RAY_SYM && !int_null) { if (acc->max_val[idx].i == INT64_MIN || - sym_lex_gt(ray_sym_vec_domain(c->agg_cols[a]), iv, acc->max_val[idx].i)) + da_sym_lex_lt(c, a, acc->max_val[idx].i, iv)) acc->max_val[idx].i = iv; } else if (!int_null) { if (iv > acc->max_val[idx].i) acc->max_val[idx].i = iv; @@ -8890,6 +9118,110 @@ static void da_accum_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t en #undef DA_PF_DIST } +/* ---- slot-partitioned accumulate -------------------------------------- + * For a pool with more workers than the per-worker slot budget allows. + * Pass 1 (nw tasks, contiguous row ranges in order): each row's group is + * computed once and its row id appended to the bucket of (range, task) + * with task = gid / span — no shared writes. Pass 2 (one task per slot span, + * ray_pool_dispatch_n): task t drains every worker's bucket t into the + * single accumulator set; no two tasks touch a slot, so nothing is merged + * and the state is one set however many workers run. The buckets hold + * int32 row ids (the caller admits tables below INT32_MAX rows). */ +typedef struct { int32_t* data; int64_t len, cap; ray_t* hdr; } da_bucket_t; +typedef struct { + da_ctx_t* c; + da_bucket_t* buckets; /* [nw * k] */ + uint32_t nw, k, span; + int64_t n_scan; + _Atomic(int) oom; +} da_part_ctx_t; + +static bool da_bucket_push(da_bucket_t* b, int32_t v) { + if (b->len == b->cap) { + int64_t ncap = b->cap ? b->cap * 2 : 1024; + ray_t* nh = NULL; + int32_t* nd = (int32_t*)scratch_alloc(&nh, (size_t)ncap * sizeof(int32_t)); + if (!nd) return false; + if (b->len) memcpy(nd, b->data, (size_t)b->len * sizeof(int32_t)); + if (b->hdr) scratch_free(b->hdr); + b->data = nd; b->hdr = nh; b->cap = ncap; + } + b->data[b->len++] = v; + return true; +} + +/* Task `t` of nw scans the t-th contiguous row range into bucket row t: + * draining rows 0..nw-1 in order then hands each slot its rows in table + * order, so the accumulation (float sums included) is the serial scan's, + * whatever the scheduling. */ +static void da_part_scatter_fn(void* raw, uint32_t wid, int64_t task, int64_t task_end) { + (void)wid; (void)task_end; + da_part_ctx_t* p = (da_part_ctx_t*)raw; + da_ctx_t* c = p->c; + da_bucket_t* mine = &p->buckets[(size_t)task * p->k]; + int64_t start = p->n_scan * task / p->nw; + int64_t end = p->n_scan * (task + 1) / p->nw; + const int64_t* match_idx = c->match_idx; + for (int64_t i = start; i < end; i++) { + int64_t r = match_idx ? match_idx[i] : i; + if (!match_idx && c->rowsel && !group_rowsel_pass(c->rowsel, r)) continue; + uint32_t t = (uint32_t)da_composite_gid(c, r) / p->span; + if (!da_bucket_push(&mine[t], (int32_t)r)) { + atomic_store_explicit(&p->oom, 1, memory_order_relaxed); + return; + } + } +} + +static void da_part_drain_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + da_part_ctx_t* p = (da_part_ctx_t*)raw; + da_ctx_t* c = p->c; + da_accum_t* acc = &c->accums[0]; + uint32_t t = (uint32_t)start; + for (uint32_t w = 0; w < p->nw; w++) { + const da_bucket_t* b = &p->buckets[(size_t)w * p->k + t]; + for (int64_t j = 0; j < b->len; j++) { + int64_t r = b->data[j]; + da_accum_row(c, acc, da_composite_gid(c, r), r); + } + } +} + +/* Returns false (accumulators untouched) when the buckets could not be + * allocated; the caller then falls back to the plain scan. */ +static bool da_accum_partitioned(da_ctx_t* c, ray_pool_t* pool, uint32_t k, int64_t n_scan) { + uint32_t nw = ray_pool_total_workers(pool); + uint32_t span = (c->n_slots + k - 1) / k; + if (span == 0) span = 1; + k = (c->n_slots + span - 1) / span; + ray_t* bh = NULL; + da_bucket_t* buckets = (da_bucket_t*)scratch_calloc(&bh, (size_t)nw * k * sizeof(da_bucket_t)); + if (!buckets) return false; + da_part_ctx_t p = { .c = c, .buckets = buckets, .nw = nw, .k = k, .span = span, + .n_scan = n_scan }; + atomic_store_explicit(&p.oom, 0, memory_order_relaxed); + ray_pool_dispatch_n(pool, da_part_scatter_fn, &p, nw); + bool ok = atomic_load_explicit(&p.oom, memory_order_relaxed) == 0; + if (ok) ray_pool_dispatch_n(pool, da_part_drain_fn, &p, k); + for (size_t i = 0; i < (size_t)nw * k; i++) + if (buckets[i].hdr) scratch_free(buckets[i].hdr); + scratch_free(bh); + return ok; +} + +/* One task per accumulator (ray_pool_dispatch_n): task i scans the i-th + * of n_accums equal row ranges into accums[i], whichever worker runs it. */ +static void da_accum_task_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { + (void)worker_id; (void)end; + da_ctx_t* c = (da_ctx_t*)ctx; + int64_t k = c->n_accums, n = c->da_n_scan; + int64_t base = n / k, rem = n % k; + int64_t lo = start * base + (start < rem ? start : rem); + int64_t hi = lo + base + (start < rem ? 1 : 0); + da_accum_fn(ctx, (uint32_t)start, lo, hi); +} + /* Parallel DA merge: merge per-worker accumulators into accums[0] by * dispatching disjoint slot ranges across pool workers. */ typedef struct { @@ -8950,6 +9282,9 @@ static void da_merge_fn(void* ctx, uint32_t wid, int64_t start, int64_t end) { } } else if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -10093,6 +10428,8 @@ typedef struct { const uint16_t* agg_ops; uint8_t n_aggs; da_val_t* partials; /* [n_tasks * n_aggs], zeroed */ + int64_t* partials_hi; /* high words of the integer partials + * (NULL unless an integer AVG) */ double* partial_sumsq; double* partial_sum_y; double* partial_sumsq_y; @@ -10132,7 +10469,7 @@ static inline double sg_prod_range(const agg_prod_t* p, int64_t r0, int64_t n, a2 += x[j + 2] * y[j + 2]; a3 += x[j + 3] * y[j + 3]; } for (; j < n; j++) a0 += x[j] * y[j]; - } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIME)) { + } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIMESTAMP)) { const double* restrict x = (const double*)pa + r0; const int64_t* restrict y = (const int64_t*)pb + r0; uint64_t s0 = 0; @@ -10299,32 +10636,45 @@ static void sg_accum_fn(void* raw, uint32_t wid, int64_t tstart, int64_t tend) { int8_t t = av->type; uint8_t at = av->attrs; uint64_t acc = 0; + int64_t hi = 0; /* 128-bit high word (integer AVG) */ double ssq = 0.0; bool need_sq = c->partial_sumsq && (op == OP_STDDEV || op == OP_STDDEV_POP || op == OP_VAR || op == OP_VAR_POP); - if (contig && (t == RAY_I64 || t == RAY_TIME)) { + if (contig && (t == RAY_I64 || t == RAY_TIMESTAMP)) { const int64_t* restrict x = (const int64_t*)p + r0; - for (int64_t j = 0; j < n; j++) { - int64_t v = x[j]; - acc += (uint64_t)v; - if (need_sq) { double d = (double)v; ssq += d * d; } + if (c->partials_hi) { + /* exact 128-bit total, still a vectorized stream */ + group_i128_sum_i64_range(x, n, &hi, &acc); + if (need_sq) + for (int64_t j = 0; j < n; j++) { double d = (double)x[j]; ssq += d * d; } + } else { + for (int64_t j = 0; j < n; j++) { + int64_t v = x[j]; + acc += (uint64_t)v; + if (need_sq) { double d = (double)v; ssq += d * d; } + } } - } else if (contig && t == RAY_I32) { + } else if (contig && (t == RAY_I32 || t == RAY_DATE || t == RAY_TIME)) { + /* the 4-byte family: I32 and the day / millisecond temporals */ const int32_t* restrict x = (const int32_t*)p + r0; for (int64_t j = 0; j < n; j++) { int64_t v = (int64_t)x[j]; acc += (uint64_t)v; if (need_sq) { double d = (double)v; ssq += d * d; } } + /* n <= SG_CHUNK_ROWS 32-bit values: the int64 sum is + * exact, its high word is the sign. */ + hi = (int64_t)acc < 0 ? -1 : 0; } else { for (int64_t j = 0; j < n; j++) { int64_t v = read_col_i64(p, rows[j], t, at); - acc += (uint64_t)v; + ray_i128_add(&hi, &acc, v); if (need_sq) { double d = (double)v; ssq += d * d; } } } c->partials[idx].i = (int64_t)acc; + if (c->partials_hi) c->partials_hi[idx] = hi; if (need_sq) c->partial_sumsq[idx] = ssq; } } @@ -10493,7 +10843,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const ray_idx_slice_t* slices = (K > 0) ? (const ray_idx_slice_t*)ray_data(g->sg_slices_hdr) : NULL; - bool need_sumsq = false, need_pair = false; + bool need_sumsq = false, need_pair = false, need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { uint16_t aop = ext->agg_ops[a]; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || @@ -10502,12 +10852,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, need_sumsq = true; if (agg_is_binary_agg(aop)) need_pair = true; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + if (aop == OP_AVG && !prod[a].enabled && agg_vecs[a] && + !group_fp_type(agg_vecs[a]->type) && + group_avg_needs_hi(agg_vecs[a]->type, ray_table_nrows(tbl))) + need_sum128 = true; } ray_t *sum_hdr = NULL, *cnt_hdr = NULL, *task_hdr = NULL, *part_hdr = NULL; ray_t *sumsq_hdr = NULL, *sum_y_hdr = NULL, *sumsq_y_hdr = NULL, *sumxy_hdr = NULL; ray_t *part_sumsq_hdr = NULL, *part_sum_y_hdr = NULL, *part_sumsq_y_hdr = NULL, *part_sumxy_hdr = NULL; + ray_t *sum_hi_hdr = NULL, *part_hi_hdr = NULL; da_val_t* sums = NULL; + int64_t* sums_hi = NULL; double *sumsq = NULL, *sum_y = NULL, *sumsq_y = NULL, *sumxy = NULL; int64_t* counts = NULL; if (K > 0) { @@ -10515,6 +10872,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, (size_t)K * n_aggs * sizeof(da_val_t)); counts = (int64_t*)scratch_calloc(&cnt_hdr, (size_t)K * sizeof(int64_t)); + if (need_sum128) + sums_hi = (int64_t*)scratch_calloc(&sum_hi_hdr, + (size_t)K * n_aggs * sizeof(int64_t)); if (need_sumsq) sumsq = (double*)scratch_calloc(&sumsq_hdr, (size_t)K * n_aggs * sizeof(double)); @@ -10533,12 +10893,16 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, n_tasks += (slices[i].n + SG_CHUNK_ROWS - 1) / SG_CHUNK_ROWS; sg_task_t* tasks = NULL; da_val_t* partials = NULL; + int64_t* part_hi = NULL; double *part_sumsq = NULL, *part_sum_y = NULL, *part_sumsq_y = NULL, *part_sumxy = NULL; if (sums && counts) { task_hdr = ray_alloc((size_t)n_tasks * (int64_t)sizeof(sg_task_t)); tasks = task_hdr ? (sg_task_t*)ray_data(task_hdr) : NULL; partials = (da_val_t*)scratch_calloc(&part_hdr, (size_t)n_tasks * n_aggs * sizeof(da_val_t)); + if (need_sum128) + part_hi = (int64_t*)scratch_calloc(&part_hi_hdr, + (size_t)n_tasks * n_aggs * sizeof(int64_t)); if (need_sumsq) part_sumsq = (double*)scratch_calloc(&part_sumsq_hdr, (size_t)n_tasks * n_aggs * sizeof(double)); @@ -10552,12 +10916,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } if (!sums || !counts || !tasks || !partials || + (need_sum128 && (!sums_hi || !part_hi)) || (need_sumsq && (!sumsq || !part_sumsq)) || (need_pair && (!sum_y || !sumsq_y || !sumxy || !part_sum_y || !part_sumsq_y || !part_sumxy))) { scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); scratch_free(part_hi_hdr); if (task_hdr) ray_free(task_hdr); scratch_free(part_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); @@ -10574,7 +10940,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } sg_ctx_t ctx = { slices, tasks, agg_vecs, agg_vecs2, prod, ext->agg_ops, - n_aggs, partials, part_sumsq, part_sum_y, + n_aggs, partials, part_hi, part_sumsq, part_sum_y, part_sumsq_y, part_sumxy, {0}, {0} }; /* Shared-stream pairing: a bare-scan SUM/AVG over the same column * a product's int side already streams rides the product loop — @@ -10589,10 +10955,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (prod[a].ta != RAY_F64) { ip = prod[a].pa; it = prod[a].ta; } else if (prod[a].tb != RAY_F64) { ip = prod[a].pb; it = prod[a].tb; } else continue; /* F64×F64 — no int side */ - if (it != RAY_I64 && it != RAY_TIME && it != RAY_I32) continue; + if (it != RAY_I64 && it != RAY_I32) continue; for (uint32_t b = 0; b < n_aggs; b++) { if (b == a || !agg_vecs[b] || ctx.fused_by[b] >= 0) continue; if (ext->agg_ops[b] != OP_SUM && ext->agg_ops[b] != OP_AVG) continue; + /* The product loop yields only the wrapped int64 side sum; + * an integer AVG needs the 128-bit total, so it streams + * its own column. */ + if (ext->agg_ops[b] == OP_AVG && part_hi) continue; if (ray_data(agg_vecs[b]) != ip || agg_vecs[b]->type != it) continue; ctx.pair_sum[a] = (int8_t)b; ctx.fused_by[b] = (int8_t)a; @@ -10619,6 +10989,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (pair || prod[a].enabled || ext->agg_ops[a] == OP_COUNT || (agg_vecs[a] && group_fp_type(agg_vecs[a]->type))) sums[di].f += partials[si].f; + else if (sums_hi) + ray_i128_add128(&sums_hi[di], (uint64_t*)&sums[di].i, + part_hi[si], (uint64_t)partials[si].i); else sums[di].i = (int64_t)((uint64_t)sums[di].i + (uint64_t)partials[si].i); @@ -10631,7 +11004,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } ray_free(task_hdr); - scratch_free(part_hdr); + scratch_free(part_hdr); scratch_free(part_hi_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); scratch_free(part_sumsq_y_hdr); scratch_free(part_sumxy_hdr); } @@ -10644,6 +11017,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } ray_t* kc = col_vec_new(key_col, K > 0 ? K : 1); @@ -10653,6 +11027,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } if (kc->type == RAY_SYM) @@ -10667,17 +11042,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } emit_agg_columns(&result, g, ext, agg_vecs, (uint32_t)K, n_aggs, - (double*)sums, (int64_t*)sums, + (double*)sums, (int64_t*)sums, sums_hi, NULL, NULL, NULL, NULL, counts, NULL, prod, sumsq, NULL, sum_y, sumsq_y, sumxy); scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } @@ -10837,6 +11214,7 @@ typedef struct { int64_t n_scan; uint8_t key_esz; bool sp_need_sum; + bool sp_need_sum128; /* an integer AVG: carry high words */ const int64_t* match_idx; ray_t* rowsel; ray_t* match_idx_block; @@ -10907,6 +11285,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { int64_t n_scan = c->n_scan; uint8_t key_esz = c->key_esz; bool sp_need_sum = c->sp_need_sum; + bool sp_need_sum128 = c->sp_need_sum128; const int64_t* match_idx = c->match_idx; ray_t* rowsel = c->rowsel; ray_t* match_idx_block = c->match_idx_block; @@ -10918,16 +11297,23 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { : (1u << 20); const uint64_t max_dense_cap = 1u << 24; bool count_only_first = (key_types[0] == RAY_SYM); - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)cap * sizeof(uint32_t)); da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ bool dyn_ok = range_count != NULL; if (dyn_ok && sp_need_sum && !count_only_first) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, (size_t)cap * n_aggs * sizeof(da_val_t)); dyn_ok = range_sum != NULL; + if (dyn_ok && sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)cap * n_aggs * sizeof(int64_t)); + dyn_ok = range_sum_hi != NULL; + } } uint64_t max_seen = 0; @@ -11044,6 +11430,19 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memset(range_sum + (size_t)old_cap * n_aggs, 0, \ (size_t)(cap - old_cap) * n_aggs * sizeof(da_val_t)); \ } \ + if (range_sum_hi) { \ + int64_t* new_hi = (int64_t*)scratch_realloc( \ + &range_hi_hdr, \ + (size_t)old_cap * n_aggs * sizeof(int64_t), \ + (size_t)cap * n_aggs * sizeof(int64_t)); \ + if (!new_hi) { \ + dyn_ok = false; \ + goto dyn_dense_done; \ + } \ + range_sum_hi = new_hi; \ + memset(range_sum_hi + (size_t)old_cap * n_aggs, 0, \ + (size_t)(cap - old_cap) * n_aggs * sizeof(int64_t)); \ + } \ } \ have_dyn_key = true; \ if (off > max_seen) max_seen = off; \ @@ -11059,6 +11458,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (range_sum_hi) \ + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11113,7 +11516,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11128,7 +11531,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11139,16 +11542,21 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11157,9 +11565,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (sp_need_sum && !range_sum) + if (sp_need_sum && !range_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, + (size_t)grp_count * n_aggs * sizeof(int64_t)); + } uint32_t gi = 0; for (uint64_t off = 0; off <= max_seen; off++) { @@ -11175,6 +11587,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } if (!range_sum) range_count[off] = gi + 1u; gi++; @@ -11198,6 +11614,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (dense_sum_hi) \ + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11243,13 +11663,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { * row is non-null and the legacy count-based divisor is * correct. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11258,7 +11678,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { return result; } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); /* Dynamic-dense probe bailed (unbounded key or no surviving row): shared @@ -11807,6 +12227,8 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, * once n_aggs exceeded its bit width. */ bool *sc_int_null_has = (bool*)(agg_types + vla_aggs); bool sc_any_nullable = false; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool sc_need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { if (agg_prod[a].enabled) { /* Fused product: F64 accumulate, no source vec. */ @@ -11837,6 +12259,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, sc_int_null_sentinel[a] = 0; sc_int_null_has[a] = false; } + if (ext->agg_ops[a] == OP_AVG && !group_fp_type(agg_types[a]) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + sc_need_sum128 = true; } if (!match_idx && !rowsel && !sc_any_nullable && n_aggs > 1) { @@ -11892,8 +12317,21 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } + /* An integer AVG needs the exact 128-bit total: take it from + * the chunk-zone metadata when the column carries it (the + * same (hi, lo) ray_avg_fn reads), otherwise skip the shortcut + * and let the parallel scan below carry the high word. */ + int64_t zone_hi = 0, zone_nn = 0; uint64_t zone_lo = 0; + bool zone_128 = false; + if (one_base_input && base_col && sc_need_sum128 && + !group_fp_type(base_type)) { + zone_128 = ray_zone_int_sum128(base_col, &zone_hi, &zone_lo, &zone_nn) + && zone_nn == nrows; + if (!zone_128) one_base_input = false; + } if (one_base_input && base_col) { - ray_t* base_sum_obj = ray_sum_fn(base_col); + ray_t* base_sum_obj = zone_128 ? ray_i64((int64_t)zone_lo) + : ray_sum_fn(base_col); if (!base_sum_obj || RAY_IS_ERR(base_sum_obj)) { scratch_free(sc_vla_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -11925,12 +12363,14 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, ray_t *sum_hdr = NULL, *cnt_hdr = NULL; double* sums_f64 = NULL; int64_t* sums_i64 = NULL; + int64_t* sums_hi = NULL; if (base_is_f64) sums_f64 = (double*)scratch_alloc(&sum_hdr, (size_t)n_aggs * sizeof(double)); else + /* [sums | high words] in one carve */ sums_i64 = (int64_t*)scratch_alloc(&sum_hdr, - (size_t)n_aggs * sizeof(int64_t)); + (size_t)2 * n_aggs * sizeof(int64_t)); int64_t* counts = (int64_t*)scratch_alloc(&cnt_hdr, sizeof(int64_t)); if ((base_is_f64 ? (sums_f64 != NULL) : (sums_i64 != NULL)) && @@ -11941,6 +12381,11 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { for (uint32_t a = 0; a < n_aggs; a++) sums_i64[a] = base_sum_i64; + if (zone_128) { + sums_hi = sums_i64 + n_aggs; + for (uint32_t a = 0; a < n_aggs; a++) + sums_hi[a] = zone_hi; + } } counts[0] = nrows; @@ -11960,7 +12405,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - sums_f64, sums_i64, + sums_f64, sums_i64, sums_hi, NULL, NULL, NULL, NULL, counts, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); @@ -12013,6 +12458,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const size_t sc_line = 64; size_t sc_words = 1; /* count[1] */ if (need_flags & DA_NEED_SUM) sc_words += n_aggs; /* sum */ + if (sc_need_sum128) sc_words += n_aggs; /* sum_hi */ if (need_flags & DA_NEED_MIN) sc_words += n_aggs; /* min_val */ if (need_flags & DA_NEED_MAX) sc_words += n_aggs; /* max_val */ if (need_flags & DA_NEED_SUMSQ) sc_words += n_aggs; /* sumsq_f64 */ @@ -12029,6 +12475,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (need_flags & DA_NEED_SUM) { sc_acc[w].sum = (da_val_t*)(void*)(blk + off); off += n_aggs; } + if (sc_need_sum128) { + sc_acc[w].sum_hi = blk + off; off += n_aggs; + } if (need_flags & DA_NEED_MIN) { sc_acc[w].min_val = (da_val_t*)(void*)(blk + off); off += n_aggs; for (uint32_t a = 0; a < n_aggs; a++) { @@ -12139,6 +12588,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { if (group_fp_type(agg_types[a])) m->sum[a].f += wa->sum[a].f; + else if (m->sum_hi) + ray_i128_add128(&m->sum_hi[a], (uint64_t*)&m->sum[a].i, + wa->sum_hi[a], (uint64_t)wa->sum[a].i); else m->sum[a].i = wrap_add_i64(m->sum[a].i, wa->sum[a].i); } @@ -12204,7 +12656,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - (double*)m->sum, (int64_t*)m->sum, + (double*)m->sum, (int64_t*)m->sum, m->sum_hi, (double*)m->min_val, (double*)m->max_val, (int64_t*)m->min_val, (int64_t*)m->max_val, m->count, agg_affine, agg_prod, m->sumsq_f64, m->nn_count, @@ -12568,6 +13020,8 @@ da_path:; int64_t da_int_null_sentinel[vla_aggs]; uint64_t agg_f64_mask = 0; uint64_t da_int_null_mask = 0; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool da_need_sum128 = false; /* Track whether any agg column can produce a null so we can * allocate per-(group, agg) non-null counts only when required. * F64 with HAS_NULLS uses NaN-skip; sentinel-typed integers @@ -12619,6 +13073,9 @@ da_path:; agg_types[a] = 0; da_int_null_sentinel[a] = 0; } + if (ext->agg_ops[a] == OP_AVG && !(agg_f64_mask & ((uint64_t)1 << a)) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + da_need_sum128 = true; } ray_pool_t* da_pool = ray_pool_get(); @@ -12629,6 +13086,7 @@ da_path:; * cells must remain O(contributing rows). */ uint32_t arrays_per_agg = 0; if (need_flags & DA_NEED_SUM) arrays_per_agg += 1; + if (da_need_sum128) arrays_per_agg += 1; /* sum_hi */ if (need_flags & DA_NEED_MIN) arrays_per_agg += 1; if (need_flags & DA_NEED_MAX) arrays_per_agg += 1; if (need_flags & DA_NEED_SUMSQ) arrays_per_agg += 1; @@ -12641,8 +13099,32 @@ da_path:; uint64_t max_workers = cells_per_worker ? (uint64_t)n_scan / cells_per_worker : 1; if (max_workers < 1) max_workers = 1; - if ((uint64_t)da_n_workers > max_workers) + /* More workers than the budget allows: keep the budget's worth + * of accumulators and give each one a contiguous row range (one + * task per accumulator, whichever worker runs it) instead of + * collapsing to a serial scan of every row. */ + bool da_ranged = false; + uint32_t da_part_tasks = 0; + if ((uint64_t)da_n_workers > max_workers && da_has_first_last) { + /* Several FIRST/LAST aggregates share one first_row/last_row + * per slot (see da_accum_row): the answer for a null-mixed + * pair depends on the order the rows arrive in, so past the + * budget those keep the serial scan in row order. */ da_n_workers = 1; + } else if ((uint64_t)da_n_workers > max_workers) { + if (n_slots >= 2 * da_n_workers && nrows <= INT32_MAX) { + /* Partition the SLOTS instead: one pass buckets the row + * ids by slot span, then one task per span drains its + * buckets into the single accumulator set. Twice as + * many spans as workers evens out skewed groups. */ + da_part_tasks = 2 * da_n_workers; + if (da_part_tasks > n_slots) da_part_tasks = n_slots; + da_n_workers = 1; + } else { + da_n_workers = (uint32_t)max_workers; + da_ranged = da_n_workers > 1; + } + } ray_t* accums_hdr; da_accum_t* accums = (da_accum_t*)scratch_calloc(&accums_hdr, @@ -12656,6 +13138,11 @@ da_path:; total * sizeof(da_val_t)); if (!accums[w].sum) { alloc_ok = false; break; } } + if (da_need_sum128) { + accums[w].sum_hi = (int64_t*)scratch_calloc(&accums[w]._h_sum_hi, + total * sizeof(int64_t)); + if (!accums[w].sum_hi) { alloc_ok = false; break; } + } if (need_flags & DA_NEED_SUMSQ) { accums[w].sumsq_f64 = (double*)scratch_calloc(&accums[w]._h_sumsq, total * sizeof(double)); @@ -12768,12 +13255,33 @@ da_path:; .n_slots = n_slots, .match_idx = match_idx, .rowsel = rowsel, + .da_n_scan = n_scan, }; + /* Pin each SYM agg column's vocabulary once for the accumulate. */ + ray_t* agg_raw_hdr = NULL; + if (n_aggs > 0) { + char* rm = (char*)scratch_calloc(&agg_raw_hdr, + (size_t)n_aggs * (sizeof(ray_sym_domain_raw_t) + 1)); + if (rm) { + da_ctx.agg_raw = (ray_sym_domain_raw_t*)rm; + da_ctx.agg_raw_ok = (uint8_t*)(rm + (size_t)n_aggs * sizeof(ray_sym_domain_raw_t)); + for (uint32_t a = 0; a < n_aggs; a++) + if (agg_vecs[a] && agg_vecs[a]->type == RAY_SYM) + da_ctx.agg_raw_ok[a] = ray_sym_domain_raw_pin( + ray_sym_vec_domain(agg_vecs[a]), &da_ctx.agg_raw[a]) ? 1 : 0; + } + } - if (da_n_workers > 1) + if (da_part_tasks > 0) { + if (!da_accum_partitioned(&da_ctx, da_pool, da_part_tasks, n_scan)) + da_accum_fn(&da_ctx, 0, 0, n_scan); + } else if (da_ranged) + ray_pool_dispatch_n(da_pool, da_accum_task_fn, &da_ctx, da_n_workers); + else if (da_n_workers > 1) ray_pool_dispatch(da_pool, da_accum_fn, &da_ctx, n_scan); else da_accum_fn(&da_ctx, 0, 0, n_scan); + if (agg_raw_hdr) scratch_free(agg_raw_hdr); /* Merge target is always accums[0] */ da_accum_t* merged = &accums[0]; @@ -12816,6 +13324,7 @@ da_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP || agg_is_binary_agg(aop)) { if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } else if (aop == OP_PROD) { /* Use per-(group, agg) non-null counts when @@ -12942,6 +13451,9 @@ da_path:; /* binary aggs accumulate Σx as double even * for integer x-columns — merge as float. */ merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -13002,6 +13514,7 @@ da_path:; da_accum_free(&accums[w]); da_val_t* da_sum = merged->sum; /* may be NULL if !DA_NEED_SUM */ + int64_t* da_sum_hi = merged->sum_hi; /* NULL unless an integer AVG */ da_val_t* da_min_val = merged->min_val; /* may be NULL if !DA_NEED_MIN */ da_val_t* da_max_val = merged->max_val; /* may be NULL if !DA_NEED_MAX */ double* da_sumsq = merged->sumsq_f64; /* may be NULL if !DA_NEED_SUMSQ */ @@ -13067,8 +13580,9 @@ da_path:; size_t dense_total = (size_t)grp_count * n_aggs; ray_t *_h_dsum = NULL, *_h_dmin = NULL, *_h_dmax = NULL; ray_t *_h_dsq = NULL, *_h_dcnt = NULL, *_h_dnn = NULL; - ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL; + ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL, *_h_dshi = NULL; da_val_t* dense_sum = da_sum ? (da_val_t*)scratch_alloc(&_h_dsum, dense_total * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = da_sum_hi ? (int64_t*)scratch_alloc(&_h_dshi, dense_total * sizeof(int64_t)) : NULL; da_val_t* dense_min_val = da_min_val ? (da_val_t*)scratch_alloc(&_h_dmin, dense_total * sizeof(da_val_t)) : NULL; da_val_t* dense_max_val = da_max_val ? (da_val_t*)scratch_alloc(&_h_dmax, dense_total * sizeof(da_val_t)) : NULL; double* dense_sumsq = da_sumsq ? (double*)scratch_alloc(&_h_dsq, dense_total * sizeof(double)) : NULL; @@ -13090,6 +13604,7 @@ da_path:; size_t si = (size_t)s * n_aggs + a; size_t di = (size_t)gi * n_aggs + a; if (dense_sum) dense_sum[di] = da_sum[si]; + if (dense_sum_hi) dense_sum_hi[di] = da_sum_hi[si]; if (dense_min_val) dense_min_val[di] = da_min_val[si]; if (dense_max_val) dense_max_val[di] = da_max_val[si]; if (dense_sumsq) dense_sumsq[di] = da_sumsq[si]; @@ -13102,7 +13617,7 @@ da_path:; } emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, (double*)dense_min_val, (double*)dense_max_val, (int64_t*)dense_min_val, (int64_t*)dense_max_val, dense_counts, agg_affine, agg_prod, dense_sumsq, @@ -13113,6 +13628,7 @@ da_path:; scratch_free(_h_dsq); scratch_free(_h_dcnt); scratch_free(_h_dnn); scratch_free(_h_dsy); scratch_free(_h_dsqy); scratch_free(_h_dxy); + scratch_free(_h_dshi); da_accum_free(&accums[0]); scratch_free(accums_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -13142,6 +13658,9 @@ da_path:; sp_eligible = false; } bool sp_need_sum = false; + /* An integer AVG divides the exact 128-bit sum: the scatter arrays + * and the sparse table then carry a high word per slot. */ + bool sp_need_sum128 = false; for (uint32_t a = 0; a < n_aggs && sp_eligible; a++) { uint16_t op = ext->agg_ops[a]; if (op == OP_COUNT) continue; @@ -13156,8 +13675,13 @@ da_path:; * accum_from_entry inherits the same nullable-agg gap.) */ if (agg_vecs[a] && ray_vec_may_have_nulls(agg_vecs[a])) sp_eligible = false; - else + else { sp_need_sum = true; + if (op == OP_AVG && !(agg_vecs[a] && group_fp_type(agg_vecs[a]->type)) && + !agg_strlen[a] && + group_avg_needs_hi(agg_vecs[a] ? agg_vecs[a]->type : 0, nrows)) + sp_need_sum128 = true; + } } } @@ -13237,7 +13761,8 @@ da_path:; .strlen_sym_count = strlen_sym_count, .agg_f64_mask = agg_f64_mask, .n_aggs = n_aggs, .n_keys = n_keys, .n_scan = n_scan, .key_esz = key_esz, - .sp_need_sum = sp_need_sum, .match_idx = match_idx, + .sp_need_sum = sp_need_sum, .sp_need_sum128 = sp_need_sum128, + .match_idx = match_idx, .rowsel = rowsel, .match_idx_block = match_idx_block, .vla_hdr = vla_hdr, .emit_filter = emit_filter, }; @@ -13269,13 +13794,14 @@ da_path:; ? (uint64_t)((uint64_t)max_key - (uint64_t)min_key + 1u) : 0u; if (have_key && key_range > 0 && key_range <= (1u << 26)) { - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)key_range * sizeof(uint32_t)); if (!range_count) goto ht_path; da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ if (sp_need_sum && key_range <= (1u << 24)) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, @@ -13284,6 +13810,16 @@ da_path:; scratch_free(cnt_hdr); goto ht_path; } + if (sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)key_range * n_aggs * sizeof(int64_t)); + if (!range_sum_hi) { + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); + scratch_free(cnt_hdr); + goto ht_path; + } + } } for (int64_t i = 0; i < n_scan; i++) { @@ -13307,6 +13843,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (range_sum_hi) + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13330,7 +13870,7 @@ da_path:; ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13346,7 +13886,7 @@ da_path:; /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13362,11 +13902,16 @@ da_path:; ? (da_val_t*)scratch_calloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_calloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13392,6 +13937,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } range_count[off] = gi + 1u; gi++; @@ -13418,6 +13967,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13426,7 +13979,7 @@ da_path:; } } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_op_ext_t* key_ext = find_ext(g, ext->keys[0]); int64_t name_id = key_ext ? key_ext->sym : 0; @@ -13437,12 +13990,12 @@ da_path:; * emit-filter range path only runs when sp_eligible was * true. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13464,7 +14017,7 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false, false)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13483,7 +14036,8 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum, + sp_need_sum128)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13510,6 +14064,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (sp_ht.sums_hi) + ray_i128_add(&sp_ht.sums_hi[(size_t)slot * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13565,15 +14123,20 @@ da_path:; } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc(&_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13583,14 +14146,17 @@ da_path:; if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (use_emit_filter && sp_need_sum) + if (use_emit_filter && sp_need_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, (size_t)grp_count * n_aggs * sizeof(int64_t)); + } sparse_i64_ht_t heavy_ht; memset(&heavy_ht, 0, sizeof(heavy_ht)); if (use_emit_filter && grp_count > 0) { - if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false, false)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13612,7 +14178,7 @@ da_path:; if (use_emit_filter) { int32_t hslot; if (!sparse_i64_touch(&heavy_ht, sp_ht.keys[s], n_aggs, false, &hslot)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&heavy_ht); sparse_i64_free(&sp_ht); @@ -13628,6 +14194,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &sp_ht.sums[(size_t)s * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &sp_ht.sums_hi[(size_t)s * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } gi++; } @@ -13655,6 +14225,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)out_gi * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13672,12 +14246,12 @@ da_path:; * and is gated to null-free agg columns (sp_eligible guard at * ~line 5737), so counts[gi] is the correct divisor. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13715,6 +14289,10 @@ ht_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_PROD || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_FIRST || aop == OP_LAST) ght_need |= GHT_NEED_SUM; + /* Integer avg divides the exact 128-bit sum: carry its high words. */ + if (aop == OP_AVG && agg_vecs[a] && !group_fp_type(agg_vecs[a]->type) && + !agg_strlen[a] && group_avg_needs_hi(agg_vecs[a]->type, nrows)) + ght_need |= GHT_NEED_SUM128; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP) { ght_need |= GHT_NEED_SUM; ght_need |= GHT_NEED_SUMSQ; } if (agg_is_binary_agg(aop)) @@ -15377,7 +15955,10 @@ sequential_fallback:; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } v = is_f64 ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (agg_affine[a].enabled) v += agg_affine[a].bias_f64; break; case OP_MIN: diff --git a/src/ops/idxop.c b/src/ops/idxop.c index 6927a52dc..cc9f9a36e 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -32,6 +32,8 @@ #include "ops/ops.h" #include "ops/rowsel.h" #include "ops/hash.h" /* ray_hash_bytes: STR hash-index key word */ +#include "core/pool.h" /* parallel hash-index build */ +#include "mem/sys.h" /* ray_sys_alloc: build scratch off the buddy heap */ #include #include #include @@ -376,9 +378,12 @@ void ray_index_release_payload(ray_index_t* ix) { ray_release(ix->u.chunk_zone.maxs); if (ix->u.chunk_zone.null_bits && !RAY_IS_ERR(ix->u.chunk_zone.null_bits)) ray_release(ix->u.chunk_zone.null_bits); + if (ix->u.chunk_zone.aggs && !RAY_IS_ERR(ix->u.chunk_zone.aggs)) + ray_release(ix->u.chunk_zone.aggs); ix->u.chunk_zone.mins = NULL; ix->u.chunk_zone.maxs = NULL; ix->u.chunk_zone.null_bits = NULL; + ix->u.chunk_zone.aggs = NULL; break; case RAY_IDX_PART: if (ix->u.part.keys && !RAY_IS_ERR(ix->u.part.keys)) ray_release(ix->u.part.keys); @@ -409,7 +414,7 @@ int ray_index_child_blocks(const ray_index_t* ix, ray_t** out, int cap) { case RAY_IDX_BLOOM: c[0] = ix->u.bloom.bits; break; case RAY_IDX_CHUNK_ZONE: c[0] = ix->u.chunk_zone.mins; c[1] = ix->u.chunk_zone.maxs; - c[2] = ix->u.chunk_zone.null_bits; + c[2] = ix->u.chunk_zone.null_bits; c[3] = ix->u.chunk_zone.aggs; break; case RAY_IDX_PART: c[0] = ix->u.part.keys; c[1] = ix->u.part.starts; c[2] = ix->u.part.lens; @@ -458,6 +463,8 @@ void ray_index_retain_payload(ray_index_t* ix) { ray_retain(ix->u.chunk_zone.maxs); if (ix->u.chunk_zone.null_bits && !RAY_IS_ERR(ix->u.chunk_zone.null_bits)) ray_retain(ix->u.chunk_zone.null_bits); + if (ix->u.chunk_zone.aggs && !RAY_IS_ERR(ix->u.chunk_zone.aggs)) + ray_retain(ix->u.chunk_zone.aggs); break; case RAY_IDX_PART: if (ix->u.part.keys && !RAY_IS_ERR(ix->u.part.keys)) ray_retain(ix->u.part.keys); @@ -598,6 +605,9 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, int64_t s = (int64_t)g * csz; int64_t e = s + csz; if (e > n) e = n; int64_t mn = INT64_MAX, mx = INT64_MIN; + uint64_t sum = 0; /* low word: wraps like the engine's int64 sum */ + int64_t hi = 0; /* high word of the exact 128-bit sum */ + int64_t nn = 0; bool any_null = false; for (int64_t i = s; i < e; i++) { if (ray_vec_is_null(v, i)) { any_null = true; continue; } @@ -611,6 +621,15 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, } if (val < mn) mn = val; if (val > mx) mx = val; + ray_i128_add(&hi, &sum, val); + nn++; + } + if (ix->u.chunk_zone.aggs) { + int64_t* ag = (int64_t*)ray_data(ix->u.chunk_zone.aggs); + ag[g] = (int64_t)sum; + ag[n_chunks + g] = nn; + if (ix->u.chunk_zone.aggs->len >= 3 * (int64_t)n_chunks) + ag[2 * (int64_t)n_chunks + g] = hi; } /* Empty (all-null) chunks keep mn=INT64_MAX / mx=INT64_MIN so * the reduce path's min(mins[*]) / max(maxs[*]) ignores them. */ @@ -807,6 +826,13 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2) { ix->u.chunk_zone.mins = mins; ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; + if (!ix->u.chunk_zone.is_f64) { + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); + if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } + aggs->len = 3 * (int64_t)n_chunks; + ix->u.chunk_zone.aggs = aggs; + } ray_err_t err = chunk_zone_scan(v, ix); if (err != RAY_OK) { @@ -857,6 +883,13 @@ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2) { ix->u.chunk_zone.mins = mins; ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; + if (!ix->u.chunk_zone.is_f64) { + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); + if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } + aggs->len = 3 * (int64_t)n_chunks; + ix->u.chunk_zone.aggs = aggs; + } ray_err_t err = chunk_zone_scan(v, ix); if (err != RAY_OK) { @@ -980,7 +1013,11 @@ static int idx_child_slots(ray_index_t* ix, ray_t** slots[4]) { case RAY_IDX_CHUNK_ZONE: slots[n++] = &ix->u.chunk_zone.mins; slots[n++] = &ix->u.chunk_zone.maxs; - slots[n++] = &ix->u.chunk_zone.null_bits; break; + slots[n++] = &ix->u.chunk_zone.null_bits; + /* 4th slot added after the first on-disk generation: older regions + * hold zero there (the payload is zeroed at alloc), which maps to + * NULL — no aggregates, nothing else changes. */ + slots[n++] = &ix->u.chunk_zone.aggs; break; case RAY_IDX_PART: slots[n++] = &ix->u.part.keys; slots[n++] = &ix->u.part.starts; slots[n++] = &ix->u.part.lens; break; @@ -1053,7 +1090,9 @@ void ray_index_inline_write(uint8_t* dst, const ray_index_t* ix) { * points at the start of the index region within the column's file mapping. * Returns NULL for a stale layout generation or a payload-size mismatch — * the caller loads the column unindexed (the index is rebuildable). */ -ray_t* ray_index_inline_map(uint8_t* region) { +ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size) { + int64_t head = IDX_ALIGN32(32 + (int64_t)sizeof(ray_index_t)); + if (region_size < head) return NULL; ray_t* idx = (ray_t*)region; if (idx->order != RAY_IDX_FORMAT_MAJOR) return NULL; if (idx->len != (int64_t)sizeof(ray_index_t)) return NULL; @@ -1062,8 +1101,34 @@ ray_t* ray_index_inline_map(uint8_t* region) { int nch = idx_child_slots(ix, slots); for (int i = 0; i < nch; i++) { int64_t o = (int64_t)(intptr_t)(*slots[i]); - *slots[i] = o ? (ray_t*)(region + o) : NULL; - } + ray_t* c = NULL; + /* A child must lie inside the region: a region re-saved by a + * binary that knows fewer child slots keeps a stale offset in a + * slot it did not write. */ + if (o >= head && o <= region_size - 32) { + ray_t* cand = (ray_t*)(region + o); + int64_t esz = ray_elem_size(cand->type); + if (cand->len >= 0 && esz > 0 && + cand->len <= (region_size - o - 32) / esz) + c = cand; + } + if (o && !c) { + /* The chunk-zone aggregates are optional; any other child out + * of bounds means the region cannot be trusted. */ + if (ix->kind == RAY_IDX_CHUNK_ZONE && slots[i] == &ix->u.chunk_zone.aggs) { + *slots[i] = NULL; + continue; + } + return NULL; + } + *slots[i] = c; + } + /* two layouts: [lo | nn] (2 per chunk) and [lo | nn | hi] (3 per chunk) */ + if (ix->kind == RAY_IDX_CHUNK_ZONE && ix->u.chunk_zone.aggs && + (ix->u.chunk_zone.aggs->type != RAY_I64 || ix->u.chunk_zone.is_f64 || + (ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks && + ix->u.chunk_zone.aggs->len != 3 * (int64_t)ix->u.chunk_zone.n_chunks))) + ix->u.chunk_zone.aggs = NULL; ix->markers |= RAY_MARK_MMAP; idx->mmod = 1; return idx; @@ -1084,6 +1149,324 @@ ray_t* ray_index_inline_map(uint8_t* region) { * hit yields the contiguous ascending slice rows[offs[gid]..offs[gid+1]). * -------------------------------------------------------------------------- */ +/* Parallel build of the CSR hash layout for numeric / SYM keys. The + * result is the serial walk's: groups numbered by first occurrence, rows + * ascending inside a group, nulls excluded — assembled in + * partition-parallel passes: + * + * A row ranges: key word per row, rows bucketed by hash partition + * (a partition's rows stay ascending: the ranges are in row order); + * B per partition: open-addressing dedupe into local groups, each + * group's first row marked in a bitmap; + * C a group's number is the rank of its first row among all marked + * rows (block popcounts + one prefix) — exactly the order the + * serial walk assigns; + * D per partition: keys and counts into the group's slots; + * E per partition: row scatter (of[] as cursor); then the key table, + * filled in group order on the calling thread so its bytes match the + * serial build. + * + * Every array a pass writes at random is faulted in beforehand from all + * workers in slices: a fresh mapping faulted at random from every worker + * serialises on the page-table locks. Returns false (nothing allocated) + * when the pool cannot be used; the caller then runs the serial walk. */ +typedef struct { + ray_t* v; + const uint8_t* base; + int64_t n; + int n_tasks; + int n_part; + int part_shift; /* partition = mix64(key) >> shift */ + uint64_t* kw; /* [n] key word (non-null rows) */ + int64_t* pr; /* [n_keys] row ids grouped by partition */ + int64_t* lg; /* [n_keys] local group of pr[j] */ + int64_t* cnt; /* [n_tasks * n_part] rows per (task, partition) */ + int64_t* part_off; /* [n_part + 1] */ + int64_t* ng_p; /* [n_part] local groups per partition */ + int64_t* gfirst; /* [n_keys] first row per local group (partition-relative) */ + int64_t* gcount; /* [n_keys] rows per local group, then its global number */ + uint64_t* bits; /* [n/64 + 1] first-row marks */ + int64_t* blk_rank; /* [n/HP_BLOCK + 1] exclusive prefix of block popcounts */ + int64_t* gk; /* [n_groups] keys */ + int64_t* of; /* [n_groups + 1] */ + int64_t* rw; /* [n_keys] */ + int64_t* tbl; /* [cap] */ + uint64_t tmask; + _Atomic(bool) oom; +} hash_par_t; + +#define HP_BLOCK 4096 + +static inline int64_t hp_task_lo(const hash_par_t* h, int64_t t) { return h->n * t / h->n_tasks; } + +static void hp_pass_a(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t lo = hp_task_lo(h, start), hi = hp_task_lo(h, start + 1); + int64_t* cnt = h->cnt + start * h->n_part; + for (int64_t i = lo; i < hi; i++) { + if (ray_vec_is_null(h->v, i)) { h->kw[i] = 0; continue; } + uint64_t k = hash_row_key_word(h->v, h->base, i); + h->kw[i] = k; + cnt[mix64(k) >> h->part_shift]++; + } +} + +static void hp_pass_a2(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t lo = hp_task_lo(h, start), hi = hp_task_lo(h, start + 1); + int64_t* cur = h->cnt + start * h->n_part; /* now the write cursors */ + for (int64_t i = lo; i < hi; i++) { + if (ray_vec_is_null(h->v, i)) continue; + h->pr[cur[mix64(h->kw[i]) >> h->part_shift]++] = i; + } +} + +static void hp_pass_b(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p], hi = h->part_off[p + 1]; + int64_t cnt = hi - lo; + h->ng_p[p] = 0; + if (cnt == 0) return; + uint64_t cap = next_pow2((uint64_t)cnt * 2 + 1); + if (cap < 16) cap = 16; + int64_t* tab = (int64_t*)ray_sys_alloc((size_t)cap * sizeof(int64_t)); + if (!tab) { atomic_store_explicit(&h->oom, true, memory_order_relaxed); return; } + memset(tab, 0, (size_t)cap * sizeof(int64_t)); + uint64_t mask = cap - 1; + int64_t* gfirst = h->gfirst + lo; + int64_t* gcount = h->gcount + lo; + int64_t ng = 0; + for (int64_t j = lo; j < hi; j++) { + int64_t i = h->pr[j]; + uint64_t k = h->kw[i]; + uint64_t slot = mix64(k) & mask; + for (;;) { + int64_t g1 = tab[slot]; + if (g1 == 0) { + tab[slot] = ng + 1; + gfirst[ng] = i; + gcount[ng] = 1; + h->lg[j] = ng; + /* the partition's rows are ascending, so i is this group's + * first row; the word is shared with other partitions */ + atomic_fetch_or_explicit((_Atomic(uint64_t)*)&h->bits[i >> 6], + (uint64_t)1 << (i & 63), memory_order_relaxed); + ng++; + break; + } + if (h->kw[gfirst[g1 - 1]] == k) { h->lg[j] = g1 - 1; gcount[g1 - 1]++; break; } + slot = (slot + 1) & mask; + } + } + h->ng_p[p] = ng; + ray_sys_free(tab); +} + +/* per-block popcount of the first-row marks (prefixed serially after) */ +static void hp_pass_c(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t w0 = start * (HP_BLOCK / 64), w1 = w0 + HP_BLOCK / 64; + int64_t nw = (h->n + 63) / 64; + if (w1 > nw) w1 = nw; + int64_t c = 0; + for (int64_t w = w0; w < w1; w++) c += __builtin_popcountll(h->bits[w]); + h->blk_rank[start] = c; +} + +static inline int64_t hp_rank(const hash_par_t* h, int64_t i) { + int64_t blk = i / HP_BLOCK; + int64_t r = h->blk_rank[blk]; + int64_t w = i >> 6; + for (int64_t x = blk * (HP_BLOCK / 64); x < w; x++) r += __builtin_popcountll(h->bits[x]); + return r + __builtin_popcountll(h->bits[w] & (((uint64_t)1 << (i & 63)) - 1)); +} + +static void hp_pass_d(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p]; + const int64_t* gfirst = h->gfirst + lo; + int64_t* gcount = h->gcount + lo; + int64_t ng = h->ng_p[p]; + for (int64_t g = 0; g < ng; g++) { + int64_t G = hp_rank(h, gfirst[g]); + h->gk[G] = (int64_t)h->kw[gfirst[g]]; + h->of[G + 1] = gcount[g]; + gcount[g] = G; /* local -> global from here on */ + } +} + +static void hp_pass_e(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hash_par_t* h = (hash_par_t*)raw; + int64_t p = start; + int64_t lo = h->part_off[p], hi = h->part_off[p + 1]; + const int64_t* gmap = h->gcount + lo; + /* rows of a partition are ascending -> ascending inside each group; + * of[] doubles as the fill cursor (the caller shifts it back); the + * partition's groups are nobody else's, so the cursors are private */ + for (int64_t j = lo; j < hi; j++) + h->rw[h->of[gmap[h->lg[j]]]++] = h->pr[j]; +} + +/* Zero (and so fault in) up to 8 fresh regions in parallel slices: each + * task owns one contiguous slice per region, so the page faults spread + * over the workers without two of them ever meeting on a page. */ +typedef struct { void* p[8]; size_t bytes[8]; int n; int n_tasks; } hp_touch_t; +static void hp_touch_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + hp_touch_t* t = (hp_touch_t*)raw; + for (int r = 0; r < t->n; r++) { + size_t lo = t->bytes[r] * (size_t)start / (size_t)t->n_tasks; + size_t hi = t->bytes[r] * (size_t)(start + 1) / (size_t)t->n_tasks; + if (hi > lo) memset((char*)t->p[r] + lo, 0, hi - lo); + } +} +static void hp_touch(ray_pool_t* pool, int n_tasks, hp_touch_t* t) { + t->n_tasks = n_tasks; + ray_pool_dispatch_n(pool, hp_touch_fn, t, (uint32_t)n_tasks); +} + +static bool hash_build_par(ray_t* v, ray_t** gkeys_out, ray_t** offs_out, + ray_t** rows_out, ray_t** table_out, uint64_t* mask_out, + int64_t* n_keys_out, int64_t* n_groups_out) { + int64_t n = v->len; + ray_pool_t* pool = ray_pool_get(); + if (!ray_pool_par_dispatch_ok(pool, n, 1 << 16)) return false; + int workers = (int)ray_pool_total_workers(pool); + + hash_par_t h; + memset(&h, 0, sizeof(h)); + h.v = v; h.base = (const uint8_t*)ray_data(v); h.n = n; + h.n_tasks = workers * 4; + if (h.n_tasks > 512) h.n_tasks = 512; + h.n_part = 2; + while (h.n_part < workers * 4 && h.n_part < 512) h.n_part <<= 1; + h.part_shift = 64; + for (int q = h.n_part; q > 1; q >>= 1) h.part_shift--; + + int64_t nw = (n + 63) / 64; + int64_t nblk = (n + HP_BLOCK - 1) / HP_BLOCK; + size_t cnt_b = (size_t)h.n_tasks * (size_t)h.n_part * sizeof(int64_t); + size_t po_b = (size_t)(h.n_part + 1) * sizeof(int64_t); + size_t bits_b = (size_t)(nw + 1) * sizeof(uint64_t); + h.kw = (uint64_t*)ray_sys_alloc((size_t)n * sizeof(uint64_t)); + h.cnt = (int64_t*)ray_sys_alloc(cnt_b); + h.part_off = (int64_t*)ray_sys_alloc(po_b); + h.ng_p = (int64_t*)ray_sys_alloc(po_b); + h.bits = (uint64_t*)ray_sys_alloc(bits_b); + h.blk_rank = (int64_t*)ray_sys_alloc((size_t)(nblk + 1) * sizeof(int64_t)); + ray_t *gkeys = NULL, *offs = NULL, *rows = NULL, *table = NULL; + bool ok = h.kw && h.cnt && h.part_off && h.ng_p && h.bits && h.blk_rank; + if (!ok) goto done; + memset(h.cnt, 0, cnt_b); memset(h.part_off, 0, po_b); memset(h.ng_p, 0, po_b); + memset(h.blk_rank, 0, (size_t)(nblk + 1) * sizeof(int64_t)); + { + hp_touch_t t = { .p = { h.kw, h.bits }, .bytes = { (size_t)n * sizeof(uint64_t), bits_b }, .n = 2 }; + hp_touch(pool, h.n_tasks, &t); + } + + /* A: key words + partition counts, then the bucketed row ids */ + ray_pool_dispatch_n(pool, hp_pass_a, &h, (uint32_t)h.n_tasks); + if (ray_interrupted()) { ok = false; goto done; } + { + int64_t run = 0; + for (int p = 0; p < h.n_part; p++) { + h.part_off[p] = run; + for (int t = 0; t < h.n_tasks; t++) { + int64_t c = h.cnt[(int64_t)t * h.n_part + p]; + h.cnt[(int64_t)t * h.n_part + p] = run; + run += c; + } + } + h.part_off[h.n_part] = run; + } + int64_t n_keys = h.part_off[h.n_part]; + size_t kb = (size_t)(n_keys > 0 ? n_keys : 1) * sizeof(int64_t); + h.pr = (int64_t*)ray_sys_alloc(kb); + h.lg = (int64_t*)ray_sys_alloc(kb); + h.gfirst = (int64_t*)ray_sys_alloc(kb); + h.gcount = (int64_t*)ray_sys_alloc(kb); + if (!h.pr || !h.lg || !h.gfirst || !h.gcount) { ok = false; goto done; } + { + hp_touch_t t = { .p = { h.pr, h.lg, h.gfirst, h.gcount }, .bytes = { kb, kb, kb, kb }, .n = 4 }; + hp_touch(pool, h.n_tasks, &t); + } + ray_pool_dispatch_n(pool, hp_pass_a2, &h, (uint32_t)h.n_tasks); + if (ray_interrupted()) { ok = false; goto done; } + + /* B: per-partition dedupe */ + ray_pool_dispatch_n(pool, hp_pass_b, &h, (uint32_t)h.n_part); + if (ray_interrupted() || atomic_load_explicit(&h.oom, memory_order_relaxed)) { ok = false; goto done; } + int64_t n_groups = 0; + for (int p = 0; p < h.n_part; p++) n_groups += h.ng_p[p]; + + /* C: first-occurrence numbering = rank of the group's first row */ + ray_pool_dispatch_n(pool, hp_pass_c, &h, (uint32_t)nblk); + { + int64_t run = 0; + for (int64_t b = 0; b < nblk; b++) { int64_t c = h.blk_rank[b]; h.blk_rank[b] = run; run += c; } + h.blk_rank[nblk] = run; + } + + gkeys = ray_vec_new(RAY_I64, n_groups > 0 ? n_groups : 1); + offs = ray_vec_new(RAY_I64, n_groups + 1); + rows = ray_vec_new(RAY_I64, n_keys > 0 ? n_keys : 1); + uint64_t cap = next_pow2((uint64_t)(n_groups < 4 ? 8 : 2 * n_groups)); + if (cap < 8) cap = 8; + table = ray_vec_new(RAY_I64, (int64_t)cap); + if (!gkeys || RAY_IS_ERR(gkeys) || !offs || RAY_IS_ERR(offs) || + !rows || RAY_IS_ERR(rows) || !table || RAY_IS_ERR(table)) { ok = false; goto done; } + gkeys->len = n_groups; offs->len = n_groups + 1; rows->len = n_keys; table->len = (int64_t)cap; + h.gk = (int64_t*)ray_data(gkeys); h.of = (int64_t*)ray_data(offs); + h.rw = (int64_t*)ray_data(rows); h.tbl = (int64_t*)ray_data(table); + h.tmask = cap - 1; + { + hp_touch_t t = { .p = { h.gk, h.of, h.rw, h.tbl }, + .bytes = { (size_t)(n_groups > 0 ? n_groups : 1) * sizeof(int64_t), + (size_t)(n_groups + 1) * sizeof(int64_t), kb, + (size_t)cap * sizeof(int64_t) }, .n = 4 }; + hp_touch(pool, h.n_tasks, &t); + } + + /* D: keys and counts; E: rows and the key table */ + ray_pool_dispatch_n(pool, hp_pass_d, &h, (uint32_t)h.n_part); + for (int64_t g = 0; g < n_groups; g++) h.of[g + 1] += h.of[g]; + ray_pool_dispatch_n(pool, hp_pass_e, &h, (uint32_t)h.n_part); + for (int64_t g = n_groups; g > 0; g--) h.of[g] = h.of[g - 1]; + h.of[0] = 0; + /* Key table filled in group order, exactly as the serial build does: + * the persisted index is then the same bytes whatever the core count. */ + for (int64_t g = 0; g < n_groups; g++) { + uint64_t slot = mix64((uint64_t)h.gk[g]) & h.tmask; + while (h.tbl[slot] != 0) slot = (slot + 1) & h.tmask; + h.tbl[slot] = g + 1; + } + if (ray_interrupted()) { ok = false; goto done; } + + *gkeys_out = gkeys; *offs_out = offs; *rows_out = rows; *table_out = table; + *mask_out = h.tmask; *n_keys_out = n_keys; *n_groups_out = n_groups; + +done: + if (!ok) { + if (gkeys && !RAY_IS_ERR(gkeys)) ray_release(gkeys); + if (offs && !RAY_IS_ERR(offs)) ray_release(offs); + if (rows && !RAY_IS_ERR(rows)) ray_release(rows); + if (table && !RAY_IS_ERR(table)) ray_release(table); + } + ray_sys_free(h.kw); ray_sys_free(h.pr); ray_sys_free(h.lg); + ray_sys_free(h.gfirst); ray_sys_free(h.gcount); ray_sys_free(h.cnt); + ray_sys_free(h.part_off); ray_sys_free(h.ng_p); ray_sys_free(h.bits); + ray_sys_free(h.blk_rank); + return ok; +} + ray_t* ray_index_attach_hash(ray_t** vp) { /* allow_str: keyed on a byte hash with payload-verified compares; * allow_sym: RAY_SYM uses domain ids. */ @@ -1092,6 +1475,34 @@ ray_t* ray_index_attach_hash(ray_t** vp) { bool is_str = (v->type == RAY_STR); int64_t n = v->len; + ray_t* table = NULL; + uint64_t mask = 0; + { + ray_t *pg = NULL, *po = NULL, *pr = NULL, *pt = NULL; + int64_t pk = 0, pn = 0; + uint64_t pm = 0; + if (!is_str && hash_build_par(v, &pg, &po, &pr, &pt, &pm, &pk, &pn)) { + if (ray_interrupted()) { + ray_release(pg); ray_release(po); ray_release(pr); ray_release(pt); + return ray_error("cancel", "interrupted"); + } + ray_t* idx = ray_index_alloc(RAY_IDX_HASH, v->type, n); + if (!idx || RAY_IS_ERR(idx)) { + ray_release(pg); ray_release(po); ray_release(pr); ray_release(pt); + return idx ? idx : ray_error("oom", NULL); + } + ray_index_t* ix = ray_index_payload(idx); + ix->u.hash.table = pt; + ix->u.hash.gkeys = pg; + ix->u.hash.offs = po; + ix->u.hash.rows = pr; + ix->u.hash.mask = pm; + ix->u.hash.n_keys = pk; + ix->u.hash.n_groups = pn; + ix->u.hash.order_sym = -1; + return attach_finalize(v, idx); + } + } /* Build-time capacity: sized by rows for O(1) inserts. */ uint64_t bcap = next_pow2((uint64_t)(n < 4 ? 8 : 2 * n)); if (bcap < 8) bcap = 8; @@ -1195,8 +1606,8 @@ ray_t* ray_index_attach_hash(ray_t** vp) { /* Attached/persisted bucket table: sized by DISTINCT keys. */ uint64_t cap = next_pow2((uint64_t)(n_groups < 4 ? 8 : 2 * n_groups)); if (cap < 8) cap = 8; - uint64_t mask = cap - 1; - ray_t* table = ray_vec_new(RAY_I64, (int64_t)cap); + mask = cap - 1; + table = ray_vec_new(RAY_I64, (int64_t)cap); if (!table || RAY_IS_ERR(table)) { ray_release(gkeys); ray_release(offs); ray_release(rows); return table ? table : ray_error("oom", NULL); @@ -2582,7 +2993,12 @@ ray_t* ray_index_drop(ray_t** vp) { * ray_alloc_copy (rc>1). Don't clobber the snapshot in that case — * the other holder still reads it. See vec_drop_index_inplace for * the same pattern. */ - bool shared = ray_atomic_load(&idx->rc) > 1; + /* A mapped index (mmod 1) rides the column file's mapping: copies of + * the column borrow it without a reference, and the mapping's owner + * unmaps it. Dropping it from a vector only detaches it — the + * snapshot stays for the other holders and nothing is released. */ + bool mapped = idx->mmod == 1; + bool shared = mapped || ray_atomic_load(&idx->rc) > 1; if (shared) { ray_index_retain_saved(ix); } @@ -2598,7 +3014,7 @@ ray_t* ray_index_drop(ray_t** vp) { /* Release the index. Per-kind children are released by the RAY_INDEX * branch of ray_release_owned_refs (added in heap.c). */ - ray_release(idx); + if (!mapped) ray_release(idx); return v; } diff --git a/src/ops/idxop.h b/src/ops/idxop.h index 2c9378444..609895a9f 100644 --- a/src/ops/idxop.h +++ b/src/ops/idxop.h @@ -165,6 +165,12 @@ typedef struct { uint8_t chunk_log2; /* chunk size = 1 << chunk_log2 (default 16 → 64 K rows) */ uint8_t is_f64; uint8_t _pad[2]; + /* Integer / temporal zones only (NULL for float zones and for + * indexes written before it existed): RAY_I64 vec of + * 2 * n_chunks — [0, n) the chunk's non-null values summed with + * int64 wraparound, [n, 2n) its non-null row count. Whole-column + * sum / count / avg answer from it in O(n_chunks). */ + ray_t* aggs; } chunk_zone; struct { /* RAY_IDX_PART */ ray_t* keys; /* distinct partition values, in ascending block order */ @@ -271,6 +277,41 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2); * compute an index for persistence without COWing a shared column. */ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2); +/* 128-bit two's-complement accumulation of int64 values: (hi, lo) += v. + * The engine's integer avg sums this way — exact for any column, and the + * same bits whatever the morsel split — and the chunk-zone metadata keeps + * the per-chunk (hi, lo) so it can answer the same value. */ +static inline void ray_i128_add(int64_t* hi, uint64_t* lo, int64_t v) { + uint64_t l = *lo + (uint64_t)v; + *hi += (v < 0 ? -1 : 0) + (l < (uint64_t)v ? 1 : 0); + *lo = l; +} +static inline void ray_i128_add128(int64_t* hi, uint64_t* lo, int64_t vhi, uint64_t vlo) { + uint64_t l = *lo + vlo; + *hi += vhi + (l < vlo ? 1 : 0); + *lo = l; +} +/* Double nearest the 128-bit value (hi, lo): the magnitude is converted + * (high word scaled by 2^64 plus the low word) and the sign reapplied, so + * a small negative total is not lost in 2^64 - |s|. One formula + * everywhere so every path agrees bit for bit. */ +static inline double ray_i128_to_f64(int64_t hi, uint64_t lo) { + bool neg = hi < 0; + uint64_t h = (uint64_t)hi, l = lo; + if (neg) { l = ~l + 1u; h = ~h + (l == 0 ? 1u : 0u); } + double d = (double)h * 18446744073709551616.0 + (double)l; + return neg ? -d : d; +} + +/* Whole-column sum (int64 wraparound) and non-null count of an integer + * column from its chunk-zone per-chunk aggregates; false when the column + * carries none for its current length. */ +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out); +/* The exact 128-bit sum (hi, lo) and non-null count from the same + * metadata: true when the per-chunk high words are stored, or when no + * partial sum can wrap int64 (then the wrapped sum is the exact one). */ +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out); + /* Build a RAY_IDX_DICT (codes + distinct values) for STR vector `v` WITHOUT * attaching it — standalone RAY_INDEX object (caller releases / attaches). * Returns RAY_ERR_NYI for non-STR. Used at splayed save to persist the dict @@ -288,7 +329,7 @@ ray_t* ray_index_attach_built(ray_t** vp, ray_t* idx); * in place and return the RAY_INDEX object (flagged RAY_MARK_MMAP). */ int64_t ray_index_inline_size(const ray_index_t* ix); void ray_index_inline_write(uint8_t* dst, const ray_index_t* ix); -ray_t* ray_index_inline_map(uint8_t* region); +ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size); /* Drop any attached index from *vp. No-op if none. Restores the * pre-attach aux state byte-for-byte. Returns *vp. */ diff --git a/src/ops/internal.h b/src/ops/internal.h index a1e1eeba8..f4381f88e 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -809,8 +809,8 @@ extern uint64_t ray_join_nullfree_keys; extern bool ray_agg_engine_v2; /* route OP_GROUP through v2 agg engine; default ON (agg_engine.c) */ void ray_expr_stats_init(void); -#define EXPR_MAX_REGS 16 -#define EXPR_MAX_INS 48 +#define EXPR_MAX_REGS 32 +#define EXPR_MAX_INS 96 #define EXPR_MORSEL RAY_MORSEL_ELEMS typedef struct { @@ -837,6 +837,8 @@ typedef struct { uint8_t col_attrs; /* column attrs — RAY_SYM width (REG_SCAN only) */ bool is_parted; /* true if this SCAN refs a parted column */ bool nullable; /* lanes may contain NULL_I64 / NaN */ + bool null_src; /* that nullability traces to a nullable column + * (else: only op-generated sentinels) */ const void* data; /* column data pointer (REG_SCAN only) */ ray_t* col_obj; /* source column vec (REG_SCAN, non-parted) — * carries the chunk-zone index for zone-skip */ @@ -1226,6 +1228,12 @@ ray_t* desc_vec_eager(ray_t* x); /* OP_PEARSON_CORR per-group accumulators: x-side piggybacks on SUM and * SUMSQ blocks; this flag enables the y-side blocks (Σy, Σy², Σxy). */ #define GHT_NEED_PEARSON 0x10 +/* Exact integer AVG: an extra int64 block (off_sum_hi) carries the high + * word of a 128-bit two's-complement sum next to each off_sum slot, so an + * integer mean never divides a wrapped int64 (ray_i128_add / ray_i128_to_f64 + * in idxop.h). Set whenever an OP_AVG agg has a non-float input; SUM keeps + * reading off_sum alone (int64 wraparound is its contract). */ +#define GHT_NEED_SUM128 0x20 /* ── ght_layout_t — inline-or-spill, fixed-size, by-value embeddable ── * @@ -1298,6 +1306,9 @@ typedef struct { uint16_t off_sum_y; uint16_t off_sumsq_y; uint16_t off_sumxy; + /* High words of the 128-bit integer sums (GHT_NEED_SUM128); 0 when the + * layout carries none. */ + uint16_t off_sum_hi; /* Earliest contributing source row for this group. Every packed entry * carries its source row in the tail slot; partition merges retain the * minimum so output order is independent of radix partition count. */ @@ -1603,6 +1614,22 @@ ray_t* exec_k_shortest(ray_graph_t* g, ray_op_t* op, /* ── pivot_exec.c ── */ ray_t* exec_if(ray_graph_t* g, ray_op_t* op); + +/* Is a descriptor view worth rebuilding over its own bytes (string.c)? */ +bool ray_str_view_should_compact(uint64_t pooled_bytes, int64_t pool_len); + +/* Shared-node memo around a sub-evaluation over a swapped g->table + * (exec.c): push sets the outer memo aside and arms one for the current + * table and sub-root; pop tears it down and restores the outer one. */ +typedef struct { + ray_t** vals; + uint32_t* uses; + uint32_t n; + ray_t* hdr; + ray_t* table; +} ray_exec_memo_save_t; +void ray_exec_memo_push(ray_graph_t* g, ray_op_t* root, ray_exec_memo_save_t* save); +void ray_exec_memo_pop(ray_graph_t* g, const ray_exec_memo_save_t* save); ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl); /* ── embedding_exec.c ── */ diff --git a/src/ops/journal.c b/src/ops/journal.c index fcf5d20d6..2495014cc 100644 --- a/src/ops/journal.c +++ b/src/ops/journal.c @@ -198,7 +198,8 @@ ray_t* ray_log_validate_fn(ray_t* path) { } ray_t* ray_log_roll_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".log.roll takes no arguments"); if (!ray_journal_is_open()) return ray_error("domain", ".log.roll: no journal open"); return err_to_ray(ray_journal_roll(), "io"); @@ -227,6 +228,7 @@ ray_t* ray_log_close_fn(ray_t** args, int64_t n) { * argument; acts on the journal .log.open/.write/.close target. Errors * with `domain` when no journal base is known (none ever opened). */ ray_t* ray_log_purge_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".log.purge takes no arguments"); return err_to_ray(ray_journal_purge(), "io"); } diff --git a/src/ops/ops.h b/src/ops/ops.h index 656b4fd57..d24ce6ae5 100644 --- a/src/ops/ops.h +++ b/src/ops/ops.h @@ -572,6 +572,23 @@ typedef struct ray_graph { * that reject an expression outright (arithmetic on a symbol) stay * quiet inside an arm and let the arm compile as it always did. */ int if_arm_depth; + + /* Result memo for shared nodes (exec.c, exec_node): a node with more + * than one consumer in the graph is executed once and its result is + * handed to every consumer as a retained ref; the memo's own ref is + * released at exec_memo_end. Set up around the flat exec_node(root) call in + * ray_execute_inner, NULL otherwise. Without it a `let`-bound string + * expression used three times ran three times. */ + ray_t** memo_vals; + uint32_t* memo_uses; + uint32_t memo_n; + ray_t* memo_hdr; + /* The table the memo was armed over. A node evaluated while g->table + * is swapped for another (an `if` branch over its compacted rows, a + * filter's right-hand side over a sub-table, a window partition) has a + * value of that table's length, so the memo neither serves nor stores + * while g->table differs. */ + ray_t* memo_table; } ray_graph_t; /* ===== Morsel Iterator ===== */ diff --git a/src/ops/pivot.c b/src/ops/pivot.c index 0be4a5579..9d25bcc30 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -252,7 +252,12 @@ static ray_t* if_eval_branch(ray_graph_t* g, ray_op_t* branch, g->table = sub; g->selection = NULL; + /* The branch's shared nodes get a memo over the branch's rows; the + * outer memo (values of the full table) is set aside meanwhile. */ + ray_exec_memo_save_t memo_save; + ray_exec_memo_push(g, branch, &memo_save); ray_t* value = exec_node(g, branch); + ray_exec_memo_pop(g, &memo_save); if (g->selection) { ray_release(g->selection); g->selection = NULL; @@ -501,6 +506,192 @@ static ray_t* if_scatter_str(ray_t* result, ray_t* value, int64_t* ids, return result; } +/* Descriptor scatter for the STR arm of the selected path. A side is a + * STR vector with one row per id (or as many rows as the table, indexed by + * the id), or a broadcast scalar. The result takes the side's 16-byte + * descriptor at ids[j] instead of appending its bytes row by row: pooled + * strings keep pointing into their pool when the rows keep most of it; + * otherwise (two different pools, a pooled scalar, or a small share of a + * big pool) the result gets a pool of exactly its own bytes. Rows not in + * either id list stay the null descriptor the caller + * zeroed. Returns NULL, the result untouched, for a side this cannot take + * (a SYM branch, a length that is neither) — the caller then falls back to + * the per-row scatter, which also reports the length error. */ +typedef struct { + ray_t* v; + int64_t* ids; + int64_t n; + bool scalar; + bool full; /* vector of nrows: row ids[j] */ + const ray_str_t* desc; + const char* bytes; /* its pool's bytes (NULL when inline-only) */ + ray_t* pool; + const char* sp; /* scalar bytes */ + size_t sl; +} if_str_side_t; + +static bool if_str_side_init(ray_t* v, int64_t* ids, int64_t n, int64_t nrows, + if_str_side_t* s) { + memset(s, 0, sizeof(*s)); + s->ids = ids; s->n = n; + if (!v || n <= 0) return true; + s->v = v; + if (v->type == -RAY_STR) { + s->scalar = true; s->sp = ray_str_ptr(v); s->sl = ray_str_len(v); + return true; + } + if (v->type != RAY_STR) return false; + if (v->len == 1) { + s->scalar = true; + s->sp = ray_str_vec_get(v, 0, &s->sl); + if (!s->sp) { s->sp = ""; s->sl = 0; } + return true; + } + if (v->len == n) s->full = false; + else if (v->len == nrows) s->full = true; + else return false; + str_resolve(v, &s->desc, &s->bytes); + s->pool = str_vec_pool_obj(v); + if (s->pool && RAY_IS_ERR(s->pool)) return false; + return true; +} + +/* The side's descriptor for its j-th id. */ +static inline ray_str_t if_str_side_desc(const if_str_side_t* s, int64_t j) { + return s->full ? s->desc[s->ids[j]] : s->desc[j]; +} + +/* Bytes the side's pooled descriptors point at. */ +static uint64_t if_str_side_pooled_bytes(const if_str_side_t* s) { + if (!s->v || s->scalar) return 0; + uint64_t sum = 0; + for (int64_t j = 0; j < s->n; j++) { + ray_str_t d = if_str_side_desc(s, j); + if (!ray_str_is_inline(&d)) sum += d.len; + } + return sum; +} + +/* Assign compact offsets to the side's pooled descriptors from *run. */ +static void if_str_side_offsets(const if_str_side_t* s, uint32_t* newoff, uint64_t* run) { + for (int64_t j = 0; j < s->n; j++) { + ray_str_t d = if_str_side_desc(s, j); + if (ray_str_is_inline(&d)) continue; + newoff[j] = (uint32_t)*run; + *run += d.len; + } +} + +typedef struct { + const int64_t* ids; + const ray_str_t* src; + bool scalar; + bool full; + ray_str_t sd; /* the scalar's descriptor */ + ray_str_t* dst; + /* compaction: pooled bytes move to dst_bytes at newoff[j] */ + const uint32_t* newoff; + const char* src_bytes; + char* dst_bytes; +} if_str_scatter_ctx_t; + +static void if_str_scatter_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const if_str_scatter_ctx_t* c = (const if_str_scatter_ctx_t*)vctx; + if (c->scalar) { + for (int64_t j = start; j < end; j++) c->dst[c->ids[j]] = c->sd; + return; + } + for (int64_t j = start; j < end; j++) { + ray_str_t d = c->full ? c->src[c->ids[j]] : c->src[j]; + if (c->newoff && !ray_str_is_inline(&d)) { + memcpy(c->dst_bytes + c->newoff[j], c->src_bytes + d.pool_off, d.len); + d.pool_off = c->newoff[j]; + } + c->dst[c->ids[j]] = d; + } +} + +static void if_str_scatter_side(const if_str_side_t* s, const uint32_t* newoff, char* dst_bytes, + uint32_t scalar_off, ray_str_t* dst) { + if (!s->v) return; + if_str_scatter_ctx_t c = { + .ids = s->ids, .src = s->desc, .scalar = s->scalar, .full = s->full, + .dst = dst, .newoff = newoff, .src_bytes = s->bytes, .dst_bytes = dst_bytes, + }; + if (s->scalar) { + memset(&c.sd, 0, sizeof(c.sd)); + c.sd.len = (uint32_t)s->sl; + if (s->sl <= RAY_STR_INLINE_MAX) { + if (s->sl) memcpy(c.sd.data, s->sp, s->sl); + } else { + memcpy(c.sd.prefix, s->sp, 4); + c.sd.pool_off = scalar_off; + } + } + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, s->n, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, if_str_scatter_fn, &c, s->n); + else + if_str_scatter_fn(&c, 0, 0, s->n); +} + +static ray_t* if_scatter_str_desc(ray_t* result, + ray_t* then_v, int64_t* t_ids, int64_t t_n, + ray_t* else_v, int64_t* e_ids, int64_t e_n, + int64_t nrows) { + if_str_side_t t, e; + if (!if_str_side_init(then_v, t_ids, t_n, nrows, &t)) return NULL; + if (!if_str_side_init(else_v, e_ids, e_n, nrows, &e)) return NULL; + + ray_t* pt = (t.v && !t.scalar) ? t.pool : NULL; + ray_t* pe = (e.v && !e.scalar) ? e.pool : NULL; + bool t_big = t.v && t.scalar && t.sl > RAY_STR_INLINE_MAX; + bool e_big = e.v && e.scalar && e.sl > RAY_STR_INLINE_MAX; + uint64_t t_ref = if_str_side_pooled_bytes(&t); + uint64_t e_ref = if_str_side_pooled_bytes(&e); + ray_str_t* dst = (ray_str_t*)ray_data(result); + + /* One pool (or none) behind both sides, and the rows keep most of it: + * point into it as is. */ + bool one_pool = (pt == pe) || !pt || !pe; + ray_t* shared = pt ? pt : pe; + if (!t_big && !e_big && one_pool && + (!shared || !ray_str_view_should_compact(t_ref + e_ref, shared->len))) { + if (shared) { ray_retain(shared); result->str_pool = shared; } + if_str_scatter_side(&t, NULL, NULL, 0, dst); + if_str_scatter_side(&e, NULL, NULL, 0, dst); + return result; + } + + /* Otherwise a pool of exactly the bytes the result points at: the two + * sides' pooled rows one after the other, then the pooled scalars. */ + uint64_t total = t_ref + e_ref + (t_big ? (uint64_t)t.sl : 0) + (e_big ? (uint64_t)e.sl : 0); + if (total > UINT32_MAX) return NULL; + int64_t n_off = (t.v && !t.scalar ? t.n : 0) + (e.v && !e.scalar ? e.n : 0); + ray_t* off_hdr = NULL; + uint32_t* newoff = (uint32_t*)scratch_alloc(&off_hdr, (size_t)(n_off > 0 ? n_off : 1) * sizeof(uint32_t)); + if (!newoff) return NULL; + uint32_t* t_off = (t.v && !t.scalar) ? newoff : NULL; + uint32_t* e_off = (e.v && !e.scalar) ? newoff + (t.v && !t.scalar ? t.n : 0) : NULL; + uint64_t run = 0; + if (t_off) if_str_side_offsets(&t, t_off, &run); + if (e_off) if_str_side_offsets(&e, e_off, &run); + uint32_t ts_off = 0, es_off = 0; + ray_t* np = ray_alloc(total > 0 ? (size_t)total : 1); + if (!np || RAY_IS_ERR(np)) { scratch_free(off_hdr); return NULL; } + np->type = RAY_U8; + np->len = (int64_t)total; + char* dst_bytes = (char*)ray_data(np); + if (t_big) { memcpy(dst_bytes + run, t.sp, t.sl); ts_off = (uint32_t)run; run += t.sl; } + if (e_big) { memcpy(dst_bytes + run, e.sp, e.sl); es_off = (uint32_t)run; run += e.sl; } + result->str_pool = np; + if_str_scatter_side(&t, t_off, dst_bytes, ts_off, dst); + if_str_scatter_side(&e, e_off, dst_bytes, es_off, dst); + scratch_free(off_hdr); + return result; +} + /* Runtime-id translation for one SYM branch of an `if`. * * A branch that scans a FILE-domain column used to go through the @@ -654,6 +845,42 @@ static bool if_branch_trivial(ray_graph_t* g, ray_op_t* op) { return false; } +/* A branch that is cheap to evaluate over ALL rows and total (no row can + * fail): scans, constants, substrings, string case/trim/length, add/sub/mul, + * comparisons, and/or/not, and `if` over such — a descriptor-only STR + * result then costs less than the selected path's compaction of the rows + * for each branch (a serial gather of the branch's columns) plus its + * scatter. Searches (str-find, like, replace, concat) and everything else + * are worth restricting to the rows that need them. */ +static bool if_branch_cheap(ray_graph_t* g, ray_op_t* op, int depth) { + if (!op || depth > 32) return false; + switch (op->opcode) { + case OP_ALIAS: case OP_MATERIALIZE: + return if_branch_cheap(g, op_child(g, op, 0), depth + 1); + case OP_SCAN: return true; + case OP_CONST: { + ray_op_ext_t* ext = find_ext(g, op->id); + return ext && ext->literal && ray_is_atom(ext->literal); + } + case OP_SUBSTR: case OP_IF: { + ray_op_ext_t* ext = find_ext(g, op->id); + if (!ext || ext->third_in >= g->node_count) return false; + for (int k = 0; k < op->arity && k < 2; k++) + if (!if_branch_cheap(g, op_child(g, op, k), depth + 1)) return false; + return if_branch_cheap(g, op_node(g, ext->third_in), depth + 1); + } + case OP_STRLEN: case OP_UPPER: case OP_LOWER: case OP_TRIM: + case OP_ADD: case OP_SUB: case OP_MUL: case OP_NEG: case OP_ABS: + case OP_EQ: case OP_NE: case OP_LT: case OP_LE: case OP_GT: case OP_GE: + case OP_AND: case OP_OR: case OP_NOT: case OP_ISNULL: + for (int k = 0; k < op->arity && k < 2; k++) + if (!if_branch_cheap(g, op_child(g, op, k), depth + 1)) return false; + return true; + default: + return false; + } +} + /* Is a trivial branch of static type `bt` filled CORRECTLY by the eager * elementwise path for result type `out`? Mixed numeric/string branch * combinations (e.g. `(if c n1 s2)` with I64 + STR) rely on the selected @@ -698,11 +925,32 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { * answer exactly as it would, or the same expression gets two types * depending on the worker count. Where it is not, this arm is the only * one there is and may report the type the rows actually have. */ - bool eager_possible = op->out_type != RAY_STR && - if_branch_trivial(g, then_op) && - if_branch_trivial(g, else_op) && - if_type_eager_ok(then_op->out_type, op->out_type) && - if_type_eager_ok(else_op->out_type, op->out_type); + ray_t* outer_sel = g->selection; + if (outer_sel) { + ray_rowsel_t* sm = ray_rowsel_meta(outer_sel); + if (!sm || sm->nrows != nrows) return NULL; + } + + int64_t selected = outer_sel ? ray_rowsel_meta(outer_sel)->total_pass : nrows; + if (selected < 0 || selected > nrows) return NULL; + /* Under a sparse outer selection (a `where:` keeping a quarter of the + * rows or fewer) the STR eager arm is a bad trade: it builds a + * descriptor for every row of the table, and the filtered result then + * keeps that whole intermediate — both parents' bytes — alive for its + * few rows. Touching only the selected rows costs less and holds only + * their bytes. Same threshold as the projection's pre-compaction. */ + bool sparse_sel = outer_sel && selected * 4 <= nrows; + + bool eager_possible = (op->out_type != RAY_STR && + if_branch_trivial(g, then_op) && + if_branch_trivial(g, else_op) && + if_type_eager_ok(then_op->out_type, op->out_type) && + if_type_eager_ok(else_op->out_type, op->out_type)) || + /* two STR vector branches that are cheap and total: + * the eager arm picks descriptors over one pass */ + (op->out_type == RAY_STR && !sparse_sel && + then_op->out_type == RAY_STR && else_op->out_type == RAY_STR && + if_branch_cheap(g, then_op, 0) && if_branch_cheap(g, else_op, 0)); { ray_pool_t* rp = ray_pool_get(); if (rp && rp->n_workers > 0 && eager_possible) @@ -714,15 +962,6 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { !if_make_branch_plan(g, else_op, &else_plan)) return NULL; - ray_t* outer_sel = g->selection; - if (outer_sel) { - ray_rowsel_t* sm = ray_rowsel_meta(outer_sel); - if (!sm || sm->nrows != nrows) return NULL; - } - - int64_t selected = outer_sel ? ray_rowsel_meta(outer_sel)->total_pass : nrows; - if (selected < 0 || selected > nrows) return NULL; - uint8_t* cond = (uint8_t*)ray_data(cond_v); int64_t true_count = if_selected_true_count(cond, nrows, outer_sel); int64_t false_count = selected - true_count; @@ -772,9 +1011,15 @@ static ray_t* exec_if_selected(ray_graph_t* g, ray_op_t* op, ray_t* cond_v) { bool ok = true; if (out_type == RAY_STR) { - result = if_scatter_str(result, then_v, true_ids, true_count, nrows); - if (result && !RAY_IS_ERR(result)) - result = if_scatter_str(result, else_v, false_ids, false_count, nrows); + ray_t* fast = if_scatter_str_desc(result, then_v, true_ids, true_count, + else_v, false_ids, false_count, nrows); + if (fast) { + result = fast; + } else { + result = if_scatter_str(result, then_v, true_ids, true_count, nrows); + if (result && !RAY_IS_ERR(result)) + result = if_scatter_str(result, else_v, false_ids, false_count, nrows); + } } else if (out_type == RAY_SYM) { ok = if_scatter_sym(result, then_v, true_ids, true_count, nrows) && if_scatter_sym(result, else_v, false_ids, false_count, nrows); @@ -893,6 +1138,36 @@ static void if_fill_par_fn(void* ctx, uint32_t wid, int64_t start, int64_t end) if_fill_range((const if_fill_ctx_t*)ctx, start, end); } +/* Descriptor fill for the STR arm of exec_if_eager: dst[r] is the chosen + * side's descriptor; a pooled else-descriptor moves by e_shift when the two + * pools were laid end to end. Rows are independent; workers write disjoint + * ranges. */ +typedef struct { + const uint8_t* cond; + const ray_str_t* t; + const ray_str_t* e; + ray_str_t* dst; + /* two pools: the chosen row's pooled bytes move to dst_bytes at newoff[r] */ + const uint32_t* newoff; + const char* t_bytes; + const char* e_bytes; + char* dst_bytes; +} if_str_desc_ctx_t; + +static void if_str_desc_fn(void* vctx, uint32_t worker_id, int64_t start, int64_t end) { + (void)worker_id; + const if_str_desc_ctx_t* c = (const if_str_desc_ctx_t*)vctx; + for (int64_t r = start; r < end; r++) { + ray_str_t d = c->cond[r] ? c->t[r] : c->e[r]; + if (c->newoff && !ray_str_is_inline(&d)) { + const char* src = c->cond[r] ? c->t_bytes : c->e_bytes; + memcpy(c->dst_bytes + c->newoff[r], src + d.pool_off, d.len); + d.pool_off = c->newoff[r]; + } + c->dst[r] = d; + } +} + static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { /* cond = inputs[0], then = inputs[1], else_id stored in ext->third_in */ ray_t* cond_v = exec_node(g, op_child(g, op, 0)); @@ -959,27 +1234,75 @@ static ray_t* exec_if_eager(ray_graph_t* g, ray_op_t* op) { uint8_t* cond_p = (uint8_t*)ray_data(cond_v); if (out_type == RAY_STR) { + /* Two STR vectors: the result is descriptors only. Each row takes + * its side's 16-byte descriptor; pooled strings keep pointing into + * their pool. One shared pool (or one side inline-only) is reused + * as is; two different pools give a pool of exactly the chosen + * rows' bytes. Nulls + * are empty descriptors and travel unchanged. No per-row append, + * no rehash; the fill runs on the worker pool. */ if (!then_scalar && !else_scalar && then_v->type == RAY_STR && else_v->type == RAY_STR && - len <= then_v->len && len <= else_v->len && - !ray_vec_may_have_nulls(then_v) && - !ray_vec_may_have_nulls(else_v)) { + len <= then_v->len && len <= else_v->len) { ray_t* then_pool = str_vec_pool_obj(then_v); ray_t* else_pool = str_vec_pool_obj(else_v); + const ray_str_t* t_desc = NULL; + const ray_str_t* e_desc = NULL; + const char* t_bytes = NULL; + const char* e_bytes = NULL; + str_resolve(then_v, &t_desc, &t_bytes); + str_resolve(else_v, &e_desc, &e_bytes); + bool ok = true; + ray_t* off_hdr = NULL; + uint32_t* newoff = NULL; if (then_pool == else_pool || !then_pool || !else_pool) { ray_t* out_pool = then_pool ? then_pool : else_pool; if (out_pool && !RAY_IS_ERR(out_pool)) { ray_retain(out_pool); result->str_pool = out_pool; } - const ray_str_t* t_desc = NULL; - const ray_str_t* e_desc = NULL; - const char* unused_pool = NULL; - str_resolve(then_v, &t_desc, &unused_pool); - str_resolve(else_v, &e_desc, &unused_pool); - ray_str_t* dst = (ray_str_t*)ray_data(result); - for (int64_t i = 0; i < len; i++) - dst[i] = cond_p[i] ? t_desc[i] : e_desc[i]; + } else if (RAY_IS_ERR(then_pool) || RAY_IS_ERR(else_pool)) { + ok = false; + } else { + /* Two pools: a pool of exactly the chosen rows' bytes (a + * serial pass assigns the offsets, the fill copies). */ + newoff = (uint32_t*)scratch_alloc(&off_hdr, (size_t)(len > 0 ? len : 1) * sizeof(uint32_t)); + if (!newoff) { + ok = false; + } else { + uint64_t run = 0; + for (int64_t r = 0; r < len; r++) { + const ray_str_t* d = cond_p[r] ? &t_desc[r] : &e_desc[r]; + if (ray_str_is_inline(d)) continue; + newoff[r] = (uint32_t)run; + run += d->len; + } + ray_t* np = (run <= UINT32_MAX) ? ray_alloc(run > 0 ? (size_t)run : 1) : NULL; + if (!np || RAY_IS_ERR(np)) { + ok = false; + } else { + np->type = RAY_U8; + np->len = (int64_t)run; + result->str_pool = np; + } + } + if (!ok && off_hdr) { scratch_free(off_hdr); off_hdr = NULL; newoff = NULL; } + } + if (ok) { + if_str_desc_ctx_t dctx = { + .cond = cond_p, .t = t_desc, .e = e_desc, + .dst = (ray_str_t*)ray_data(result), + .newoff = newoff, .t_bytes = t_bytes, .e_bytes = e_bytes, + .dst_bytes = result->str_pool ? (char*)ray_data(result->str_pool) : NULL, + }; + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, len, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, if_str_desc_fn, &dctx, len); + else + if_str_desc_fn(&dctx, 0, 0, len); + if (off_hdr) scratch_free(off_hdr); + if (ray_vec_may_have_nulls(then_v) || ray_vec_may_have_nulls(else_v)) + result->attrs |= RAY_ATTR_HAS_NULLS; ray_release(cond_v); ray_release(then_v); ray_release(else_v); return result; } @@ -1284,6 +1607,12 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { uint8_t need_flags = GHT_NEED_SUM; /* always need sum (used for FIRST/LAST too) */ if (agg_op == OP_MIN) need_flags |= GHT_NEED_MIN; if (agg_op == OP_MAX) need_flags |= GHT_NEED_MAX; + /* Integer avg divides the exact 128-bit sum (high words in off_sum_hi), + * like every group engine — never a wrapped int64. */ + if (agg_op == OP_AVG && (vcol->type == RAY_I64 || vcol->type == RAY_TIMESTAMP || + (vcol->type != RAY_F64 && vcol->type != RAY_F32 && + nrows >= ((int64_t)1 << 31)))) + need_flags |= GHT_NEED_SUM128; /* n_keys/n_aggs are no longer capped: ght_compute_layout spills to an * owned heap block (ly.spill_hdr) whenever n_keys exceeds GHT_INLINE @@ -1820,7 +2149,10 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } v = val_is_f64 ? ROW_RD_F64(row, ly.off_sum, s) / nn - : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; + : (ly.need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly.off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly.off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; break; case OP_MIN: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } diff --git a/src/ops/query.c b/src/ops/query.c index f5bc3071e..028c53bc2 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -528,6 +528,79 @@ static uint16_t resolve_agg_opcode(int64_t sym_id) { return 0; } +/* See the call site in ray_select_impl. NULL: some output is not answerable + * from metadata (the caller plans the query as usual). */ +static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t dict_n, + int64_t from_id) { + int64_t nrows = ray_table_nrows(tbl); + if (nrows <= 0) return NULL; /* empty tables keep the planner's answers */ + int64_t n_out = 0; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + if (dict_elems[i]->i64 == from_id) continue; + ray_t* e = dict_elems[i + 1]; + if (!e || e->type != RAY_LIST || ray_len(e) != 2) return NULL; + ray_t** el = (ray_t**)ray_data(e); + if (el[0]->type != -RAY_SYM || (el[0]->attrs & ATTR_QUOTED)) return NULL; + if (el[1]->type != -RAY_SYM || (el[1]->attrs & ATTR_QUOTED)) return NULL; + ray_t* col = ray_table_get_col(tbl, el[1]->i64); + if (!col || !ray_is_vec(col) || RAY_IS_PARTED(col->type) || col->type == RAY_MAPCOMMON || + (col->attrs & RAY_ATTR_SLICE) || col->len != nrows) + return NULL; + uint16_t op = resolve_agg_opcode(el[0]->i64); + bool int_col = col->type == RAY_I64 || col->type == RAY_I32 || + col->type == RAY_I16 || col->type == RAY_U8; + int64_t zs, zn, zh; uint64_t zl; + switch (op) { + case OP_COUNT: break; + case OP_MIN: case OP_MAX: { + if (ray_index_kind(col) != RAY_IDX_CHUNK_ZONE) return NULL; + ray_index_t* ix = ray_index_payload(col->index); + if (ix->built_for_len != col->len || ix->u.chunk_zone.is_f64 || + !ix->u.chunk_zone.mins || !ix->u.chunk_zone.maxs) return NULL; + break; + } + case OP_SUM: + if (!int_col || !ray_zone_int_sum(col, &zs, &zn)) return NULL; + break; + case OP_AVG: + if (!int_col || !ray_zone_int_sum128(col, &zh, &zl, &zn)) return NULL; + break; + default: return NULL; + } + n_out++; + } + if (n_out == 0) return NULL; + + ray_t* res = ray_table_new(n_out); + if (!res || RAY_IS_ERR(res)) return NULL; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid == from_id) continue; + ray_t** el = (ray_t**)ray_data(dict_elems[i + 1]); + ray_t* col = ray_table_get_col(tbl, el[1]->i64); + uint16_t op = resolve_agg_opcode(el[0]->i64); + ray_t* atom = op == OP_COUNT ? ray_i64(nrows) + : op == OP_MIN ? ray_min_fn(col) + : op == OP_MAX ? ray_max_fn(col) + : op == OP_SUM ? ray_sum_fn(col) + : ray_avg_fn(col); + if (!atom || RAY_IS_ERR(atom) || !ray_is_atom(atom)) { + if (atom && RAY_IS_ERR(atom)) ray_error_free(atom); else if (atom) ray_release(atom); + ray_release(res); + return NULL; + } + ray_t* v = ray_vec_new(-atom->type, 1); + if (v && !RAY_IS_ERR(v)) v = ray_vec_append(v, &atom->i64); + if (v && !RAY_IS_ERR(v) && RAY_ATOM_IS_NULL(atom)) ray_vec_set_null(v, 0, true); + ray_release(atom); + if (!v || RAY_IS_ERR(v)) { ray_release(res); return NULL; } + res = ray_table_add_col(res, kid, v); + ray_release(v); + if (!res || RAY_IS_ERR(res)) return NULL; + } + return res; +} + static bool agg_name_is_percentile(int64_t sym_id) { ray_t* s = ray_sym_str(sym_id); return s && ray_str_len(s) == 10 && @@ -1018,6 +1091,38 @@ static ray_op_t* sel_alias_lookup(ray_graph_t* g, int64_t sym) { return NULL; } +/* The op a sort key binds to. When the sort runs over a projection + * (`root` is its SELECT), the key is the projected column that scans the + * source column — an output, or the hidden key appended for it — so the + * sort reads it by POSITION and no name lookup over the projected table + * (scans named by source, expressions `_e`) can pick another column. + * Otherwise a fresh scan resolved by name, as before. */ +static ray_op_t* select_sort_key_op(ray_graph_t* g, ray_op_t* root, int64_t sym, const char* name) { + if (root && root->opcode == OP_SELECT) { + ray_op_ext_t* se = find_ext(g, root->id); + if (se && se->base.opcode == OP_SELECT) + for (uint32_t c = 0; c < se->sort.n_cols; c++) { + ray_op_t* col = &g->nodes[se->sort.columns[c]]; + if (col->opcode != OP_SCAN) continue; + ray_op_ext_t* ce = find_ext(g, col->id); + if (ce && ce->base.opcode == OP_SCAN && ce->sym == sym) return col; + } + } + return ray_scan(g, name); +} + +/* Index of the first projected column that is a bare scan of source + * column `sym`, or -1. Used to bind a sort key to the projected column + * that already carries it. */ +static int64_t select_scan_output(ray_graph_t* g, ray_op_t** col_ops, int64_t nc, int64_t sym) { + for (int64_t c = 0; c < nc; c++) { + if (!col_ops[c] || col_ops[c]->opcode != OP_SCAN) continue; + ray_op_ext_t* ce = find_ext(g, col_ops[c]->id); + if (ce && ce->base.opcode == OP_SCAN && ce->sym == sym) return c; + } + return -1; +} + /* Takes the error compile_expr_dag left on the graph (see ops.h), or * NULL. Callers that report a compile failure use it so the message * names the actual problem when there is one. */ @@ -3069,6 +3174,228 @@ static int64_t derived_key_chunk_rows(void) { #endif return DERIVED_KEY_CHUNK; } + +/* ---- chunk STR build: the entries at positions dv[lo, lo+n) as one STR + * vector over one pool. Lengths and byte pointers are read on the workers + * (the raw snapshot needs no lock); a position past the file prefix — a + * runtime-appended entry — is resolved through the domain on the calling + * thread. A serial prefix sum places the pooled rows, then the workers + * write the descriptors and copy the pooled bytes into disjoint ranges. */ +typedef struct { + const void* dv; + uint8_t dattrs; + int64_t lo; + const ray_sym_domain_raw_t* raw; + const char** ptr; + uint32_t* len; + uint32_t* off; + ray_str_t* dst; + char* pool; + atomic_int late; +} dk_chunk_ctx_t; + +static void dk_chunk_scan_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dk_chunk_ctx_t* c = (dk_chunk_ctx_t*)vctx; + int late = 0; + for (int64_t i = start; i < end; i++) { + int64_t pos = ray_read_sym(c->dv, c->lo + i, RAY_SYM, c->dattrs); + if (pos >= 0 && pos < c->raw->count) { + size_t sl = 0; + c->ptr[i] = ray_sym_domain_raw_str(c->raw, pos, &sl); + c->len[i] = (uint32_t)sl; + } else { + c->ptr[i] = NULL; + c->len[i] = UINT32_MAX; + late++; + } + } + if (late) atomic_fetch_add_explicit(&c->late, late, memory_order_relaxed); +} + +static void dk_chunk_fill_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_chunk_ctx_t* c = (const dk_chunk_ctx_t*)vctx; + for (int64_t i = start; i < end; i++) { + ray_str_t* d = &c->dst[i]; + uint32_t l = c->len[i]; + memset(d, 0, sizeof(*d)); + d->len = l; + if (l == 0) continue; + if (l <= RAY_STR_INLINE_MAX) { + memcpy(d->data, c->ptr[i], l); + } else { + memcpy(c->pool + c->off[i], c->ptr[i], l); + memcpy(d->prefix, c->ptr[i], 4); + d->pool_off = c->off[i]; + } + } +} + +static ray_t* derived_key_chunk_strs(const void* dv, uint8_t dattrs, int64_t lo, int64_t n, + const ray_sym_domain_raw_t* raw, + struct ray_sym_domain_s* dom) { + ray_t* aux_hdr = NULL; + size_t ptr_sz = (size_t)n * sizeof(const char*); + size_t u32_sz = (size_t)n * sizeof(uint32_t); + char* mem = (char*)scratch_alloc(&aux_hdr, ptr_sz + 2 * u32_sz); + if (!mem) return NULL; + dk_chunk_ctx_t c; + memset(&c, 0, sizeof(c)); + c.dv = dv; c.dattrs = dattrs; c.lo = lo; c.raw = raw; + c.ptr = (const char**)mem; + c.len = (uint32_t*)(mem + ptr_sz); + c.off = (uint32_t*)(mem + ptr_sz + u32_sz); + atomic_store_explicit(&c.late, 0, memory_order_relaxed); + + ray_pool_t* pool = ray_pool_get(); + bool par = ray_pool_par_dispatch_ok(pool, n, RAY_PARALLEL_THRESHOLD); + if (par) ray_pool_dispatch(pool, dk_chunk_scan_fn, &c, n); + else dk_chunk_scan_fn(&c, 0, 0, n); + if (atomic_load_explicit(&c.late, memory_order_relaxed)) { + for (int64_t i = 0; i < n; i++) { + if (c.len[i] != UINT32_MAX) continue; + int64_t pos = ray_read_sym(dv, lo + i, RAY_SYM, dattrs); + ray_t* a = ray_sym_domain_str(dom, pos); + size_t sl = a ? ray_str_len(a) : 0; + if (sl > UINT32_MAX) { scratch_free(aux_hdr); return NULL; } + c.ptr[i] = a ? ray_str_ptr(a) : ""; + c.len[i] = (uint32_t)sl; + } + } + uint64_t total = 0; + for (int64_t i = 0; i < n; i++) { + if (c.len[i] <= RAY_STR_INLINE_MAX) continue; + c.off[i] = (uint32_t)total; + total += c.len[i]; + if (total > UINT32_MAX) { scratch_free(aux_hdr); return NULL; } + } + ray_t* sv = ray_vec_new(RAY_STR, n); + if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); scratch_free(aux_hdr); return NULL; } + sv->len = n; + if (total > 0) { + ray_t* sp = ray_alloc((size_t)total); + if (!sp || RAY_IS_ERR(sp)) { ray_release(sv); scratch_free(aux_hdr); return NULL; } + sp->type = RAY_U8; + sp->len = (int64_t)total; + sv->str_pool = sp; + c.pool = (char*)ray_data(sp); + } + c.dst = (ray_str_t*)ray_data(sv); + if (par) ray_pool_dispatch(pool, dk_chunk_fill_fn, &c, n); + else dk_chunk_fill_fn(&c, 0, 0, n); + scratch_free(aux_hdr); + return sv; +} + +/* ---- chunk result interning: each distinct string of the chunk's key + * vector is interned once, all of them under one lock, and its id spread + * over the rows that hold it. The hashes are the intern table's own + * (ray_hash_bytes), computed on the workers; the dedupe is an + * open-addressing table over the distinct ordinals, on scratch. */ +typedef struct { + const ray_str_t* desc; + const char* pool; + uint32_t* hash; +} dk_hash_ctx_t; + +static void dk_hash_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_hash_ctx_t* c = (const dk_hash_ctx_t*)vctx; + for (int64_t i = start; i < end; i++) { + const ray_str_t* d = &c->desc[i]; + c->hash[i] = (uint32_t)ray_hash_bytes(ray_str_t_ptr(d, c->pool), d->len); + } +} + +static bool derived_key_intern_chunk(ray_t* kc, int64_t n, int64_t* out) { + const ray_str_t* desc = NULL; + const char* pool = NULL; + str_resolve(kc, &desc, &pool); + int64_t slots = 1024; + while (slots < 2 * n) slots <<= 1; + ray_t* hdr = NULL; + size_t hash_sz = (size_t)n * sizeof(uint32_t); + size_t rep_sz = (size_t)n * sizeof(int32_t); + size_t tab_sz = (size_t)slots * sizeof(int32_t); + size_t dstr_sz = (size_t)n * sizeof(const char*); + size_t dlen_sz = (size_t)n * sizeof(size_t); + size_t dhsh_sz = (size_t)n * sizeof(uint32_t); + size_t did_sz = (size_t)n * sizeof(int64_t); + /* One carve; the 8-byte arrays go first so every field stays aligned. */ + char* mem = (char*)scratch_alloc(&hdr, hash_sz + rep_sz + tab_sz + dstr_sz + dlen_sz + dhsh_sz + did_sz); + if (!mem) return false; + const char** dstr = (const char**)mem; mem += dstr_sz; + size_t* dlen = (size_t*)mem; mem += dlen_sz; + int64_t* did = (int64_t*)mem; mem += did_sz; + uint32_t* hash = (uint32_t*)mem; mem += hash_sz; + int32_t* rep = (int32_t*)mem; mem += rep_sz; + int32_t* tab = (int32_t*)mem; mem += tab_sz; + uint32_t* dhsh = (uint32_t*)mem; + memset(tab, 0xff, tab_sz); + + dk_hash_ctx_t hc = { .desc = desc, .pool = pool, .hash = hash }; + ray_pool_t* rp = ray_pool_get(); + if (ray_pool_par_dispatch_ok(rp, n, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(rp, dk_hash_fn, &hc, n); + else + dk_hash_fn(&hc, 0, 0, n); + + uint64_t mask = (uint64_t)slots - 1; + int64_t nd = 0; + for (int64_t i = 0; i < n; i++) { + uint32_t h = hash[i]; + const ray_str_t* d = &desc[i]; + const char* sp = ray_str_t_ptr(d, pool); + uint64_t s = ((uint64_t)h * 0x9E3779B97F4A7C15ull >> 32) & mask; + for (;;) { + int32_t r = tab[s]; + if (r < 0) { + tab[s] = (int32_t)nd; + rep[i] = (int32_t)nd; + dstr[nd] = sp; dlen[nd] = d->len; dhsh[nd] = h; + nd++; + break; + } + if (dhsh[r] == h && dlen[r] == d->len && + (d->len == 0 || memcmp(dstr[r], sp, d->len) == 0)) { + rep[i] = r; + break; + } + s = (s + 1) & mask; + } + } + /* Key strings are values, not names: interned without the dotted- + * segment caching that a name with '.' in it gets (a host or URL would + * otherwise intern every one of its segments too). */ + if (ray_sym_intern_batch_no_split(dhsh, dstr, dlen, nd, did) < 0) { scratch_free(hdr); return false; } + for (int64_t i = 0; i < n; i++) out[i] = did[rep[i]]; + scratch_free(hdr); + return true; +} + +/* ---- the spread pass of derived_key_over_sym_domain: each row takes the + * key of its symbol's slot (workers, disjoint ranges). */ +typedef struct { + const void* cd; + uint8_t attrs; + int64_t dn; + const int32_t* pos; + const int64_t* key; /* spread: key per slot; NULL = write the slot */ + int64_t* out; +} dk_rows_ctx_t; + +static void dk_spread_fn(void* vctx, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dk_rows_ctx_t* c = (const dk_rows_ctx_t*)vctx; + if (c->key) { + for (int64_t r = start; r < end; r++) + c->out[r] = c->key[c->pos[ray_read_sym(c->cd, r, RAY_SYM, c->attrs)]]; + } else { + for (int64_t r = start; r < end; r++) + c->out[r] = c->pos[ray_read_sym(c->cd, r, RAY_SYM, c->attrs)]; + } +} static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom_vec, struct ray_sym_domain_s* dom, int64_t du) { if (!dom || dom == ray_sym_runtime_domain() || du <= 0) return NULL; @@ -3102,21 +3429,8 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom const int64_t chunk = derived_key_chunk_rows(); for (int64_t lo = 0; lo < du; lo += chunk) { int64_t n = du - lo < chunk ? du - lo : chunk; - ray_t* sv = ray_vec_new(RAY_STR, n); - if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); goto fail; } - for (int64_t i = 0; i < n; i++) { - int64_t pos = ray_read_sym(dv, lo + i, dom_vec->type, dom_vec->attrs); - const char* sp = NULL; - size_t sl = 0; - if (pos >= 0 && pos < raw.count) { - sp = ray_sym_domain_raw_str(&raw, pos, &sl); - } else { - ray_t* a = ray_sym_domain_str(dom, pos); - if (a) { sp = ray_str_ptr(a); sl = ray_str_len(a); } - } - sv = ray_str_vec_append(sv, sp ? sp : "", sp ? sl : 0); - if (!sv || RAY_IS_ERR(sv)) { if (sv) ray_error_free(sv); goto fail; } - } + ray_t* sv = derived_key_chunk_strs(dv, dom_vec->attrs, lo, n, &raw, dom); + if (!sv) goto fail; ray_t* mini = ray_table_new(0); if (mini && !RAY_IS_ERR(mini)) mini = ray_table_add_col(mini, col_sym, sv); ray_release(sv); @@ -3134,13 +3448,7 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom if (!kc || RAY_IS_ERR(kc)) { if (kc) ray_error_free(kc); goto fail; } if (!ray_is_vec(kc) || kc->len != n) { ray_release(kc); goto fail; } if (kc->type == RAY_STR) { - for (int64_t i = 0; i < n; i++) { - size_t sl = 0; - const char* sp = ray_str_vec_get(kc, i, &sl); - int64_t id = ray_sym_intern(sp ? sp : "", sp ? sl : 0); - if (id < 0) { ray_release(kc); goto fail; } - kd[lo + i] = id; - } + if (!derived_key_intern_chunk(kc, n, kd + lo)) { ray_release(kc); goto fail; } } else if (RAY_IS_SYM(kc->type)) { /* The STR evaluation still produced symbols (e.g. a literal * symbol branch): take them cell by cell as runtime ids. */ @@ -3160,6 +3468,397 @@ static ray_t* derived_key_str_chunks(ray_t* by_expr, int64_t col_sym, ray_t* dom ray_sym_domain_raw_unpin(dom); return NULL; } + +/* ---- aggregates over the vocabulary ------------------------------------- + * A grouping whose key is derived from one SYM column C and whose every + * aggregate reads only C — count(C), min(C), max(C), sum/avg(strlen C) — + * is decided by the distinct values of C, not by the rows: a group's count + * is the sum of its values' row counts, its min the smallest of its values, + * its strlen sum the length-weighted count. So: one grouping of the rows by + * C itself (the engine's fastest shape: a column key, a count) gives the + * distinct values with their counts; the key expression runs once per + * distinct value (derived_key_str_chunks); and the aggregates are rewritten + * over that per-value table. Nothing per row is evaluated or accumulated + * beyond the count histogram. count(C) counts null rows too, avg(strlen C) + * divides by the non-null rows, min/max skip null — the row-wise rules. + * Returns NULL for every shape it does not take (the caller runs the row + * path). */ +enum { DKV_COUNT = 1, DKV_MIN, DKV_MAX, DKV_SUM_LEN, DKV_AVG_LEN }; +#define DKV_MAX_AGGS 32 + +static ray_t* dkv_sym(const char* name) { return ray_sym(ray_sym_intern(name, strlen(name))); } + +/* (head a) / (head a b) as an expression list; the arguments are consumed. */ +static ray_t* dkv_call(const char* head, ray_t* a, ray_t* b) { + ray_t* l = ray_list_new(b ? 3 : 2); + if (!l || RAY_IS_ERR(l)) { if (a) ray_release(a); if (b) ray_release(b); return NULL; } + ray_t* h = dkv_sym(head); + l = ray_list_append(l, h); ray_release(h); + l = ray_list_append(l, a); ray_release(a); + if (b) { l = ray_list_append(l, b); ray_release(b); } + if (!l || RAY_IS_ERR(l)) { if (l) ray_error_free(l); return NULL; } + return l; +} + +/* (select {keys[i]: vals[i] …}); keys are runtime sym ids, vals consumed. */ +static ray_t* dkv_select(const int64_t* keys, ray_t** vals, int64_t n) { + ray_t* kv = ray_sym_vec_new(RAY_SYM_W64, n); + ray_t* vl = ray_list_new(n); + if (!kv || RAY_IS_ERR(kv) || !vl || RAY_IS_ERR(vl)) { + for (int64_t i = 0; i < n; i++) if (vals[i]) ray_release(vals[i]); + if (kv && !RAY_IS_ERR(kv)) ray_release(kv); + if (vl && !RAY_IS_ERR(vl)) ray_release(vl); + return NULL; + } + kv->len = n; + for (int64_t i = 0; i < n; i++) { + ((int64_t*)ray_data(kv))[i] = keys[i]; + vl = ray_list_append(vl, vals[i]); + ray_release(vals[i]); + } + ray_t* d = ray_dict_new(kv, vl); + if (!d || RAY_IS_ERR(d)) { if (d) ray_error_free(d); return NULL; } + ray_t* sel = ray_list_new(2); + if (!sel || RAY_IS_ERR(sel)) { ray_release(d); return NULL; } + ray_t* h = dkv_sym("select"); + sel = ray_list_append(sel, h); ray_release(h); + sel = ray_list_append(sel, d); ray_release(d); + return sel; +} + +static bool dkv_is_col(ray_t* e, int64_t col) { + return e && e->type == -RAY_SYM && !(e->attrs & ATTR_QUOTED) && e->i64 == col; +} + +/* Which decomposable aggregate over C is `v`, or 0. */ +static int dkv_agg_kind(ray_t* v, int64_t col) { + if (!v || v->type != RAY_LIST || ray_len(v) != 2) return 0; + ray_t** e = (ray_t**)ray_data(v); + if (!e[0] || e[0]->type != -RAY_SYM) return 0; + ray_t* hs = ray_sym_str(e[0]->i64); + if (!hs) return 0; + const char* h = ray_str_ptr(hs); size_t hl = ray_str_len(hs); + bool is = false; + #define DKV_IS(lit) (hl == sizeof(lit) - 1 && memcmp(h, lit, hl) == 0) + if (DKV_IS("count") || DKV_IS("min") || DKV_IS("max")) { + if (!dkv_is_col(e[1], col)) return 0; + return DKV_IS("count") ? DKV_COUNT : DKV_IS("min") ? DKV_MIN : DKV_MAX; + } + is = DKV_IS("sum") || DKV_IS("avg"); + if (!is) return 0; + ray_t* a = e[1]; + if (!a || a->type != RAY_LIST || ray_len(a) != 2) return 0; + ray_t** ae = (ray_t**)ray_data(a); + if (!ae[0] || ae[0]->type != -RAY_SYM) return 0; + ray_t* as = ray_sym_str(ae[0]->i64); + if (!as || ray_str_len(as) != 6 || memcmp(ray_str_ptr(as), "strlen", 6) != 0) return 0; + if (!dkv_is_col(ae[1], col)) return 0; + return DKV_IS("sum") ? DKV_SUM_LEN : DKV_AVG_LEN; + #undef DKV_IS +} + +/* Per distinct value (workers): its non-null row count and its + * length-weighted row count, the length read off the pinned mapping. A + * position past the file prefix is marked (lenw = INT64_MIN) and resolved + * through the domain on the calling thread. */ +typedef struct { + const void* hc; + uint8_t attrs; + const int64_t* cnt; + int64_t* nn; + int64_t* lenw; + struct ray_sym_domain_s* dom; + ray_sym_domain_raw_t raw; + bool raw_ok; + atomic_int late; +} dkv_len_ctx_t; + +static void dkv_len_fn(void* vctx, uint32_t wid, int64_t lo, int64_t hi) { + (void)wid; + dkv_len_ctx_t* c = (dkv_len_ctx_t*)vctx; + int late = 0; + for (int64_t i = lo; i < hi; i++) { + int64_t pos = ray_read_sym(c->hc, i, RAY_SYM, c->attrs); + c->nn[i] = pos > 0 ? c->cnt[i] : 0; + if (pos <= 0) { c->lenw[i] = 0; continue; } + if (c->raw_ok && pos < c->raw.count) { + size_t l; (void)ray_sym_domain_raw_str(&c->raw, pos, &l); + c->lenw[i] = c->cnt[i] * (int64_t)l; + } else { + c->lenw[i] = INT64_MIN; + late++; + } + } + if (late) atomic_fetch_add_explicit(&c->late, late, memory_order_relaxed); +} + +/* H (distinct values of C with their counts) in C's domain-position order. + * The grouping that produced H emits in an order that depends on the core + * count; the rewrite's output order follows H, and a derived-key grouping + * on the row path comes out in the same order at every core count (the + * first-seen order of the key values, which follows the positions of the + * values they come from). Positions are distinct, so the order is the + * rank of each position among those present: a bitmap over the domain, + * block popcounts, then a scatter. Returns a new table, or NULL (H kept). */ +static ray_t* dkv_order_by_position(ray_t* H, int64_t dom_count) { + ray_t* Hc = ray_table_get_col_idx(H, 0); + ray_t* Hn = ray_table_get_col_idx(H, 1); + int64_t du = ray_table_nrows(H); + if (dom_count <= 0 || du <= 1) return NULL; + int64_t nw = (dom_count + 63) / 64; + int64_t nb = (nw + 63) / 64; /* rank blocks of 64 words */ + ray_t *bh = NULL, *rh = NULL; + uint64_t* bits = (uint64_t*)scratch_calloc(&bh, (size_t)nw * sizeof(uint64_t)); + int64_t* rank = (int64_t*)scratch_alloc(&rh, (size_t)(nb + 1) * sizeof(int64_t)); + if (!bits || !rank) { scratch_free(bh); scratch_free(rh); return NULL; } + const void* cd = ray_data(Hc); + for (int64_t i = 0; i < du; i++) { + int64_t pos = ray_read_sym(cd, i, RAY_SYM, Hc->attrs); + if (pos < 0 || pos >= dom_count) { scratch_free(bh); scratch_free(rh); return NULL; } + bits[pos >> 6] |= (uint64_t)1 << (pos & 63); + } + int64_t run = 0; + for (int64_t b = 0; b < nb; b++) { + rank[b] = run; + int64_t w1 = (b + 1) * 64 < nw ? (b + 1) * 64 : nw; + for (int64_t w = b * 64; w < w1; w++) run += __builtin_popcountll(bits[w]); + } + ray_t* nc = ray_sym_vec_new(Hc->attrs & RAY_SYM_W_MASK, du); + ray_t* nn = ray_vec_new(RAY_I64, du); + if (!nc || RAY_IS_ERR(nc) || !nn || RAY_IS_ERR(nn)) { + if (nc && !RAY_IS_ERR(nc)) ray_release(nc); + if (nn && !RAY_IS_ERR(nn)) ray_release(nn); + scratch_free(bh); scratch_free(rh); + return NULL; + } + ray_sym_vec_adopt_domain(nc, Hc); + nc->len = du; nn->len = du; + void* ncd = ray_data(nc); + int64_t* nnd = (int64_t*)ray_data(nn); + const int64_t* hnd = (const int64_t*)ray_data(Hn); + for (int64_t i = 0; i < du; i++) { + int64_t pos = ray_read_sym(cd, i, RAY_SYM, Hc->attrs); + int64_t w = pos >> 6; + int64_t r = rank[w >> 6]; + for (int64_t x = (w >> 6) * 64; x < w; x++) r += __builtin_popcountll(bits[x]); + r += __builtin_popcountll(bits[w] & (((uint64_t)1 << (pos & 63)) - 1)); + ray_write_sym(ncd, r, (uint64_t)pos, RAY_SYM, nc->attrs); + nnd[r] = hnd[i]; + } + if (Hc->attrs & RAY_ATTR_HAS_NULLS) nc->attrs |= RAY_ATTR_HAS_NULLS; + scratch_free(bh); scratch_free(rh); + ray_t* out = ray_table_new(2); + if (!out || RAY_IS_ERR(out)) { ray_release(nc); ray_release(nn); return NULL; } + out = ray_table_add_col(out, ray_table_col_name(H, 0), nc); + out = ray_table_add_col(out, ray_table_col_name(H, 1), nn); + ray_release(nc); ray_release(nn); + if (!out || RAY_IS_ERR(out)) return NULL; + return out; +} + +static ray_t* derived_key_vocab_aggs(ray_t* tbl, ray_t* by_expr, ray_t* where_expr, + ray_t** dict_elems, int64_t dict_n, + int64_t from_id, int64_t by_id, int64_t where_id, + int64_t take_id, int64_t asc_id, int64_t desc_id, + int64_t nearest_id) { + if (!by_expr || by_expr->type != RAY_LIST || !tbl) return NULL; + int64_t ref_syms[2]; + if (collect_col_refs(by_expr, tbl, ref_syms, 2, 0) != 1) return NULL; + int64_t col = ref_syms[0]; + int64_t bound[32]; + if (!derived_key_expr_ok(by_expr, tbl, col, bound, 0)) return NULL; + ray_t* C = ray_table_get_col(tbl, col); + int64_t nrows = ray_table_nrows(tbl); + if (!C || C->type != RAY_SYM || !ray_is_vec(C) || C->len != nrows || nrows < 65536) return NULL; + struct ray_sym_domain_s* dom = ray_sym_vec_domain(C); + if (!dom || dom == ray_sym_runtime_domain()) return NULL; + + /* Every output is one of the decomposable aggregates over C; sort/take + * clauses may only name output aliases. */ + int64_t alias[DKV_MAX_AGGS]; int kind[DKV_MAX_AGGS]; int n_aggs = 0; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid == from_id || kid == by_id || kid == where_id || kid == take_id || + kid == asc_id || kid == desc_id) continue; + if (kid == nearest_id) return NULL; + if (n_aggs >= DKV_MAX_AGGS) return NULL; + int k = dkv_agg_kind(dict_elems[i + 1], col); + if (!k) return NULL; + alias[n_aggs] = kid; kind[n_aggs] = k; n_aggs++; + } + if (n_aggs == 0) return NULL; + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* v = dict_elems[i + 1]; + int64_t nk = (v && v->type == -RAY_SYM) ? 1 : (v && v->type == RAY_SYM) ? ray_len(v) : -1; + if (nk < 0) return NULL; + for (int64_t j = 0; j < nk; j++) { + int64_t sid = (v->type == -RAY_SYM) ? v->i64 : sym_cell_runtime_id(v, j); + bool ok = false; + for (int a = 0; a < n_aggs; a++) if (alias[a] == sid) ok = true; + if (!ok) return NULL; + } + } + + /* 1. the rows grouped by C itself: distinct values with their counts */ + int64_t s_t = ray_sym_intern("__dkv_t", 7), s_cnt = ray_sym_intern("__dkv_cnt", 9); + int64_t s_s = ray_sym_intern("__dkv_s", 7), s_k = ray_sym_intern("__dkv_k", 7); + int64_t s_ref = ray_sym_intern("__dkv_ref", 9), s_nn = ray_sym_intern("__dkv_nn", 8); + int64_t s_lenw = ray_sym_intern("__dkv_lenw", 10); + ray_t* H = NULL; + { + int64_t keys[4]; ray_t* vals[4]; int64_t n = 0; + keys[n] = from_id; vals[n++] = ray_sym(s_t); + keys[n] = by_id; vals[n++] = ray_sym(col); + keys[n] = s_cnt; vals[n++] = dkv_call("count", ray_sym(col), NULL); + if (where_expr) { keys[n] = where_id; ray_retain(where_expr); vals[n++] = where_expr; } + for (int64_t i = 0; i < n; i++) if (!vals[i]) { for (int64_t j = 0; j < n; j++) if (vals[j]) ray_release(vals[j]); return NULL; } + ray_t* q = dkv_select(keys, vals, n); + if (!q) return NULL; + if (ray_env_push_query_scope() != RAY_OK) { ray_release(q); return NULL; } + ray_env_set_query_local(s_t, tbl); + H = ray_eval(q); + ray_env_pop_scope(); + ray_release(q); + } + if (!H || RAY_IS_ERR(H) || H->type != RAY_TABLE || ray_table_ncols(H) != 2) { + if (H && !RAY_IS_ERR(H)) ray_release(H); else if (H) ray_error_free(H); + return NULL; + } + ray_t* Hc = ray_table_get_col_idx(H, 0); + ray_t* Hn = ray_table_get_col_idx(H, 1); + int64_t du = ray_table_nrows(H); + if (!Hc || !Hn || Hc->type != RAY_SYM || Hn->type != RAY_I64 || du <= 0 || + ray_sym_vec_domain(Hc) != dom) { ray_release(H); return NULL; } + { + ray_t* Ho = dkv_order_by_position(H, ray_sym_domain_count(dom)); + if (Ho) { + ray_release(H); + H = Ho; + Hc = ray_table_get_col_idx(H, 0); + Hn = ray_table_get_col_idx(H, 1); + } + } + + /* 2. the key once per distinct value */ + ray_t* key_dom = derived_key_str_chunks(by_expr, col, Hc, dom, du); + if (!key_dom || RAY_IS_ERR(key_dom)) { if (key_dom) ray_error_free(key_dom); ray_release(H); return NULL; } + + /* 3. the per-value table: key, count, the value, its non-null count, + * its length-weighted count */ + ray_t* S = NULL; + { + ray_t* nn = ray_vec_new(RAY_I64, du); + ray_t* lenw = ray_vec_new(RAY_I64, du); + if (!nn || RAY_IS_ERR(nn) || !lenw || RAY_IS_ERR(lenw)) { + if (nn && !RAY_IS_ERR(nn)) ray_release(nn); + if (lenw && !RAY_IS_ERR(lenw)) ray_release(lenw); + ray_release(key_dom); ray_release(H); return NULL; + } + nn->len = du; lenw->len = du; + dkv_len_ctx_t lc = { + .hc = ray_data(Hc), .attrs = Hc->attrs, .cnt = (const int64_t*)ray_data(Hn), + .nn = (int64_t*)ray_data(nn), .lenw = (int64_t*)ray_data(lenw), .dom = dom, + }; + lc.raw_ok = ray_sym_domain_raw_pin(dom, &lc.raw); + atomic_store_explicit(&lc.late, 0, memory_order_relaxed); + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, du, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, dkv_len_fn, &lc, du); + else + dkv_len_fn(&lc, 0, 0, du); + if (atomic_load_explicit(&lc.late, memory_order_relaxed)) { + /* positions past the file prefix: resolved through the domain on + * the calling thread */ + for (int64_t i = 0; i < du; i++) { + if (lc.lenw[i] != INT64_MIN) continue; + int64_t pos = ray_read_sym(lc.hc, i, RAY_SYM, lc.attrs); + ray_t* a = ray_sym_domain_str(dom, pos); + lc.lenw[i] = lc.cnt[i] * (a ? (int64_t)ray_str_len(a) : 0); + } + } + S = ray_table_new(5); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_k, key_dom); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_cnt, Hn); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_ref, Hc); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_nn, nn); + if (S && !RAY_IS_ERR(S)) S = ray_table_add_col(S, s_lenw, lenw); + ray_release(nn); ray_release(lenw); + } + ray_release(key_dom); + ray_release(H); + if (!S || RAY_IS_ERR(S)) { if (S) ray_error_free(S); return NULL; } + + /* 4. the aggregates rewritten over the per-value table, sort/take as + * written */ + ray_t* R = NULL; + { + int64_t keys[DKV_MAX_AGGS + 8]; ray_t* vals[DKV_MAX_AGGS + 8]; int64_t n = 0; + bool bad = false; + for (int64_t i = 0; i + 1 < dict_n && !bad; i += 2) { + int64_t kid = dict_elems[i]->i64; + ray_t* v = NULL; + if (kid == from_id) v = ray_sym(s_s); + else if (kid == by_id) v = ray_sym(s_k); + else if (kid == where_id) continue; + else if (kid == take_id || kid == asc_id || kid == desc_id) { v = dict_elems[i + 1]; ray_retain(v); } + else { + int k = 0; + for (int a = 0; a < n_aggs; a++) if (alias[a] == kid) k = kind[a]; + switch (k) { + case DKV_COUNT: v = dkv_call("sum", ray_sym(s_cnt), NULL); break; + case DKV_MIN: v = dkv_call("min", ray_sym(s_ref), NULL); break; + case DKV_MAX: v = dkv_call("max", ray_sym(s_ref), NULL); break; + case DKV_SUM_LEN: v = dkv_call("sum", ray_sym(s_lenw), NULL); break; + case DKV_AVG_LEN: v = dkv_call("/", dkv_call("sum", ray_sym(s_lenw), NULL), + dkv_call("sum", ray_sym(s_nn), NULL)); break; + default: v = NULL; + } + } + if (!v) { bad = true; break; } + keys[n] = kid; vals[n++] = v; + } + if (bad) { for (int64_t j = 0; j < n; j++) ray_release(vals[j]); ray_release(S); return NULL; } + ray_t* q = dkv_select(keys, vals, n); + if (!q) { ray_release(S); return NULL; } + if (ray_env_push_query_scope() != RAY_OK) { ray_release(q); ray_release(S); return NULL; } + ray_env_set_query_local(s_s, S); + R = ray_eval(q); + ray_env_pop_scope(); + ray_release(q); + } + ray_release(S); + if (!R || RAY_IS_ERR(R)) { if (R) ray_error_free(R); return NULL; } + if (ray_is_lazy(R)) R = ray_lazy_materialize(R); + if (!R || RAY_IS_ERR(R) || R->type != RAY_TABLE || ray_table_ncols(R) == 0) { + if (R && !RAY_IS_ERR(R)) ray_release(R); else if (R) ray_error_free(R); + return NULL; + } + /* The engine emits a post-aggregate expression (the avg quotient) + * after the plain aggregates: put the outputs back in the order they + * were written, key first. */ + int64_t rc = ray_table_ncols(R); + if (rc != n_aggs + 1) { ray_release(R); return NULL; } + ray_t* O = ray_table_new(rc); + if (!O || RAY_IS_ERR(O)) { if (O) ray_error_free(O); ray_release(R); return NULL; } + O = ray_table_add_col(O, ray_table_col_name(R, 0), ray_table_get_col_idx(R, 0)); + for (int a = 0; a < n_aggs && O && !RAY_IS_ERR(O); a++) { + ray_t* col = ray_table_get_col(R, alias[a]); + if (!col) { ray_release(O); ray_release(R); return NULL; } + O = ray_table_add_col(O, alias[a], col); + } + ray_release(R); + if (!O || RAY_IS_ERR(O)) { if (O) ray_error_free(O); return NULL; } + /* the key column is named as the row path names a computed key */ + int64_t kname = derived_key_name(by_expr); + for (int64_t c = 1; c < rc; c++) + if (ray_table_col_name(O, c) == kname) { kname = ray_sym_intern("key", 3); break; } + ray_table_set_col_name(O, 0, kname); + agg_route_note_key_domain(); + return O; +} + static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { if (!by_expr || by_expr->type != RAY_LIST || !tbl) return NULL; int64_t ref_syms[2]; @@ -3215,6 +3914,9 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { ray_sym_vec_adopt_domain(dom_vec, C); int64_t du = 0, du_max = nrows / 2; if (du_max > INT32_MAX) du_max = INT32_MAX; /* slots are int32 */ + /* Slots in first-seen row order: the interned key ids follow the slot + * order, and with them the order the groups come out in — the same + * order the row-wise evaluation gives. */ bool ok = true; for (int64_t r = 0; r < nrows; r++) { int64_t id = ray_read_sym(cd, r, C->type, C->attrs); @@ -3228,6 +3930,11 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { } if (!ok || du == 0) { ray_release(dom_vec); scratch_free(pos_hdr); return NULL; } dom_vec->len = du; + dk_rows_ctx_t rc; + memset(&rc, 0, sizeof(rc)); + rc.cd = cd; rc.attrs = C->attrs; rc.dn = dn; + ray_pool_t* rpool = ray_pool_get(); + bool rows_par = ray_pool_par_dispatch_ok(rpool, nrows, RAY_PARALLEL_THRESHOLD); /* Evaluate the expression over the du distinct symbols through the * same DAG compiler the row-wise key would take, against a one-column @@ -3254,20 +3961,37 @@ static ray_t* derived_key_over_sym_domain(ray_t* by_expr, ray_t* tbl) { if (!ray_is_vec(key_dom) || key_dom->len != du) { ray_release(key_dom); scratch_free(pos_hdr); return NULL; } } - /* Pass 2: spread by slot. */ - ray_t* ids = ray_vec_new(RAY_I64, nrows); - if (!ids || RAY_IS_ERR(ids)) { if (ids) ray_error_free(ids); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } - ids->len = nrows; - int64_t* idp = (int64_t*)ray_data(ids); - for (int64_t r = 0; r < nrows; r++) - idp[r] = pos[ray_read_sym(cd, r, C->type, C->attrs)]; - scratch_free(pos_hdr); - ray_t* spread = ray_at_fn(key_dom, ids); - ray_release(ids); - ray_release(key_dom); - if (spread && !RAY_IS_ERR(spread) && ray_is_lazy(spread)) spread = ray_lazy_materialize(spread); - if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); return NULL; } - if (!ray_is_vec(spread) || spread->len != nrows) { ray_release(spread); return NULL; } + /* Pass 2: spread by slot — each row takes the key of its symbol's slot, + * written straight into the result (no index vector, no gather). */ + if (!ray_is_vec(key_dom) || key_dom->len != du) { ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + ray_t* spread = NULL; + if (key_dom->type == RAY_SYM && (key_dom->attrs & RAY_SYM_W_MASK) == RAY_SYM_W64) { + spread = ray_sym_vec_new(RAY_SYM_W64, nrows); + if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + spread->len = nrows; + ray_sym_vec_adopt_domain(spread, key_dom); + rc.pos = pos; rc.key = (const int64_t*)ray_data(key_dom); rc.out = (int64_t*)ray_data(spread); + if (rows_par) ray_pool_dispatch(rpool, dk_spread_fn, &rc, nrows); + else dk_spread_fn(&rc, 0, 0, nrows); + scratch_free(pos_hdr); + ray_release(key_dom); + } else { + /* Any other key vector (the one-shot SYM evaluation's width, or a + * non-SYM result): index by slot and gather. */ + ray_t* ids = ray_vec_new(RAY_I64, nrows); + if (!ids || RAY_IS_ERR(ids)) { if (ids) ray_error_free(ids); ray_release(key_dom); scratch_free(pos_hdr); return NULL; } + ids->len = nrows; + rc.pos = pos; rc.key = NULL; rc.out = (int64_t*)ray_data(ids); + if (rows_par) ray_pool_dispatch(rpool, dk_spread_fn, &rc, nrows); + else dk_spread_fn(&rc, 0, 0, nrows); + scratch_free(pos_hdr); + spread = ray_at_fn(key_dom, ids); + ray_release(ids); + ray_release(key_dom); + if (spread && !RAY_IS_ERR(spread) && ray_is_lazy(spread)) spread = ray_lazy_materialize(spread); + if (!spread || RAY_IS_ERR(spread)) { if (spread) ray_error_free(spread); return NULL; } + if (!ray_is_vec(spread) || spread->len != nrows) { ray_release(spread); return NULL; } + } agg_route_note_key_domain(); return spread; } @@ -7091,6 +7815,24 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (n_out == 0 && !where_expr && !by_expr && !take_expr && !has_sort && !nearest_expr) { DICT_VIEW_CLOSE(dv); return tbl; } + /* Whole-table aggregates answered from column metadata: + * `(select {from: T a: (sum c) b: (min d) …})` with no other clause, each + * output one of count / min / max / sum / avg over a plain column that + * the metadata answers exactly — count is the row count; min/max read + * the chunk-zone extrema of an integer or temporal column; sum and avg + * read its per-chunk sums (avg only when no partial sum can leave + * double's integer range, so the value is the row-wise one). Any + * output the metadata cannot answer leaves the query to the planner. */ + if (!where_expr && !by_expr && !take_expr && !has_sort && !nearest_expr && n_out > 0 && + tbl->type == RAY_TABLE) { + ray_t* meta_res = select_aggs_from_metadata(tbl, dict_elems, dict_n, from_id); + if (meta_res) { + ray_release(tbl); + DICT_VIEW_CLOSE(dv); + return meta_res; + } + } + /* Streaming parted ORDER BY: `(select {…} from: PARTED asc/desc: KEY)` * with no by:/take:/nearest: over a table whose partitions are already * internally sorted on KEY and whose partition key-ranges are globally @@ -7135,8 +7877,8 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * intermediate filtered table materialised. Closes a large * latency gap on ORDER BY + LIMIT shapes that were previously * dominated by the filtered-table materialisation step. */ - if (where_expr && take_expr && has_sort && !by_expr && !nearest_expr) { - if (ray_fused_topk_supported(where_expr, tbl)) { + if (take_expr && (has_sort || where_expr) && !by_expr && !nearest_expr) { + if (!where_expr || ray_fused_topk_supported(where_expr, tbl)) { /* Walk the dict and check: exactly one asc/desc clause naming * a single scalar column, take is an atom K, and every * output column is a -RAY_SYM source-column reference (no @@ -7184,6 +7926,16 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * column. The dict key is the alias the result publishes; * the value names the source column to gather from. */ if (n_out_syms >= 255) { bad_clause = 1; break; } + /* Projections resolve left to right: a name bound by an + * earlier output is that output's value, not the source + * column. The fused paths gather source columns, so such + * a shape is left to the planner. */ + if (v && v->type == -RAY_SYM && !(v->attrs & ATTR_QUOTED)) { + bool alias_ref = false; + for (uint8_t o = 0; o < n_out_syms; o++) + if (out_aliases[o] == v->i64 && out_syms[o] != v->i64) alias_ref = true; + if (alias_ref) { bad_clause = 1; break; } + } if (v && v->type == -RAY_SYM && !(v->attrs & ATTR_QUOTED)) { ray_t* oc = ray_table_get_col(tbl, v->i64); if (!oc) { bad_clause = 1; break; } @@ -7232,13 +7984,43 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (!kc) bad_clause = 1; } } - if (!bad_clause && n_sort_keys > 0 && n_out_syms > 0) { + /* Without a sort the positional path takes only a literal K: + * an expression is left to the general path, which evaluates + * it exactly once. */ + if (n_sort_keys == 0 && !(ray_is_atom(take_expr) && + (take_expr->type == -RAY_I64 || take_expr->type == -RAY_I32))) + bad_clause = 1; + if (!bad_clause && n_out_syms > 0) { ray_t* tv = ray_eval(take_expr); if (tv && !RAY_IS_ERR(tv) && ray_is_atom(tv) && (tv->type == -RAY_I64 || tv->type == -RAY_I32)) { int64_t k = (tv->type == -RAY_I64) ? tv->i64 : tv->i32; ray_release(tv); - if (k > 0 && k <= FPK_MAX_K && k < ray_table_nrows(tbl)) { + /* Positional take under a filter, and an ascending + * take on a column known to be sorted (no nulls): the + * answer is the first |k| passing rows from one end of + * the table, found without scanning the rest. */ + bool sorted_asc = false; + if (n_sort_keys == 1 && !sort_descs[0]) { + ray_t* kc = ray_table_get_col(tbl, sort_key_syms[0]); + sorted_asc = kc && (kc->attrs & RAY_ATTR_SORTED) && + !(kc->attrs & RAY_ATTR_SLICE) && + !ray_vec_may_have_nulls(kc); + } + if (where_expr && (n_sort_keys == 0 || (sorted_asc && k > 0)) && + k != 0 && k != INT64_MIN && + (k < 0 ? -k : k) <= FPK_MAX_K && + (k < 0 ? -k : k) < ray_table_nrows(tbl)) { + ray_t* res = ray_fused_take_select(tbl, where_expr, k, + out_syms, out_aliases, + n_out_syms); + if (res && !RAY_IS_ERR(res)) { + ray_release(tbl); + DICT_VIEW_CLOSE(dv); return res; + } + if (res && RAY_IS_ERR(res)) ray_release(res); + } + if (n_sort_keys > 0 && k > 0 && k <= FPK_MAX_K && k < ray_table_nrows(tbl)) { ray_t* res = ray_fused_topk_select(tbl, where_expr, sort_key_syms, sort_descs, @@ -7277,6 +8059,10 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * FLAT path so nothing there changes. */ bool parted_bydict_deferred = false; bool computed_single_key = false; /* by: is one expression, compiled or const */ + /* Sort keys that are source columns no output scans: carried through + * the projection (appended after the outputs) so the sort can read + * them, dropped from the result after it ran. */ + int n_hidden_sort = 0; ray_t* deferred_bydict = NULL; int64_t deferred_nk = 0; int64_t dep_key_base_sym = -1; @@ -10171,7 +10957,18 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { } else { /* Single key expression. Over a lone SYM column evaluate it per * distinct symbol and feed the spread key as a constant node, - * named the way the eval-level path names a computed key. */ + * named the way the eval-level path names a computed key. When + * every aggregate reads that column too, the whole grouping is + * decided over the distinct values (derived_key_vocab_aggs). */ + { + ray_t* vres = derived_key_vocab_aggs(tbl, by_expr, where_expr, dict_elems, dict_n, + from_id, by_id, where_id, take_id, + asc_id, desc_id, nearest_id); + if (vres) { + ray_graph_free(g); ray_release(tbl); scratch_free(sel_slots_hdr); DICT_VIEW_CLOSE(dv); + return vres; + } + } ray_t* dom_key = derived_key_over_sym_domain(by_expr, tbl); if (dom_key) { key_ops[0] = ray_const_vec(g, dom_key); @@ -11169,6 +11966,15 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { * eval fallback via the width check below; now only the other * use_eval_fallback trigger (a failed compile) applies. */ int64_t nc_max = select_output_count(dict_elems, dict_n); + /* plus one slot per sort key name: any of them may have to be + * carried through the projection as a hidden column */ + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* val = dict_elems[i + 1]; + if (val->type == -RAY_SYM) nc_max += 1; + else if (ray_is_vec(val) && val->type == RAY_SYM) nc_max += ray_len(val); + } if (nc_max < 1) nc_max = 1; ray_t* colops_hdr = NULL; ray_op_t** col_ops = (ray_op_t**)scratch_alloc(&colops_hdr, @@ -11209,6 +12015,35 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { g->sel_alias_syms = NULL; g->sel_alias_ids = NULL; g->sel_alias_n = 0; + /* Sort keys name SOURCE columns (like where: and by:, they are + * alias-blind). A key some output scans bare is read from that + * output (the sort binds to the projected column by position, + * see exec_sort); a key no output scans is projected too, after + * the outputs, and dropped from the result afterwards. Output + * aliases are not consulted: `{b: a ... asc: b}` sorts by the + * source column b, not by the output that renames a. */ + if (!use_eval_fallback && has_sort) { + for (int64_t i = 0; i + 1 < dict_n; i += 2) { + int64_t kid = dict_elems[i]->i64; + if (kid != asc_id && kid != desc_id) continue; + ray_t* val = dict_elems[i + 1]; + int64_t nk = val->type == -RAY_SYM ? 1 + : (ray_is_vec(val) && val->type == RAY_SYM) ? val->len : 0; + for (int64_t k = 0; k < nk; k++) { + int64_t ks = val->type == -RAY_SYM ? val->i64 : sym_cell_runtime_id(val, k); + if (select_scan_output(g, col_ops, nc, ks) >= 0) continue; + if (!ray_table_get_col(tbl, ks) || nc >= nc_max) continue; + ray_t* nm = ray_sym_str(ks); + ray_op_t* sc = nm ? ray_scan(g, ray_str_ptr(nm)) : NULL; + if (!sc) continue; + col_ops[nc] = sc; + alias_syms[nc] = ks; + alias_ids[nc] = sc->id; + nc++; + n_hidden_sort++; + } + } + } if (use_eval_fallback) { if (g->compile_err) { ray_release(g->compile_err); g->compile_err = NULL; } /* The fallback evaluates projections directly over `tbl`, @@ -11362,14 +12197,14 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { if (val->type == -RAY_SYM) { /* Single column name */ ray_t* s = ray_sym_str(val->i64); - sort_keys[n_sort] = ray_scan(g, ray_str_ptr(s)); + sort_keys[n_sort] = select_sort_key_op(g, root, val->i64, ray_str_ptr(s)); sort_descs[n_sort] = is_desc; n_sort++; } else if (ray_is_vec(val) && val->type == RAY_SYM) { /* Multiple column names — cell-data via the vec's domain */ for (int64_t c = 0; c < val->len; c++) { ray_t* s = ray_sym_vec_cell(val, c); - sort_keys[n_sort] = ray_scan(g, ray_str_ptr(s)); + sort_keys[n_sort] = select_sort_key_op(g, root, sym_cell_runtime_id(val, c), ray_str_ptr(s)); sort_descs[n_sort] = is_desc; n_sort++; } @@ -11472,6 +12307,21 @@ static ray_t* ray_select_impl(ray_t** args, int64_t n, bool aliases_resolved) { /* Optimize and execute */ root = ray_optimize(g, root); ray_t* result = ray_execute(g, root); + if (n_hidden_sort > 0 && result && !RAY_IS_ERR(result)) { + if (ray_is_lazy(result)) result = ray_lazy_materialize(result); + if (result && !RAY_IS_ERR(result) && result->type == RAY_TABLE) { + /* the hidden keys were projected last: drop the trailing + * columns (by position — an output may carry the same name) */ + int64_t rc = ray_table_ncols(result); + int64_t keep_n = rc - n_hidden_sort; + ray_t* kept = keep_n >= 0 ? ray_table_new(keep_n > 0 ? keep_n : 1) : NULL; + for (int64_t c = 0; kept && !RAY_IS_ERR(kept) && c < keep_n; c++) + kept = ray_table_add_col(kept, ray_table_col_name(result, c), + ray_table_get_col_idx(result, c)); + if (kept && !RAY_IS_ERR(kept)) { ray_release(result); result = kept; } + else if (kept) ray_release(kept); + } + } /* A computed key takes its column name from its op's ext sym, which is * not a name (a const node's slot holds the literal; an expression node's * ext resolves to whatever shares its id) — name it the way the diff --git a/src/ops/sort.c b/src/ops/sort.c index b835f71cb..3e8a02edf 100644 --- a/src/ops/sort.c +++ b/src/ops/sort.c @@ -3572,6 +3572,26 @@ ray_t* ray_sort(ray_t** cols, uint8_t* descs, uint8_t* nulls_first, return result; } +/* The column of `tbl` a scan sort key reads. When the sort runs directly + * over a SELECT and the key is one of that SELECT's columns, bind by + * POSITION: the projection names scans by their source column and + * expressions `_e`, so a hidden key or a renamed output can share a + * name with another projected column and a name lookup would read the + * wrong one. Any other key resolves by name. Borrowed. */ +static ray_t* sort_key_scan_col(ray_graph_t* g, ray_op_t* op, ray_t* tbl, + uint32_t key_id, int64_t key_sym) { + ray_op_t* child = g ? op_child(g, op, 0) : NULL; + if (child && child->opcode == OP_SELECT) { + ray_op_ext_t* se = find_ext(g, child->id); + if (se && se->base.opcode == OP_SELECT && + (int64_t)se->sort.n_cols == ray_table_ncols(tbl)) + for (uint32_t c = 0; c < se->sort.n_cols; c++) + if (se->sort.columns[c] == key_id) + return ray_table_get_col_idx(tbl, c); + } + return ray_table_get_col(tbl, key_sym); +} + ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { if (!tbl || RAY_IS_ERR(tbl)) return tbl; @@ -3603,7 +3623,7 @@ ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { ray_op_t* key_op = op_node(g, ext->sort.columns[k]); ray_op_ext_t* key_ext = find_ext(g, key_op->id); if (key_ext && key_ext->base.opcode == OP_SCAN) { - key_cols[k] = ray_table_get_col(tbl, key_ext->sym); + key_cols[k] = sort_key_scan_col(g, op, tbl, key_op->id, key_ext->sym); if (!key_cols[k]) { all_scan = 0; break; } } else { all_scan = 0; @@ -3655,7 +3675,7 @@ ray_t* exec_sort(ray_graph_t* g, ray_op_t* op, ray_t* tbl, int64_t limit) { ray_op_t* key_op = op_node(g, ext->sort.columns[k]); ray_op_ext_t* key_ext = find_ext(g, key_op->id); if (key_ext && key_ext->base.opcode == OP_SCAN) { - sort_vecs[k] = ray_table_get_col(tbl, key_ext->sym); + sort_vecs[k] = sort_key_scan_col(g, op, tbl, key_op->id, key_ext->sym); } else { ray_t* saved = g->table; g->table = tbl; diff --git a/src/ops/string.c b/src/ops/string.c index 501e89431..bf01958dd 100644 --- a/src/ops/string.c +++ b/src/ops/string.c @@ -987,171 +987,173 @@ static bool substr_scalar_arg(ray_t* v, int64_t* out) { } } -static ray_t* substr_str_scalar_view(ray_t* input, int64_t start, int64_t length) { - if (!input || input->type != RAY_STR) - return NULL; - - int64_t nrows = input->len; - ray_t* result = ray_vec_new(RAY_STR, nrows); - if (!result || RAY_IS_ERR(result)) return result ? result : ray_error("oom", NULL); - result->len = nrows; +/* True when a whole-column (scalar) start/length argument is null. Such an + * argument applies to every row, so the entire result is null — and a null + * integer scalar is INT64_MIN, which must never reach the `scalar - 1` + * subtraction in the row loop (signed-overflow UB). Only scalars are tested + * here; the per-row vector case is handled inline via ray_vec_is_null. */ +static bool substr_scalar_is_null(ray_t* v) { + if (!v) return false; + if (ray_is_atom(v)) return RAY_ATOM_IS_NULL(v); + if (ray_is_vec(v) && v->len == 1 && ray_vec_may_have_nulls(v)) + return ray_vec_is_null(v, 0); + return false; +} - const ray_str_t* src = NULL; - const char* pool = NULL; - str_resolve(input, &src, &pool); - ray_t* owner = (input->attrs & RAY_ATTR_SLICE) ? input->slice_parent : input; - ray_t* pool_obj = owner ? owner->str_pool : NULL; - if (pool_obj && !RAY_IS_ERR(pool_obj)) { - ray_retain(pool_obj); - result->str_pool = pool_obj; +/* Per-row start / length argument of substr: an atom, a 1-element vector + * (scalar), or a vector with one value per row (I64 / I32 / F64). `ok` is + * false for a null. */ +typedef struct { + int64_t scalar; + const int64_t* i64; + const int32_t* i32; + const double* f64; + ray_t* v; /* the vector, for null checks; NULL for scalars */ + bool all_null; +} substr_arg_t; + +static bool substr_arg_init(ray_t* v, int64_t nrows, substr_arg_t* a) { + memset(a, 0, sizeof(*a)); + int64_t sc; + if (substr_scalar_arg(v, &sc)) { a->scalar = sc; return true; } + if (substr_scalar_is_null(v)) { a->all_null = true; return true; } + if (!ray_is_vec(v) || v->len != nrows) return false; + a->v = v; + switch (v->type) { + case RAY_I64: a->i64 = (const int64_t*)ray_data(v); return true; + case RAY_I32: a->i32 = (const int32_t*)ray_data(v); return true; + case RAY_F64: a->f64 = (const double*)ray_data(v); return true; + default: return false; } +} - ray_str_t* dst = (ray_str_t*)ray_data(result); - for (int64_t i = 0; i < nrows; i++) { - const ray_str_t* s = &src[i]; - ray_str_t* d = &dst[i]; - memset(d, 0, sizeof(*d)); +static inline bool substr_arg_at(const substr_arg_t* a, int64_t r, int64_t* out) { + if (a->all_null) return false; + if (!a->v) { *out = a->scalar; return true; } + if (a->i64) { int64_t x = a->i64[r]; if (x == NULL_I64) return false; *out = x; return true; } + if (a->i32) { int32_t x = a->i32[r]; if (x == NULL_I32) return false; *out = (int64_t)x; return true; } + double d = a->f64[r]; + if (d != d) return false; + *out = (int64_t)d; + return true; +} - int64_t st = start - 1; +/* Substring of a STR column as descriptors over the column's own pool: + * an inline result copies its bytes, a longer one points into the parent + * pool at the shifted offset. No bytes are copied and no pool is built, + * so the pass is descriptor-bound and runs on the worker pool. A null + * start or length gives a null (empty) row; the empty result of a start + * past the end is an empty row. Returns NULL for shapes it does not take + * (the caller keeps the general loop). */ +typedef struct { + const ray_str_t* src; + const char* pool; + ray_str_t* dst; + substr_arg_t start; + substr_arg_t len; + _Atomic(uint32_t) any_null; + _Atomic(uint32_t) range_err; + _Atomic(uint64_t) pooled_bytes; /* bytes the pooled results point at */ +} substr_view_ctx_t; + +static void substr_view_fn(void* vctx, uint32_t worker_id, int64_t lo, int64_t hi) { + (void)worker_id; + substr_view_ctx_t* c = (substr_view_ctx_t*)vctx; + bool null_seen = false, range_seen = false; + uint64_t pooled = 0; + for (int64_t i = lo; i < hi; i++) { + ray_str_t* d = &c->dst[i]; + memset(d, 0, sizeof(*d)); + int64_t st, ln; + if (!substr_arg_at(&c->start, i, &st) || !substr_arg_at(&c->len, i, &ln)) { + null_seen = true; + continue; + } + const ray_str_t* s = &c->src[i]; int64_t sl = (int64_t)s->len; + st -= 1; /* 1-based → 0-based */ if (st < 0) st = 0; if (st >= sl) continue; - - int64_t ln = length; if (ln < 0 || ln > sl - st) ln = sl - st; if (ln <= 0) continue; - + const char* sp = ray_str_t_ptr(s, c->pool) + st; d->len = (uint32_t)ln; - if (!ray_str_is_inline(s) && !pool) { - ray_release(result); - return NULL; - } - const char* sp = ray_str_t_ptr(s, pool) + st; if (ln <= RAY_STR_INLINE_MAX) { memcpy(d->data, sp, (size_t)ln); - } else if (!ray_str_is_inline(s) && pool_obj) { - if ((uint64_t)s->pool_off + (uint64_t)st > UINT32_MAX) { - ray_release(result); - return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); - } + } else { + /* a pooled result needs a pooled source (an inline source is at + * most 12 bytes, so a longer result never comes from one) */ + if ((uint64_t)s->pool_off + (uint64_t)st > UINT32_MAX) { range_seen = true; d->len = 0; continue; } memcpy(d->prefix, sp, 4); d->pool_off = s->pool_off + (uint32_t)st; - ray_str_t_cache_hash(d, pool); - } else { - ray_release(result); - return NULL; + pooled += (uint64_t)ln; + /* hash32 stays 0: a consumer that needs it computes it once + * (ray_str_t_hash32); hashing every substring here paid a pass + * over the bytes that most consumers never used. */ } } - - return result; + if (null_seen) atomic_store_explicit(&c->any_null, 1, memory_order_relaxed); + if (range_seen) atomic_store_explicit(&c->range_err, 1, memory_order_relaxed); + if (pooled) atomic_fetch_add_explicit(&c->pooled_bytes, pooled, memory_order_relaxed); } -static bool substr_len_at(ray_t* len_v, int64_t row, int64_t* out) { - if (!len_v || !out) return false; - if (ray_vec_may_have_nulls(len_v)) { - if (ray_vec_is_null(len_v, row)) return false; - } - switch (len_v->type) { - case RAY_I64: { - int64_t v = ((const int64_t*)ray_data(len_v))[row]; - if (v == NULL_I64) return false; - *out = v; - return true; - } - case RAY_I32: { - int32_t v = ((const int32_t*)ray_data(len_v))[row]; - if (v == NULL_I32) return false; - *out = (int64_t)v; - return true; - } - default: - return false; - } +/* A view whose bytes are a small share (under an eighth) of the pool it + * points into is worth rebuilding over its own bytes: the copy costs the + * few bytes it keeps, the pool it would otherwise pin costs the rest. A + * larger share stays a view — the parent pool is usually alive anyway (a + * column, a sibling intermediate), and copying most of it would only add + * a second copy for the view's lifetime. */ +bool ray_str_view_should_compact(uint64_t pooled_bytes, int64_t pool_len) { + return pool_len > 0 && pooled_bytes * 8 < (uint64_t)pool_len; } -static ray_t* substr_str_scalar_start_len_view(ray_t* input, - int64_t start, - ray_t* len_v) { - if (!input || input->type != RAY_STR || !len_v) - return NULL; - if (len_v->type != RAY_I64 && len_v->type != RAY_I32) - return NULL; +static ray_t* substr_str_view(ray_t* input, ray_t* start_v, ray_t* len_v) { + if (!input || input->type != RAY_STR) return NULL; int64_t nrows = input->len; - if (len_v->len != nrows) - return NULL; + substr_view_ctx_t ctx; + memset(&ctx, 0, sizeof(ctx)); + if (!substr_arg_init(start_v, nrows, &ctx.start)) return NULL; + if (!substr_arg_init(len_v, nrows, &ctx.len)) return NULL; ray_t* result = ray_vec_new(RAY_STR, nrows); if (!result || RAY_IS_ERR(result)) return result ? result : ray_error("oom", NULL); result->len = nrows; - - const ray_str_t* src = NULL; - const char* pool = NULL; - str_resolve(input, &src, &pool); + str_resolve(input, &ctx.src, &ctx.pool); ray_t* owner = (input->attrs & RAY_ATTR_SLICE) ? input->slice_parent : input; ray_t* pool_obj = owner ? owner->str_pool : NULL; if (pool_obj && !RAY_IS_ERR(pool_obj)) { ray_retain(pool_obj); result->str_pool = pool_obj; } + ctx.dst = (ray_str_t*)ray_data(result); + atomic_store_explicit(&ctx.any_null, 0, memory_order_relaxed); + atomic_store_explicit(&ctx.range_err, 0, memory_order_relaxed); + atomic_store_explicit(&ctx.pooled_bytes, 0, memory_order_relaxed); - ray_str_t* dst = (ray_str_t*)ray_data(result); - int64_t st0 = start - 1; - if (st0 < 0) st0 = 0; - for (int64_t i = 0; i < nrows; i++) { - ray_str_t* d = &dst[i]; - memset(d, 0, sizeof(*d)); - - int64_t ln = 0; - if (!substr_len_at(len_v, i, &ln)) { - ray_vec_set_null(result, i, true); - continue; - } - - const ray_str_t* s = &src[i]; - int64_t sl = (int64_t)s->len; - if (st0 >= sl) continue; - - if (ln < 0 || ln > sl - st0) ln = sl - st0; - if (ln <= 0) continue; - - d->len = (uint32_t)ln; - if (!ray_str_is_inline(s) && !pool) { - ray_release(result); - return NULL; - } - const char* sp = ray_str_t_ptr(s, pool) + st0; - if (ln <= RAY_STR_INLINE_MAX) { - memcpy(d->data, sp, (size_t)ln); - } else if (!ray_str_is_inline(s) && pool_obj) { - if ((uint64_t)s->pool_off + (uint64_t)st0 > UINT32_MAX) { - ray_release(result); - return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); - } - memcpy(d->prefix, sp, 4); - d->pool_off = s->pool_off + (uint32_t)st0; - ray_str_t_cache_hash(d, pool); - } else { - ray_release(result); - return NULL; - } + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, nrows, RAY_PARALLEL_THRESHOLD)) + ray_pool_dispatch(pool, substr_view_fn, &ctx, nrows); + else + substr_view_fn(&ctx, 0, 0, nrows); + if (atomic_load_explicit(&ctx.range_err, memory_order_relaxed)) { + ray_release(result); + return ray_error("range", "substr: pool offset exceeds %lld bytes", (long long)UINT32_MAX); + } + if (atomic_load_explicit(&ctx.any_null, memory_order_relaxed)) + result->attrs |= RAY_ATTR_HAS_NULLS; + /* A result with no pooled descriptor (every substring fits inline) + * has nothing in the parent pool to keep alive. A view that does + * point into it stays a view: the column is alive anyway, and a + * sibling `if` over two such views can pick either side without + * copying (a compacted view would give it two different pools). */ + if (result->str_pool && + atomic_load_explicit(&ctx.pooled_bytes, memory_order_relaxed) == 0) { + ray_release(result->str_pool); + result->str_pool = NULL; } - return result; } -/* True when a whole-column (scalar) start/length argument is null. Such an - * argument applies to every row, so the entire result is null — and a null - * integer scalar is INT64_MIN, which must never reach the `scalar - 1` - * subtraction in the row loop (signed-overflow UB). Only scalars are tested - * here; the per-row vector case is handled inline via ray_vec_is_null. */ -static bool substr_scalar_is_null(ray_t* v) { - if (!v) return false; - if (ray_is_atom(v)) return RAY_ATOM_IS_NULL(v); - if (ray_is_vec(v) && v->len == 1 && ray_vec_may_have_nulls(v)) - return ray_vec_is_null(v, 0); - return false; -} - ray_t* exec_substr(ray_graph_t* g, ray_op_t* op) { ray_t* input = exec_node(g, op_child(g, op, 0)); ray_t* start_v = exec_node(g, op_child(g, op, 1)); @@ -1167,26 +1169,14 @@ ray_t* exec_substr(ray_graph_t* g, ray_op_t* op) { bool is_str = (input->type == RAY_STR); if (is_str) { - int64_t s_const = 0, l_const = 0; - if (substr_scalar_arg(start_v, &s_const) && - substr_scalar_arg(len_v, &l_const)) { - ray_t* view = substr_str_scalar_view(input, s_const, l_const); - if (view) { - ray_release(input); - ray_release(start_v); - ray_release(len_v); - return view; - } - } - if (substr_scalar_arg(start_v, &s_const) && - !ray_is_atom(len_v) && len_v->len == nrows) { - ray_t* view = substr_str_scalar_start_len_view(input, s_const, len_v); - if (view) { - ray_release(input); - ray_release(start_v); - ray_release(len_v); - return view; - } + /* Descriptors over the column's own pool, for every scalar / per-row + * combination of start and length that reads as an integer. */ + ray_t* view = substr_str_view(input, start_v, len_v); + if (view) { + ray_release(input); + ray_release(start_v); + ray_release(len_v); + return view; } } diff --git a/src/ops/system.c b/src/ops/system.c index 6c0a2627f..922b45ba8 100644 --- a/src/ops/system.c +++ b/src/ops/system.c @@ -680,6 +680,7 @@ static bool objsize_push_index_children(ray_objsize_walk_t* w, ray_index_t* ix) OBJSIZE_PUSH(ix->u.chunk_zone.mins); OBJSIZE_PUSH(ix->u.chunk_zone.maxs); OBJSIZE_PUSH(ix->u.chunk_zone.null_bits); + OBJSIZE_PUSH(ix->u.chunk_zone.aggs); break; case RAY_IDX_PART: OBJSIZE_PUSH(ix->u.part.keys); OBJSIZE_PUSH(ix->u.part.starts); @@ -873,7 +874,8 @@ ray_t* ray_mem_ts_fn(ray_t** args, int64_t n) { * pages/pools. Rayforce values are reference-counted, so this is allocator * GC rather than a tracing collector. Variadic to allow `(.sys.gc)`. */ ray_t* ray_gc_fn(ray_t** args, int64_t n) { - (void)args; (void)n; + (void)args; + if (n != 0) return ray_error("arity", ".sys.gc takes no arguments"); ray_heap_gc(); /* Same statement-boundary rule as the REPL: an explicit maintenance * call is also a chance to notice the process has gone quiet. */ diff --git a/src/ops/tblop.c b/src/ops/tblop.c index 375d4ae3b..2e65da07a 100644 --- a/src/ops/tblop.c +++ b/src/ops/tblop.c @@ -1341,6 +1341,9 @@ ray_t* ray_alter_fn(ray_t** args, int64_t n) { return ray_alter_set_cow_fail(original_var, cow_result, idx, val, name_sym); } var = cow_result; + /* A value written in place can break the order a `sorted` marker + * promises (consumers trust it without re-checking). */ + var->attrs &= (uint8_t)~RAY_ATTR_SORTED; /* Validate idx shape + (for the atom case) bounds BEFORE we * touch any state. The accelerator-index drop below would diff --git a/src/store/col.c b/src/store/col.c index 27a6b7516..295a17364 100644 --- a/src/store/col.c +++ b/src/store/col.c @@ -1748,14 +1748,35 @@ static ray_t* col_mmap_impl(const char* path, struct ray_sym_domain_s* dom, * column's single mapping. ray_free reads the full mapping size from the * reserved _idx_pad slot to munmap the whole region (payload + index). */ if (cm.has_index) { - ray_t* idx = ray_index_inline_map((uint8_t*)cm.mapped + cm.index_offset); + ray_t* idx = ray_index_inline_map((uint8_t*)cm.mapped + cm.index_offset, + (int64_t)cm.mapped_size - (int64_t)cm.index_offset); if (idx) { /* NULL = stale index layout generation: load unindexed */ ray_t* r = ray_index_attach_built(&vec, idx); if (r && !RAY_IS_ERR(r)) vec = r; } - /* The munmap size is derived from the column + index at free time - * (ray_free), so no aux slot is needed — leaving str_pool intact on STR. */ /* Attach failure → column loads unindexed; correctness unaffected. */ + /* The mapping is longer than the payload by the index region, and + * ray_free can only size it from the attached index. Once the + * column drops that index — an in-place edit of the loaded column's + * only reference detaches a mapped index without releasing it — or + * when ray_index_inline_map discarded a child that is still in the + * file, the tail would stay mapped for the life of the process. So + * a mapped column that carries an index registers its region under + * its own address with the true length, exactly as str_pool_cow does + * for a string column whose pool pointer stops leading to it; a + * string column already reaches the region through its pool. */ + if (vec->type != RAY_STR && (vec->attrs & RAY_ATTR_HAS_INDEX)) { + ray_file_map_t* m = (ray_file_map_t*)ray_sys_alloc(sizeof(*m)); + if (m) { + m->base = cm.mapped; + m->len = cm.mapped_size; + m->rc = 1; + m->next = NULL; + ray_file_map_register(vec, m); + } + /* No descriptor (oom): the free falls back to sizing from the + * index, the behaviour before this registration existed. */ + } } return vec; diff --git a/src/store/hnsw.c b/src/store/hnsw.c index a350c4335..05ed802d7 100644 --- a/src/store/hnsw.c +++ b/src/store/hnsw.c @@ -908,6 +908,57 @@ bool ray_hnsw_vec_size_valid(int64_t n_nodes, int32_t dim) { return (uint64_t)n_nodes <= SIZE_MAX / sizeof(float) / (uint64_t)dim; } +/* Validate the graph topology before making the loaded index available to + * search. All ids in these files are untrusted: a bad neighbor used to + * reach hnsw_greedy_closest() or hnsw_search_layer() becomes an unchecked + * offset into idx->vectors. The node_ids mapping is equally important — a + * duplicate, missing, or wrong-level entry can make a valid-looking layer + * resolve the wrong neighbor block. + */ +static bool hnsw_persisted_layers_valid(const ray_hnsw_t* idx) { + if (!idx || !idx->node_level || idx->n_nodes <= 0 || idx->n_layers <= 0) + return false; + + for (int64_t id = 0; id < idx->n_nodes; id++) { + if (idx->node_level[id] < 0 || idx->node_level[id] >= idx->n_layers) + return false; + } + + uint8_t* seen = (uint8_t*)ray_sys_alloc((size_t)idx->n_nodes); + if (!seen) return false; + + bool valid = true; + for (int32_t l = 0; l < idx->n_layers && valid; l++) { + const ray_hnsw_layer_t* layer = &idx->layers[l]; + int64_t expected = 0; + for (int64_t id = 0; id < idx->n_nodes; id++) + if (idx->node_level[id] >= l) expected++; + if (layer->n_nodes != expected || !layer->node_ids || !layer->neighbors) { + valid = false; + break; + } + + memset(seen, 0, (size_t)idx->n_nodes); + for (int64_t i = 0; i < layer->n_nodes && valid; i++) { + int64_t id = layer->node_ids[i]; + if (id < 0 || id >= idx->n_nodes || idx->node_level[id] < l || seen[id]) { + valid = false; + break; + } + seen[id] = 1; + } + + size_t nb_count = (size_t)layer->n_nodes * (size_t)layer->M_max; + for (size_t i = 0; i < nb_count && valid; i++) { + int64_t id = layer->neighbors[i]; + if (id != -1 && (id < 0 || id >= idx->n_nodes)) valid = false; + } + } + + ray_sys_free(seen); + return valid; +} + static ray_hnsw_t* hnsw_load_impl(const char* dir, bool use_mmap) { if (!dir) return NULL; (void)use_mmap; /* mmap optimization deferred — both paths read into memory */ @@ -1001,6 +1052,11 @@ static ray_hnsw_t* hnsw_load_impl(const char* dir, bool use_mmap) { fclose(f); } + if (!hnsw_persisted_layers_valid(idx)) { + ray_hnsw_free(idx); + return NULL; + } + /* Read vectors */ snprintf(path, sizeof(path), "%s/hnsw_vectors.bin", dir); f = fopen(path, "rb"); diff --git a/src/store/splay.c b/src/store/splay.c index d59bd0b2e..4fe29bc10 100644 --- a/src/store/splay.c +++ b/src/store/splay.c @@ -23,6 +23,8 @@ #include "splay.h" #include "core/runtime.h" +#include "core/pool.h" +#include "mem/sys.h" #include "store/col.h" #include "store/fileio.h" #include "store/serde.h" @@ -512,12 +514,16 @@ ray_t* ray_splay_load(const char* dir, const char* sym_path) { * rewrite), so a later mmap load gets the same block-skip an in-memory build * has. Best-effort and idempotent-ish: ray_col_append_index refuses a file * that is not exactly payload-sized (already indexed), so re-runs are no-ops. */ -void ray_splay_build_indexes(const char* dir, ray_t* tbl) { - if (!dir || !tbl || RAY_IS_ERR(tbl) || tbl->type != RAY_TABLE) return; - int64_t nc = ray_table_ncols(tbl); - for (int64_t c = 0; c < nc; c++) { +/* Index one column of a just-written splayed table (see + * ray_splay_build_indexes). Columns are independent — each reads and + * appends to its own file — so the caller runs one task per column. */ +/* deferred: when non-NULL and the column qualifies for a hash index, the + * computed zone is stored there instead of persisted and nothing is written; + * splay_persist_hash_or_zone finishes the column. */ +static void splay_build_index_col(const char* dir, ray_t* tbl, int64_t c, ray_t** deferred) { + { ray_t* col = ray_table_get_col_idx(tbl, c); - if (!col || RAY_IS_ERR(col)) continue; + if (!col || RAY_IS_ERR(col)) return; /* Explicit SYM index: a SYM column carrying a grouped (hash) index in * memory gets a hash index persisted inline — regardless of length (the @@ -557,7 +563,7 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { } } } - continue; + return; } /* Explicit STR index: a grouped / unique hash on a STR column is keyed @@ -575,10 +581,10 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { (void)ray_col_append_index(path, ray_index_payload(col->index), col->len, RAY_STR); } - continue; + return; } - if (col->len < (1 << 16)) continue; + if (col->len < (1 << 16)) return; /* STR columns get a dictionary (group on int codes); numeric/temporal * get the per-chunk min/max for block-skip. @@ -593,11 +599,18 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { ray_t* idx = (col->type == RAY_STR) ? ray_index_dict_compute(col) : ray_index_chunk_zone_compute(col, 16); - if (!idx || RAY_IS_ERR(idx)) { if (idx) ray_error_free(idx); continue; } + if (!idx || RAY_IS_ERR(idx)) { if (idx) ray_error_free(idx); return; } if (col->type != RAY_STR && ray_csv_hash_upgrade_check(col->type, col->len, ray_index_payload(idx))) { + if (deferred) { + /* Inside a per-column task: the hash build has its own + * parallel path that needs the pool, so hand the zone + * back and let the caller build the hash afterwards. */ + *deferred = idx; + return; + } ray_t* hi = ray_idx_hash_fn(col); if (hi && !RAY_IS_ERR(hi) && (hi->attrs & RAY_ATTR_HAS_INDEX)) { ray_release(idx); /* zone sacrificed for the hash */ @@ -612,7 +625,7 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { ray_index_payload(hi->index), hi->len, hi->type); } ray_release(hi); - continue; + return; } if (hi) { if (RAY_IS_ERR(hi)) ray_error_free(hi); else ray_release(hi); } /* Hash build failed — fall through and persist the zone. */ @@ -631,6 +644,77 @@ void ray_splay_build_indexes(const char* dir, ray_t* tbl) { } } +/* Persist a deferred column: its hash index when the build succeeded + * (hashed[c]), else the zone it was computed with (deferred[c]). */ +typedef struct { const char* dir; ray_t* tbl; ray_t** deferred; ray_t** hashed; } splay_index_ctx_t; + +static void splay_persist_deferred(splay_index_ctx_t* x, int64_t c) { + ray_t* col = ray_table_get_col_idx(x->tbl, c); + ray_t* nstr = ray_sym_str(ray_table_col_name(x->tbl, c)); + char path[1100]; + int n = (nstr && !RAY_IS_ERR(nstr)) + ? snprintf(path, sizeof(path), "%s/%.*s", x->dir, (int)ray_str_len(nstr), ray_str_ptr(nstr)) + : -1; + bool have_path = n > 0 && n < (int)sizeof(path); + ray_t* hi = x->hashed[c]; + if (hi) { + if (have_path) + (void)ray_col_append_index(path, ray_index_payload(hi->index), hi->len, hi->type); + ray_release(hi); + } else if (have_path) { + (void)ray_col_append_index(path, ray_index_payload(x->deferred[c]), col->len, col->type); + } + ray_release(x->deferred[c]); + x->hashed[c] = NULL; x->deferred[c] = NULL; +} + +static void splay_persist_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + splay_index_ctx_t* x = (splay_index_ctx_t*)raw; + if (x->deferred[start]) splay_persist_deferred(x, start); +} +static void splay_build_index_task(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; (void)end; + splay_index_ctx_t* c = (splay_index_ctx_t*)raw; + splay_build_index_col(c->dir, c->tbl, start, &c->deferred[start]); +} + +void ray_splay_build_indexes(const char* dir, ray_t* tbl) { + + if (!dir || !tbl || RAY_IS_ERR(tbl) || tbl->type != RAY_TABLE) return; + int64_t nc = ray_table_ncols(tbl); + if (nc <= 0) return; + /* One task per column: the zone / dictionary / hash builds are per-row + * scans of each column and used to run one after another on the + * calling thread — the longest serial stretch of a CSV → splayed load. */ + ray_pool_t* pool = ray_pool_get(); + if (ray_pool_par_dispatch_ok(pool, nc, 2)) { + /* Zones / dictionaries per column in parallel; the columns that + * qualify for a hash index come back deferred and are built one + * after another on this thread, each hash build parallel inside. */ + ray_t** deferred = (ray_t**)ray_sys_alloc((size_t)nc * 2 * sizeof(ray_t*)); + if (!deferred) { + for (int64_t c = 0; c < nc; c++) splay_build_index_col(dir, tbl, c, NULL); + return; + } + memset(deferred, 0, (size_t)nc * 2 * sizeof(ray_t*)); + splay_index_ctx_t ctx = { .dir = dir, .tbl = tbl, .deferred = deferred, .hashed = deferred + nc }; + ray_pool_dispatch_n(pool, splay_build_index_task, &ctx, (uint32_t)nc); + /* Hash builds one after another (each parallel inside), then the + * writes of all deferred columns together. */ + for (int64_t c = 0; c < nc; c++) { + if (!deferred[c]) continue; + ray_t* hi = ray_idx_hash_fn(ray_table_get_col_idx(tbl, c)); + if (hi && !RAY_IS_ERR(hi) && (hi->attrs & RAY_ATTR_HAS_INDEX)) ctx.hashed[c] = hi; + else if (hi) { if (RAY_IS_ERR(hi)) ray_error_free(hi); else ray_release(hi); } + } + ray_pool_dispatch_n(pool, splay_persist_task, &ctx, (uint32_t)nc); + ray_sys_free(deferred); + } else { + for (int64_t c = 0; c < nc; c++) splay_build_index_col(dir, tbl, c, NULL); + } +} + ray_t* ray_read_splayed(const char* dir, const char* sym_path) { return splay_load_impl(dir, sym_path, true); } diff --git a/src/table/domain.c b/src/table/domain.c index 0e9f58697..a7d13f77a 100644 --- a/src/table/domain.c +++ b/src/table/domain.c @@ -63,6 +63,8 @@ #include "mem/arena.h" /* ray_arena_t / ray_arena_str — domain atom storage */ #include "store/fileio.h" /* flock + tmp/rename protocol for flush */ #include "ops/hash.h" /* ray_hash_bytes (same hash family as g_sym) */ +#include "core/pool.h" /* batch intern: parallel read-only probe */ +#include "sym.h" /* ray_sym_intern_prehashed (runtime fallback) */ #include #include #include @@ -177,6 +179,10 @@ struct ray_sym_domain_s { * inserts incrementally; growth rebuilds). Guarded by g_dom_lock. */ uint64_t* buckets; uint64_t bucket_mask; /* cap - 1; 0 = not built yet */ + /* Batch interns currently probing a snapshot of `buckets` outside the + * lock (guarded by g_dom_lock). While non-zero a replaced table is + * retired instead of freed. */ + int32_t batch_inflight; dom_retired_t* retired; /* replaced atom arrays + old LUTs */ @@ -292,6 +298,15 @@ static bool dom_retire(ray_sym_domain_t* d, void* p) { d->retired = r; return true; } + +/* Drop a replaced reverse-index table: freed at once unless a batch intern + * is probing a snapshot outside the lock, then retired with the domain. */ +static bool dom_drop_buckets_locked(ray_sym_domain_t* d, uint64_t* old) { + if (!old) return true; + if (d->batch_inflight == 0) { ray_sys_free(old); return true; } + return dom_retire(d, old); +} + /* Publish (map, offsets, count) as the domain's raw snapshot; the previous * record is retired. Called with the file's current image in place. */ static bool dom_publish_raw_snap(ray_sym_domain_t* d) { @@ -516,7 +531,17 @@ static bool dom_extend_from_file_locked(ray_sym_domain_t* d, size_t st_size) { * the grown count (OOB reads for lock-free consumers) — the * same corner dom_append_locked hits; mirror its loud abort * (the count is already published, there is no clean undo). */ - if (d->buckets) { ray_sys_free(d->buckets); d->buckets = NULL; d->bucket_mask = 0; } + if (d->buckets) { + /* A batch intern may be probing a snapshot of this + * table outside the lock. */ + if (!dom_drop_buckets_locked(d, d->buckets)) { + fprintf(stderr, "rayforce: sym domain '%s': OOM retiring " + "reverse index after external extend\n", + d->path ? d->path : "?"); + abort(); + } + d->buckets = NULL; d->bucket_mask = 0; + } int64_t* lut = atomic_load_explicit(&d->runtime_lut, memory_order_relaxed); if (lut) { if (!dom_retire(d, lut)) { @@ -763,15 +788,33 @@ static bool dom_build_index_locked(ray_sym_domain_t* d, int64_t extra) { if (!buckets) return false; uint64_t mask = cap - 1; - for (int64_t i = 0; i < count; i++) { - ray_t* a = dom_atom_at_locked(d, i); - if (!a) { ray_sys_free(buckets); return false; } - uint32_t h = (uint32_t)ray_hash_bytes(ray_str_ptr(a), ray_str_len(a)); - uint64_t slot = h & mask; - while (buckets[slot] != 0) slot = (slot + 1) & mask; - buckets[slot] = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)i + 1); + if (d->buckets) { + /* Growth: the old table covers every published entry and carries + * the hashes — re-slot its entries, no string hashing. */ + uint64_t old_cap = d->bucket_mask + 1; + for (uint64_t i = 0; i < old_cap; i++) { + uint64_t e = d->buckets[i]; + if (e == 0) continue; + uint64_t slot = (uint32_t)(e >> 32) & mask; + while (buckets[slot] != 0) slot = (slot + 1) & mask; + buckets[slot] = e; + } + } else { + for (int64_t i = 0; i < count; i++) { + ray_t* a = dom_atom_at_locked(d, i); + if (!a) { ray_sys_free(buckets); return false; } + uint32_t h = (uint32_t)ray_hash_bytes(ray_str_ptr(a), ray_str_len(a)); + uint64_t slot = h & mask; + while (buckets[slot] != 0) slot = (slot + 1) & mask; + buckets[slot] = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)i + 1); + } + } + /* A batch intern may be probing a snapshot of the old table outside + * the lock (see ray_sym_domain_intern_batch). */ + if (d->buckets && !dom_drop_buckets_locked(d, d->buckets)) { + ray_sys_free(buckets); + return false; } - ray_sys_free(d->buckets); d->buckets = buckets; d->bucket_mask = mask; return true; @@ -919,6 +962,345 @@ int64_t ray_sym_domain_intern(ray_sym_domain_t* dom, const char* str, size_t len return pos; } +/* ---- batch intern ---------------------------------------------------------- */ + +/* Read-only probe over a snapshot of the reverse index taken under the + * lock. Everything the snapshot points at outlives it: replaced bucket + * tables and atom arrays are retired, not freed, and the file prefix is + * read through the pinned raw snapshot. Entries appended after the + * snapshot (pos >= count) are ignored here and resolved under the lock. */ +typedef struct { + const uint64_t* buckets; + uint64_t mask; + ray_t* const* atoms; + int64_t count; + ray_sym_domain_raw_t raw; /* raw.count == 0: no file prefix */ + const char* const* strs; + const size_t* lens; + const uint32_t* hashes; + int64_t* out_pos; + /* misses, grouped by hash partition (hash >> part_shift) */ + int64_t* miss; /* [n_miss] batch indices */ + int64_t* uniq; /* [n_miss] first occurrences, per partition segment */ + int64_t* part_off; /* [n_part + 1] */ + int64_t* uniq_n; /* [n_part] distinct misses per partition */ + int64_t* bytes_p; /* [n_part] arena bytes the partition's atoms need */ + void** region; /* [n_part] arena region per partition */ + ray_t** atoms_w; /* current atom array (fill target) */ + /* Set when a probe met an entry it could not compare (no atom, no + * raw bytes): the misses are then resolved under the lock instead. */ + _Atomic(bool) unsure; + int part_shift; + _Atomic(bool) oom; +} dom_batch_ctx_t; + +static void dom_batch_probe_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + const dom_batch_ctx_t* b = (const dom_batch_ctx_t*)raw; + for (int64_t i = start; i < end; i++) { + uint32_t h = b->hashes[i]; + size_t len = b->lens[i]; + const char* s = b->strs[i]; + uint64_t slot = h & b->mask; + int64_t found = -1; + for (;;) { + uint64_t e = atomic_load_explicit((_Atomic(uint64_t)*)&b->buckets[slot], + memory_order_relaxed); + if (e == 0) break; + if ((uint32_t)(e >> 32) == h) { + int64_t pos = (int64_t)(uint32_t)e - 1; + if (pos < b->count) { + const char* p = NULL; + size_t l = 0; + ray_t* a = atomic_load_explicit((_Atomic(ray_t*)*)&b->atoms[pos], + memory_order_acquire); + if (a) { p = ray_str_ptr(a); l = ray_str_len(a); } + else if (pos < b->raw.count) p = ray_sym_domain_raw_str(&b->raw, pos, &l); + else atomic_store_explicit((_Atomic(bool)*)&b->unsure, true, memory_order_relaxed); + if (p && l == len && (len == 0 || memcmp(p, s, len) == 0)) { + found = pos; + break; + } + } + } + slot = (slot + 1) & b->mask; + } + b->out_pos[i] = found; + } +} + +/* Dedupe the misses of one hash partition among themselves: the first + * occurrence stays a miss (out_pos -1) and is listed in the partition's + * segment of `uniq`; a repeat records its representative as -(rep + 2). + * Partitions are disjoint by hash, so no two tasks ever see the same + * string. */ +static void dom_batch_dedupe_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dom_batch_ctx_t* b = (dom_batch_ctx_t*)raw; + for (int64_t p = start; p < end; p++) { + int64_t lo = b->part_off[p], hi = b->part_off[p + 1]; + int64_t cnt = hi - lo; + if (cnt == 0) { b->uniq_n[p] = 0; continue; } + uint64_t cap = 16; + while ((uint64_t)cnt * 2 > cap) cap <<= 1; + int64_t* tab = (int64_t*)ray_sys_alloc((size_t)cap * sizeof(int64_t)); + if (!tab) { atomic_store_explicit(&b->oom, true, memory_order_relaxed); b->uniq_n[p] = 0; continue; } + memset(tab, 0xff, (size_t)cap * sizeof(int64_t)); /* -1 = empty */ + uint64_t mask = cap - 1; + int64_t k = 0; + int64_t bytes = 0; + for (int64_t j = lo; j < hi; j++) { + int64_t i = b->miss[j]; + uint32_t h = b->hashes[i]; + uint64_t slot = h & mask; + int64_t rep = -1; + while (tab[slot] >= 0) { + int64_t r = tab[slot]; + if (b->hashes[r] == h && b->lens[r] == b->lens[i] && + (b->lens[i] == 0 || memcmp(b->strs[r], b->strs[i], b->lens[i]) == 0)) { + rep = r; + break; + } + slot = (slot + 1) & mask; + } + if (rep >= 0) { + b->out_pos[i] = -(rep + 2); + } else { + tab[slot] = i; + b->uniq[lo + k++] = i; + bytes += (int64_t)ray_arena_str_bytes(b->lens[i]); + } + } + b->uniq_n[p] = k; + b->bytes_p[p] = bytes; + ray_sys_free(tab); + } +} + +/* Append the partition's distinct strings: build the atoms in the + * partition's arena region at the positions the caller assigned (batch + * order), publish them in the reverse index (CAS on the empty slot keeps + * concurrent partitions from claiming one slot twice) and resolve the + * partition's repeats. Runs under the domain lock; the count is + * published by the caller once every partition is done. */ +static void dom_batch_insert_fn(void* raw, uint32_t wid, int64_t start, int64_t end) { + (void)wid; + dom_batch_ctx_t* b = (dom_batch_ctx_t*)raw; + for (int64_t p = start; p < end; p++) { + int64_t lo = b->part_off[p]; + int64_t hi = lo + b->uniq_n[p]; + char* at = (char*)b->region[p]; + for (int64_t j = lo; j < hi; j++) { + int64_t i = b->uniq[j]; + int64_t pos = b->out_pos[i]; /* assigned in batch order */ + ray_t* s = ray_arena_str_at(at, b->strs[i], b->lens[i]); + at += ray_arena_str_bytes(b->lens[i]); + b->atoms_w[pos] = s; + uint32_t h = b->hashes[i]; + uint64_t e = ((uint64_t)h << 32) | ((uint64_t)(uint32_t)pos + 1); + uint64_t slot = h & b->mask; + for (;;) { + uint64_t cur = 0; + if (atomic_compare_exchange_strong_explicit((_Atomic(uint64_t)*)&b->buckets[slot], + &cur, e, memory_order_relaxed, memory_order_relaxed)) + break; + slot = (slot + 1) & b->mask; + } + } + for (int64_t j = lo; j < b->part_off[p + 1]; j++) { + int64_t i = b->miss[j]; + int64_t v = b->out_pos[i]; + if (v < -1) b->out_pos[i] = b->out_pos[-(v + 2)]; + } + } +} + +/* Serial fallback for the misses: find-or-append one by one under the lock + * (the domain changed under us, or the parallel path ran out of memory). */ +static bool dom_batch_append_serial_locked(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos) { + for (int64_t i = 0; i < n; i++) { + if (out_pos[i] >= 0) continue; + int64_t pos = dom_probe_locked(dom, hashes[i], strs[i], lens[i]); + if (pos < 0) pos = dom_append_locked(dom, hashes[i], strs[i], lens[i]); + if (pos < 0) return false; + out_pos[i] = pos; + } + return true; +} + +bool ray_sym_domain_intern_batch(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos) { + if (!dom || n < 0) return false; + if (n == 0) return true; + if (dom->kind == DOM_RUNTIME) { + for (int64_t i = 0; i < n; i++) { + int64_t id = ray_sym_intern_prehashed(hashes[i], strs[i], lens[i]); + if (id < 0) return false; + out_pos[i] = id; + } + return true; + } + + dom_batch_ctx_t b; + memset(&b, 0, sizeof(b)); + + dom_lock(); + int64_t count = atomic_load_explicit(&dom->count, memory_order_relaxed); + /* Headroom for the whole batch up front so no rebuild happens while + * the batch is in flight. */ + if (!dom->buckets || + (double)(count + n + 1) > 0.7 * (double)(dom->bucket_mask + 1)) { + if (!dom_build_index_locked(dom, n + 1)) { dom_unlock(); return false; } + } + if (count == 0) { + uint32_t h0 = (uint32_t)ray_hash_bytes("", 0); + if (dom_append_locked(dom, h0, "", 0) != 0) { dom_unlock(); return false; } + } + b.buckets = dom->buckets; + b.mask = dom->bucket_mask; + b.atoms = atomic_load_explicit(&dom->atoms, memory_order_acquire); + b.count = atomic_load_explicit(&dom->count, memory_order_acquire); + dom->batch_inflight++; + dom_unlock(); + + if (!ray_sym_domain_raw_pin(dom, &b.raw)) b.raw.count = 0; + b.strs = strs; b.lens = lens; b.hashes = hashes; b.out_pos = out_pos; + + ray_pool_t* pool = ray_pool_get(); + bool par = ray_pool_par_dispatch_ok(pool, n, 4096); + if (par) ray_pool_dispatch(pool, dom_batch_probe_fn, &b, n); + else dom_batch_probe_fn(&b, 0, 0, n); + + /* Misses, grouped by hash partition. */ + int n_part = 1; + if (par) { + int64_t want = (int64_t)ray_pool_total_workers(pool) * 4; + while (n_part < want && n_part < 1024) n_part <<= 1; + } + b.part_shift = 31; + for (int p = n_part; p > 2; p >>= 1) b.part_shift--; + if (n_part == 1) n_part = 2; /* keep the shift below the type width */ + int64_t n_miss = 0; + for (int64_t i = 0; i < n; i++) n_miss += (out_pos[i] < 0); + if (n_miss == 0) { + dom_lock(); + dom->batch_inflight--; + dom_unlock(); + return true; + } + + b.miss = (int64_t*)ray_sys_alloc((size_t)n_miss * sizeof(int64_t)); + b.uniq = (int64_t*)ray_sys_alloc((size_t)n_miss * sizeof(int64_t)); + b.part_off = (int64_t*)ray_sys_alloc((size_t)(n_part + 1) * sizeof(int64_t)); + b.uniq_n = (int64_t*)ray_sys_alloc((size_t)n_part * sizeof(int64_t)); + b.bytes_p = (int64_t*)ray_sys_alloc((size_t)n_part * sizeof(int64_t)); + b.region = (void**)ray_sys_alloc((size_t)n_part * sizeof(void*)); + bool ok = b.miss && b.uniq && b.part_off && b.uniq_n && b.bytes_p && b.region; + if (ok) { + memset(b.part_off, 0, (size_t)(n_part + 1) * sizeof(int64_t)); + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < 0) b.part_off[(hashes[i] >> b.part_shift) + 1]++; + for (int p = 0; p < n_part; p++) b.part_off[p + 1] += b.part_off[p]; + { + int64_t* fill = b.uniq_n; /* scratch cursor per partition */ + memcpy(fill, b.part_off, (size_t)n_part * sizeof(int64_t)); + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < 0) b.miss[fill[hashes[i] >> b.part_shift]++] = i; + } + if (par) ray_pool_dispatch_n(pool, dom_batch_dedupe_fn, &b, (uint32_t)n_part); + else dom_batch_dedupe_fn(&b, 0, 0, n_part); + if (atomic_load_explicit(&b.oom, memory_order_relaxed)) { + /* Undo the repeat marks; the serial path resolves everything. */ + for (int64_t i = 0; i < n; i++) if (out_pos[i] < -1) out_pos[i] = -1; + ok = false; + } + } + + dom_lock(); + dom->batch_inflight--; /* no probe outside the lock past this point */ + bool unchanged = ok && dom->buckets == b.buckets && + atomic_load_explicit(&dom->count, memory_order_relaxed) == b.count && + !atomic_load_explicit(&b.unsure, memory_order_relaxed); + if (unchanged) { + int64_t total = 0; + for (int p = 0; p < n_part; p++) total += b.uniq_n[p]; + int64_t base = b.count; + ok = base + total < (int64_t)UINT32_MAX; + /* Atom array: grow once by replacement (lock-free readers may hold + * the old pointer). */ + if (ok && base + total > dom->atoms_cap) { + int64_t ncap = dom->atoms_cap < 8 ? 8 : dom->atoms_cap; + while (ncap < base + total) ncap *= 2; + ray_t** narr = (ray_t**)ray_sys_alloc((size_t)ncap * sizeof(ray_t*)); + if (!narr) ok = false; + else { + ray_t** old = atomic_load_explicit(&dom->atoms, memory_order_relaxed); + if (base > 0) memcpy(narr, old, (size_t)base * sizeof(ray_t*)); + if (!dom_retire(dom, old)) { ray_sys_free(narr); ok = false; } + else { + atomic_store_explicit(&dom->atoms, narr, memory_order_release); + dom->atoms_cap = ncap; + } + } + } + if (ok) { + /* One arena region per partition; positions in partition order. + * Nothing is published until every partition has built its + * atoms, so a failed reservation costs only arena space. */ + /* New strings take positions in batch order (first occurrence), + * so the symfile does not depend on how the batch was split. */ + int64_t pos = base; + for (int64_t i = 0; i < n; i++) + if (out_pos[i] == -1) out_pos[i] = pos++; + for (int p = 0; p < n_part; p++) { + b.region[p] = NULL; + if (b.bytes_p[p] > 0) { + b.region[p] = ray_arena_alloc_raw(dom->arena, (size_t)b.bytes_p[p]); + if (!b.region[p]) { ok = false; break; } + } + } + if (ok) { + b.atoms_w = atomic_load_explicit(&dom->atoms, memory_order_relaxed); + if (par) ray_pool_dispatch_n(pool, dom_batch_insert_fn, &b, (uint32_t)n_part); + else dom_batch_insert_fn(&b, 0, 0, n_part); + atomic_store_explicit(&dom->count, pos, memory_order_release); + /* Same invalidation as dom_append_locked: the runtime LUT + * no longer covers the vocabulary. */ + int64_t* lut = atomic_load_explicit(&dom->runtime_lut, memory_order_relaxed); + if (lut) { + if (!dom_retire(dom, lut)) { + fprintf(stderr, "rayforce: sym domain '%s': OOM retiring runtime " + "LUT after batch append\n", dom->path ? dom->path : "?"); + abort(); + } + atomic_store_explicit(&dom->runtime_lut, NULL, memory_order_release); + } + } else { + /* Nothing published: undo the assigned positions and the + * repeat marks; the serial path resolves them again. */ + for (int64_t i = 0; i < n; i++) + if (out_pos[i] < -1 || out_pos[i] >= base) out_pos[i] = -1; + } + } + } else { + for (int64_t i = 0; i < n; i++) if (out_pos[i] < -1) out_pos[i] = -1; + } + if (!unchanged || !ok) + ok = dom_batch_append_serial_locked(dom, n, strs, lens, hashes, out_pos); + dom_unlock(); + + ray_sys_free(b.miss); + ray_sys_free(b.uniq); + ray_sys_free(b.part_off); + ray_sys_free(b.uniq_n); + ray_sys_free(b.bytes_p); + ray_sys_free(b.region); + return ok; +} + int64_t ray_sym_domain_count(ray_sym_domain_t* dom) { if (!dom) return 0; if (dom->kind == DOM_RUNTIME) return (int64_t)ray_sym_count(); @@ -1013,18 +1395,37 @@ ray_err_t ray_sym_domain_flush(ray_sym_domain_t* dom, bool durable) { if (fwrite(&magic, 4, 1, f) != 1 || fwrite(&count, 8, 1, f) != 1) err = RAY_ERR_IO; written_size = 12; + /* Records are packed into a large buffer first: two stdio calls per + * entry dominated the flush of a big vocabulary. */ + enum { FLUSH_BUF = 1u << 20 }; + uint8_t* wb = (uint8_t*)ray_sys_alloc(FLUSH_BUF); + size_t wn = 0; + if (!wb) err = RAY_ERR_OOM; for (int64_t i = 0; err == RAY_OK && i < count; i++) { ray_t* s = atoms[i]; size_t slen = ray_str_len(s); if (slen > UINT32_MAX) { err = RAY_ERR_RANGE; break; } uint32_t len32 = (uint32_t)slen; - if (fwrite(&len32, 4, 1, f) != 1 || - (slen > 0 && fwrite(ray_str_ptr(s), 1, slen, f) != slen)) { - err = RAY_ERR_IO; - break; + if (wn + 4 + slen > FLUSH_BUF) { + if (wn && fwrite(wb, 1, wn, f) != wn) { err = RAY_ERR_IO; break; } + wn = 0; + } + if (4 + slen > FLUSH_BUF) { + /* Oversized record: straight through. */ + if (fwrite(&len32, 4, 1, f) != 1 || + fwrite(ray_str_ptr(s), 1, slen, f) != slen) { + err = RAY_ERR_IO; + break; + } + } else { + memcpy(wb + wn, &len32, 4); + if (slen) memcpy(wb + wn + 4, ray_str_ptr(s), slen); + wn += 4 + slen; } written_size += 4 + slen; } + if (err == RAY_OK && wn && fwrite(wb, 1, wn, f) != wn) err = RAY_ERR_IO; + ray_sys_free(wb); if (fclose(f) != 0 && err == RAY_OK) err = RAY_ERR_IO; } if (err != RAY_OK) goto fail_tmp; diff --git a/src/table/domain.h b/src/table/domain.h index e5c0f9e6b..b9ce66105 100644 --- a/src/table/domain.h +++ b/src/table/domain.h @@ -164,6 +164,17 @@ const int64_t* ray_sym_domain_runtime_lut(ray_sym_domain_t* dom); * the shared object immediately; ray_sym_domain_flush persists them. */ int64_t ray_sym_domain_intern(ray_sym_domain_t* dom, const char* str, size_t len); +/* Batch find-or-append of n strings (hashes[i] = (uint32_t)ray_hash_bytes + * of strs[i]); out_pos[i] receives the position. Repeats inside the + * batch resolve to one position. FILE: the lookup of the existing + * vocabulary runs in parallel on the pool; new strings are appended in + * batch order (a string's first occurrence), whatever the worker count. RUNTIME: one ray_sym_intern per entry. + * Returns false on OOM (out_pos then holds -1 for the entries not + * resolved). */ +bool ray_sym_domain_intern_batch(ray_sym_domain_t* dom, int64_t n, + const char* const* strs, const size_t* lens, + const uint32_t* hashes, int64_t* out_pos); + /* Number of entries in the domain. */ int64_t ray_sym_domain_count(ray_sym_domain_t* dom); diff --git a/src/table/sym.c b/src/table/sym.c index 2843827de..7c55cdf1c 100644 --- a/src/table/sym.c +++ b/src/table/sym.c @@ -824,6 +824,27 @@ int64_t ray_sym_intern_batch(const uint32_t* hashes, const char* const* strs, return 0; } +/* Intern n pre-hashed, already-deduplicated strings under one lock without + * caching dotted segments: for VALUES (a derived group key's strings — hosts, + * URLs), which are not namespace paths. Such a symbol is not dotted until + * the same string is interned as a name (the probe-hit path of + * sym_intern_nolock caches the segments then) or ray_sym_rebuild_segments + * runs — the contract of ray_sym_intern_no_split. Same ids as + * ray_sym_intern_batch. */ +int64_t ray_sym_intern_batch_no_split(const uint32_t* hashes, const char* const* strs, + const size_t* lens, int64_t n, int64_t* out_ids) { + if (!atomic_load_explicit(&g_sym_inited, memory_order_acquire)) return -1; + if (n <= 0) return 0; + sym_lock(); + for (int64_t i = 0; i < n; i++) { + int64_t id = sym_intern_nolock_noseg(hashes[i], strs[i], lens[i]); + if (id < 0) { sym_unlock(); return -1; } + out_ids[i] = id; + } + sym_unlock(); + return 0; +} + /* -------------------------------------------------------------------------- * ray_sym_intern_no_split — persistence-only bulk intern * -------------------------------------------------------------------------- */ diff --git a/src/table/sym.h b/src/table/sym.h index 15418a179..7ed5e0bf7 100644 --- a/src/table/sym.h +++ b/src/table/sym.h @@ -129,6 +129,9 @@ int64_t ray_sym_intern_prehashed(uint32_t hash, const char* str, size_t len); * any intern failed, in which case out_ids is only partially filled. */ int64_t ray_sym_intern_batch(const uint32_t* hashes, const char* const* strs, const size_t* lens, int64_t n, int64_t* out_ids); +/* The same for VALUES: no dotted-segment caching (see sym.c). */ +int64_t ray_sym_intern_batch_no_split(const uint32_t* hashes, const char* const* strs, + const size_t* lens, int64_t n, int64_t* out_ids); /* Monotonic counter bumped by ray_sym_init and ray_sym_destroy. A cache * keyed on sym ids is valid only while this is unchanged: ids are stable diff --git a/src/vec/vec.c b/src/vec/vec.c index f5be8d985..9a342d571 100644 --- a/src/vec/vec.c +++ b/src/vec/vec.c @@ -102,7 +102,12 @@ static inline void vec_drop_index_inplace(ray_t* v) { if (!(v->attrs & RAY_ATTR_HAS_INDEX)) return; ray_t* idx = v->index; ray_index_t* ix = ray_index_payload(idx); - bool shared = ray_atomic_load(&idx->rc) > 1; + /* A mapped index (mmod 1) rides the column file's mapping: copies of + * the column borrow it without a reference, and the mapping's owner + * unmaps it. Dropping it from a vector only detaches it — the + * snapshot stays for the other holders and nothing is released. */ + bool mapped = idx->mmod == 1; + bool shared = mapped || ray_atomic_load(&idx->rc) > 1; if (shared) { /* Take our own retained references to the saved-pointer slots @@ -120,7 +125,7 @@ static inline void vec_drop_index_inplace(ray_t* v) { ix->saved_attrs = 0; } v->attrs &= (uint8_t)~RAY_ATTR_HAS_INDEX; - ray_release(idx); + if (!mapped) ray_release(idx); } /* -------------------------------------------------------------------------- @@ -515,22 +520,11 @@ ray_t* ray_vec_concat(ray_t* a, ray_t* b) { } } - /* Propagate null bitmaps from a and b. - * Slices don't carry RAY_ATTR_HAS_NULLS — check RAY_ATTR_SLICE too. */ - if (ray_vec_may_have_nulls(a) || - ray_vec_may_have_nulls(b)) { - for (int64_t i = 0; i < a->len; i++) { - if (ray_vec_is_null((ray_t*)a, i)) { - ray_err_t err = ray_vec_set_null_checked(result, i, true); - if (err != RAY_OK) { ray_release(result); return ray_error(ray_err_code_str(err), NULL); } - } - } - for (int64_t i = 0; i < b->len; i++) { - if (ray_vec_is_null((ray_t*)b, i)) { - ray_err_t err = ray_vec_set_null_checked(result, a->len + i, true); - if (err != RAY_OK) { ray_release(result); return ray_error(ray_err_code_str(err), NULL); } - } - } + /* Canonical empty payloads were copied above; only the null hint + * remains. Scan payloads (including slices), not input hint bits. */ + if (ray_vec_text_has_nulls(a) || ray_vec_text_has_nulls(b)) { + vec_drop_index_inplace(result); + result->attrs |= RAY_ATTR_HAS_NULLS; } return result; @@ -650,10 +644,15 @@ ray_t* ray_vec_concat(ray_t* a, ray_t* b) { (size_t)b->len * esz); } - /* Propagate null bitmaps from a and b. - * Slices don't carry RAY_ATTR_HAS_NULLS — check RAY_ATTR_SLICE too. */ - if (ray_vec_may_have_nulls(a) || - ray_vec_may_have_nulls(b)) { + /* SYM's zero id survives copying, widening and domain translation. + * Numeric sentinels retain their existing propagation path below. */ + if (result->type == RAY_SYM) { + if (ray_vec_text_has_nulls(a) || ray_vec_text_has_nulls(b)) { + vec_drop_index_inplace(result); + result->attrs |= RAY_ATTR_HAS_NULLS; + } + } else if (ray_vec_may_have_nulls(a) || + ray_vec_may_have_nulls(b)) { for (int64_t i = 0; i < a->len; i++) { if (ray_vec_is_null((ray_t*)a, i)) { ray_err_t err = ray_vec_set_null_checked(result, i, true); diff --git a/test/rfl/collection/collection_branch_cov.rfl b/test/rfl/collection/collection_branch_cov.rfl index dc0e28bb8..24708f789 100644 --- a/test/rfl/collection/collection_branch_cov.rfl +++ b/test/rfl/collection/collection_branch_cov.rfl @@ -218,12 +218,11 @@ ;; STR equality (lines 145-151) (count (union ["aa" "bb"] ["bb" "cc"])) -- 3 -;; Cross-type comparison via atom_eq fallback (lines 159-165) -;; except where vec1 is I64 and vec2 is F64 → cross-type hs_eq_rows -;; NOTE: hash values differ across types (I64 vs F64), so the hashset -;; probe may miss. The typed-vec path uses type-specific hashing, so -;; cross-type except returns all of vec1 (no matches found). -(count (except [1 2 3 4] (as 'F64 [2 3]))) -- 4 +;; Cross-type comparison: except where vec1 is I64 and vec2 is F64 → +;; cross-type hs_eq_rows. The first I64 probe switches the F64-built set +;; into num_f64 mode so 2 and 2.0 share a bucket; this used to return all +;; of vec1 because the two sides hashed through different functions (#645). +(count (except [1 2 3 4] (as 'F64 [2 3]))) -- 2 ;; ══════════════════════════════════════════════════════════════════════ ;; Section 10: hashset_grow — trigger by exceeding load factor (lines 202-226) diff --git a/test/rfl/collection/in.rfl b/test/rfl/collection/in.rfl index 026417e41..9f25838cf 100644 --- a/test/rfl/collection/in.rfl +++ b/test/rfl/collection/in.rfl @@ -86,9 +86,10 @@ (in (list) [1h 2h]) -- (list) ;; ========== TEXT (SYM/STR) NULL SEMANTICS (#593) ========== -;; `in` admits the typed verdict-LUT kernel only when BOTH sides are exactly -;; null-free, because that kernel uses null-matches-nothing semantics while -;; this path is null-equals-null. The gate used to ask +;; `in` admits the typed verdict-LUT kernel unless BOTH sides carry a null: +;; the kernel uses null-matches-nothing semantics while the hashset path is +;; null-equals-null, and the two disagree only on a null row against a null +;; set element. The gate used to ask ;; ray_vec_may_have_nulls, which is unconditionally true for SYM and STR ;; (their null is a payload value, not an attribute bit) — so the kernel was ;; unreachable for text columns and every symbol `in` fell through to the @@ -102,8 +103,8 @@ ;; null in the column, non-null needle — the null matches nothing else (in (as 'SYM ["a" "" "b"]) (as 'SYM ["a"])) -- [true false false] -;; null only in the NEEDLES: the column is null-free but the gate must still -;; reject, since a null needle needs null-equals-null against nothing +;; null only in the NEEDLES: a null-free column has no row for the null +;; needle to equal, so both semantics agree and the kernel is admitted (in (as 'SYM ["a" "b"]) (as 'SYM ["" "a"])) -- [true false] ;; nulls on both sides @@ -113,6 +114,38 @@ (in (as 'SYM ["a" "b" "c"]) (as 'SYM ["b" "c"])) -- [false true true] (in ["a" "b" "c"] ["b" "c"]) -- [false true true] +;; ========== ONE-SIDED NULLS REACH THE KERNEL (#593) ========== +;; A nullable column against null-free needles (or the reverse) is admitted: +;; a null row can only match a null needle. Before, one null row sent the +;; whole column to the hashset, whose i64/f64 hashes disagree across numeric +;; types — so the mixed-type rows below also answered wrongly. + +;; nullable column, null-free needles, every null-bearing width +(in [1 0Nl 3] [1 3]) -- [true false true] +(in [1i 0Ni 3i] [1 3]) -- [true false true] +(in [1h 0Nh 3h] [1h]) -- [true false false] +(in [1.0 0Nf 3.0] [1.0 3.0]) -- [true false true] +(in [2024.01.01 0Nd] [2024.01.01]) -- [true false] +(in ["a" "" "b"] ["a"]) -- [true false false] + +;; the i32 null sentinel is INT32_MIN: a non-null needle of that value must +;; not match the null row +(in [1i 0Ni 3i] [-2147483648 3]) -- [false false true] + +;; null-free column, null needles: the null needle matches nothing +(in [1 2 3] [0Nl 3]) -- [false false true] +(in [1.0 2.0 3.0] [0Nf 3.0]) -- [false false true] +(in [2024.01.01 2024.01.02] [0Nd 2024.01.02]) -- [false true] +(in ["a" "b"] ["" "a"]) -- [true false] + +;; mixed int/float with a null on one side — was [true false false] and +;; [false false false] on the hashset path +(in [1.0 0Nf 3.0] [1 3]) -- [true false true] +(in [1 0Nl 3] [1.0 3.0]) -- [true false true] + +;; nulls on both sides still take the null-equals-null path +(in [1 0Nl 3] [0Nl 3]) -- [false true true] + ;; empty needle set over a null-free symbol column (in (as 'SYM ["a" "b"]) (as 'SYM [])) -- [false false] @@ -123,3 +156,64 @@ ;; duplicate needles must not double-count or change the verdict (in (as 'SYM ["a" "b"]) (as 'SYM ["a" "a" "a"])) -- [true false] + +;; ========== TEMPORAL EQUALS ONLY ITS OWN TYPE (#593 review) ========== +;; atom_eq never matches a DATE to an int or to a TIMESTAMP, and neither do +;; find / except or the hashset `in`. The typed kernel compared raw +;; payloads — day 0 matched 0i, a DATE's day count matched a TIMESTAMP's +;; nanoseconds — so widening its gate made the answer depend on where the +;; nulls were. Every shape below answers the same with nulls on either +;; side, none, or (for the hashset) both. + +;; temporal column, int needles: nothing matches +(in [2000.01.01 2000.01.03] [0i 2i]) -- [false false] +(in [2000.01.01 0Nd 2000.01.03] [0i 2i]) -- [false false false] +(in [2000.01.01 2000.01.03] [0Ni 2i]) -- [false false] +(in [2000.01.01 0Nd] [0]) -- [false false] +(in [00:00:01.000 0Nt] [1000i]) -- [false false] +(in [2000.01.01D00:00:00.000000000 0Np] [0]) -- [false false] + +;; int column, temporal needles: nothing matches +(in [0i 2i] [2000.01.01 2000.01.03]) -- [false false] +(in [0 0Nl] [2000.01.01]) -- [false false] +(in [0 1] [2000.01.01 0Nd]) -- [false false] + +;; distinct temporal types never match — not even by raw payload +(in [2000.01.02] [2000.01.01D00:00:00.000000001]) -- [false] +(in [2000.01.02 0Nd] [2000.01.02D00:00:00.000000000]) -- [false false] +(in [2000.01.01D00:00:00.000000000 0Np] [2000.01.01]) -- [false false] +(in [00:00:00.001] [2000.01.01D00:00:00.000000001]) -- [false] + +;; same with nulls on both sides: only null equals null +(in [2000.01.01 0Nd] [0Ni 0i]) -- [false true] + +;; same-type temporal still matches, with or without nulls +(in [2000.01.01 0Nd 2000.01.03] [2000.01.03]) -- [false false true] +(in [2000.01.01 2000.01.03] [2000.01.01 0Nd]) -- [true false] + +;; the fused where: and find agree with the bare in +(count (select {from: (table [d] (list [2000.01.01 2000.01.03 0Nd])) where: (in d [0i 2i])})) -- 0 +(find [2000.01.01 2000.01.03] [0i]) -- [0Nl] + +;; ========== LARGE NEEDLE SETS HASH INSIDE THE KERNEL ========== +;; Past the SIMD small-set size (8) the kernel hashes the needles instead of +;; scanning them per row, so a nullable column against thousands of needles +;; stays O(rows + needles). Pins the boundary and the table's edge values. + +;; 8 needles (SIMD path) and 9 (hash path) answer alike +(in [1 2 3 0Nl 20] [1 3 5 7 9 11 13 15]) -- [true false true false false] +(in [1 2 3 0Nl 20] [1 3 5 7 9 11 13 15 20]) -- [true false true false true] +(in [1.0 2.0 0Nf 20.0] [1.0 3.0 5.0 7.0 9.0 11.0 13.0 15.0 20.0]) -- [true false false true] + +;; -0.0 and +0.0 share a slot; NaN / null never match +(in [-0.0 0.0 1.5] (as 'F64 (til 20))) -- [true true false] +(in [0Nf 2.5] (concat (as 'F64 (til 20)) [2.5])) -- [false true] + +;; duplicates in a large set, and a large set against a nullable column +(sum (in (concat (til 1000) [0Nl]) (take (til 100) 5000))) -- 100 +(sum (in (concat (til 1000) [0Nl]) (* 2 (til 1000)))) -- 500 +(sum (in (concat (as 'F64 (til 1000)) [0Nf]) (as 'F64 (* 2 (til 1000))))) -- 500 + +;; a large set of narrow ints against a narrow column +(sum (in (as 'I32 (til 1000)) (as 'I32 (* 3 (til 100))))) -- 100 +(sum (in (as 'I16 (til 1000)) (as 'I16 (* 3 (til 100))))) -- 100 diff --git a/test/rfl/collection/mixed_numeric.rfl b/test/rfl/collection/mixed_numeric.rfl new file mode 100644 index 000000000..028d45da9 --- /dev/null +++ b/test/rfl/collection/mixed_numeric.rfl @@ -0,0 +1,59 @@ +;; Set operations over mixed int/float typed vectors (#645). +;; +;; Numeric equality crosses widths and int/float: (== 3 3.0) is true, and +;; atom_eq compares any two numerics through f64. The row hashset used by +;; find/except/union/sect/in hashed int cells with ray_hash_i64 and float +;; cells with ray_hash_f64, so equal values landed in different buckets and +;; the probe missed; a correct answer was a hash collision. The set now +;; switches to hashing int cells through f64 when it meets a float probe. + +;; the contract these pin +(== [1 2 3] [1.0 2.0 3.0]) -- [true true true] + +;; find — both directions, and a haystack larger than the needles +(find [1.0 2.0 3.0] [3 1]) -- [2 0] +(find [1 2 3] [3.0 1.0]) -- [2 0] +(find (as 'F64 (til 100)) [3 1]) -- [3 1] +(find (til 100) [3.0 1.0 2.5]) -- [3 1 0Nl] + +;; except +(except [1 2 3] [1.0 3.0]) -- [2] +(except [1.0 2.0 3.0] [1 3]) -- [2.0] +(except [1i 2i 3i] [2.0]) -- [1i 3i] +(except [1h 2h] [2.0]) -- [1h] + +;; union — the shared value appears once +(count (union [1 2] [2.0 3.0])) -- 3 + +;; sect +(sect [1 2 3] [2.0 3.0]) -- [2 3] +(sect [1.0 2.0 3.0] [3 9]) -- [3.0] + +;; in with a null on BOTH sides still reaches the hashset +(in [1 0Nl 3] [1.0 0Nf]) -- [true true false] +(in [1.0 0Nf 3.0] [0Nl 3]) -- [false true true] + +;; a fractional float matches no int +(find [1 2 3] [2.5]) -- [0Nl] +(except [1 2 3] [2.5]) -- [1 2 3] + +;; same-type sets are unaffected +(find [1 2 3] [3 1]) -- [2 0] +(except [1 2 3] [1 3]) -- [2] +(find [1.0 2.0] [2.0]) -- [1] + +;; a float-built set probed by ints (no rehash needed), then by floats +(set _fs (as 'F64 (til 1000))) +(count (except (til 2000) _fs)) -- 1000 +(count (except (as 'F64 (til 2000)) _fs)) -- 1000 + +;; beyond 2^53 an int and a double compare through f64, so an int matches +;; the double its neighbour rounds to — (== 9007199254740993 +;; 9007199254740992.0) is true too. Same-type ints that share a double +;; image are still told apart. +(find [9007199254740993 9007199254740992 9007199254740994] [9007199254740992.0 9007199254740994.0]) -- [0 2] +(except [9007199254740993 9007199254740992 9007199254740994] [9007199254740992]) -- [9007199254740993 9007199254740994] + +;; enough rows to force hashset growth before and after the switch +(count (except (til 1000) (as 'F64 (til 500)))) -- 500 +(sum (find (as 'F64 (til 1000)) (til 1000))) -- 499500 diff --git a/test/rfl/expr/narrow_binary.rfl b/test/rfl/expr/narrow_binary.rfl index c9280abf5..5f3f77eca 100644 --- a/test/rfl/expr/narrow_binary.rfl +++ b/test/rfl/expr/narrow_binary.rfl @@ -237,3 +237,20 @@ ;; null is the least value in narrow-int comparison: -1 is NOT < null (< [-1h] [0Nh]) -- [false] (type (< [-1h] [0Nh])) -- 'B8 + +;; =================================================================== +;; Fused select: an I32/I16 narrowing cast or a comparison result used +;; as an operand of an i64 op (or of another comparison) is widened to +;; i64 lanes before the kernel reads it — the 4-, 2- or 1-byte scratch +;; buffer must never be read as 8-byte lanes. +;; =================================================================== +(set NB (table [a b c] (list [14 20 30] [1 2 3] [0Nl 2 3]))) +(at (select {from: NB z: (- a (as 'I32 -1))}) 'z) -- [15 21 31] +(at (select {from: NB z: (+ a (as 'I32 -1))}) 'z) -- [13 19 29] +(at (select {from: NB z: (neg (as 'I32 b))}) 'z) -- [-1 -2 -3] +(at (select {from: NB z: (abs (as 'I16 (neg b)))}) 'z) -- [1 2 3] +(at (select {from: NB z: (+ (> a 15) 1)}) 'z) -- [1 2 2] +(at (select {from: NB z: (< (as 'I32 b) (as 'I32 a))}) 'z) -- [true true true] +(count (select {from: NB where: (< (as 'I32 a) (as 'I16 25))})) -- 2 +;; the narrow null sentinel widens to the i64 null +(at (select {from: NB z: (+ a (as 'I32 c))}) 'z) -- [0Nl 22 33] diff --git a/test/rfl/expr/null_propagation.rfl b/test/rfl/expr/null_propagation.rfl index 837ac824c..e90615842 100644 --- a/test/rfl/expr/null_propagation.rfl +++ b/test/rfl/expr/null_propagation.rfl @@ -133,3 +133,37 @@ ;; Arithmetic: null DOES propagate (count (+ [0Nl 1 2] 1)) -- 3 (sum (map nil? (+ [0Nl 1 2] 1))) -- 1 + +;; =================================================================== +;; Fused select: a null produced in the middle of the program (x/0, +;; overflow, |INT64_MIN|) survives every downstream op — abs/neg/floor, +;; cast to i64, further i64 arithmetic — and the output column carries +;; HAS_NULLS, so aggregates skip those lanes instead of being poisoned. +;; =================================================================== +(set i (til 200000)) +(set NT (table [h b a c] (list (- (% i 300) 150) (% i 13) (% i 97) (% i 11)))) +(set H (at NT 'h)) (set B (at NT 'b)) (set A (at NT 'a)) (set C (at NT 'c)) +;; every 13th b is zero +(count (where (== B 0))) -- 15385 +(count (where (nil? (at (select {from: NT x: (abs (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (neg (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (floor (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (as 'I64 (/ h b))}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (+ (as 'I64 (/ h b)) 1)}) 'x)))) -- 15385 +(count (where (nil? (at (select {from: NT x: (+ (div h b) 1)}) 'x)))) -- 15385 +;; sums equal the column-at-a-time answers (nulls skipped, not summed) +(sum (at (select {from: NT x: (abs (/ h b))}) 'x)) -- (sum (abs (/ H B))) +(sum (at (select {from: NT x: (as 'I64 (/ h b))}) 'x)) -- (sum (as 'I64 (/ H B))) +(sum (at (select {from: NT x: (+ (div h b) 1)}) 'x)) -- (sum (+ (div H B) 1)) +;; the same through a program wider than 16 registers +(set WX (at (select {from: NT x: (abs (+ (/ h b) (+ a (+ c (+ a (+ c (+ a (+ c (+ a (+ c (+ a c)))))))))))}) 'x)) +(count (where (nil? WX))) -- 15385 +(sum WX) -- (sum (abs (+ (/ H B) (+ A (+ C (+ A (+ C (+ A (+ C (+ A (+ C (+ A C)))))))))))) +;; grouped aggregates over the fused column match a pre-materialized one +(set G1 (select {from: NT by: {k: c} s: (sum (abs (/ h b))) m: (avg (abs (/ h b)))})) +(set NX (table [c x] (list C (abs (/ H B))))) +(set G2 (select {from: NX by: {k: c} s: (sum x) m: (avg x)})) +(at G1 's) -- (at G2 's) +(at G1 'm) -- (at G2 'm) +;; a null-free program keeps HAS_NULLS off (no spurious null-aware path) +(count (where (nil? (at (select {from: NT x: (abs (+ h 1.5))}) 'x)))) -- 0 diff --git a/test/rfl/expr/wide_predicate.rfl b/test/rfl/expr/wide_predicate.rfl new file mode 100644 index 000000000..5c2ce0667 --- /dev/null +++ b/test/rfl/expr/wide_predicate.rfl @@ -0,0 +1,33 @@ +;; Filters and computed columns wider than 16 expression registers compile +;; to one fused expression (EXPR_MAX_REGS) instead of falling back to +;; column-at-a-time evaluation. Every answer is checked against the same +;; conditions combined step by step. +(set n 200000) +(set i (til n)) +(set T (table [a b c d e f] (list (% i 97) (% i 13) (% (* i 7) 1000) (as 'DATE (% i 60)) (% i 2) (% i 3)))) +;; four conditions, one a `within` (the shape that used to exceed the limit) +(set W4 (select {from: T where: (and (== a 5) (within d [2000.01.10 2000.01.20]) (== e 0) (== f 1))})) +(set S1 (select {from: T where: (== a 5)})) +(set S2 (select {from: S1 where: (within d [2000.01.10 2000.01.20])})) +(set S3 (select {from: S2 where: (== e 0)})) +(set S4 (select {from: S3 where: (== f 1)})) +(count W4) -- (count S4) +(at W4 'c) -- (at S4 'c) +;; six conditions with arithmetic on both sides +(set W6 (select {from: T where: (and (> (+ a b) 20) (< (* c 2) 1500) (within d [2000.01.05 2000.02.19]) (!= e 1) (>= (- f 1) 0) (<= (+ a (* 2 b)) 100))})) +(set R6 (select {from: T where: (> (+ a b) 20)})) +(set R6 (select {from: R6 where: (< (* c 2) 1500)})) +(set R6 (select {from: R6 where: (within d [2000.01.05 2000.02.19])})) +(set R6 (select {from: R6 where: (!= e 1)})) +(set R6 (select {from: R6 where: (>= (- f 1) 0)})) +(set R6 (select {from: R6 where: (<= (+ a (* 2 b)) 100)})) +(count W6) -- (count R6) +(sum (at W6 'c)) -- (sum (at R6 'c)) +;; a grouping under a four-condition filter (grouped by a computed key) +(set G4 (select {from: T by: {k: (xbar c 100)} n: (count a) where: (and (== a 5) (within d [2000.01.10 2000.01.20]) (== e 0) (== f 1))})) +(set GS (select {from: S4 by: {k: (xbar c 100)} n: (count a)})) +(at (select {from: G4 asc: k}) 'n) -- (at (select {from: GS asc: k}) 'n) +(at (select {from: G4 asc: k}) 'k) -- (at (select {from: GS asc: k}) 'k) +;; a wide computed column +(set X (select {from: T x: (+ (+ (+ (* a 3) (* b 5)) (+ (* c 7) (- a b))) (+ (+ (- c a) (+ b c)) (* (+ a 1) (+ b 2))))})) +(at (at X 'x) 12345) -- (+ (+ (+ (* 26 3) (* 8 5)) (+ (* 415 7) (- 26 8))) (+ (+ (- 415 26) (+ 8 415)) (* (+ 26 1) (+ 8 2)))) diff --git a/test/rfl/fused/fused_take_early_stop.rfl b/test/rfl/fused/fused_take_early_stop.rfl new file mode 100644 index 000000000..afd1895f5 --- /dev/null +++ b/test/rfl/fused/fused_take_early_stop.rfl @@ -0,0 +1,59 @@ +;; Fused filter + positional take (src/ops/fused_topk.c, ray_fused_take_select). +;; +;; `(select {cols from: T where: pred take: K})` with no ordering answers +;; with the first K passing rows (K > 0) or the last |K| (K < 0) in table +;; order, and stops scanning once they are known. `asc: key take: K` on a +;; column marked sorted (no nulls) goes the same way. The table spans +;; many scan chunks so the per-worker lists and the cutoff are exercised; +;; every answer is known by construction (x = row, c = x mod 1000). + +(set N 1000000) +(set X (til N)) +(set TK (table [x c s] (list X (% X 1000) (.attr.set 'sorted (til N))))) + +;; first K passing rows +(at (select {x: x from: TK where: (== c 7) take: 10}) 'x) -- [7 1007 2007 3007 4007 5007 6007 7007 8007 9007] +;; last |K| passing rows, in table order +(at (select {x: x from: TK where: (== c 7) take: -10}) 'x) -- [990007 991007 992007 993007 994007 995007 996007 997007 998007 999007] +;; AND predicate +(at (select {x: x from: TK where: (and (== c 7) (> x 500000)) take: 3}) 'x) -- [500007 501007 502007] +(at (select {x: x from: TK where: (and (== c 7) (< x 500000)) take: -2}) 'x) -- [498007 499007] +;; fewer passing rows than K, and none +(at (select {x: x from: TK where: (and (== c 7) (< x 3000)) take: 10}) 'x) -- [7 1007 2007] +(count (select {x: x from: TK where: (> x 5000000) take: 10})) -- 0 +;; all columns, aliased column +(count (select {from: TK where: (== c 7) take: 4})) -- 4 +(at (select {y: x from: TK where: (== c 7) take: 2}) 'y) -- [7 1007] +;; K beyond one worker's chunk of matches: identical to filter-then-take +(set FA (at (select {x: x from: TK where: (== c 7) take: 700}) 'x)) +(set FB (take (at (select {x: x from: TK where: (== c 7)}) 'x) 700)) +(count FA) -- 700 +(sum (as 'I64 (== FA FB))) -- 700 +(set LA (at (select {x: x from: TK where: (== c 7) take: -700}) 'x)) +(set LB (take (at (select {x: x from: TK where: (== c 7)}) 'x) -700)) +(sum (as 'I64 (== LA LB))) -- 700 +;; ascending take on a sorted key column = first K passing rows +(at (select {x: x from: TK where: (== c 7) asc: s take: 5}) 'x) -- [7 1007 2007 3007 4007] +;; the same key without the sorted marker still orders correctly +(at (select {x: x from: TK where: (== c 7) asc: x take: 3}) 'x) -- [7 1007 2007] +;; descending take on the sorted column stays on the ordered path +(at (select {x: x from: TK where: (== c 7) desc: s take: 3}) 'x) -- [999007 998007 997007] +;; a value written with alter set can break a `sorted` marker's order: the +;; marker is cleared, and the ascending take orders by value again +(set sv (.attr.set 'sorted (til 10))) +(alter 'sv set 8 -3) +(set TS (table [x s c] (list (til 10) sv (% (til 10) 2)))) +(at (select {x: x from: TS where: (== c 0) asc: s take: 2}) 'x) -- [8 0] +;; a take given as an expression is evaluated once +(set tcnt 0) +(count (select {x: x from: TK where: (== c 7) take: (do (set tcnt (+ tcnt 1)) 9)})) -- 9 +tcnt -- 1 +;; projections resolve left to right: an output that names an earlier +;; output's alias reads that output's value, not the source column — the +;; fused paths gather source columns and leave such a shape to the planner +(set TA (table [a b c] (list [3 1 2 4] [30 10 20 40] [0 0 0 0]))) +(at (select {b: a z: b from: TA where: (== c 0) take: 2}) 'z) -- [3 1] +(at (select {b: a z: b from: TA where: (== c 0) take: -2}) 'z) -- [2 4] +(at (select {b: a z: b from: TA where: (== c 0) asc: a take: 2}) 'z) -- [1 2] +(at (select {b: a z: b from: TA asc: a take: 2}) 'z) -- [1 2] +(at (select {b: b z: b from: TA where: (== c 0) take: 2}) 'z) -- [30 10] diff --git a/test/rfl/fused/topk_like_nullable_nowhere.rfl b/test/rfl/fused/topk_like_nullable_nowhere.rfl new file mode 100644 index 000000000..ef9c53fa0 --- /dev/null +++ b/test/rfl/fused/topk_like_nullable_nowhere.rfl @@ -0,0 +1,37 @@ +;; The fused filter + top-k takes a `like` over a text column that holds +;; empty strings (a null text cell is the empty string on every like path), +;; and a sort + take with no where: at all — both used to fall back to a full +;; filter / sort of the table. Results are checked against the unfused +;; spelling. +(set N 120000) +(set i (til N)) +(set url (as 'SYMBOL (map (fn [k] (if (== 0 (% k 97)) (format "http://google.com/q/%" k) (if (== 0 (% k 41)) "" (format "http://site%.example.com/p/%" (% k 500) k)))) i))) +(set us (as 'STR url)) +(set ts (% (* i 7919) 100003)) +(set T (table [url us ts k] (list url us ts (% i 17)))) +(count (select {from: T where: (nil? url)})) -- 2896 +;; like on the nullable SYM column: fused select equals the direct like +(count (select {from: T where: (like url "*google*")})) -- 1238 +(== (count (select {from: T where: (like url "*google*")})) (sum (as 'I64 (like url "*google*")))) -- true +(== (count (select {from: T where: (like us "*google*")})) (sum (as 'I64 (like us "*google*")))) -- true +;; top-k over the like: same rows as sorting the filtered table +(set A (select {from: T where: (like url "*google*") asc: ts take: 10})) +(set B (select {from: (select {from: T where: (like url "*google*")}) asc: ts take: 10})) +(all (== (at A 'ts) (at B 'ts))) -- true +(all (== (at A 'url) (at B 'url))) -- true +(count (cols A)) -- 4 +;; the sort key need not be in the projection +(all (== (at (select {from: T u: url where: (like url "*google*") asc: ts take: 10}) 'u) (at B 'url))) -- true +(all (== (at (select {from: T u: us where: (like us "*google*") desc: ts take: 10}) 'u) (at (select {from: (select {from: T where: (like url "*google*")}) desc: ts take: 10}) 'us))) -- true +;; a pattern the empty string matches admits the null rows on both paths +(== (count (select {from: T where: (like url "*") asc: ts take: 200000})) (sum (as 'I64 (like url "*")))) -- true +(== (count (select {from: T where: (like url "*") asc: ts take: 200000})) 120000) -- true +;; and with a projection list +(all (== (at (select {from: T u: url t: ts where: (like us "*google*") asc: ts take: 10}) 't) (at B 'ts))) -- true +;; no where: at all — the whole table's top-k, ascending and descending +(all (== (at (select {from: T asc: ts take: 10}) 'ts) (at (select {from: (select {from: T asc: ts}) take: 10}) 'ts))) -- true +(all (== (at (select {from: T desc: ts take: 10}) 'ts) (at (select {from: (select {from: T desc: ts}) take: 10}) 'ts))) -- true +(first (at (select {from: T asc: ts take: 1}) 'ts)) -- 0 +(count (select {from: T desc: ts take: 25})) -- 25 +;; empty strings sort first as null text +(first (at (select {from: T asc: us take: 1}) 'us)) -- "" diff --git a/test/rfl/fused/topk_zone_prune.rfl b/test/rfl/fused/topk_zone_prune.rfl new file mode 100644 index 000000000..776093314 --- /dev/null +++ b/test/rfl/fused/topk_zone_prune.rfl @@ -0,0 +1,68 @@ +;; Fused top-k over a stored column with a chunk-zone index skips chunks +;; whose extremum cannot beat the current K-th key. The time column here +;; runs in two long ascending blocks (like a table sorted by (counter, +;; time)), so most chunks are prunable; every answer must equal the same +;; query over an unindexed in-memory copy, for asc/desc, two keys, K near +;; the number of matches, and a nullable key (below). +(.sys.exec "rm -rf rf_test_zprune rf_test_zprune.csv") -- 0 +(set n 600000) +(set i (til n)) +(set t (+ 1000000 (* 7 (% i 300000)))) +(set s (as 'SYMBOL (map (fn [j] (if (== 0 (% j 3)) "" (format "p%" (% j 101)))) i))) +(set v (% (* i 13) 1000)) +(.csv.write (table [t s v] (list t s v)) "rf_test_zprune.csv") -- 0 +(set T (.csv.splayed [t s v] [I64 SYM I64] "rf_test_zprune.csv" "rf_test_zprune/")) +(set Tm (.csv.read [t s v] [I64 SYM I64] "rf_test_zprune.csv")) +(at (.idx.info (at T 't)) 'kind) -- 'chunk_zone +(set eqt (fn [a b] (all (map (fn [c] (all (== (as 'STR (at a c)) (as 'STR (at b c))))) (cols a))))) +(set q1 (fn [x] (select {s: s t: t v: v from: x where: (!= s "") asc: t take: 10}))) +(set q2 (fn [x] (select {s: s t: t from: x where: (!= s "") asc: [t s] take: 10}))) +(set q3 (fn [x] (select {v: v t: t from: x where: (> v 500) desc: t take: 7}))) +(set q4 (fn [x] (select {v: v t: t from: x where: (> v 990) asc: t take: 50}))) +(set q5 (fn [x] (select {v: v t: t from: x where: (== v 3) asc: t take: 700}))) +(eqt (q1 T) (q1 Tm)) -- true +(eqt (q2 T) (q2 Tm)) -- true +(eqt (q3 T) (q3 Tm)) -- true +(eqt (q4 T) (q4 Tm)) -- true +(eqt (q5 T) (q5 Tm)) -- true +(at (q1 T) 't) -- [1000007 1000007 1000014 1000014 1000028 1000028 1000035 1000035 1000049 1000049] +;; a nullable key: one chunk entirely null and every 1000th row null — +;; nulls sort first under asc (their chunks are never skipped) and last +;; under desc +(.sys.exec "rm -rf rf_test_zprune_n") -- 0 +(set kn (as 'I64 (map (fn [j] (if (or (== 0 (% j 1000)) (and (>= j 131072) (< j 196608))) 0N (+ 5 (% j 300000)))) i))) +(.db.splayed.set "rf_test_zprune_n/" (table [kn v] (list kn v))) +(set TN (.db.splayed.get "rf_test_zprune_n/")) +(set TNm (table [kn v] (list kn v))) +(set qn1 (fn [x] (select {kn: kn v: v from: x where: (> v 100) asc: kn take: 20}))) +(set qn2 (fn [x] (select {kn: kn v: v from: x where: (> v 100) desc: kn take: 20}))) +(set qn3 (fn [x] (select {kn: kn v: v from: x where: (== v 7) asc: [kn v] take: 5}))) +(eqt (qn1 TN) (qn1 TNm)) -- true +(eqt (qn2 TN) (qn2 TNm)) -- true +(eqt (qn3 TN) (qn3 TNm)) -- true +(.sys.exec "rm -rf rf_test_zprune_n") -- 0 +(.sys.exec "rm -rf rf_test_zprune rf_test_zprune.csv") -- 0 +;; best-chunk-first: the chunks are visited in the order of their zone +;; extremum, so an `asc` over a descending column (its best chunk is the +;; last one) and a key whose best chunk sits in the middle prune the rest +;; whatever the physical order; answers equal the in-memory copy and the +;; planner's full sort +(.sys.exec "rm -rf rf_test_zprune_o") -- 0 +(set kd (- n i)) +(set km (abs (- i 300000))) +(.db.splayed.set "rf_test_zprune_o/" (table [kd km v] (list kd km v))) +(set TO (.db.splayed.get "rf_test_zprune_o/")) +(set TOm (table [kd km v] (list kd km v))) +(set qo1 (fn [x] (select {kd: kd v: v from: x asc: kd take: 8000}))) +(set qo2 (fn [x] (select {kd: kd v: v from: x where: (> v 500) asc: kd take: 100}))) +(set qo3 (fn [x] (select {km: km v: v from: x where: (< v 900) asc: km take: 300}))) +(set qo4 (fn [x] (select {km: km v: v from: x desc: km take: 5}))) +(set qo5 (fn [x] (select {km: km kd: kd from: x where: (== v 7) asc: [km kd] take: 40}))) +(eqt (qo1 TO) (qo1 TOm)) -- true +(eqt (qo2 TO) (qo2 TOm)) -- true +(eqt (qo3 TO) (qo3 TOm)) -- true +(eqt (qo4 TO) (qo4 TOm)) -- true +(eqt (qo5 TO) (qo5 TOm)) -- true +(at (qo1 TO) 'kd) -- (+ 1 (til 8000)) +(at (qo4 TO) 'km) -- [300000 299999 299999 299998 299998] +(.sys.exec "rm -rf rf_test_zprune_o") -- 0 diff --git a/test/rfl/group/avg_exact_i128.rfl b/test/rfl/group/avg_exact_i128.rfl new file mode 100644 index 000000000..0217ee916 --- /dev/null +++ b/test/rfl/group/avg_exact_i128.rfl @@ -0,0 +1,139 @@ +;; avg of an integer column divides the EXACT 128-bit sum in every group +;; engine: keyless, dense/hash/radix keyed (v2), the float-key legacy hash +;; path, the small-table serial finishes, the slice-indexed path, pivot, +;; nullable inputs and a splayed copy — all the same bits as (avg vec). +;; SUM keeps its int64 wraparound contract next to the exact mean. +;; A column of 2^62 averages to exactly 2^62; alternating +(2^63-1-i) / +;; -(2^63-1-i) pairs sum to exactly -1 per pair, so their mean is -0.5 +;; whenever a group holds whole pairs (keys derive from (div d 2)). +(set n 70000) +(set d (til n)) +(set hc (take [4611686018427387904] n)) +(set hv (* (- (* 2 (% d 2)) 1) (- 9223372036854775807 d))) +(set p (div d 2)) +(set k3 (% p 3)) +(set f3 (as 'F64 k3)) +(set k350 (% p 350)) +(set ksp (* (% p 350) 1000003)) +(set k35k (% p 35000)) +(set T (table [c v d k3 f3 k350 ksp k35k] (list hc hv d k3 f3 k350 ksp k35k))) +(set E 4611686018427387904.0) +(set allE (fn [x] (== (count x) (sum (== x E))))) +(set allH (fn [x] (== (count x) (sum (== x -0.5))))) +;; the bare column +(avg hc) -- 4611686018427387904.0 +(avg hv) -- -0.5 +;; keyless select: one agg, a where, two aggs over one column (sum stays wrapped) +(at (select {from: T b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v)}) 'b) -- [-0.5] +(== (first (at (select {from: T b: (avg v)}) 'b)) (avg hv)) -- true +(== (first (at (select {from: T b: (avg c)}) 'b)) (avg hc)) -- true +(at (select {from: T where: (>= d 2) b: (avg v)}) 'b) -- [-0.5] +(at (select {from: T where: (>= d 2) b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg c) s: (sum c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v) s: (sum v)}) 'b) -- [-0.5] +(at (select {from: T b: (avg v) s: (sum v)}) 's) -- [-35000] +(at (select {from: T b: (avg c) m: (max c)}) 'b) -- [4611686018427387904.0] +;; few groups (dense), 350 groups, 35000 groups of two rows (each group's +;; own total overflows int64), sparse keys, a where, two keys +(allE (at (select {from: T by: k3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: k350 b: (avg v)}) 'b)) -- true +(count (select {from: T by: k35k b: (avg c)})) -- 35000 +(allE (at (select {from: T by: k35k b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k35k b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: ksp b: (avg v)}) 'b)) -- true +(allE (at (select {from: T by: ksp b: (avg c) s: (sum c)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: k3 b: k350} m: (avg v)}) 'm)) -- true +;; avg exact while sum of the same column wraps (23334, 23334, 23332 rows of 2^62) +(at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 's) -- [0Nl 0Nl 0] +(allE (at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 'b)) -- true +;; float keys take the legacy hash engine (radix merge + row emit) +(allE (at (select {from: T by: f3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v) s: (sum v)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: f3 b: (as 'F64 k350)} m: (avg v)}) 'm)) -- true +;; small tables: the serial finishes +(allH (at (select {from: T where: (< d 60) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T where: (< d 60) by: f3 b: (avg v)}) 'b)) -- true +(at (select {from: T where: (< d 60) b: (avg v)}) 'b) -- [-0.5] +;; narrow and temporal inputs agree with the bare vector avg +(set i32 (as 'I32 (* (- (* 2 (% d 2)) 1) (- 2147483647 (% d 1000))))) +(set i16 (as 'I16 (* (- (* 2 (% d 2)) 1) (- 32767 (% d 100))))) +(set u8 (as 'U8 (% d 256))) +(set bo (== (% d 3) 0)) +(set dt (as 'DATE (- 2147483647 (% d 1000)))) +(set tm (as 'TIME (- 2147483647 (% d 1000)))) +(set ts (as 'TIMESTAMP hv)) +(set TN (table [k3 f3 i32 i16 u8 bo dt tm ts] (list k3 f3 i32 i16 u8 bo dt tm ts))) +(at (select {from: TN a: (avg i32)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg i16)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg ts)}) 'a) -- [-0.5] +(== (first (at (select {from: TN a: (avg u8)}) 'a)) (avg u8)) -- true +(== (first (at (select {from: TN a: (avg bo)}) 'a)) (avg bo)) -- true +(== (first (at (select {from: TN a: (avg dt)}) 'a)) (avg dt)) -- true +(== (first (at (select {from: TN a: (avg tm)}) 'a)) (avg tm)) -- true +(at (select {from: TN by: k3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(== (at (select {from: TN by: k3 asc: k3 a: (avg u8)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg u8)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg bo)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg bo)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg dt)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg dt)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg tm)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg tm)}) 'a)) -- [true true true] +;; nulls are skipped; an all-null group is null +(set hcn (* hc (at [1 0N] (as 'I64 (== (% d 7) 0))))) +(set TNu (table [c d k3 f3] (list hcn d k3 f3))) +(at (select {from: TNu a: (avg c)}) 'a) -- [4611686018427387904.0] +(== (first (at (select {from: TNu a: (avg c)}) 'a)) (avg hcn)) -- true +(at (select {from: TNu a: (avg c) s: (sum c)}) 'a) -- [4611686018427387904.0] +(at (select {from: TNu where: (>= d 2) a: (avg c)}) 'a) -- [4611686018427387904.0] +(allE (at (select {from: TNu by: k3 a: (avg c)}) 'a)) -- true +(allE (at (select {from: TNu by: f3 a: (avg c)}) 'a)) -- true +(set TA (table [c k3 f3] (list (* hc (at [1 0N] (as 'I64 (== k3 2)))) k3 f3))) +(at (select {from: TA by: k3 asc: k3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: TA by: f3 asc: f3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: (select {from: TA where: (== k3 2)}) a: (avg c)}) 'a) -- [0Nf] +;; pivot +(set TP (table [r pc c v] (list (% p 3) (% (div d 6) 2) hc hv))) +(set P (pivot TP 'r 'pc 'v avg)) +(at P (at (cols P) 1)) -- [-0.5 -0.5 -0.5] +(at P (at (cols P) 2)) -- [-0.5 -0.5 -0.5] +(set Q (pivot TP 'r 'pc 'c avg)) +(allE (at Q (at (cols Q) 1))) -- true +(allE (at Q (at (cols Q) 2))) -- true +;; slice-indexed group (hash index on the SYM key + in-filter) +(set TS (update {from: (table [s v c] (list (as 'SYM (% p 5)) hv hc)) s: (.idx.hash s)})) +(at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg v)}) 'a) -- [-0.5 -0.5 -0.5] +(allE (at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg c) s2: (sum c)}) 'a)) -- true +;; single-key sparse scatter path (wide I32 key, avg over strlen of a STR +;; column, a count, a where): the mean of the lengths, bit for bit +(set cid (as 'I32 (* 35000 (% (* d 2654435761) 6000)))) +(set url (concat "http://example.com/some/path/" (as 'STR (* d (% d 977))))) +(set H (table [cid url c] (list cid url hc))) +(set R (select {from: H by: cid asc: cid l: (avg (strlen url)) n: (count url) where: (!= url "")})) +(count R) -- 6000 +(== (first (at R 'l)) (avg (strlen (at (select {from: H where: (== cid 0)}) 'url)))) -- true +(== (at R 'l) (at (select {from: H by: cid asc: cid l: (avg (strlen url)) where: (!= url "")}) 'l)) -- (take [true] 6000) +(allE (at (select {from: H by: cid l: (avg c) n: (count url)}) 'l)) -- true +(allE (at (select {from: H by: cid l: (avg c) s: (sum c) where: (!= url "")}) 'l)) -- true +;; a splayed copy answers exactly what the in-memory table answers +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 +(set Tm (table [c v d k3 f3 k350] (list hc hv d k3 f3 k350))) +(.db.splayed.set "rf_test_avg_i128/" Tm) +(set Ts (.db.splayed.get "rf_test_avg_i128/")) +(at (select {from: Ts a: (avg v) b: (avg c)}) 'a) -- [-0.5] +(at (select {from: Ts a: (avg v) b: (avg c)}) 'b) -- [4611686018427387904.0] +(== (at (select {from: Ts a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm a: (avg v) b: (avg c)}) 'b)) -- [true] +(== (at (select {from: Ts a: (avg v)}) 'a) (at (select {from: Tm a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts where: (>= d 2) a: (avg v)}) 'a) (at (select {from: Tm where: (>= d 2) a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a)) -- [true true true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(allH (at (select {from: Ts by: k350 a: (avg v)}) 'a)) -- true +(== (at (select {from: Ts by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(== (avg (at Ts 'c)) (first (at (select {from: Ts a: (avg c)}) 'a))) -- true +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 diff --git a/test/rfl/group/dense_slot_split.rfl b/test/rfl/group/dense_slot_split.rfl new file mode 100644 index 000000000..229ff1b25 --- /dev/null +++ b/test/rfl/group/dense_slot_split.rfl @@ -0,0 +1,54 @@ +;; The direct-array group path with more workers than its slot budget allows: +;; the slots are split across tasks (each scans every row for its own range +;; of groups, one accumulator set). 200k rows over ~250k key slots with a +;; nullable min/sum/count keep cells * workers above the row count at any +;; worker count from two up. The oracle is the same grouping forced onto +;; the hash path by one far-away key (the range then exceeds the slot +;; budget); every group but that one must agree. +(set N 200000) +(set i (til N)) +(set k (% (* i 7919) 130003)) +(set s (as 'SYMBOL (map (fn [j] (if (== 0 (% j 13)) "" (format "host%.example.org" (% (* j 31) 9973)))) i))) +(set v (as 'I64 (map (fn [j] (if (== 0 (% j 7)) 0N (% (* j 17) 1000))) i))) +(set f (% (* i 3) 100)) +(set T (table [k s v f] (list k s v f))) +(set T2 (table [k s v f] (list (concat k [1000000000]) (concat s (as 'SYMBOL ["zz"])) (concat v [5]) (concat f [7])))) +(set R (select {from: T by: k c: (count s) m: (min s) x: (max s) sv: (sum v) av: (avg v) fi: (first f) la: (last f)})) +(set O (select {from: T2 by: k c: (count s) m: (min s) x: (max s) sv: (sum v) av: (avg v) fi: (first f) la: (last f) where: (< k 1000000000)})) +(count R) -- 130003 +(== (count R) (count O)) -- true +(set RS (select {from: R asc: k})) +(set OS (select {from: O asc: k})) +(all (== (at RS 'k) (at OS 'k))) -- true +(all (== (at RS 'c) (at OS 'c))) -- true +(all (== (as 'STR (at RS 'm)) (as 'STR (at OS 'm)))) -- true +(all (== (as 'STR (at RS 'x)) (as 'STR (at OS 'x)))) -- true +(all (== (at RS 'sv) (at OS 'sv))) -- true +(all (== (at RS 'av) (at OS 'av))) -- true +(all (== (at RS 'fi) (at OS 'fi))) -- true +(all (== (at RS 'la) (at OS 'la))) -- true +;; strlen over the SYM column with empty cells, and a two-key composite +(set R2 (select {from: T by: k l: (avg (strlen s)) c: (count k)})) +(set O2 (select {from: T2 by: k l: (avg (strlen s)) c: (count k) where: (< k 1000000000)})) +(all (== (at (select {from: R2 asc: k}) 'l) (at (select {from: O2 asc: k}) 'l))) -- true +(set R3 (select {from: T by: [k f] c: (count s) m: (min s) sv: (sum v)})) +(set O3 (select {from: T2 by: [k f] c: (count s) m: (min s) sv: (sum v) where: (< k 1000000000)})) +(== (count R3) (count O3)) -- true +(== (sum (at R3 'sv)) (sum (at O3 'sv))) -- true +(== (sum (as 'I64 (strlen (as 'STR (at R3 'm))))) (sum (as 'I64 (strlen (as 'STR (at O3 'm)))))) -- true +;; the same under a where: selection (the tasks walk the selection too) +(set R4 (select {from: T by: k c: (count s) m: (min s) sv: (sum v) where: (> f 9)})) +(set O4 (select {from: T2 by: k c: (count s) m: (min s) sv: (sum v) where: (and (> f 9) (< k 1000000000))})) +(== (count R4) (count O4)) -- true +(all (== (at (select {from: R4 asc: k}) 'sv) (at (select {from: O4 asc: k}) 'sv))) -- true +(all (== (as 'STR (at (select {from: R4 asc: k}) 'm)) (as 'STR (at (select {from: O4 asc: k}) 'm)))) -- true +;; float sums over the partitioned accumulate add each group's rows in +;; table order: 1e16 + 1 - 1e16 + 1 rounds to 1.0 in that order, whatever +;; the core count (a scheduler-dependent order gave 0.0 for some groups) +(set NF 200000) +(set iF (til NF)) +(set TF (table [k f s] (list (% iF 50000) (at [1e16 1.0 -1e16 1.0] (as 'I64 (floor (/ iF 50000)))) (as 'SYMBOL (map (fn [j] (format "h%" (% j 97))) iF))))) +(set RF (select {from: TF by: k l: (sum (strlen s)) sf: (sum f) c: (count f)})) +(distinct (at RF 'sf)) -- [1.0] +(count RF) -- 50000 + diff --git a/test/rfl/group/derived_key_vocab_aggs.rfl b/test/rfl/group/derived_key_vocab_aggs.rfl new file mode 100644 index 000000000..d55740411 --- /dev/null +++ b/test/rfl/group/derived_key_vocab_aggs.rfl @@ -0,0 +1,59 @@ +;; A grouping keyed by an expression over one FILE-domain SYM column whose +;; every aggregate reads that column — count, min, max, sum/avg of strlen — +;; is decided over the column's distinct values: the rows are grouped by the +;; column once (a count per value), the key runs once per value, and the +;; aggregates are rewritten over that. Every answer is checked against the +;; row-wise evaluation, which one aggregate over another column forces. +(.sys.exec "rm -rf /tmp/rfl_dkv/") -- 0 +(set N 200000) +(set i (til N)) +(set D 3000) +(set refs (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 70) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 70) k))))) (til D))) +(set ref (as 'SYMBOL (map (fn [k] (at refs (% (* k 7919) D))) i))) +(set v (% (* i 31) 1000)) +(set T (table [ref v] (list ref v))) +(.db.splayed.set "/tmp/rfl_dkv/" T) +(set F (.db.splayed.get "/tmp/rfl_dkv/")) +(set key (fn [t] (select {from: t by: (let p (str-find ref "://") (let s (substr ref (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr ref 1 4) "http") (not (nil? sl))) (substr r 1 sl) ref))))) l: (avg (strlen ref)) c: (count ref) mn: (min ref) mx: (max ref) sl: (sum (strlen ref)) where: (!= ref "")}))) +(set oracle (fn [t] (select {from: t by: (let p (str-find ref "://") (let s (substr ref (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr ref 1 4) "http") (not (nil? sl))) (substr r 1 sl) ref))))) l: (avg (strlen ref)) c: (count ref) mn: (min ref) mx: (max ref) sl: (sum (strlen ref)) sv: (sum v) where: (!= ref "")}))) +(set same (fn [R O col] (all (== (at (select {from: R asc: c}) col) (at (select {from: O asc: c}) col))))) +(set R (key F)) +(set O (oracle F)) +(count R) -- 80 +(== (count R) (count O)) -- true +(cols R) -- [p l c mn mx sl] +(type (at R 'l)) -- 'F64 +(type (at R 'mn)) -- 'SYM +(same R O 'c) -- true +(same R O 'l) -- true +(same R O 'sl) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'mn)) (as 'STR (at (select {from: O asc: c}) 'mn)))) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'mx)) (as 'STR (at (select {from: O asc: c}) 'mx)))) -- true +(all (== (as 'STR (at (select {from: R asc: c}) 'p)) (as 'STR (at (select {from: O asc: c}) 'p)))) -- true +;; the same table in memory (runtime domain) takes the row path and agrees +(set RM (key T)) +(== (count RM) (count R)) -- true +(same RM R 'c) -- true +(same RM R 'l) -- true +;; no where: the null group is there, its avg is the null float, its min the empty symbol +(set R2 (select {from: F by: (substr ref 1 4) c: (count ref) l: (avg (strlen ref)) mn: (min ref)})) +(set O2 (select {from: F by: (substr ref 1 4) c: (count ref) l: (avg (strlen ref)) mn: (min ref) sv: (sum v)})) +(count R2) -- 12 +(same R2 O2 'c) -- true +(all (== (nil? (at (select {from: R2 asc: c}) 'l)) (nil? (at (select {from: O2 asc: c}) 'l)))) -- true +(sum (at R2 'c)) -- 200000 +;; sort and take on an output alias +(set R3 (select {from: F by: (substr ref 1 4) c: (count ref) where: (!= ref "") desc: c take: 3})) +(set O3 (select {from: F by: (substr ref 1 4) c: (count ref) sv: (sum v) where: (!= ref "") desc: c take: 3})) +(all (== (at R3 'c) (at O3 'c))) -- true +(all (== (as 'STR (at R3 'ref)) (as 'STR (at O3 'ref)))) -- true +;; an alias that spells the source column's name renames the key to `key` +(cols (select {from: F by: (substr ref 1 4) ref: (count ref) where: (!= ref "")})) -- [key ref] +;; unsorted output: groups in the order their first value appears in the +;; column's vocabulary, the same at every core count (the per-value counts +;; come out of a grouping whose order depends on the core count, and are +;; put back in vocabulary order first) +(set RU (select {from: F by: (substr ref 1 4) c: (count ref)})) +(as 'STR (at RU 'ref)) -- ["" "http" "ab5" "ab0" "ab9" "ab4" "ab3" "ab8" "ab2" "ab1" "ab6" "ab7"] +(at RU 'c) -- [18203 164471 1733 1736 1733 1733 1734 1731 1731 1731 1732 1732] +(.sys.exec "rm -rf /tmp/rfl_dkv/") -- 0 diff --git a/test/rfl/io/csv_splayed_dedupe_overflow.rfl b/test/rfl/io/csv_splayed_dedupe_overflow.rfl new file mode 100644 index 000000000..5364d72db --- /dev/null +++ b/test/rfl/io/csv_splayed_dedupe_overflow.rfl @@ -0,0 +1,27 @@ +;; Splayed CSV load of a SYM column whose per-partition dictionary +;; overflows (src/io/csv.c: csv_dedup_task caps a partition at +;; CSV_DEDUP_MAX_ENTS distinct strings — 8192 in debug builds — and the +;; column then interns row by row into the symfile domain, +;; csv_intern_dicts_domain's fallback). 140k distinct strings exceed the +;; cap in at least one partition however the column is split (at most 16 +;; partitions). Release builds keep the cap at 1M and take the batched +;; path; the answers are the same either way. + +(.sys.exec "rm -rf rf_test_splayed_ovf rf_test_splayed_ovf.csv") -- 0 +(.sys.exec "seq 0 139999 | awk '{print $1\",k\"$1}' > rf_test_splayed_ovf.csv") -- 0 +(set Tovf (.csv.splayed [id s] [I64 SYM] "rf_test_splayed_ovf.csv" "rf_test_splayed_ovf/")) +(count Tovf) -- 140000 +(sum (at Tovf 'id)) -- 9799930000 +(count (distinct (at Tovf 's))) -- 140000 +(at (at Tovf 's) 0) -- 'k0 +(at (at Tovf 's) 8192) -- 'k8192 +(at (at Tovf 's) 139999) -- 'k139999 +;; every row's symbol is the one written for its id +(count (select {from: Tovf where: (== s (as 'SYM "k77777"))})) -- 1 +(at (at (select {id: id from: Tovf where: (== s (as 'SYM "k77777"))}) 'id) 0) -- 77777 +;; reload from disk agrees +(set Rovf (.db.splayed.get "rf_test_splayed_ovf/")) +(count Rovf) -- 140000 +(count (distinct (at Rovf 's))) -- 140000 +(at (at Rovf 's) 139999) -- 'k139999 +(.sys.exec "rm -rf rf_test_splayed_ovf rf_test_splayed_ovf.csv") -- 0 diff --git a/test/rfl/journal/ops_journal_purge.rfl b/test/rfl/journal/ops_journal_purge.rfl index f21da7139..c570073e1 100644 --- a/test/rfl/journal/ops_journal_purge.rfl +++ b/test/rfl/journal/ops_journal_purge.rfl @@ -41,6 +41,7 @@ ;; State was reset: a second purge has no base to act on -> domain error. (.log.purge) !- domain +(.log.purge 1) !- arity ;; ════════════════════════════════════════════════════════════════════════ ;; 2. The issue #279 workflow: open, write, close, then purge WITHOUT diff --git a/test/rfl/journal/ops_journal_wrappers.rfl b/test/rfl/journal/ops_journal_wrappers.rfl index 9826f1495..9592562fd 100644 --- a/test/rfl/journal/ops_journal_wrappers.rfl +++ b/test/rfl/journal/ops_journal_wrappers.rfl @@ -54,6 +54,7 @@ ;; closed → err_to_ray RAY_OK arm → null. ;; ════════════════════════════════════════════════════════════════════════ (.log.roll) !- domain +(.log.roll 1) !- arity (.log.snapshot) !- domain (nil? (.log.sync)) -- true (nil? (.log.close)) -- true diff --git a/test/rfl/query/shared_node_memo.rfl b/test/rfl/query/shared_node_memo.rfl new file mode 100644 index 000000000..239a5f7ef --- /dev/null +++ b/test/rfl/query/shared_node_memo.rfl @@ -0,0 +1,32 @@ +;; A DAG node with several consumers is executed once and its value shared +;; (exec_node memo). The answers must equal the per-consumer evaluation the +;; engine used to do, including when one consumer's kernel would have been +;; free to reuse the input buffer in place (rc == 1) while a sibling still +;; read it. +(set N 70000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 500) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 500) k))))) i))) +(set T (table [S] (list S))) +;; a let-bound string search used twice and three times equals its explicit form +(all (== (at (select {from: T x: (let p (str-find S "://") (+ p p))}) 'x) (at (select {from: T x: (+ (str-find S "://") (str-find S "://"))}) 'x))) -- true +(all (== (at (select {from: T x: (let p (str-find S "://") (+ p (+ p p)))}) 'x) (at (select {from: T x: (* 3 (str-find S "://"))}) 'x))) -- true +;; the shared node feeds both an arithmetic consumer and a comparison; the +;; comparison must see the original values +(set R (select {from: T s: S x1: (let p (str-find S "://") (as 'I64 (within p [4 5]))) x3: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (as 'I64 (not (nil? sl))))))) x4: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (as 'I64 (and (within p [4 5]) (not (nil? sl)))))))) pp: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (+ p (* 0 sl)))))) p0: (str-find S "://")})) +(count (select {from: R where: (!= x4 (* x1 x3))})) -- 0 +(count (select {from: R where: (!= pp p0)})) -- 0 +(sum (at R 'x4)) -- 57576 +;; the whole derived host expression equals its unshared spelling +(set host (fn [] (select {from: T k: (let p (str-find S "://") (let s (substr S (+ p 4) -1) (let r (if (== (str-find s "www.") 0) (substr s 5 -1) s) (let sl (str-find r "/") (if (and (within p [4 5]) (== (substr S 1 4) "http") (not (nil? sl))) (substr r 1 sl) S)))))}))) +(count (distinct (at (host) 'k))) -- 511 +(sum (as 'I64 (strlen (at (host) 'k)))) -- 1099454 +;; a shared node used both outside an `if` and inside its branches: the +;; branch runs over its own rows only, so the value computed there must not +;; be handed to the consumer that runs over the whole table (and vice versa) +(set c (< (% i 7) 3)) +(set U (table [S c] (list S c))) +(all (== (at (select {from: U r: (let x (upper S) (if c (concat x "?") (concat x "!")))}) 'r) (at (select {from: U r: (if c (concat (upper S) "?") (concat (upper S) "!"))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (strlen S) (if c x (+ x 1000)))}) 'r) (at (select {from: U r: (if c (strlen S) (+ (strlen S) 1000))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (strlen S) (+ x (if c x (+ x 1000))))}) 'r) (at (select {from: U r: (+ (strlen S) (if c (strlen S) (+ (strlen S) 1000)))}) 'r))) -- true +(all (== (at (select {from: U r: (let x (substr S 8 5) (if c x (concat x "-")))}) 'r) (at (select {from: U r: (if c (substr S 8 5) (concat (substr S 8 5) "-"))}) 'r))) -- true +(sum (at (select {from: U r: (let x (strlen S) (+ x (if c x (+ x 1000))))}) 'r)) -- 41272796 diff --git a/test/rfl/query/sort_key_not_projected.rfl b/test/rfl/query/sort_key_not_projected.rfl new file mode 100644 index 000000000..91ac3221e --- /dev/null +++ b/test/rfl/query/sort_key_not_projected.rfl @@ -0,0 +1,60 @@ +;; A select that sorts by a source column it does not output. The key is +;; carried through the projection for the sort and dropped from the result; +;; before, the sort looked the key up in the projected table and failed with +;; `nyi` (only the fused top-k shapes, which read the source table, worked). +(set T (table [a b c s] (list (til 100) (% (til 100) 7) (reverse (til 100)) (as 'SYMBOL (map (fn [i] (format "x%" (% i 13))) (til 100)))))) + +;; plain sort, no filter, no take +(take (at (select {a: a from: T asc: c}) 'a) 5) -- [99 98 97 96 95] +(cols (select {a: a from: T asc: c})) -- ['a] +(count (select {a: a from: T asc: c})) -- 100 +;; a filter the fused path does not take, with and without take +(at (select {a: a from: T where: (> (+ b 1) 3) asc: c take: 3}) 'a) -- [97 96 95] +(cols (select {a: a from: T where: (> (+ b 1) 3) asc: c take: 3})) -- ['a] +(take (at (select {a: a from: T where: (> (+ b 1) 3) desc: c}) 'a) 4) -- [3 4 5 6] +(at (select {a: a from: T where: (in s ['x1 'x2]) desc: c take: 3}) 'a) -- [1 2 14] +;; two keys, neither projected +(at (select {a: a from: T where: (> (+ b 1) 3) asc: [b c] take: 4}) 'a) -- [94 87 80 73] +;; a computed output alongside +(at (select {a: a d: (+ a 1) from: T where: (> b 2) desc: c take: 2}) 'd) -- [4 5] +(cols (select {a: a d: (+ a 1) from: T where: (> b 2) desc: c take: 2})) -- ['a 'd] +;; a key that is also an output is not duplicated +(cols (select {a: a c: c from: T asc: c take: 2})) -- ['a 'c] +(at (select {a: a c: c from: T asc: c take: 2}) 'c) -- [0 1] +;; an output that renames the key column: the sort reads that output (the +;; key binds to the projected column that scans it), nothing extra is +;; added or dropped +(at (select {b: a from: T asc: a}) 'b) -- (til 100) +(take (at (select {b: a from: T desc: a}) 'b) 3) -- [99 98 97] +(take (at (select {b: a from: T where: (> (+ b 1) 3) desc: a}) 'b) 3) -- [97 96 95] +(select {x: c y: a from: T desc: a take: [1 2]}) -- (table [x y] (list [1 2] [98 97])) + +;; Sort keys name SOURCE columns, never output aliases: `b: a asc: b` sorts +;; by the source b (hidden, since no output scans it) and returns a under +;; the name b; a computed output under the key's name changes nothing +(set HB (table [a b] (list [3 1 2] [0 2 1]))) +(at (select {b: a from: HB asc: b}) 'b) -- [3 2 1] +(cols (select {b: a from: HB asc: b})) -- ['b] +(at (select {b: a from: HB desc: b take: 2}) 'b) -- [1 2] +(at (select {b: (+ a 1) from: HB asc: b}) 'b) -- [4 3 2] +(at (select {c: b from: HB asc: b}) 'c) -- [0 1 2] +;; the key binds to the projected column by position, so a hidden key whose +;; name collides with the projection's internal name for a computed output +;; (`_e`) still sorts by the source column — with and without take, one +;; and two keys, either direction +(set HC (table [a _e0 _e1] (list [3 1 2 5 4] [0 2 1 4 3] [1 1 0 0 1]))) +(at (select {z: (+ a 10) from: HC asc: _e0}) 'z) -- [13 12 11 14 15] +(cols (select {z: (+ a 10) from: HC asc: _e0})) -- ['z] +(at (select {z: (+ a 10) from: HC asc: _e0 take: 3}) 'z) -- [13 12 11] +(at (select {z: (+ a 10) from: HC asc: [_e1 _e0]}) 'z) -- [12 15 13 11 14] +(at (select {z: (+ a 10) from: HC asc: [_e1 _e0] take: 3}) 'z) -- [12 15 13] +(at (select {z: (+ a 10) w: (* a 2) from: HC desc: _e1 asc: _e0}) 'z) -- [13 11 14 12 15] +(at (select {z: (+ a 10) w: (* a 2) from: HC desc: _e1 asc: _e0 take: 2}) 'w) -- [6 2] +;; more hidden keys than the former fixed capacity of 16 +(set i (til 1000)) +(set T17 (table [a k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16] (list i (- 0 (% (* i 3) 5)) (- 0 (% (* i 4) 5)) (- 0 (% (* i 5) 5)) (- 0 (% (* i 6) 5)) (- 0 (% (* i 7) 5)) (- 0 (% (* i 8) 5)) (- 0 (% (* i 9) 5)) (- 0 (% (* i 10) 5)) (- 0 (% (* i 11) 5)) (- 0 (% (* i 12) 5)) (- 0 (% (* i 13) 5)) (- 0 (% (* i 14) 5)) (- 0 (% (* i 15) 5)) (- 0 (% (* i 16) 5)) (- 0 (% (* i 17) 5)) (- 0 (% (* i 18) 5)) (- 0 (% (* i 19) 5))))) +(set R17 (select {z: (+ a 10) from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16]})) +(set M17 (select {z: (+ a 10) k0: k0 k1: k1 k2: k2 k3: k3 k4: k4 k5: k5 k6: k6 k7: k7 k8: k8 k9: k9 k10: k10 k11: k11 k12: k12 k13: k13 k14: k14 k15: k15 k16: k16 from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16]})) +(cols R17) -- ['z] +(at R17 'z) -- (at M17 'z) +(at (select {z: (+ a 10) from: T17 asc: [k0 k1 k2 k3 k4 k5 k6 k7 k8 k9 k10 k11 k12 k13 k14 k15 k16] take: 5}) 'z) -- (take (at M17 'z) 5) diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl new file mode 100644 index 000000000..e340e9c3b --- /dev/null +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -0,0 +1,72 @@ +;; Whole-table aggregates over a splayed table answered from the chunk-zone +;; metadata (per-chunk min/max, per-chunk sums and non-null counts): every +;; answer must equal the same query over an in-memory copy of the data, +;; nulls and types included. Four chunks of 64k rows. +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 +(set n 200000) +(set i (til n)) +(set M (table [a16 a32 a64 big d f] (list (as 'I16 (% i 3000)) (as 'I32 (- (% (* i 7) 100000) 50000)) (- i 100000) (* (- i 100000) 92233720368547) (as 'DATE (% i 900)) (/ (as 'F64 i) 7.0)))) +(.csv.write M "rf_test_zone_aggs.csv") -- 0 +;; blank the first field of every row whose a16 value ends in 0: nulls +(.sys.exec "sed -i.bak 's/^\\([0-9]*\\)0,/,/' rf_test_zone_aggs.csv") -- 0 +(set T (.csv.splayed [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv" "rf_test_zone_aggs/")) +(set Tm (.csv.read [a16 a32 a64 big d f] [I16 I32 I64 I64 DATE F64] "rf_test_zone_aggs.csv")) +(at (.idx.info (at T 'a16)) 'kind) -- 'chunk_zone +(set q (fn [t] (select {from: t s16: (sum a16) c: (count a16) av: (avg a16) s32: (sum a32) av32: (avg a32) s64: (sum a64) av64: (avg a64) mn: (min d) mx: (max d) m16: (min a16) x32: (max a32)}))) +(set R (q T)) +(set O (q Tm)) +(cols R) -- (cols O) +(all (map (fn [c] (== (at (at R c) 0) (at (at O c) 0))) (cols R))) -- true +(all (map (fn [c] (== (type (at R c)) (type (at O c)))) (cols R))) -- true +;; the pinned values +(at (at R 's16) 0) -- (sum (at Tm 'a16)) +(at (at R 'mn) 0) -- 2000.01.01 +(at (at R 'c) 0) -- 200000 +;; sums past double's integer range: the metadata keeps the exact 128-bit +;; per-chunk sums and the row-wise reduction accumulates the same way, so +;; avg agrees bit for bit; a float column and a filter take the usual path +(at (select {from: T a: (avg big)}) 'a) -- (at (select {from: Tm a: (avg big)}) 'a) +(avg (at T 'big)) -- (avg (at Tm 'big)) +(at (select {from: T a: (sum a16) b: (sum f)}) 'b) -- (at (select {from: Tm a: (sum a16) b: (sum f)}) 'b) +(at (select {from: T a: (sum a16) where: (> a32 0)}) 'a) -- (at (select {from: Tm a: (sum a16) where: (> a32 0)}) 'a) +;; scalar forms read the same metadata +(sum (at T 'a32)) -- (sum (at Tm 'a32)) +(avg (at T 'a16)) -- (avg (at Tm 'a16)) +;; editing a copy of a loaded column drops the copy's index; the loaded +;; column's own metadata stays intact and still answers +(set U (delete {from: T where: (== a64 5)})) +(set w (at T 'a32)) +(alter 'w set 0 7) +(at (select {from: T m: (min a32) x: (max a32) s: (sum a32)}) 's) -- (at (select {from: Tm m: (min a32) x: (max a32) s: (sum a32)}) 's) +(at (select {from: T m: (min a32)}) 'm) -- (at (select {from: Tm m: (min a32)}) 'm) +;; a column whose every value is the int64 maximum is not "all null" +(.sys.exec "rm -rf rf_test_zone_max") -- 0 +(.db.splayed.set "rf_test_zone_max/" (table [m] (list (take [9223372036854775807] 70000)))) +(set TX (.db.splayed.get "rf_test_zone_max/")) +(at (select {from: TX a: (min m) b: (max m)}) 'a) -- [9223372036854775807] +(min (at TX 'm)) -- 9223372036854775807 +(.sys.exec "rm -rf rf_test_zone_max") -- 0 +;; the exact mean of values that overflow int64 within every chunk: +;; alternating +(2^63-1-i) / -(2^63-1-i) pairs sum to exactly one per pair, +;; and a column of 2^62 averages to exactly 2^62 — from the metadata, from +;; the row-wise reduction, and from a parted copy +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 +(set hn 150000) +(set hi (til hn)) +(set hv (* (- (* 2 (% hi 2)) 1) (- 9223372036854775807 hi))) +(set hc (take [4611686018427387904] hn)) +(.db.splayed.set "rf_test_zone_huge/" (table [hv hc] (list hv hc))) +(set TH (.db.splayed.get "rf_test_zone_huge/")) +(set THm (table [hv hc] (list hv hc))) +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(avg hv) -- -0.5 +(avg (at TH 'hv)) -- -0.5 +(avg (at TH 'hc)) -- 4611686018427387904.0 +(at (select {from: TH a: (avg hv) where: (>= hi 0)}) 'a) -- (at (select {from: THm a: (avg hv) where: (>= hi 0)}) 'a) +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 +;; an empty table keeps the planner's answers +(at (select {from: (select {from: T where: (< a64 -1000000)}) c: (count a16)}) 'c) -- (at (select {from: (select {from: Tm where: (< a64 -1000000)}) c: (count a16)}) 'c) +(.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 diff --git a/test/rfl/strop/str_view_pool_compact.rfl b/test/rfl/strop/str_view_pool_compact.rfl new file mode 100644 index 000000000..a6e2e7b8b --- /dev/null +++ b/test/rfl/strop/str_view_pool_compact.rfl @@ -0,0 +1,78 @@ +;; A STR result that is a view into another vector's pool keeps only the +;; bytes it points at: an `if` that kept a few rows of two big columns, or a +;; substring of a few bytes, must not hold (or copy) the whole parent pools. +;; direct-bytes counts the big (32 MB and up) pool allocations, so a result +;; that pins or copies a whole parent pool shows up there and one that keeps +;; its few bytes does not. +(set N 500000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) i))) +(set S2 (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) i))) +(set st (+ 1 (% i 50))) +(set c (< (% i 7) 3)) +(set T (table [S S2 st c] (list S S2 st c))) +(set direct (fn [] (do (.sys.gc) (at (.sys.mem) 'direct-bytes)))) +(set base (direct)) +(>= base 33554432) -- true +;; if over two pools under a where: keeping one row in fifty +(set R (select {from: T x: (if c S S2) where: (== st 1)})) +(count R) -- 10000 +(< (- (direct) base) 1000000) -- true +(all (== (at R 'x) (at (select {from: T x: (if c S S2) where: (== st 1)}) 'x))) -- true +(all (map (fn [j] (== (at (at R 'x) j) (if (at c (* j 50)) (at S (* j 50)) (at S2 (* j 50))))) (til 200))) -- true +(set R 0) +;; a filter that keeps most rows: the same values whichever arm ran +(set R (select {from: T x: (if c S S2) where: (> st 10)})) +(count R) -- 400000 +(all (== (at R 'x) (at (at (select {from: T x: (if c S S2)}) 'x) (where (> st 10))))) -- true +(set R 0) +;; the same over two substring views +(set R (select {from: T x: (if c (substr S 2 30) (substr S2 2 30)) where: (== st 1)})) +(< (- (direct) base) 1000000) -- true +(at (at R 'x) 0) -- "ttps://host0.example.org/path/" +(set R 0) +;; a pooled scalar side under a where +(set R (select {from: T x: (if c S "a-rather-long-literal-string") where: (== st 1)})) +(< (- (direct) base) 1000000) -- true +(set R 0) +;; a per-row condition over two whole columns with no filter (two cheap STR +;; branches take the eager arm, which picks descriptors over one pass): the +;; result holds less than the two parent pools together +(set R (select {from: T x: (if c S S2)})) +(set grow (- (direct) base)) +(< grow 40000000) -- true +(all (map (fn [j] (== (at (at R 'x) j) (if (at c j) (at S j) (at S2 j)))) (til 20000))) -- true +(set R 0) +;; the eager arm: a scalar condition (a variable, not a column) is not a +;; row mask, so the selected path declines and the eager fill runs over two +;; different pools. It builds a pool of the chosen side's bytes only, and +;; the result keeps its values after every source is released — at 500k rows +;; (the fill runs on the worker pool) and at 1000 rows (below the parallel +;; threshold, serial fill) +(set pick true) +(set RE (select {from: T x: (if pick S S2)})) +(set grow (- (direct) base)) +(< grow 40000000) -- true +(set pick false) +(set RF (select {from: T x: (if pick S S2)})) +(set Ts (take T 1000)) +(set pick true) +(set RS (select {from: Ts x: (if pick S S2)})) +(set pick false) +(set RS2 (select {from: Ts x: (if pick S S2)})) +;; a substring of three bytes is inline everywhere: no pool at all, so the +;; result outlives the table without holding its bytes +(set R (select {from: T x: (substr S st 3)})) +(set T 0) (set S 0) (set S2 0) (set i 0) (set st 0) (set c 0) (set Ts 0) +;; the eager results still read their strings, every row checked +(set ix (til 500000)) +(all (== (at RE 'x) (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) ix)))) -- true +(all (== (at RF 'x) (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) ix)))) -- true +(all (== (at RS 'x) (as 'STR (map (fn [k] (format "https://host%.example.org/path/%/page?id=%" (% k 500) k (* k 13))) (til 1000))))) -- true +(all (== (at RS2 'x) (as 'STR (map (fn [k] (format "second-pool-string-%-%" k (* k 7))) (til 1000))))) -- true +;; each eager result holds one side's bytes, not both parents' pools +(set RE 0) (set RS 0) (set RS2 0) +(< (direct) 40000000) -- true +(set RF 0) +(< (direct) 1000000) -- true +(count R) -- 500000 diff --git a/test/rfl/strop/substr_if_str_views.rfl b/test/rfl/strop/substr_if_str_views.rfl new file mode 100644 index 000000000..13774ac48 --- /dev/null +++ b/test/rfl/strop/substr_if_str_views.rfl @@ -0,0 +1,33 @@ +;; substr over a STR column with per-row start / length is a descriptor view +;; over the column's pool, and `if` over two STR columns is a descriptor pick +;; whatever pools the sides come from. Every answer is checked against the +;; per-row evaluation of the same expression. +(set N 120000) +(set i (til N)) +(set S (as 'STR (map (fn [k] (if (== 0 (% k 11)) "" (if (== 0 (% k 3)) (format "http://www.host%.example.org/path/%/page?id=%" (% k 500) k (* k 13)) (if (== 0 (% k 7)) (format "ab%" (% k 10)) (format "https://host%.example.org/p/%" (% k 500) k))))) i))) +(set st (% (* i 7) 19)) +(set ln (- (% (* i 5) 23) 3)) +(set stn (as 'I64 (map (fn [j] (if (== 0 (% j 13)) 0N (% (* j 7) 19))) i))) +(set kk (% i 17)) +(set T (table [S st ln stn k] (list S st ln stn kk))) +(set oracle (fn [f] (as 'STR (map f i)))) +;; vector start, whole tail +(all (== (at (select {from: T x: (substr S st -1)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) -1))))) -- true +;; vector start and vector length (negative lengths, past-the-end starts) +(all (== (at (select {from: T x: (substr S st ln)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) (at ln j)))))) -- true +;; a float start vector reads as its integer image +(all (== (at (select {from: T x: (substr S (as 'F64 st) -1)}) 'x) (oracle (fn [j] (substr (at S j) (at st j) -1))))) -- true +;; null starts give empty (null) rows +(all (== (at (select {from: T x: (substr S stn 4)}) 'x) (oracle (fn [j] (if (nil? (at stn j)) "" (substr (at S j) (at stn j) 4)))))) -- true +;; scalar start and length, the old view shape +(all (== (at (select {from: T x: (substr S 5 7)}) 'x) (oracle (fn [j] (substr (at S j) 5 7))))) -- true +;; a substring of a substring stays a view; the pool is shared, not copied +(all (== (at (select {from: T x: (substr (substr S 9 -1) 2 5)}) 'x) (oracle (fn [j] (substr (substr (at S j) 9 -1) 2 5))))) -- true +;; if over two STR sides from different pools, with empty strings on both +(all (== (at (select {from: T x: (if (== 0 (% k 2)) (substr S 9 -1) S)}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 2)) (substr (at S j) 9 -1) (at S j)))))) -- true +;; if over two views of one pool +(all (== (at (select {from: T x: (if (== 0 (% k 3)) (substr S 5 -1) (substr S 2 6))}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 3)) (substr (at S j) 5 -1) (substr (at S j) 2 6)))))) -- true +;; the picked side's empty strings are nulls of the result +(count (select {from: (select {from: T x: (if (== 0 (% k 2)) (substr S 9 -1) S)}) where: (nil? x)})) -- 16410 +;; a scalar side keeps the per-row path +(all (== (at (select {from: T x: (if (== 0 (% k 3)) S "zzz")}) 'x) (oracle (fn [j] (if (== 0 (% (at kk j) 3)) (at S j) "zzz"))))) -- true diff --git a/test/rfl/system/reserved_namespace.rfl b/test/rfl/system/reserved_namespace.rfl index fd0d72382..69a451442 100644 --- a/test/rfl/system/reserved_namespace.rfl +++ b/test/rfl/system/reserved_namespace.rfl @@ -5,10 +5,10 @@ ;; `'?` regardless of arity — so we exercise by calling, which is ;; what actually matters here.) ;; .sys.gc returns 0 on success — doubles as a binding-exists probe. -;; Registered as variadic so both (.sys.gc) and (.sys.gc 0) work; -;; other .sys.* info builtins follow the same convention. +;; Registered as variadic so the zero-argument call remains valid while +;; the implementation can reject unexpected arguments explicitly. (.sys.gc) -- 0 -(.sys.gc 0) -- 0 +(.sys.gc 0) !- arity (count (.sys.info)) -- 5 (count (.sys.mem)) -- 16 (count (.sys.build)) -- 2 @@ -116,7 +116,7 @@ internals !- name (del .sys.gc) !- reserve (del .os.getenv) !- reserve ;; Built-in still resolves after the blocked shadow attempts. -(.sys.gc 0) -- 0 +(.sys.gc) -- 0 (nil? .os.getenv) -- false ;; User-level dotted writes under a non-`.` root still work. (set myns.x 1) diff --git a/test/rfl/system/system_branch_cov.rfl b/test/rfl/system/system_branch_cov.rfl index bb67d9ed6..d07fe3aed 100644 --- a/test/rfl/system/system_branch_cov.rfl +++ b/test/rfl/system/system_branch_cov.rfl @@ -180,8 +180,8 @@ ;; ray_gc_fn (line 539) ;; ══════════════════════════════════════════════════════════════════════ (.sys.gc) -- 0 -(.sys.gc 0) -- 0 -(.sys.gc "x") -- 0 +(.sys.gc 0) !- arity +(.sys.gc "x") !- arity ;; ────────────── teardown ────────────── (.sys.exec "rm -rf /tmp/rfl_sys_bc_*") diff --git a/test/test_agg_contract.c b/test/test_agg_contract.c index 2134755cf..3f61880c3 100644 --- a/test/test_agg_contract.c +++ b/test/test_agg_contract.c @@ -1734,7 +1734,13 @@ static test_result_t test_cancelled_group(void) { * 20-worker pool and a 100k-slot slab the raw replication (20 slabs) leaves * most caches, and the run must use at most floor(0.75 * LLC / slab) task * slabs (never fewer than the pool when everything fits). The result is - * identical either way. */ + * identical either way. + * + * The LLC is pinned to 32 MB (the size the route assumes when none is + * reported) for the routed query: ~10 slabs fit, so the bound bites but the + * run is not cache-starved. Unpinned, a small-cache runner (a macOS CI VM + * reports a few MB of L2) fits fewer than three slabs and the route rightly + * switches to partition ownership, failing the strategy assertion. */ static test_result_t test_dense_cache_bound(void) { ray_pool_destroy(); TEST_ASSERT_EQ_I(ray_pool_init_total(20), RAY_OK); @@ -1743,21 +1749,24 @@ static test_result_t test_dense_cache_bound(void) { "(set cb_t (table [k v] (list (as 'I32 (% (* cb_i 7919) 100000)) (% cb_i 13))))"); TEST_ASSERT_NOT_NULL(setup); TEST_ASSERT_FALSE(RAY_IS_ERR(setup)); ray_release(setup); agg_route_reset(); + ray_cache_llc_set_for_test(32ull << 20); ray_t* r = ray_eval_str("(select {from:cb_t by:k s:(sum v)})"); - TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_FALSE(RAY_IS_ERR(r)); agg_route_stats_t stats = agg_route_stats(); + uint64_t llc = ray_cache_llc_bytes(); + ray_cache_llc_set_for_test(0); /* before any assert can return */ + TEST_ASSERT_NOT_NULL(r); TEST_ASSERT_FALSE(RAY_IS_ERR(r)); TEST_ASSERT_EQ_I(stats.routes[AGG_ROUTE_V2_DENSE], 1); TEST_ASSERT_EQ_I(stats.dense_strategy, AGG_DENSE_TASK_LOCAL); TEST_ASSERT_TRUE(stats.dense_tasks >= 2 && stats.dense_tasks <= 20); TEST_ASSERT_EQ_I(ray_table_nrows(r), 100000); - uint64_t llc = ray_cache_llc_bytes(); - if (llc > 0) { + { size_t block = agg_resolve(OP_SUM, RAY_I64)->state_size; double slots = (double)stats.dense_local_slots / stats.dense_tasks; double slab = slots * (block + sizeof(int64_t) + 1); double budget = (double)llc * 0.75; uint32_t cap = slab * 20 > budget ? (uint32_t)(budget / slab) : 20; if (cap < 2) cap = 2; + TEST_ASSERT_TRUE(cap >= 3 && cap < 20); /* the pin makes the bound bite */ TEST_ASSERT_EQ_I(stats.dense_tasks, cap); } /* The bounded run computes the same sums as the serial engine. */ diff --git a/test/test_agg_engine.c b/test/test_agg_engine.c index ebd94475c..3adb1a8ba 100644 --- a/test/test_agg_engine.c +++ b/test/test_agg_engine.c @@ -2103,6 +2103,101 @@ static test_result_t test_group_values_f64(void) { PASS(); } + +/* ══════════════════════════════════════════════════════════════════════ + * Exact integer AVG on the legacy engines (v2 OFF) and on v2 (ON). + * A column of 2^62 averages to exactly 2^62 and alternating + * +(2^63-1-i) / -(2^63-1-i) pairs to exactly -0.5 whenever a group holds + * whole pairs — only the 128-bit sum gives those bits (the wrapped int64 + * total is garbage, a double running sum loses the low bits). The key + * type and the row count steer the legacy ladder: an I64 key over 70000 + * rows takes the direct-array path, the same key over 60 rows the serial + * finish, an F64 key the hash/radix row layout, and a sparse I64 key + * (range far above the row count) the single-key sparse paths, which now + * hand an integer AVG to the row layout. */ +static ray_t* avg128_make(int64_t n, bool f64_key, bool sparse_key) { + ray_t* kvec = ray_vec_new(f64_key ? RAY_F64 : RAY_I64, n); kvec->len = n; + ray_t* vvec = ray_vec_new(RAY_I64, n); vvec->len = n; + ray_t* cvec = ray_vec_new(RAY_I64, n); cvec->len = n; + int64_t* vd = (int64_t*)ray_data(vvec); + int64_t* cd = (int64_t*)ray_data(cvec); + for (int64_t i = 0; i < n; i++) { + int64_t k = (i / 2) % 3; + if (sparse_key) k *= 1000000007LL; + if (f64_key) ((double*)ray_data(kvec))[i] = (double)k; + else ((int64_t*)ray_data(kvec))[i] = k; + int64_t mag = INT64_MAX - i; + vd[i] = (i & 1) ? mag : -mag; + cd[i] = (int64_t)1 << 62; + } + ray_t* tbl = ray_table_new(3); + tbl = ray_table_add_col(tbl, ray_sym_intern("k", 1), kvec); ray_release(kvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("v", 1), vvec); ray_release(vvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("c", 1), cvec); ray_release(cvec); + return tbl; +} +static ray_op_t* gb_avg128(ray_graph_t* g) { + ray_op_t* k = ray_scan(g, "k"); ray_op_t* v = ray_scan(g, "v"); ray_op_t* c = ray_scan(g, "c"); + uint16_t ops[] = { OP_AVG, OP_AVG, OP_SUM, OP_COUNT }; + ray_op_t* ins[] = { v, c, c, c }; ray_op_t* keys[] = { k }; + return ray_group(g, keys, 1, ops, ins, 4); +} +/* Run gb_avg128 with the v2 flag as given; the 2nd and 3rd result columns + * are the two means. Fails unless every group is exactly -0.5 / 2^62. */ +static test_result_t avg128_check(ray_t* tbl, bool v2, const char* what) { + ray_agg_engine_v2 = v2; + ray_graph_t* g = ray_graph_new(tbl); + ray_t* r = ray_execute(g, gb_avg128(g)); + if (r && ray_is_lazy(r)) r = ray_lazy_materialize(r); + ray_agg_engine_v2 = true; /* restore default */ + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + if (!r || RAY_IS_ERR(r) || r->type != RAY_TABLE || ray_table_ncols(r) != 5) { + res = (test_result_t){ TEST_FAIL, "avg128: bad result shape" }; + } else { + ray_t* mv = ray_table_get_col_idx(r, 1); + ray_t* mc = ray_table_get_col_idx(r, 2); + int64_t ng = ray_table_nrows(r); + if (ng != 3 || !mv || !mc || mv->type != RAY_F64 || mc->type != RAY_F64) { + res = (test_result_t){ TEST_FAIL, "avg128: expected 3 groups of F64 means" }; + } else { + for (int64_t i = 0; i < ng; i++) { + double a = ((const double*)ray_data(mv))[i]; + double b = ((const double*)ray_data(mc))[i]; + if (a != -0.5 || b != 4611686018427387904.0) { + snprintf(ray_test_fail_buf, sizeof ray_test_fail_buf, + "%s: group %lld mean(v)=%.17g mean(c)=%.17g (want -0.5, 2^62)", + what, (long long)i, a, b); + res = (test_result_t){ TEST_FAIL, ray_test_fail_buf }; + break; + } + } + } + } + if (r && !RAY_IS_ERR(r)) ray_release(r); + ray_graph_free(g); + return res; +} +static test_result_t test_avg_exact_i128_engines(void) { + ray_heap_init(); (void)ray_sym_init(); + struct { int64_t n; bool f64; bool sparse; const char* name; } shapes[] = { + { HC_N, false, false, "i64 key, 70000 rows" }, + { 60, false, false, "i64 key, 60 rows" }, + { HC_N, true, false, "f64 key, 70000 rows" }, + { 60, true, false, "f64 key, 60 rows" }, + { HC_N, false, true, "sparse i64 key, 70000 rows" }, + { 60, false, true, "sparse i64 key, 60 rows" }, + }; + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + for (size_t i = 0; i < sizeof(shapes) / sizeof(shapes[0]) && res.status == TEST_PASS; i++) { + ray_t* tbl = avg128_make(shapes[i].n, shapes[i].f64, shapes[i].sparse); + res = avg128_check(tbl, false, shapes[i].name); + if (res.status == TEST_PASS) res = avg128_check(tbl, true, shapes[i].name); + ray_release(tbl); + } + ray_sym_destroy(); ray_heap_destroy(); + return res; +} + const test_entry_t agg_engine_entries[] = { { "pearson_old_engine_r_vs_r2", test_pearson_old_engine_r_vs_r2, NULL, NULL }, { "diff_group_pearson_1k", test_diff_group_pearson_1k, NULL, NULL }, @@ -2179,5 +2274,6 @@ const test_entry_t agg_engine_entries[] = { { "group_keys_i_i32", test_group_keys_i_i32, NULL, NULL }, { "group_keys_multi", test_group_keys_multi, NULL, NULL }, { "agg_run_one_i64", test_agg_run_one_i64, NULL, NULL }, + { "avg_exact_i128_engines", test_avg_exact_i128_engines, NULL, NULL }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_domain.c b/test/test_domain.c index ce7186ac6..7da142c70 100644 --- a/test/test_domain.c +++ b/test/test_domain.c @@ -39,6 +39,9 @@ #include "table/domain.h" #include "store/col.h" #include "store/serde.h" +#include "core/pool.h" /* the batch intern probes on the pool */ +#include "ops/hash.h" /* ray_hash_bytes: batch intern takes prehashed entries */ +#include "mem/sys.h" #include "ops/ops.h" /* RAY_PARTED_BASE (parted-flatten adoption test) */ #include #include @@ -401,6 +404,108 @@ static test_result_t test_domain_open_basic(void) { PASS(); } +/* ray_sym_domain_intern_batch: one batch with repeats spread over the hash + * partitions, half the vocabulary already interned one by one. Every + * entry resolves to the position find() reports, repeats agree, the + * pre-interned half keeps its positions, the count grows by exactly the + * new distinct strings, a second identical batch changes nothing and "" + * is position 0. */ +static test_result_t test_domain_intern_batch(void) { + unlink(TMP_DOM_SYM_PATH); + unlink(TMP_DOM_SYM_PATH ".lk"); + TEST_ASSERT_NOT_NULL(ray_pool_get()); /* parallel probe path */ + + ray_sym_domain_t* dom = ray_sym_domain_open_or_create(TMP_DOM_SYM_PATH); + TEST_ASSERT_NOT_NULL(dom); + + enum { NV = 20000, NB = 60000, SL = 16 }; + char* vocab = (char*)ray_sys_alloc((size_t)NV * SL); + int64_t* pre = (int64_t*)ray_sys_alloc((size_t)NV * sizeof(int64_t)); + int64_t* seen = (int64_t*)ray_sys_alloc((size_t)NV * sizeof(int64_t)); + const char** strs = (const char**)ray_sys_alloc((size_t)NB * sizeof(char*)); + size_t* lens = (size_t*)ray_sys_alloc((size_t)NB * sizeof(size_t)); + uint32_t* hashes = (uint32_t*)ray_sys_alloc((size_t)NB * sizeof(uint32_t)); + int64_t* pos = (int64_t*)ray_sys_alloc((size_t)NB * sizeof(int64_t)); + int64_t* pos2 = (int64_t*)ray_sys_alloc((size_t)NB * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(vocab); TEST_ASSERT_NOT_NULL(pre); TEST_ASSERT_NOT_NULL(seen); + TEST_ASSERT_NOT_NULL(strs); TEST_ASSERT_NOT_NULL(lens); TEST_ASSERT_NOT_NULL(hashes); + TEST_ASSERT_NOT_NULL(pos); TEST_ASSERT_NOT_NULL(pos2); + + for (int v = 0; v < NV; v++) { + snprintf(vocab + (size_t)v * SL, SL, "v%05d_%c", v, 'a' + v % 26); + seen[v] = -1; + pre[v] = -1; + } + for (int v = 0; v < NV / 2; v++) { + const char* sv = vocab + (size_t)v * SL; + pre[v] = ray_sym_domain_intern(dom, sv, strlen(sv)); + TEST_ASSERT(pre[v] > 0, "pre-intern gets a position"); + } + int64_t count_before = ray_sym_domain_count(dom); + TEST_ASSERT_EQ_I(count_before, NV / 2 + 1); /* + reserved "" */ + + for (int j = 0; j < NB; j++) { + int v = (int)(((int64_t)j * 7919) % NV); /* every string ~3 times */ + const char* sv = vocab + (size_t)v * SL; + strs[j] = sv; + lens[j] = strlen(sv); + hashes[j] = (uint32_t)ray_hash_bytes(sv, lens[j]); + pos[j] = -7; + } + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, NB, strs, lens, hashes, pos)); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 1); + + for (int j = 0; j < NB; j++) { + int v = (int)(((int64_t)j * 7919) % NV); + TEST_ASSERT(pos[j] > 0 && pos[j] <= NV, "position in range, never 0"); + if (seen[v] < 0) seen[v] = pos[j]; + TEST_ASSERT_EQ_I(pos[j], seen[v]); /* repeats agree */ + if (pre[v] >= 0) TEST_ASSERT_EQ_I(pos[j], pre[v]); /* hits keep their position */ + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, strs[j], lens[j]), pos[j]); + ray_t* a = ray_sym_domain_str(dom, pos[j]); + TEST_ASSERT_NOT_NULL(a); + TEST_ASSERT_EQ_U(ray_str_len(a), lens[j]); + TEST_ASSERT_MEM_EQ(lens[j], ray_str_ptr(a), strs[j]); + } + /* distinct positions: every vocabulary entry got exactly one */ + for (int v = 0; v < NV; v++) TEST_ASSERT(seen[v] > 0, "every string appeared"); + + /* the same batch again: all hits, nothing appended */ + for (int j = 0; j < NB; j++) pos2[j] = -7; + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, NB, strs, lens, hashes, pos2)); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 1); + for (int j = 0; j < NB; j++) TEST_ASSERT_EQ_I(pos2[j], pos[j]); + + /* "" resolves to the reserved position 0; a fresh string still appends */ + { + const char* two[2] = { "", "brand_new_entry" }; + size_t tl[2] = { 0, strlen("brand_new_entry") }; + uint32_t th[2] = { (uint32_t)ray_hash_bytes("", 0), (uint32_t)ray_hash_bytes(two[1], tl[1]) }; + int64_t tp[2] = { -7, -7 }; + TEST_ASSERT_TRUE(ray_sym_domain_intern_batch(dom, 2, two, tl, th, tp)); + TEST_ASSERT_EQ_I(tp[0], 0); + TEST_ASSERT_EQ_I(tp[1], NV + 1); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), NV + 2); + } + + /* the flushed file reopens with the same vocabulary */ + TEST_ASSERT_EQ_I(ray_sym_domain_flush(dom, false), RAY_OK); + ray_sym_domain_release(dom); + ray_sym_domain_t* re = ray_sym_domain_open(TMP_DOM_SYM_PATH); + TEST_ASSERT_NOT_NULL(re); + TEST_ASSERT_EQ_I(ray_sym_domain_count(re), NV + 2); + for (int j = 0; j < NB; j += 997) + TEST_ASSERT_EQ_I(ray_sym_domain_find(re, strs[j], lens[j]), pos[j]); + ray_sym_domain_release(re); + + ray_sys_free(vocab); ray_sys_free(pre); ray_sys_free(seen); + ray_sys_free(strs); ray_sys_free(lens); ray_sys_free(hashes); + ray_sys_free(pos); ray_sys_free(pos2); + unlink(TMP_DOM_SYM_PATH); + unlink(TMP_DOM_SYM_PATH ".lk"); + PASS(); +} + /* Task 7b: open_or_create on a missing file yields an empty writable * domain; "" is seeded at position 0 by the first intern; flush creates * the file; verify-base-unchanged makes a racing writer LOUD. */ @@ -1804,6 +1909,47 @@ static test_result_t test_domain_runtime_lut(void) { PASS(); } +static test_result_t test_domain_concat_text_nulls(void) { + ray_sym_domain_t* dom = NULL; + int64_t pos_a = -1, pos_b = -1; + TEST_ASSERT_TRUE(build_divergent_qsym_fixture(&dom, &pos_a, &pos_b)); + /* Fixture includes the builtin vocabulary, so its positions need W16. */ + const uint8_t widths[] = {RAY_SYM_W16, RAY_SYM_W32, RAY_SYM_W64}; + for (int w = 0; w < 3; w++) { + ray_t* file = ray_sym_vec_new(widths[w], 4); + ray_sym_domain_release(file->sym_domain); + ray_sym_domain_retain(dom); + file->sym_domain = dom; + file->len = 4; + const int64_t vals[] = {pos_b, pos_a, 0, pos_b}; + for (int i = 0; i < 4; i++) ray_write_sym(ray_data(file), i, vals[i], RAY_SYM, file->attrs); + int64_t ids[] = {0, ray_sym_intern("dq_b", 4)}; + ray_t* runtime = ray_vec_from_raw(RAY_SYM, ids, 2); + ray_t* slice = ray_vec_slice(file, 1, 2); + for (int side = 0; side < 2; side++) { + ray_t* out = ray_vec_concat(side ? runtime : slice, side ? slice : runtime); + TEST_ASSERT_NOT_NULL(out); + TEST_ASSERT_FALSE(RAY_IS_ERR(out)); + TEST_ASSERT_EQ_PTR(ray_sym_vec_domain(out), ray_sym_runtime_domain()); + TEST_ASSERT_EQ_I(out->attrs & RAY_SYM_W_MASK, RAY_SYM_W64); + TEST_ASSERT_TRUE(out->attrs & RAY_ATTR_HAS_NULLS); + TEST_ASSERT_FALSE(out->attrs & (RAY_ATTR_SLICE | RAY_ATTR_HAS_INDEX)); + int64_t expected[] = {ray_sym_intern("dq_a", 4), 0, 0, ids[1]}; + for (int i = 0; i < 4; i++) { + int64_t value = expected[(i + (side ? 2 : 0)) % 4]; + TEST_ASSERT_EQ_I(((int64_t*)ray_data(out))[i], value); + TEST_ASSERT_EQ_I(ray_vec_is_null(out, i), value == 0); + } + ray_release(out); + } + ray_release(slice); ray_release(file); ray_release(runtime); + } + ray_sym_domain_release(dom); + unlink(TMP_DOM_QSYM_PATH); + unlink(TMP_DOM_QSYM_PATH ".lk"); + PASS(); +} + #define TMP_DOM_BADSYM_PATH "/tmp/rayforce_test_domain_badsym" /* Position-0 reservation: ray_sym_save-produced files carry "" at @@ -1986,8 +2132,10 @@ const test_entry_t domain_entries[] = { { "domain/parted_flatten_adopts", test_domain_parted_flatten_adopts, domain_rt_setup, domain_rt_teardown }, { "domain/str_eager_lockfree", test_domain_str_eager_lockfree, domain_setup, domain_teardown }, { "domain/raw_pin", test_domain_raw_pin, domain_setup, domain_teardown }, + { "domain/concat_text_nulls", test_domain_concat_text_nulls, domain_rt_setup, domain_rt_teardown }, { "domain/runtime_lut", test_domain_runtime_lut, domain_rt_setup, domain_rt_teardown }, { "domain/open_position0_validation", test_domain_open_position0_validation, domain_setup, domain_teardown }, + { "domain/intern_batch", test_domain_intern_batch, domain_setup, domain_teardown }, { "domain/dict_upsert_file_keys", test_domain_dict_upsert_file_keys, domain_rt_setup, domain_rt_teardown }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_embedding.c b/test/test_embedding.c index 89037ec4a..091e2c52f 100644 --- a/test/test_embedding.c +++ b/test/test_embedding.c @@ -36,6 +36,7 @@ #include "lang/internal.h" #include "lang/format.h" #include "store/hnsw.h" +#include "store/fileio.h" #include #include #include @@ -1005,6 +1006,32 @@ static test_result_t test_hnsw_vec_size_valid_guard(void) { PASS(); } +/* A persisted neighbor id is an array index during greedy descent. A + * malformed file must be rejected by both load entry points before a query + * can turn it into an out-of-bounds read / process crash. */ +static test_result_t test_hnsw_load_rejects_bad_neighbor(void) { + const char* dir = "/tmp/ray_hnsw_bad_neighbor"; + float vecs[2 * 2] = { 1.0f, 0.0f, 0.0f, 1.0f }; + ray_hnsw_t* idx = ray_hnsw_build(vecs, 2, 2, RAY_HNSW_L2, 4, 50); + TEST_ASSERT_NOT_NULL(idx); + TEST_ASSERT_EQ_I(ray_hnsw_save(idx, dir), RAY_OK); + ray_hnsw_free(idx); + + char path[256]; + snprintf(path, sizeof(path), "%s/hnsw_layer_0.bin", dir); + FILE* f = fopen(path, "r+b"); + TEST_ASSERT_NOT_NULL(f); + /* Layer metadata is two int64 values; overwrite the first neighbor. */ + TEST_ASSERT_EQ_I(fseek(f, (long)(2 * sizeof(int64_t)), SEEK_SET), 0); + int64_t bad_id = 1000000000; + TEST_ASSERT_EQ_U(fwrite(&bad_id, sizeof(bad_id), 1, f), 1); + TEST_ASSERT_EQ_I(fclose(f), 0); + + TEST_ASSERT_NULL(ray_hnsw_load(dir)); + TEST_ASSERT_NULL(ray_hnsw_mmap(dir)); + PASS(); +} + /* Trigger the maxheap_sift_down / results-replacement path in hnsw_search_layer. * * The replacement branch (lines 342-344) fires when: @@ -2002,6 +2029,7 @@ const test_entry_t embedding_entries[] = { { "embedding/hnsw_mmap_load", test_hnsw_mmap_load, emb_setup, emb_teardown }, { "embedding/hnsw_build_overflow_rejected", test_hnsw_build_overflow_rejected, emb_setup, emb_teardown }, { "embedding/hnsw_vec_size_valid_guard", test_hnsw_vec_size_valid_guard, emb_setup, emb_teardown }, + { "embedding/hnsw_load_rejects_bad_neighbor", test_hnsw_load_rejects_bad_neighbor, emb_setup, emb_teardown }, { "embedding/hnsw_search_sift_down", test_hnsw_search_sift_down, emb_setup, emb_teardown }, /* rerank coverage (S7) */ @@ -2051,4 +2079,3 @@ const test_entry_t embedding_entries[] = { { NULL, NULL, NULL, NULL }, }; - diff --git a/test/test_expr_null.c b/test/test_expr_null.c index 89f7fa674..7a2601fe7 100644 --- a/test/test_expr_null.c +++ b/test/test_expr_null.c @@ -324,18 +324,24 @@ static test_result_t test_nullfree_promotion_invariance(void) { ray_expr_t ex; TEST_ASSERT(expr_compile(g, tbl, build_i64_plus_f64(g), &ex), "null-free promotion compiles"); - /* No INPUT is nullable, so the compiler marks no instruction null_aware and - * no reg nullable. Single-null float model: an F64 producer that may yield - * 0Nf from finite inputs (overflow) gets HAS_NULLS via a PRECISE post-scan - * in expr_eval_full (expr_last_op_produces_f64_null + - * mark_f64_nonfinite_as_null) — NOT the compile-time nullable flag — so - * this null-free-promotion compile invariant is preserved unchanged. */ + /* No INPUT is nullable, so the compiler marks no instruction null_aware + * and no reg null_src (column-derived nullability). Single-null float + * model: the F64 ADD may yield 0Nf from finite inputs (overflow), so its + * destination IS nullable — downstream kernels would have to honour the + * sentinel — but HAS_NULLS on the output comes from a PRECISE post-scan + * in expr_eval_full (expr_flag_output_nulls + mark_f64_nonfinite_as_null), + * NOT from a conservative flag, so a finite result stays null-free. */ for (uint8_t i = 0; i < ex.n_ins; i++) TEST_ASSERT(ex.ins[i].null_aware == 0, "no null_aware on null-free promotion"); for (uint8_t r2 = 0; r2 < ex.n_regs; r2++) - TEST_ASSERT(!ex.regs[r2].nullable, - "no nullable regs on null-free promotion"); + TEST_ASSERT(!ex.regs[r2].null_src, + "no column-derived nullable regs on null-free promotion"); + for (uint8_t r2 = 0; r2 < ex.n_regs; r2++) + if (ex.regs[r2].kind != REG_SCRATCH || r2 == ex.out_reg) continue; + else TEST_ASSERT(!ex.regs[r2].nullable, "promotion cast is not nullable"); + TEST_ASSERT(ex.regs[ex.out_reg].nullable, + "F64 ADD destination is nullable (overflow -> 0Nf generator)"); ray_graph_free(g); ray_release(tbl); ray_sym_destroy(); ray_heap_destroy(); diff --git a/test/test_index.c b/test/test_index.c index af3a02135..bcba719ff 100644 --- a/test/test_index.c +++ b/test/test_index.c @@ -26,6 +26,8 @@ #include "test.h" #include #include "mem/heap.h" +#include "mem/sys.h" +#include "core/pool.h" #include "mem/cow.h" #include "vec/vec.h" #include "table/sym.h" @@ -33,6 +35,9 @@ #include "ops/rowsel.h" #include "store/col.h" #include +#include +#include +#include #include #include #include @@ -327,6 +332,83 @@ static test_result_t test_index_hash_with_nulls_preserved(void) { PASS(); } +/* Large column: the build runs partition-parallel above 64k rows and must + * produce the serial walk's layout — groups in first-occurrence order, rows + * ascending inside a group, nulls excluded — checked against a reference + * computed the obvious way. */ +static test_result_t test_index_hash_large_parallel(void) { + ray_heap_init(); + /* The parallel build needs the pool; create it before the attach so + * the test does not silently take the serial fallback. */ + ray_pool_t* pool = ray_pool_get(); + TEST_ASSERT_NOT_NULL(pool); + const int64_t n = 300000, kmax = 5003; + ray_t* v = ray_vec_new(RAY_I64, n); + TEST_ASSERT_NOT_NULL(v); + int64_t* xs = (int64_t*)ray_data(v); + for (int64_t i = 0; i < n; i++) + xs[i] = (int64_t)(((uint64_t)i * 2654435761ull) % (uint64_t)kmax) - 17; + v->len = n; + /* every 977th row null */ + for (int64_t i = 0; i < n; i += 977) + TEST_ASSERT_EQ_I(ray_vec_set_null_checked(v, i, true), RAY_OK); + + /* reference: first-occurrence group ids and counts */ + int64_t* gid_of_key = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + int64_t* ref_key = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + int64_t* ref_cnt = (int64_t*)ray_sys_alloc((size_t)kmax * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(gid_of_key); TEST_ASSERT_NOT_NULL(ref_key); TEST_ASSERT_NOT_NULL(ref_cnt); + for (int64_t k = 0; k < kmax; k++) { gid_of_key[k] = -1; ref_cnt[k] = 0; } + int64_t ref_groups = 0, ref_keys = 0; + for (int64_t i = 0; i < n; i++) { + if (ray_vec_is_null(v, i)) continue; + int64_t k = xs[i] + 17; + if (gid_of_key[k] < 0) { gid_of_key[k] = ref_groups; ref_key[ref_groups++] = xs[i]; } + ref_cnt[gid_of_key[k]]++; + ref_keys++; + } + + ray_t* w = v; + ray_t* r = ray_index_attach_hash(&w); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + ray_index_t* ix = ray_index_payload(w->index); + TEST_ASSERT_EQ_I((int)ix->kind, RAY_IDX_HASH); + TEST_ASSERT_EQ_I(ix->u.hash.n_keys, ref_keys); + TEST_ASSERT_EQ_I(ix->u.hash.n_groups, ref_groups); + const int64_t* gk = (const int64_t*)ray_data(ix->u.hash.gkeys); + const int64_t* of = (const int64_t*)ray_data(ix->u.hash.offs); + const int64_t* rw = (const int64_t*)ray_data(ix->u.hash.rows); + TEST_ASSERT_EQ_I(of[0], 0); + TEST_ASSERT_EQ_I(of[ref_groups], ref_keys); + /* Serial layout: groups in first-occurrence order, each with its count, + * rows ascending and all storing the group's key. */ + for (int64_t g = 0; g < ref_groups; g++) { + TEST_ASSERT_EQ_I(gk[g], ref_key[g]); + TEST_ASSERT_EQ_I(of[g + 1] - of[g], ref_cnt[g]); + for (int64_t j = of[g]; j < of[g + 1]; j++) { + TEST_ASSERT_EQ_I(xs[rw[j]], ref_key[g]); + if (j > of[g]) TEST_ASSERT_TRUE(rw[j] > rw[j - 1]); + } + } + /* table probes: every key resolves to its group, an absent key misses */ + for (int64_t k = 0; k < kmax; k += 61) { + const int64_t* grows = NULL; + int64_t gn = 0; + TEST_ASSERT_EQ_I(ray_index_hash_group(w, k - 17, &grows, &gn), 1); + TEST_ASSERT_EQ_I(gn, ref_cnt[gid_of_key[k]]); + } + { + const int64_t* grows = NULL; + int64_t gn = 0; + TEST_ASSERT_EQ_I(ray_index_hash_group(w, kmax + 1000, &grows, &gn), 0); + } + + ray_sys_free(gid_of_key); ray_sys_free(ref_key); ray_sys_free(ref_cnt); + ray_release(w); + ray_heap_destroy(); + PASS(); +} + /* ─── Sort index ──────────────────────────────────────────────────── */ static test_result_t test_index_sort_attach_drop(void) { @@ -502,6 +584,67 @@ static test_result_t test_index_persistence_roundtrip(void) { PASS(); } +/* ─── Mapped column drops its own index: the whole mapping is unmapped ── + * + * A column loaded by mmap with an inline index region is longer than its + * payload. Dropping the index from the loaded column itself (the sole + * reference: an in-place edit does exactly this) used to leave the index + * tail mapped for the life of the process, because ray_free sized the + * unmap from the index it no longer had. The file is laid out so the + * region crosses into a page of its own; after the free that page must + * be gone (msync reports ENOMEM on an unmapped range). */ +static test_result_t test_index_mapped_drop_unmaps_tail(void) { + ray_heap_init(); + /* Lay the file out so the inline index region crosses into a page of + * its own whatever the page size (4 KiB on Linux, 16 KiB on Apple + * silicon): the payload ends 64 bytes short of the second page. */ + long pg = sysconf(_SC_PAGESIZE); + TEST_ASSERT_TRUE(pg >= 4096); + int64_t n = (2 * (int64_t)pg - 96) / 8; + ray_t* v = ray_vec_new(RAY_I64, n); + for (int64_t i = 0; i < n; i++) { int64_t x = i * 3; v = ray_vec_append(v, &x); } + TEST_ASSERT_FALSE(RAY_IS_ERR(v)); + ray_t* w = v; + TEST_ASSERT_FALSE(RAY_IS_ERR(ray_index_attach_chunk_zone(&w, 8))); + + char path[] = "/tmp/idx_drop_unmap_XXXXXX"; + int fd = mkstemp(path); + TEST_ASSERT_TRUE(fd >= 0); + close(fd); + TEST_ASSERT_EQ_I(ray_col_save(w, path), RAY_OK); /* writes the inline index region too */ + ray_release(w); + + struct stat st; + TEST_ASSERT_EQ_I(stat(path, &st), 0); + TEST_ASSERT_TRUE(st.st_size > 2 * pg); /* the region reaches a further page */ + size_t mapped = ((size_t)st.st_size + (size_t)pg - 1) & ~((size_t)pg - 1); + + ray_t* m = ray_col_mmap(path); + TEST_ASSERT_FALSE(RAY_IS_ERR(m)); + TEST_ASSERT_EQ_U(m->mmod, 1); + TEST_ASSERT_TRUE(m->attrs & RAY_ATTR_HAS_INDEX); + TEST_ASSERT_EQ_I((int)ray_index_payload(m->index)->kind, RAY_IDX_CHUNK_ZONE); + char* last_page = (char*)m + mapped - (size_t)pg; + TEST_ASSERT_EQ_I(msync(last_page, (size_t)pg, MS_ASYNC), 0); /* mapped while loaded */ + + /* Sole reference: the drop detaches the mapped index in place. */ + ray_t* d = m; + ray_t* r = ray_index_drop(&d); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + TEST_ASSERT_TRUE(d == m); + TEST_ASSERT_FALSE(d->attrs & RAY_ATTR_HAS_INDEX); + int64_t* data = (int64_t*)ray_data(d); + TEST_ASSERT_EQ_I(data[n - 1], (n - 1) * 3); + + ray_release(d); + errno = 0; + int rc = msync(last_page, (size_t)pg, MS_ASYNC); + TEST_ASSERT_TRUE(rc == -1 && errno == ENOMEM); /* the tail page is unmapped */ + unlink(path); + ray_heap_destroy(); + PASS(); +} + /* ─── Slice null detection on indexed/parent vec ───────────────────── */ static test_result_t test_index_aux_helper_slice(void) { @@ -3697,6 +3840,7 @@ const test_entry_t index_entries[] = { { "index/unsupported_type", test_index_unsupported_type, NULL, NULL }, { "index/hash_attach_drop", test_index_hash_attach_drop, NULL, NULL }, { "index/hash_with_nulls_preserved", test_index_hash_with_nulls_preserved, NULL, NULL }, + { "index/hash_large_parallel", test_index_hash_large_parallel, NULL, NULL }, { "index/sort_attach_drop", test_index_sort_attach_drop, NULL, NULL }, { "index/bloom_attach_drop", test_index_bloom_attach_drop, NULL, NULL }, { "index/replace_cross_kind", test_index_replace_cross_kind, NULL, NULL }, @@ -3704,6 +3848,7 @@ const test_entry_t index_entries[] = { { "index/null_readers_through_helper", test_index_null_readers_through_helper, NULL, NULL }, { "index/aux_helper_slice", test_index_aux_helper_slice, NULL, NULL }, { "index/drop_under_shared_cow", test_index_drop_under_shared_cow, NULL, NULL }, + { "index/mapped_drop_unmaps_tail", test_index_mapped_drop_unmaps_tail, NULL, NULL }, { "index/persistence_roundtrip", test_index_persistence_roundtrip, NULL, NULL }, { "index/bool_zone_and_hash", test_index_bool_zone_and_hash, NULL, NULL }, { "index/i16_zone_and_hash", test_index_i16_zone_and_hash, NULL, NULL }, diff --git a/test/test_splay.c b/test/test_splay.c index 34188b468..d037cf5eb 100644 --- a/test/test_splay.c +++ b/test/test_splay.c @@ -40,6 +40,10 @@ #include "lang/internal.h" /* ray_set/get_splayed_fn (surface resolver) */ #include "mem/heap.h" #include "table/sym.h" +#include "table/domain.h" /* symfile positions of a streamed CSV load */ +#include "io/csv.h" /* ray_csv_save_splayed_named_opts */ +#include "mem/sys.h" +#include "core/pool.h" #include #include #include @@ -68,6 +72,139 @@ static void rm_rf(const char* path) { (void)ray_test_rm_rf(path); } +/* Streamed CSV -> splayed load, several chunks. The symfile positions of + * the SYM columns are the ones the cell-by-cell writer gives: per chunk, + * columns in order, each column's strings by first occurrence — whatever + * the worker count or how the dictionaries were split into partitions. + * Chunks of 20k rows with >4k new strings each take the parallel batch. */ +static test_result_t test_csv_splayed_symfile_order(void) { + TEST_ASSERT_NOT_NULL(ray_pool_get()); + const char* dir = TMP_SPLAY_BASE "/csvorder"; + const char* csv = TMP_SPLAY_BASE "/csvorder.csv"; + rm_rf(dir); + mkdir(TMP_SPLAY_BASE, 0755); + + enum { NROWS = 60000, CHUNK = 20000, NA = 13000, NB = 9000 }; + FILE* f = fopen(csv, "wb"); + TEST_ASSERT_NOT_NULL(f); + fputs("a,b,v\n", f); + for (int r = 0; r < NROWS; r++) { + int ai = (int)(((int64_t)r * 7919) % NA); + if (r % 5 == 0) fprintf(f, "a%d,a%d,%d\n", ai, (ai + 11) % NA, r); /* b reuses a's strings */ + else fprintf(f, "a%d,b%d,%d\n", ai, (int)(((int64_t)r * 104729) % NB), r); + } + fclose(f); + + int8_t types[] = { RAY_SYM, RAY_SYM, RAY_I64 }; + ray_err_t err = ray_csv_save_splayed_named_opts(csv, ',', true, types, 3, NULL, 0, dir, CHUNK); + TEST_ASSERT_EQ_I(err, RAY_OK); + + /* expected positions: walk as the writer does */ + int64_t* pa = (int64_t*)ray_sys_alloc(NA * sizeof(int64_t)); + int64_t* pb = (int64_t*)ray_sys_alloc(NB * sizeof(int64_t)); + TEST_ASSERT_NOT_NULL(pa); TEST_ASSERT_NOT_NULL(pb); + for (int i = 0; i < NA; i++) pa[i] = -1; + for (int i = 0; i < NB; i++) pb[i] = -1; + int64_t next = 1; /* 0 is "" */ + for (int c0 = 0; c0 < NROWS; c0 += CHUNK) { + for (int r = c0; r < c0 + CHUNK; r++) { /* column a */ + int ai = (int)(((int64_t)r * 7919) % NA); + if (pa[ai] < 0) pa[ai] = next++; + } + for (int r = c0; r < c0 + CHUNK; r++) { /* column b */ + int ai = (int)(((int64_t)r * 7919) % NA); + if (r % 5 == 0) { int x = (ai + 11) % NA; if (pa[x] < 0) pa[x] = next++; } + else { int bi = (int)(((int64_t)r * 104729) % NB); if (pb[bi] < 0) pb[bi] = next++; } + } + } + + char sym_path[256]; + snprintf(sym_path, sizeof(sym_path), "%s/.sym", dir); + ray_sym_domain_t* dom = ray_sym_domain_open(sym_path); + TEST_ASSERT_NOT_NULL(dom); + TEST_ASSERT_EQ_I(ray_sym_domain_count(dom), next); + char buf[32]; + for (int i = 0; i < NA; i++) { + if (pa[i] < 0) continue; + int n = snprintf(buf, sizeof(buf), "a%d", i); + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, buf, (size_t)n), pa[i]); + } + for (int i = 0; i < NB; i++) { + if (pb[i] < 0) continue; + int n = snprintf(buf, sizeof(buf), "b%d", i); + TEST_ASSERT_EQ_I(ray_sym_domain_find(dom, buf, (size_t)n), pb[i]); + } + ray_sym_domain_release(dom); + + /* and the cells read back the strings written */ + ray_t* t = ray_read_splayed(dir, sym_path); + TEST_ASSERT_FALSE(RAY_IS_ERR(t)); + TEST_ASSERT_EQ_I(ray_table_nrows(t), NROWS); + ray_t* ca = ray_table_get_col_idx(t, 0); + for (int r = 0; r < NROWS; r += 997) { + ray_t* cell = ray_sym_vec_cell(ca, r); + TEST_ASSERT_NOT_NULL(cell); + int n = snprintf(buf, sizeof(buf), "a%d", (int)(((int64_t)r * 7919) % NA)); + TEST_ASSERT_EQ_U(ray_str_len(cell), (size_t)n); + TEST_ASSERT_MEM_EQ((size_t)n, ray_str_ptr(cell), buf); + } + ray_release(t); + ray_sys_free(pa); ray_sys_free(pb); + rm_rf(dir); + unlink(csv); + PASS(); +} + +/* A chunk whose byte window holds no quote, in a file that has quotes in + * another chunk, is split into rows exactly as the whole file is: a lone + * '\r' ends a row. (The quote-free fast path of the parallel scanner did + * not treat it so, and merged two rows, losing a value.) */ +static test_result_t test_csv_splayed_quote_mode_per_file(void) { + TEST_ASSERT_NOT_NULL(ray_pool_get()); + const char* dir = TMP_SPLAY_BASE "/csvquote"; + const char* csv = TMP_SPLAY_BASE "/csvquote.csv"; + rm_rf(dir); + mkdir(TMP_SPLAY_BASE, 0755); + + enum { NROWS = 60000, CHUNK = 20000, LONE = 45007 }; + FILE* f = fopen(csv, "wb"); + TEST_ASSERT_NOT_NULL(f); + fputs("s,v\n", f); + fputs("\"q,1\",0\n", f); /* quotes only in chunk 0 */ + for (int r = 1; r < NROWS; r++) { + if (r == LONE) fprintf(f, "left\rright,%d\n", r); /* chunk 2 */ + else fprintf(f, "r%d,%d\n", r, r); + } + fclose(f); + + int8_t types[] = { RAY_SYM, RAY_I64 }; + ray_t* mem = ray_read_csv_named_opts(csv, ',', true, types, 2, NULL, 0); + TEST_ASSERT_FALSE(RAY_IS_ERR(mem)); + int64_t want = ray_table_nrows(mem); + TEST_ASSERT_EQ_I(want, NROWS + 1); /* the lone \r splits a row */ + + ray_err_t err = ray_csv_save_splayed_named_opts(csv, ',', true, types, 2, NULL, 0, dir, CHUNK); + TEST_ASSERT_EQ_I(err, RAY_OK); + char sym_path[256]; + snprintf(sym_path, sizeof(sym_path), "%s/.sym", dir); + ray_t* t = ray_read_splayed(dir, sym_path); + TEST_ASSERT_FALSE(RAY_IS_ERR(t)); + TEST_ASSERT_EQ_I(ray_table_nrows(t), want); + /* every row agrees with the in-memory read */ + ray_t* vm = ray_table_get_col_idx(mem, 1); + ray_t* vt = ray_table_get_col_idx(t, 1); + for (int64_t r = 0; r < want; r++) { + TEST_ASSERT_EQ_I(ray_vec_is_null(vt, r), ray_vec_is_null(vm, r)); + if (!ray_vec_is_null(vm, r)) + TEST_ASSERT_EQ_I(((int64_t*)ray_data(vt))[r], ((int64_t*)ray_data(vm))[r]); + } + ray_release(t); + ray_release(mem); + rm_rf(dir); + unlink(csv); + PASS(); +} + /* ========================================================================= * 1. ray_splay_save: NULL dir → RAY_ERR_IO * ========================================================================= */ @@ -2277,5 +2414,7 @@ const test_entry_t splay_entries[] = { { "splay/empty_sym_table_roundtrip", test_empty_sym_table_roundtrip, splay_setup, splay_teardown }, { "splay/resolution_order_independence", test_resolution_order_independence, splay_setup, splay_teardown }, { "splay/resolution_explicit_wins", test_resolution_explicit_wins, splay_setup, splay_teardown }, + { "splay/csv_symfile_order", test_csv_splayed_symfile_order, splay_setup, splay_teardown }, + { "splay/csv_quote_mode_per_file", test_csv_splayed_quote_mode_per_file, splay_setup, splay_teardown }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_vec.c b/test/test_vec.c index 458adf7c4..986e537e7 100644 --- a/test/test_vec.c +++ b/test/test_vec.c @@ -25,6 +25,7 @@ #include #include "mem/heap.h" #include "vec/vec.h" +#include "vec/str.h" #include "vec/embedding.h" #include "table/sym.h" #include "core/platform.h" @@ -1782,6 +1783,94 @@ static test_result_t test_vec_concat_str_null(void) { PASS(); } +/* Canonical text payloads must propagate even without HAS_NULLS. Exercise + * both operands, chunk/tail boundaries, all SYM width pairs and slices whose + * parent has nulls outside the selected range. */ +static test_result_t test_vec_concat_text_null_scan(void) { + const uint8_t widths[] = {RAY_SYM_W8, RAY_SYM_W16, RAY_SYM_W32, RAY_SYM_W64}; + const int64_t positions[] = {-1, 0, 255, 256, 258}; + for (int kind = 0; kind < 2; kind++) { + for (int wa = 0; wa < (kind ? 4 : 1); wa++) { + for (int wb = 0; wb < (kind ? 4 : 1); wb++) { + for (int side = 0; side < 2; side++) { + for (int p = 0; p < 5; p++) { + ray_t* v[2]; + for (int s = 0; s < 2; s++) { + v[s] = kind ? ray_sym_vec_new(widths[s ? wb : wa], 261) + : ray_vec_new(RAY_STR, 261); + TEST_ASSERT_NOT_NULL(v[s]); + TEST_ASSERT_FALSE(RAY_IS_ERR(v[s])); + if (kind) { + v[s]->len = 261; + for (int64_t i = 0; i < 261; i++) + ray_write_sym(ray_data(v[s]), i, 1, RAY_SYM, v[s]->attrs); + } else { + for (int64_t i = 0; i < 261; i++) + v[s] = ray_str_vec_append(v[s], "a pooled string longer than inline", 33); + } + ray_vec_set_null(v[s], 0, true); + ray_vec_set_null(v[s], 260, true); + if (s == side && positions[p] >= 0) + ray_vec_set_null(v[s], positions[p] + 1, true); + v[s]->attrs &= (uint8_t)~RAY_ATTR_HAS_NULLS; + } + ray_t* a = ray_vec_slice(v[0], 1, 259); + ray_t* b = ray_vec_slice(v[1], 1, 259); + ray_t* c = ray_vec_concat(a, b); + TEST_ASSERT_NOT_NULL(c); + TEST_ASSERT_FALSE(RAY_IS_ERR(c)); + TEST_ASSERT_EQ_I(c->len, 518); + TEST_ASSERT_EQ_I(!!(c->attrs & RAY_ATTR_HAS_NULLS), positions[p] >= 0); + TEST_ASSERT_FALSE(c->attrs & (RAY_ATTR_SLICE | RAY_ATTR_HAS_INDEX | RAY_ATTR_SORTED)); + /* Result owns its pool/domain independently of inputs. */ + ray_release(a); ray_release(b); + ray_release(v[0]); ray_release(v[1]); + for (int64_t i = 0; i < c->len; i++) { + bool is_null = positions[p] >= 0 && i == side * 259 + positions[p]; + TEST_ASSERT_EQ_I(ray_vec_is_null(c, i), is_null); + if (kind) { + TEST_ASSERT_EQ_I(ray_read_sym(ray_data(c), i, RAY_SYM, c->attrs), !is_null); + } else if (!is_null) { + size_t len; + const char* str = ray_str_vec_get(c, i, &len); + TEST_ASSERT_EQ_U(len, 33); + TEST_ASSERT_MEM_EQ(33, str, "a pooled string longer than inline"); + } else { + ray_str_t zero = {0}; + TEST_ASSERT_MEM_EQ(sizeof(zero), &((ray_str_t*)ray_data(c))[i], &zero); + } + } + ray_release(c); + } + } + } + } + } + PASS(); +} + +static test_result_t test_vec_concat_text_empty(void) { + for (int kind = 0; kind < 2; kind++) { + ray_t* empty = ray_vec_new(kind ? RAY_SYM : RAY_STR, 0); + ray_t* nulls = ray_vec_new(kind ? RAY_SYM : RAY_STR, 2); + nulls->len = 2; + ray_vec_set_null(nulls, 0, true); + ray_vec_set_null(nulls, 1, true); + nulls->attrs &= (uint8_t)~RAY_ATTR_HAS_NULLS; + for (int side = 0; side < 3; side++) { + ray_t* c = ray_vec_concat(side == 0 ? nulls : empty, side == 1 ? nulls : empty); + TEST_ASSERT_NOT_NULL(c); + TEST_ASSERT_FALSE(RAY_IS_ERR(c)); + TEST_ASSERT_EQ_I(c->len, side == 2 ? 0 : 2); + TEST_ASSERT_EQ_I(!!(c->attrs & RAY_ATTR_HAS_NULLS), side != 2); + for (int64_t i = 0; i < c->len; i++) TEST_ASSERT_TRUE(ray_vec_is_null(c, i)); + ray_release(c); + } + ray_release(nulls); ray_release(empty); + } + PASS(); +} + /* ---- str_vec_append: pool growth across many large strings ------------- */ static test_result_t test_str_vec_append_pool_grow(void) { @@ -2372,6 +2461,8 @@ const test_entry_t vec_entries[] = { { "vec/concat_empty", test_vec_concat_empty, vec_setup, vec_teardown }, { "vec/concat_str", test_vec_concat_str, vec_setup, vec_teardown }, { "vec/concat_str_null", test_vec_concat_str_null, vec_setup, vec_teardown }, + { "vec/concat_text_null_scan", test_vec_concat_text_null_scan, vec_setup, vec_teardown }, + { "vec/concat_text_empty", test_vec_concat_text_empty, vec_setup, vec_teardown }, { "vec/str_append_pool_grow", test_str_vec_append_pool_grow, vec_setup, vec_teardown }, { "vec/str_set_paths", test_str_vec_set_paths, vec_setup, vec_teardown }, { "vec/str_guards", test_str_vec_guards, vec_setup, vec_teardown },