diff --git a/src/mem/heap.c b/src/mem/heap.c index 452440569..02e41fe55 100644 --- a/src/mem/heap.c +++ b/src/mem/heap.c @@ -1683,8 +1683,11 @@ void ray_free(ray_t* v) { * pool before appending — has to ask the registry, and only a * string column was ever registered. So the lock stays off the * ordinary free entirely, and a mutated column pays it once. */ + /* A column loaded with an inline index registered its region too + * (col.c): the index may have been detached since, and only the + * descriptor still knows the mapped length. */ ray_file_map_t* m = col_map; - if (!m && v->type == RAY_STR) m = ray_file_map_lookup(v); + if (!m) m = ray_file_map_lookup(v); if (m) { ray_file_map_release(m); if (h) RAY_STAT(h->stats.free_count++); @@ -2942,9 +2945,14 @@ void ray_parallel_end(void) { * else is the page-rounded payload plus an inline passenger index. */ static size_t mapped_block_bytes(const ray_t* v) { if (v->type == RAY_TABLE || v->type == RAY_DICT || v->type == RAY_LIST) return 0; - if (v->type == RAY_STR) { + { + /* Same route as ray_free: the pool's descriptor for a string + * column, else the registry — which also holds every indexed + * mapped column (col.c), whether or not its index is still + * attached. */ ray_file_map_t* m = NULL; - if (v->str_pool && !RAY_IS_ERR(v->str_pool) && v->str_pool->mmod == 3) + if (v->type == RAY_STR && v->str_pool && !RAY_IS_ERR(v->str_pool) && + v->str_pool->mmod == 3) m = v->str_pool->file_map; if (!m) m = ray_file_map_lookup(v); if (m) return m->len; diff --git a/src/ops/agg.c b/src/ops/agg.c index 138afb654..fe7d7f9d4 100644 --- a/src/ops/agg.c +++ b/src/ops/agg.c @@ -169,12 +169,14 @@ static ray_t* agg_parted_avg(ray_t* x) { if (!agg_parted_numeric_base(base)) return ray_error("type", "avg expects a numeric or temporal parted column, got %s", ray_type_name(base)); ray_t** segs = (ray_t**)ray_data(x); double sum = 0.0; + int64_t hi = 0; uint64_t lo = 0; /* integer segments: exact 128-bit sum */ int64_t cnt = 0; + bool fp = base == RAY_F64 || base == RAY_F32; for (int64_t s = 0; s < x->len; s++) { ray_t* seg = segs[s]; if (!seg) continue; int has_nulls = ray_vec_may_have_nulls(seg); - if (base == RAY_F64 || base == RAY_F32) { + if (fp) { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; if (base == RAY_F64) sum += ((double*)ray_data(seg))[i]; @@ -184,12 +186,12 @@ static ray_t* agg_parted_avg(ray_t* x) { } else { for (int64_t i = 0; i < seg->len; i++) { if (has_nulls && ray_vec_is_null(seg, i)) continue; - sum += (double)agg_read_i64(seg, i); cnt++; + ray_i128_add(&hi, &lo, agg_read_i64(seg, i)); cnt++; } } } if (cnt == 0) return ray_typed_null(-RAY_F64); - return make_f64(sum / (double)cnt); + return make_f64((fp ? sum : ray_i128_to_f64(hi, lo)) / (double)cnt); } static ray_t* agg_parted_prod(ray_t* x) { @@ -404,36 +406,63 @@ static ray_t* agg_pair_vec(ray_t* x, ray_t* y, uint16_t op) { return ray_f64(ray_f64_fin(num / sqrt(dx * dy))); } -/* Whole-column sum / non-null count of an integer column from its - * chunk-zone index (per-chunk int64 sums with wraparound and non-null - * counts), in O(n_chunks). `exact_f64` reports whether every partial sum - * an accumulation in double could meet stays below 2^53 in magnitude, i.e. - * the sum converted to double equals any double accumulation of the rows. - * Returns false when the column has no such index for its current length. */ -bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64) { - if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return false; +/* Per-chunk aggregates of an integer column's chunk-zone index, or NULL: + * [sum low words | non-null counts | sum high words] (the last block only + * in the current layout). */ +static const int64_t* zone_aggs(ray_t* x, uint32_t* n_out, bool* have_hi) { + if (!x || !ray_is_vec(x) || ray_index_kind(x) != RAY_IDX_CHUNK_ZONE) return NULL; ray_index_t* ix = ray_index_payload(x->index); if (ix->built_for_len != x->len || ix->u.chunk_zone.is_f64 || !ix->u.chunk_zone.aggs) - return false; + return NULL; uint32_t n = ix->u.chunk_zone.n_chunks; - if (ix->u.chunk_zone.aggs->len != 2 * (int64_t)n) return false; - const int64_t* ag = (const int64_t*)ray_data(ix->u.chunk_zone.aggs); - const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); - const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + int64_t len = ix->u.chunk_zone.aggs->len; + if (len != 2 * (int64_t)n && len != 3 * (int64_t)n) return NULL; + *n_out = n; + *have_hi = len == 3 * (int64_t)n; + return (const int64_t*)ray_data(ix->u.chunk_zone.aggs); +} + +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; uint64_t sum = 0; int64_t nn = 0; - double bound = 0.0; for (uint32_t g = 0; g < n; g++) { sum += (uint64_t)ag[g]; nn += ag[n + g]; - if (ag[n + g] > 0) { - double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); - bound += (a > b ? a : b) * (double)ag[n + g]; - } } *sum_out = (int64_t)sum; *nn_out = nn; - if (exact_f64) *exact_f64 = bound < 9007199254740992.0; /* 2^53 */ + return true; +} + +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out) { + uint32_t n; bool have_hi; + const int64_t* ag = zone_aggs(x, &n, &have_hi); + if (!ag) return false; + ray_index_t* ix = ray_index_payload(x->index); + const int64_t* mins = (const int64_t*)ray_data(ix->u.chunk_zone.mins); + const int64_t* maxs = (const int64_t*)ray_data(ix->u.chunk_zone.maxs); + int64_t hi = 0; uint64_t lo = 0; int64_t nn = 0; + double bound = 0.0; + for (uint32_t g = 0; g < n; g++) { + int64_t cnt = ag[n + g]; + nn += cnt; + if (have_hi) { + ray_i128_add128(&hi, &lo, ag[2 * (int64_t)n + g], (uint64_t)ag[g]); + } else { + /* no high words stored: the wrapped low word is the chunk's + * exact sum only when no partial sum could leave int64 */ + if (cnt > 0) { + double a = fabs((double)mins[g]), b = fabs((double)maxs[g]); + bound += (a > b ? a : b) * (double)cnt; + } + ray_i128_add(&hi, &lo, ag[g]); + } + } + if (!have_hi && !(bound < 9223372036854775808.0)) return false; /* 2^63 */ + *hi_out = hi; *lo_out = lo; *nn_out = nn; return true; } @@ -453,7 +482,7 @@ ray_t* ray_sum_fn(ray_t* x) { /* Integer columns with per-chunk sums in their zone index. */ if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { int64_t zs, zn; - if (ray_zone_int_sum(x, &zs, &zn, NULL)) return make_i64(zs); + if (ray_zone_int_sum(x, &zs, &zn)) return make_i64(zs); } /* Narrow/temporal types need specific return constructors that the * DAG executor doesn't provide — use scalar path for these. TIMESTAMP @@ -608,14 +637,14 @@ ray_t* ray_avg_fn(ray_t* x) { /* Canonical admission: numeric + temporal (→ F64); SYM/STR/GUID are * non-numeric → type error (the DAG path otherwise averaged raw ids). */ if (!agg_type_admitted(OP_AVG, x->type)) return ray_error("type", "avg expects a numeric or temporal vector, got %s", ray_type_name(x->type)); - /* Integer columns with per-chunk sums: exact when no partial sum can - * leave double's integer range (then every accumulation order in - * double gives the same value). */ + /* Integer columns with per-chunk sums: the exact 128-bit total is + * what the row-wise reduction computes too, so the answers agree + * bit for bit. */ if (x->type == RAY_I64 || x->type == RAY_I32 || x->type == RAY_I16 || x->type == RAY_U8) { - int64_t zs, zn; bool exact = false; - if (ray_zone_int_sum(x, &zs, &zn, &exact) && exact) { + int64_t zh, zn; uint64_t zl; + if (ray_zone_int_sum128(x, &zh, &zl, &zn)) { if (zn == 0) return ray_typed_null(-RAY_F64); - return make_f64((double)zs / (double)zn); + return make_f64(ray_i128_to_f64(zh, zl) / (double)zn); } } AGG_VEC_VIA_DAG(x, ray_avg); @@ -663,7 +692,7 @@ ray_t* ray_min_fn(ray_t* x) { * zone has them; the sentinel alone cannot tell a * column of INT64_MAX values from an empty one. */ int64_t zs_, zn_; - bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); if (have_nn ? zn_ == 0 : mn == INT64_MAX) return ray_typed_null(-x->type); /* Preserve the column's storage width on the result. */ switch (x->type) { @@ -723,7 +752,7 @@ ray_t* ray_max_fn(ray_t* x) { * zone has them; the sentinel alone cannot tell a * column of INT64_MIN values from an empty one. */ int64_t zs_, zn_; - bool have_nn = ray_zone_int_sum(x, &zs_, &zn_, NULL); + bool have_nn = ray_zone_int_sum(x, &zs_, &zn_); if (have_nn ? zn_ == 0 : mx == INT64_MIN) return ray_typed_null(-x->type); switch (x->type) { case RAY_BOOL: return ray_bool((bool)mx); diff --git a/src/ops/agg_engine.c b/src/ops/agg_engine.c index 81ffe6651..984faff60 100644 --- a/src/ops/agg_engine.c +++ b/src/ops/agg_engine.c @@ -184,6 +184,14 @@ agg_v2_reason_t agg_v2_admission(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { const agg_vtable_t* vt = agg_resolve(ext->agg_ops[a], ic->type); if (!vt) return AGG_V2_AGG_TYPE; if (vt->kind != ACC_STREAMING) return AGG_V2_BUFFERED; + /* The streaming 64-bit integer avg packs its 128-bit high word with + * the group count in one word (agg_stream.c): exact while every + * group has fewer than 2^32 rows; the narrower kernels keep a plain + * int64 sum, exact below 2^31 rows. A table that large keeps the + * legacy engines, whose accumulators carry a full high word. */ + if (ext->agg_ops[a] == OP_AVG && ic->type != RAY_F64 && ic->type != RAY_F32 && + ray_table_nrows(tbl) >= ((int64_t)1 << ((ic->type == RAY_I64 || ic->type == RAY_TIMESTAMP) ? 32 : 31))) + return AGG_V2_AGG_TYPE; } return AGG_V2_ADMITTED; } diff --git a/src/ops/agg_stream.c b/src/ops/agg_stream.c index 8912d1860..083b5a3c9 100644 --- a/src/ops/agg_stream.c +++ b/src/ops/agg_stream.c @@ -5,6 +5,7 @@ #include "ops/ops.h" #include "ops/internal.h" /* ray_f64_fin (single-null float model) */ #include "lang/internal.h" /* ray_median_dbl_inplace */ +#include "ops/idxop.h" /* ray_i128_add / ray_i128_to_f64: exact integer avg */ #include #include #include /* realloc/free for the buffered median accumulator */ @@ -311,6 +312,80 @@ static const agg_vtable_t AVG_F64 = { .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, }; +/* ---- avg, integer inputs: exact 128-bit sum --------------------------- + * Every integer/temporal avg kernel below keeps the exact sum and divides + * ray_i128_to_f64(hi, lo) by the count — the same bits as the keyless + * reduction, the legacy group engines and the chunk-zone metadata, for any + * column (a double running sum loses low bits past 2^53 and an int64 one + * wraps). + * + * The state stays 16 bytes (the dense plans budget slot traffic by state + * size): `lo` is a signed int64 running sum and `hc` packs, as hi:32 | + * cnt:32, the number of times that sum wrapped past +/-2^63 with the row + * count — the total is wraps * 2^64 + lo. A signed-overflow test per row + * is a never-taken branch on ordinary data, cheaper than a carry chain. + * Exact for any int64 values while a group holds fewer than 2^32 rows + * (|sum| < 2^95, so the wrap count fits 32 signed bits); agg_v2_admission + * keeps tables of 2^32 rows or more off the v2 engine when an integer avg + * is present. */ +typedef struct { int64_t lo; uint64_t hc; } avg_i128_state; +#define AVG_I128_WRAPS(hc) ((int64_t)(int32_t)(uint32_t)((hc) >> 32)) +#define AVG_I128_CNT(hc) ((int64_t)((hc) & 0xffffffffu)) +#define AVG_I128_ONE_WRAP ((uint64_t)1 << 32) +static void avg_i128_init(void* s) { + avg_i128_state* st = (avg_i128_state*)s; st->lo = 0; st->hc = 0; +} +static inline void avg_i128_add(avg_i128_state* st, int64_t v) { + int64_t o = st->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + st->lo = nw; + st->hc += 1u; + /* signed overflow: o and v share a sign the result lacks */ + if (RAY_UNLIKELY(((o ^ nw) & (v ^ nw)) < 0)) + st->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static void avg_i128_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; avg_i128_state* dd = (avg_i128_state*)d; const avg_i128_state* ss = (const avg_i128_state*)s; + int64_t o = dd->lo, v = ss->lo; + int64_t nw = (int64_t)((uint64_t)o + (uint64_t)v); + dd->lo = nw; + dd->hc += ss->hc; + if (((o ^ nw) & (v ^ nw)) < 0) + dd->hc += (v < 0) ? (uint64_t)0 - AVG_I128_ONE_WRAP : AVG_I128_ONE_WRAP; +} +static double avg_i128_final_result(const void* s) { + const avg_i128_state* st = s; + int64_t cnt = AVG_I128_CNT(st->hc); + if (!cnt) return NULL_F64; + /* two's-complement (hi, lo): the signed low word borrows one from the + * wrap count when negative */ + int64_t hi = AVG_I128_WRAPS(st->hc) + (st->lo < 0 ? -1 : 0); + return ray_f64_fin(ray_i128_to_f64(hi, (uint64_t)st->lo) / (double)cnt); +} +AGG_SCALAR_FINAL(avg_i128_final, double, ray_f64, value != value) +#define AVG_I128_UPDATE_BODY \ + avg_i128_add((avg_i128_state*)((char*)base + (size_t)gids[i]*stride), (int64_t)d[i]) + +/* ---- avg, inputs of at most 32 bits: exact int64 sum ------------------- + * Fewer than 2^31 values of at most 32 bits sum to less than 2^63, so a + * plain int64 running sum is exact (agg_v2_admission keeps larger tables + * off v2 for these kernels) and (double)sum / cnt is bit for bit what the + * 128-bit form would give. */ +typedef struct { int64_t sum; int64_t cnt; } avg_i64_state; +static void avg_i64_init(void* s) { ((avg_i64_state*)s)->sum = 0; ((avg_i64_state*)s)->cnt = 0; } +static void avg_i64_merge(void* d, const void* s, acc_arena_t* a) { + (void)a; ((avg_i64_state*)d)->sum += ((const avg_i64_state*)s)->sum; + ((avg_i64_state*)d)->cnt += ((const avg_i64_state*)s)->cnt; +} +static double avg_i64_final_result(const void* s) { + const avg_i64_state* st = s; + return st->cnt ? ray_f64_fin((double)st->sum / (double)st->cnt) : NULL_F64; +} +AGG_SCALAR_FINAL(avg_i64_final, double, ray_f64, value != value) +#define AVG_I64_UPDATE_BODY \ + avg_i64_state* st = (avg_i64_state*)((char*)base + (size_t)gids[i]*stride); \ + st->sum += (int64_t)d[i]; st->cnt++ + /* ---- variance family, I64 (sumsq as int64 unsigned-wrap; formula group.c:2190) -- */ /* Shifted-data accumulator: sums are of (v - k), where k is the first * value this state sees. The textbook one-pass form sumsq/n - mean^2 @@ -797,15 +872,14 @@ static void avg_bool_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_BOOL_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_bool_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_bool_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_bool_native_update(void* base, size_t stride, const uint32_t* gids, @@ -910,15 +984,14 @@ static void avg_u8_native_update(void* base, size_t stride, const uint32_t* gids int64_t n, acc_arena_t* a) { (void)a; const uint8_t* d = (const uint8_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_U8_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_u8_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_u8_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_u8_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1023,15 +1096,14 @@ static void avg_i16_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int16_t* d = (const int16_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I16_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i16_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i16_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i16_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1136,15 +1208,14 @@ static void avg_i32_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_I32_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i32_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_i32_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_i32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1203,15 +1274,14 @@ static void avg_i64_native_update(void* base, size_t stride, const uint32_t* gid int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_I64_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_i64_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_i64_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void min_f32_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1363,15 +1433,14 @@ static void avg_date_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_DATE_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_date_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_date_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_date_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1459,15 +1528,14 @@ static void avg_time_native_update(void* base, size_t stride, const uint32_t* gi int64_t n, acc_arena_t* a) { (void)a; const int32_t* d = (const int32_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I64_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIME_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_time_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i64_init, .update_batch = avg_time_native_update, + .merge = avg_i64_merge, .finalize = avg_i64_final, .finalize_value = avg_i64_final_value, }; static void var_time_native_update(void* base, size_t stride, const uint32_t* gids, @@ -1575,15 +1643,14 @@ static void avg_timestamp_native_update(void* base, size_t stride, const uint32_ int64_t n, acc_arena_t* a) { (void)a; const int64_t* d = (const int64_t*)vals; AGG_UPDATE_LOOP(valid, n, { - avg_f64_state* st = (avg_f64_state*)((char*)base + (size_t)gids[i]*stride); - st->sum += d[i]; st->cnt++; + AVG_I128_UPDATE_BODY; }); } static const agg_vtable_t AVG_TIMESTAMP_NATIVE = { - .state_size = sizeof(avg_f64_state), .kind = ACC_STREAMING, .out_type = RAY_F64, - .init = avg_f64_init, .update_batch = avg_timestamp_native_update, - .merge = avg_f64_merge, .finalize = avg_f64_final, .finalize_value = avg_f64_final_value, + .state_size = sizeof(avg_i128_state), .kind = ACC_STREAMING, .out_type = RAY_F64, + .init = avg_i128_init, .update_batch = avg_timestamp_native_update, + .merge = avg_i128_merge, .finalize = avg_i128_final, .finalize_value = avg_i128_final_value, }; static void var_timestamp_native_update(void* base, size_t stride, const uint32_t* gids, diff --git a/src/ops/group.c b/src/ops/group.c index bb737c18a..5226285f2 100644 --- a/src/ops/group.c +++ b/src/ops/group.c @@ -43,6 +43,19 @@ static inline bool group_fp_type(int8_t t) { return t == RAY_F32 || t == RAY_F64; } +/* Does an integer AVG over this input need the 128-bit high word next to + * its int64 sum? Only a 64-bit input can push a group's total past int64 + * in fewer than 2^31 rows; a narrower input (and a strlen fusion, whose + * lengths are smaller still) stays exact in the wrapped int64 sum, so it + * skips the carry and the per-slot high word entirely — the dense + * direct-array path in particular carves that word per worker. An + * unknown type (0: a linear plan without a source vector) is treated as + * 64-bit. */ +static inline bool group_avg_needs_hi(int8_t t, int64_t nrows) { + return t == RAY_I64 || t == RAY_TIMESTAMP || t == RAY_SYM || t == 0 || + nrows >= ((int64_t)1 << 31); +} + /* * group_key_f64_bits -- read an F64 GROUP BY key's bits, canonicalised. * @@ -131,10 +144,13 @@ int64_t ray_group_perpart_runs(void) { typedef struct { double sum_f, min_f, max_f, prod_f, first_f, last_f, sum_sq_f; int64_t sum_i, min_i, max_i, prod_i, first_i, last_i, sum_sq_i; - /* Parallel f64 sum of the integer stream — used by AVG so the - * mean of an i64 column whose sum exceeds 2^63 stays accurate - * instead of being whatever (uint64) wrap left in sum_i. */ + /* f64 sum of the integer stream (variance / stddev of integers). */ double sum_d; + /* High word of the exact 128-bit sum of the integer stream; sum_i is + * its low word. AVG of an integer column divides this exact total, + * so the mean is the same bits whatever the morsel split and equals + * the value the chunk-zone metadata computes from its per-chunk sums. */ + int64_t sum_hi; int64_t cnt; int64_t zero_count; bool has_first; @@ -145,7 +161,7 @@ static void reduce_acc_init(reduce_acc_t* acc) { acc->prod_f = 1.0; acc->first_f = 0; acc->last_f = 0; acc->sum_sq_f = 0; acc->sum_i = 0; acc->min_i = INT64_MAX; acc->max_i = INT64_MIN; acc->prod_i = 1; acc->first_i = 0; acc->last_i = 0; acc->sum_sq_i = 0; - acc->sum_d = 0; + acc->sum_d = 0; acc->sum_hi = 0; acc->cnt = 0; acc->zero_count = 0; acc->has_first = false; } @@ -329,13 +345,14 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, #define RED_NEED_MAX (1u << 7) #define RED_NEED_FIRST (1u << 8) #define RED_NEED_LAST (1u << 9) +#define RED_NEED_SUM_128 (1u << 10) /* integer stream: exact 128-bit sum in (sum_hi, sum_i) */ #define RED_MASK_ANY (RED_NEED_COUNT | RED_NEED_ZERO) #define RED_MASK_MIN (RED_NEED_COUNT | RED_NEED_MIN) #define RED_MASK_MAX (RED_NEED_COUNT | RED_NEED_MAX) #define RED_MASK_FIRST (RED_NEED_COUNT | RED_NEED_FIRST) #define RED_MASK_LAST (RED_NEED_COUNT | RED_NEED_LAST) -#define RED_MASK_AVG_I (RED_NEED_SUM_D | RED_NEED_COUNT) +#define RED_MASK_AVG_I (RED_NEED_SUM_128 | RED_NEED_COUNT) #define RED_MASK_AVG_F (RED_NEED_SUM | RED_NEED_COUNT) #define RED_MASK_STATS_I (RED_NEED_SUM_D | RED_NEED_SUM_SQ | RED_NEED_COUNT) #define RED_MASK_STATS_F (RED_NEED_SUM | RED_NEED_SUM_SQ | RED_NEED_COUNT) @@ -368,6 +385,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -396,6 +418,11 @@ static ray_t* agg_wide_reduce(ray_t* input, uint16_t op, if ((NEEDS) & RED_NEED_PROD) \ (acc)->prod_i = (int64_t)((uint64_t)(acc)->prod_i * (uint64_t)v); \ if ((NEEDS) & RED_NEED_SUM_D) (acc)->sum_d += (double)v; \ + if ((NEEDS) & RED_NEED_SUM_128) { \ + uint64_t lo_ = (uint64_t)(acc)->sum_i + (uint64_t)v; \ + (acc)->sum_hi += (v < 0 ? -1 : 0) + (lo_ < (uint64_t)v ? 1 : 0); \ + (acc)->sum_i = (int64_t)lo_; \ + } \ if (((NEEDS) & RED_NEED_ZERO) && v == 0) (acc)->zero_count++; \ if (((NEEDS) & RED_NEED_MIN) && v < (acc)->min_i) (acc)->min_i = v; \ if (((NEEDS) & RED_NEED_MAX) && v > (acc)->max_i) (acc)->max_i = v; \ @@ -662,7 +689,11 @@ static void reduce_merge(reduce_acc_t* dst, const reduce_acc_t* src, int8_t in_t break; case OP_AVG: if (fp) dst->sum_f += src->sum_f; - else dst->sum_d += src->sum_d; + else { + uint64_t lo = (uint64_t)dst->sum_i + (uint64_t)src->sum_i; + dst->sum_hi += src->sum_hi + (lo < (uint64_t)src->sum_i ? 1 : 0); + dst->sum_i = (int64_t)lo; + } dst->cnt += src->cnt; break; case OP_VAR: case OP_VAR_POP: case OP_STDDEV: case OP_STDDEV_POP: @@ -4268,7 +4299,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: result = ray_i64(scan_n); break; - case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : merged.sum_d / merged.cnt)) : ray_typed_null(-RAY_F64); break; + case OP_AVG: result = merged.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? merged.sum_f / merged.cnt : ray_i128_to_f64(merged.sum_hi, (uint64_t)merged.sum_i) / merged.cnt)) : ray_typed_null(-RAY_F64); break; case OP_FIRST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.first_f) : reduction_i64_result(merged.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_LAST: result = merged.has_first ? (group_fp_type(in_type) ? ray_f64(merged.last_f) : reduction_i64_result(merged.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); break; case OP_VAR: case OP_VAR_POP: @@ -4309,7 +4340,7 @@ ray_t* exec_reduction(ray_graph_t* g, ray_op_t* op, ray_t* input) { /* COUNT returns total length including nulls — matches ray_count_fn's * "count all elements" semantics, not SQL's COUNT(col) non-null count. */ case OP_COUNT: return ray_i64(scan_n); - case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : acc.sum_d / acc.cnt)) : ray_typed_null(-RAY_F64); + case OP_AVG: return acc.cnt > 0 ? ray_f64(ray_f64_fin((in_type == RAY_F64 || in_type == RAY_F32) ? acc.sum_f / acc.cnt : ray_i128_to_f64(acc.sum_hi, (uint64_t)acc.sum_i) / acc.cnt)) : ray_typed_null(-RAY_F64); case OP_FIRST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.first_f) : reduction_i64_result(acc.first_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_LAST: return acc.has_first ? (group_fp_type(in_type) ? ray_f64(acc.last_f) : reduction_i64_result(acc.last_i, in_type, in_type == RAY_SYM ? input : NULL)) : ray_typed_null(-(op->out_type ? op->out_type : in_type)); case OP_VAR: case OP_VAR_POP: @@ -4683,6 +4714,8 @@ bool ght_compute_layout(ght_layout_t* out, uint32_t n_keys, uint32_t n_aggs, if (need_flags & GHT_NEED_MIN) { out->off_min = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_MAX) { out->off_max = (uint16_t)off; off += block; } if (need_flags & GHT_NEED_SUMSQ) { out->off_sumsq = (uint16_t)off; off += block; } + /* High words of the exact 128-bit integer sums (integer AVG). */ + if (need_flags & GHT_NEED_SUM128) { out->off_sum_hi = (uint16_t)off; off += block; } /* Per-slot row-index bounds for FIRST/LAST. Two int64 blocks of * n_agg_vals slots each, allocated only when needed. */ if (has_first_last) { @@ -5310,6 +5343,14 @@ static inline void init_accum_from_entry(char* row, const char* entry, } else { memcpy(row + ly->off_sum + s * 8, agg_data + s * 8, 8); } + /* 128-bit sum: the high word sign-extends the first integer + * value (the row was zeroed above, so 0 is right for F64, + * truthy and non-negative slots). */ + if ((nf & GHT_NEED_SUM128) && !(af & GHT_AF_F64) && + !(aflags2[a] & GHT_AF2_TRUTHY)) { + int64_t v; memcpy(&v, agg_data + s * 8, 8); + if (v < 0) { int64_t hi = -1; memcpy(row + ly->off_sum_hi + s * 8, &hi, 8); } + } } if (nf & GHT_NEED_MIN) memcpy(row + ly->off_min + s * 8, agg_data + s * 8, 8); if (nf & GHT_NEED_MAX) memcpy(row + ly->off_max + s * 8, agg_data + s * 8, 8); @@ -5432,6 +5473,7 @@ static inline void accum_from_entry(char* row, const char* entry, else if (af & GHT_AF_LAST) { if (take_last) memcpy(row + ly->off_sum + s * 8, val, 8); } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -5560,6 +5602,7 @@ static void accum_from_entry_nullable(char* row, const char* entry, } } else if (aflags2[a] & GHT_AF2_TRUTHY) { ROW_WR_I64(row, ly->off_sum, s) += (v != 0); } else if (af & GHT_AF_PROD) { ROW_WR_I64(row, ly->off_sum, s) = (int64_t)((uint64_t)ROW_RD_I64(row, ly->off_sum, s) * (uint64_t)v); } + else if (nf & GHT_NEED_SUM128) { ray_i128_add(&ROW_WR_I64(row, ly->off_sum_hi, s), (uint64_t*)&ROW_WR_I64(row, ly->off_sum, s), v); } else { ROW_WR_I64(row, ly->off_sum, s) = wrap_add_i64(ROW_RD_I64(row, ly->off_sum, s), v); } } if (nf & GHT_NEED_MIN) { @@ -6713,7 +6756,10 @@ static void radix_phase3_fn(void* ctx, uint32_t worker_id, int64_t start, int64_ case OP_AVG: if (nn == 0) { v = NULL_F64; grp_set_null(ao->vec, di); break; } v = sf ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (ao->affine) v += ao->bias_f64; break; case OP_MIN: @@ -6926,7 +6972,9 @@ static inline uint32_t group_merge_row(group_ht_t* ht, const uint8_t* const aflags = ly->agg_flags; const int8_t* const vslot = ly->agg_val_slot; uint16_t off_sum = ly->off_sum; + uint16_t off_sum_hi = ly->off_sum_hi; bool need_sum = (ly->need_flags & GHT_NEED_SUM) != 0; + bool need_sum128 = (ly->need_flags & GHT_NEED_SUM128) != 0; for (;;) { uint32_t sv = ht->slots[slot]; if (sv == HT_EMPTY) { @@ -6965,10 +7013,16 @@ static inline uint32_t group_merge_row(group_ht_t* ht, memcpy(&sv_f, src_row + off, 8); // cppcheck-suppress invalidPointerCast // GHT rows are 8-byte aligned and off_sum/s*8 are multiples of 8 *(double*)(row + off) += sv_f; + } else if (need_sum128) { + int64_t sv_i, sv_hi; + memcpy(&sv_i, src_row + off, 8); + memcpy(&sv_hi, src_row + off_sum_hi + (size_t)s * 8, 8); + ray_i128_add128((int64_t*)(row + off_sum_hi + (size_t)s * 8), + (uint64_t*)(row + off), sv_hi, (uint64_t)sv_i); } else { int64_t sv_i; memcpy(&sv_i, src_row + off, 8); - *(int64_t*)(row + off) += sv_i; + *(int64_t*)(row + off) = wrap_add_i64(*(int64_t*)(row + off), sv_i); } } } @@ -7107,7 +7161,7 @@ static void radix_v2_phase1_fn(void* ctx, uint32_t worker_id, * NOTE: "hashing is identical" is load-bearing and is enforced by the * v2_packed branch in the staging loop — see the comment there. */ if (!wide_any && !inline_str && !nullable && ly->null_words == 0 && - (ly->need_flags & ~(uint32_t)GHT_NEED_SUM) == 0 && + (ly->need_flags & ~(uint32_t)(GHT_NEED_SUM | GHT_NEED_SUM128)) == 0 && !(ly->agg_flags_any & (GHT_AF_FIRST | GHT_AF_LAST | GHT_AF_BINARY | GHT_AF_HOLISTIC)) && nk >= 1 && nk <= 2 && ly->entry_stride <= 64) { @@ -7515,6 +7569,11 @@ typedef union { double f; int64_t i; } da_val_t; typedef struct { da_val_t* sum; /* SUM/AVG/FIRST/LAST [n_slots * n_aggs] */ + int64_t* sum_hi; /* high words of the exact 128-bit integer sums + * [n_slots * n_aggs]; allocated only when an + * integer AVG is present (sum[].i is then the low + * word — still the wrapped int64 SUM), NULL + * otherwise so SUM-only queries pay nothing. */ da_val_t* min_val; /* MIN [n_slots * n_aggs] */ da_val_t* max_val; /* MAX [n_slots * n_aggs] */ double* sumsq_f64; /* sum-of-squares for STDDEV/VAR */ @@ -7535,6 +7594,7 @@ typedef struct { double* sumxy; /* Σxy */ /* Arena headers */ ray_t* _h_sum; + ray_t* _h_sum_hi; ray_t* _h_min; ray_t* _h_max; ray_t* _h_sumsq; @@ -7549,6 +7609,7 @@ typedef struct { static inline void da_accum_free(da_accum_t* a) { scratch_free(a->_h_sum); + scratch_free(a->_h_sum_hi); scratch_free(a->_h_min); scratch_free(a->_h_max); scratch_free(a->_h_sumsq); @@ -7566,11 +7627,18 @@ static inline void da_accum_free(da_accum_t* a) { * non-NULL) carries the per-(group, agg) non-null row count: AVG/VAR/ * STDDEV use it as the divisor and MIN/MAX/PROD/FIRST/LAST emit a typed * null when it is zero. Pass NULL to keep the legacy count[gid]-divisor - * behaviour (callers without HAS_NULLS aggs need not allocate it). */ + * behaviour (callers without HAS_NULLS aggs need not allocate it). + * sum_hi64 (if non-NULL) carries the high words of the exact 128-bit + * integer sums: an integer AVG then divides ray_i128_to_f64(hi, lo), the + * same bits as the keyless reduction and the chunk-zone metadata. A + * caller passing NULL must not emit an integer AVG. The strlen fusion + * keeps the plain wrapped add (a sum of lengths never leaves int64), so a + * high word of 0 is exact there. */ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* ext, ray_t* const* agg_vecs, uint32_t grp_count, uint32_t n_aggs, const double* sum_f64, const int64_t* sum_i64, + const int64_t* sum_hi64, const double* min_f64, const double* max_f64, const int64_t* min_i64, const int64_t* max_i64, const int64_t* counts, @@ -7657,7 +7725,9 @@ static void emit_agg_columns(ray_t** result, ray_graph_t* g, const ray_op_ext_t* break; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } - v = is_f64 ? sum_f64[idx] / nn : (double)sum_i64[idx] / nn; + v = is_f64 ? sum_f64[idx] / nn + : sum_hi64 ? ray_i128_to_f64(sum_hi64[idx], (uint64_t)sum_i64[idx]) / nn + : (double)sum_i64[idx] / nn; if (affine && affine[a].enabled) v += affine[a].bias_f64; break; @@ -7872,12 +7942,14 @@ typedef struct { int64_t* keys; int64_t* counts; da_val_t* sums; + int64_t* sums_hi; /* high words of the 128-bit integer sums (integer AVG) */ uint32_t cap; uint32_t size; ray_t* _h_used; ray_t* _h_keys; ray_t* _h_counts; ray_t* _h_sums; + ray_t* _h_sums_hi; } sparse_i64_ht_t; static inline uint64_t sparse_i64_mix(uint64_t x) { @@ -7906,6 +7978,7 @@ static void sparse_i64_free(sparse_i64_ht_t* ht) { scratch_free(ht->_h_keys); scratch_free(ht->_h_counts); scratch_free(ht->_h_sums); + scratch_free(ht->_h_sums_hi); memset(ht, 0, sizeof(*ht)); } @@ -8025,7 +8098,7 @@ static int64_t da_count_emit_keep_min_u32(const uint32_t* counts, } static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, - bool need_sum) { + bool need_sum, bool need_sum128) { memset(ht, 0, sizeof(*ht)); if (cap < 1024) cap = 1024; cap = sparse_i64_pow2(cap); @@ -8037,7 +8110,12 @@ static bool sparse_i64_init(sparse_i64_ht_t* ht, uint32_t cap, uint8_t n_aggs, ht->sums = (da_val_t*)scratch_calloc(&ht->_h_sums, (size_t)cap * n_aggs * sizeof(da_val_t)); } - if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums)) { + if (need_sum && need_sum128) { + ht->sums_hi = (int64_t*)scratch_calloc(&ht->_h_sums_hi, + (size_t)cap * n_aggs * sizeof(int64_t)); + } + if (!ht->used || !ht->keys || !ht->counts || (need_sum && !ht->sums) || + (need_sum && need_sum128 && !ht->sums_hi)) { sparse_i64_free(ht); return false; } @@ -8057,7 +8135,7 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, bool need_sum) { sparse_i64_ht_t old = *ht; sparse_i64_ht_t nw; - if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum)) + if (!sparse_i64_init(&nw, old.cap * 2u, n_aggs, need_sum, old.sums_hi != NULL)) return false; for (uint32_t i = 0; i < old.cap; i++) { if (!old.used[i]) continue; @@ -8068,6 +8146,9 @@ static bool sparse_i64_rehash(sparse_i64_ht_t* ht, uint8_t n_aggs, if (need_sum) memcpy(&nw.sums[(size_t)s * n_aggs], &old.sums[(size_t)i * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (nw.sums_hi) + memcpy(&nw.sums_hi[(size_t)s * n_aggs], &old.sums_hi[(size_t)i * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); nw.size++; } sparse_i64_free(&old); @@ -8089,6 +8170,9 @@ static bool sparse_i64_touch(sparse_i64_ht_t* ht, int64_t key, uint8_t n_aggs, if (need_sum) memset(&ht->sums[(size_t)s * n_aggs], 0, (size_t)n_aggs * sizeof(da_val_t)); + if (ht->sums_hi) + memset(&ht->sums_hi[(size_t)s * n_aggs], 0, + (size_t)n_aggs * sizeof(int64_t)); ht->size++; } *out_slot = s; @@ -8392,16 +8476,67 @@ static inline int64_t scalar_i64_at(const void* ptr, int8_t type, int64_t r) { return read_col_i64(ptr, r, type, 0); /* attrs=0: agg columns are numeric, never SYM */ } +/* Exact 128-bit sum of a contiguous int64 run, kept vectorizable. + * Blocks of 1024 rows, three forms, escalating once for the rest of the + * run when a block fails a check (a column's values are alike): + * 0. non-negative values below 2^53 (ids, counts, timestamps): one OR of + * the raw values next to the wrapped sum proves every value is in + * [0, 2^53), so the block sum cannot have overflowed int64; + * 1. values in [-2^53, 2^53): the same with the values biased by 2^53 + * (one extra add); + * 2. anything: two plain running sums, the wrapped total and the sum of + * the (arithmetically shifted) high 32-bit halves; with n < 2^32 the + * low halves' sum is below 2^64, so it is recovered exactly from the + * wrapped total, and the block's value is sh * 2^32 + low. + * A failed check re-reads only that block (8 KiB, L1-hot). Same bits as + * ray_i128_add per element (idxop.h) at a fraction of its cost. */ +static inline void group_i128_sum_i64_range(const int64_t* restrict x, int64_t n, + int64_t* hi, uint64_t* lo) { + const int64_t blk = 1024; + const uint64_t bias = (uint64_t)1 << 53; + int mode = 0; + for (int64_t b = 0; b < n; b += blk) { + int64_t e = (n - b > blk) ? b + blk : n; + if (mode == 0) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u; } + if ((o >> 53) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 1; + } + if (mode == 1) { + uint64_t s = 0, o = 0; + for (int64_t r = b; r < e; r++) { uint64_t u = (uint64_t)x[r]; s += u; o |= u + bias; } + if ((o >> 54) == 0) { ray_i128_add(hi, lo, (int64_t)s); continue; } + mode = 2; + } + uint64_t sl = 0; int64_t sh = 0; + for (int64_t r = b; r < e; r++) { + int64_t v = x[r]; + sl += (uint64_t)v; + sh += v >> 32; + } + uint64_t low = sl - ((uint64_t)sh << 32); /* sum of the low halves */ + ray_i128_add128(hi, lo, sh >> 32, (uint64_t)sh << 32); + ray_i128_add128(hi, lo, 0, low); + } +} + /* Tight SIMD-friendly loop for single SUM/AVG on i64 (no mask). - * Note: int64 sum can overflow; caller responsibility to use appropriate types. */ + * Note: the int64 sum wraps (SUM's contract); an integer AVG also carries + * the high word (acc->sum_hi) so it divides the exact 128-bit total. */ static void scalar_sum_i64_fn(void* ctx, uint32_t worker_id, int64_t start, int64_t end) { scalar_ctx_t* c = (scalar_ctx_t*)ctx; da_accum_t* acc = &c->accums[worker_id]; const int64_t* restrict data = (const int64_t*)c->agg_ptrs[0]; - int64_t sum = 0; - for (int64_t r = start; r < end; r++) - sum = wrap_add_i64(sum, data[r]); - acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + if (acc->sum_hi) { + group_i128_sum_i64_range(data + start, end - start, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else { + int64_t sum = 0; + for (int64_t r = start; r < end; r++) + sum = wrap_add_i64(sum, data[r]); + acc->sum[0].i = wrap_add_i64(acc->sum[0].i, sum); + } acc->count[0] += end - start; } @@ -8424,6 +8559,41 @@ static void scalar_sum_linear_i64_fn(void* ctx, uint32_t worker_id, int64_t star const agg_linear_t* lin = &c->agg_linear[0]; int64_t n = end - start; + if (acc->sum_hi) { + /* Integer AVG needs the exact 128-bit total. A bare column (one + * term, coefficient 1, no bias — the common keyless avg) streams + * the vectorized block sum; a narrow column's block total cannot + * leave int64 at all. Any other linear shape walks the rows: the + * per-row value wraps like the materialized expression would and + * the 128-bit total of those rows is exact. */ + bool bare = lin->n_terms == 1 && lin->coeff_i64[0] == 1 && + lin->bias_i64 == 0 && lin->term_ptrs[0]; + int8_t t0 = lin->term_types[0]; + if (bare && (t0 == RAY_I64 || t0 == RAY_TIMESTAMP)) { + group_i128_sum_i64_range((const int64_t*)lin->term_ptrs[0] + start, n, + &acc->sum_hi[0], (uint64_t*)&acc->sum[0].i); + } else if (bare && t0 != RAY_SYM) { + /* <= 32-bit values: 2^30 of them stay far inside int64 */ + const int64_t blk = (int64_t)1 << 30; + for (int64_t b = start; b < end; b += blk) { + int64_t e = (end - b > blk) ? b + blk : end; + int64_t sum = 0; + for (int64_t r = b; r < e; r++) + sum += scalar_i64_at(lin->term_ptrs[0], t0, r); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, sum); + } + } else { + for (int64_t r = start; r < end; r++) { + int64_t iv = lin->bias_i64; + for (uint8_t t = 0; t < lin->n_terms; t++) + iv = wrap_add_i64(iv, wrap_mul_i64(lin->coeff_i64[t], + scalar_i64_at(lin->term_ptrs[t], lin->term_types[t], r))); + ray_i128_add(&acc->sum_hi[0], (uint64_t*)&acc->sum[0].i, iv); + } + } + acc->count[0] += n; + return; + } int64_t sum = wrap_mul_i64(lin->bias_i64, n); /* n_terms is bounded by AGG_LINEAR_MAX_TERMS (8, internal.h) — an * unrelated fixed cap on linear-expression arity, not a GROUP n_keys/ @@ -8500,7 +8670,8 @@ static inline void scalar_accum_row(scalar_ctx_t* c, da_accum_t* acc, int64_t r) if (nn) nn[a]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[a], (uint64_t*)&acc->sum[a].i, iv); + else acc->sum[a].i = wrap_add_i64(acc->sum[a].i, iv); if (acc->sumsq_f64) acc->sumsq_f64[a] += fv * fv; if (nn) nn[a]++; } @@ -8654,7 +8825,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 uint8_t v_attrs = c->agg_cols[a] ? c->agg_cols[a]->attrs : 0; int64_t v = read_col_i64(c->agg_ptrs[a], r, c->agg_types[a], v_attrs); if (RAY_LIKELY(!((inm >> a) & 1) || v != c->agg_int_null_sentinel[a])) { - acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, v); + else acc->sum[idx].i = wrap_add_i64(acc->sum[idx].i, v); if (nn) nn[idx]++; } } @@ -8718,7 +8890,8 @@ static inline void da_accum_row(da_ctx_t* c, da_accum_t* acc, int32_t gid, int64 if (nn) nn[idx]++; } } else if (RAY_LIKELY(!int_null)) { - acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); + if (acc->sum_hi) ray_i128_add(&acc->sum_hi[idx], (uint64_t*)&acc->sum[idx].i, iv); + else acc->sum[idx].i = (int64_t)((uint64_t)acc->sum[idx].i + (uint64_t)iv); if (acc->sumsq_f64) acc->sumsq_f64[idx] += fv * fv; if (nn) nn[idx]++; } @@ -9109,6 +9282,9 @@ static void da_merge_fn(void* ctx, uint32_t wid, int64_t start, int64_t end) { } } else if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -10252,6 +10428,8 @@ typedef struct { const uint16_t* agg_ops; uint8_t n_aggs; da_val_t* partials; /* [n_tasks * n_aggs], zeroed */ + int64_t* partials_hi; /* high words of the integer partials + * (NULL unless an integer AVG) */ double* partial_sumsq; double* partial_sum_y; double* partial_sumsq_y; @@ -10291,7 +10469,7 @@ static inline double sg_prod_range(const agg_prod_t* p, int64_t r0, int64_t n, a2 += x[j + 2] * y[j + 2]; a3 += x[j + 3] * y[j + 3]; } for (; j < n; j++) a0 += x[j] * y[j]; - } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIME)) { + } else if (ta == RAY_F64 && (tb == RAY_I64 || tb == RAY_TIMESTAMP)) { const double* restrict x = (const double*)pa + r0; const int64_t* restrict y = (const int64_t*)pb + r0; uint64_t s0 = 0; @@ -10458,32 +10636,45 @@ static void sg_accum_fn(void* raw, uint32_t wid, int64_t tstart, int64_t tend) { int8_t t = av->type; uint8_t at = av->attrs; uint64_t acc = 0; + int64_t hi = 0; /* 128-bit high word (integer AVG) */ double ssq = 0.0; bool need_sq = c->partial_sumsq && (op == OP_STDDEV || op == OP_STDDEV_POP || op == OP_VAR || op == OP_VAR_POP); - if (contig && (t == RAY_I64 || t == RAY_TIME)) { + if (contig && (t == RAY_I64 || t == RAY_TIMESTAMP)) { const int64_t* restrict x = (const int64_t*)p + r0; - for (int64_t j = 0; j < n; j++) { - int64_t v = x[j]; - acc += (uint64_t)v; - if (need_sq) { double d = (double)v; ssq += d * d; } + if (c->partials_hi) { + /* exact 128-bit total, still a vectorized stream */ + group_i128_sum_i64_range(x, n, &hi, &acc); + if (need_sq) + for (int64_t j = 0; j < n; j++) { double d = (double)x[j]; ssq += d * d; } + } else { + for (int64_t j = 0; j < n; j++) { + int64_t v = x[j]; + acc += (uint64_t)v; + if (need_sq) { double d = (double)v; ssq += d * d; } + } } - } else if (contig && t == RAY_I32) { + } else if (contig && (t == RAY_I32 || t == RAY_DATE || t == RAY_TIME)) { + /* the 4-byte family: I32 and the day / millisecond temporals */ const int32_t* restrict x = (const int32_t*)p + r0; for (int64_t j = 0; j < n; j++) { int64_t v = (int64_t)x[j]; acc += (uint64_t)v; if (need_sq) { double d = (double)v; ssq += d * d; } } + /* n <= SG_CHUNK_ROWS 32-bit values: the int64 sum is + * exact, its high word is the sign. */ + hi = (int64_t)acc < 0 ? -1 : 0; } else { for (int64_t j = 0; j < n; j++) { int64_t v = read_col_i64(p, rows[j], t, at); - acc += (uint64_t)v; + ray_i128_add(&hi, &acc, v); if (need_sq) { double d = (double)v; ssq += d * d; } } } c->partials[idx].i = (int64_t)acc; + if (c->partials_hi) c->partials_hi[idx] = hi; if (need_sq) c->partial_sumsq[idx] = ssq; } } @@ -10652,7 +10843,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const ray_idx_slice_t* slices = (K > 0) ? (const ray_idx_slice_t*)ray_data(g->sg_slices_hdr) : NULL; - bool need_sumsq = false, need_pair = false; + bool need_sumsq = false, need_pair = false, need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { uint16_t aop = ext->agg_ops[a]; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || @@ -10661,12 +10852,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, need_sumsq = true; if (agg_is_binary_agg(aop)) need_pair = true; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + if (aop == OP_AVG && !prod[a].enabled && agg_vecs[a] && + !group_fp_type(agg_vecs[a]->type) && + group_avg_needs_hi(agg_vecs[a]->type, ray_table_nrows(tbl))) + need_sum128 = true; } ray_t *sum_hdr = NULL, *cnt_hdr = NULL, *task_hdr = NULL, *part_hdr = NULL; ray_t *sumsq_hdr = NULL, *sum_y_hdr = NULL, *sumsq_y_hdr = NULL, *sumxy_hdr = NULL; ray_t *part_sumsq_hdr = NULL, *part_sum_y_hdr = NULL, *part_sumsq_y_hdr = NULL, *part_sumxy_hdr = NULL; + ray_t *sum_hi_hdr = NULL, *part_hi_hdr = NULL; da_val_t* sums = NULL; + int64_t* sums_hi = NULL; double *sumsq = NULL, *sum_y = NULL, *sumsq_y = NULL, *sumxy = NULL; int64_t* counts = NULL; if (K > 0) { @@ -10674,6 +10872,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, (size_t)K * n_aggs * sizeof(da_val_t)); counts = (int64_t*)scratch_calloc(&cnt_hdr, (size_t)K * sizeof(int64_t)); + if (need_sum128) + sums_hi = (int64_t*)scratch_calloc(&sum_hi_hdr, + (size_t)K * n_aggs * sizeof(int64_t)); if (need_sumsq) sumsq = (double*)scratch_calloc(&sumsq_hdr, (size_t)K * n_aggs * sizeof(double)); @@ -10692,12 +10893,16 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, n_tasks += (slices[i].n + SG_CHUNK_ROWS - 1) / SG_CHUNK_ROWS; sg_task_t* tasks = NULL; da_val_t* partials = NULL; + int64_t* part_hi = NULL; double *part_sumsq = NULL, *part_sum_y = NULL, *part_sumsq_y = NULL, *part_sumxy = NULL; if (sums && counts) { task_hdr = ray_alloc((size_t)n_tasks * (int64_t)sizeof(sg_task_t)); tasks = task_hdr ? (sg_task_t*)ray_data(task_hdr) : NULL; partials = (da_val_t*)scratch_calloc(&part_hdr, (size_t)n_tasks * n_aggs * sizeof(da_val_t)); + if (need_sum128) + part_hi = (int64_t*)scratch_calloc(&part_hi_hdr, + (size_t)n_tasks * n_aggs * sizeof(int64_t)); if (need_sumsq) part_sumsq = (double*)scratch_calloc(&part_sumsq_hdr, (size_t)n_tasks * n_aggs * sizeof(double)); @@ -10711,12 +10916,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } if (!sums || !counts || !tasks || !partials || + (need_sum128 && (!sums_hi || !part_hi)) || (need_sumsq && (!sumsq || !part_sumsq)) || (need_pair && (!sum_y || !sumsq_y || !sumxy || !part_sum_y || !part_sumsq_y || !part_sumxy))) { scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); scratch_free(part_hi_hdr); if (task_hdr) ray_free(task_hdr); scratch_free(part_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); @@ -10733,7 +10940,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } sg_ctx_t ctx = { slices, tasks, agg_vecs, agg_vecs2, prod, ext->agg_ops, - n_aggs, partials, part_sumsq, part_sum_y, + n_aggs, partials, part_hi, part_sumsq, part_sum_y, part_sumsq_y, part_sumxy, {0}, {0} }; /* Shared-stream pairing: a bare-scan SUM/AVG over the same column * a product's int side already streams rides the product loop — @@ -10748,10 +10955,14 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (prod[a].ta != RAY_F64) { ip = prod[a].pa; it = prod[a].ta; } else if (prod[a].tb != RAY_F64) { ip = prod[a].pb; it = prod[a].tb; } else continue; /* F64×F64 — no int side */ - if (it != RAY_I64 && it != RAY_TIME && it != RAY_I32) continue; + if (it != RAY_I64 && it != RAY_I32) continue; for (uint32_t b = 0; b < n_aggs; b++) { if (b == a || !agg_vecs[b] || ctx.fused_by[b] >= 0) continue; if (ext->agg_ops[b] != OP_SUM && ext->agg_ops[b] != OP_AVG) continue; + /* The product loop yields only the wrapped int64 side sum; + * an integer AVG needs the 128-bit total, so it streams + * its own column. */ + if (ext->agg_ops[b] == OP_AVG && part_hi) continue; if (ray_data(agg_vecs[b]) != ip || agg_vecs[b]->type != it) continue; ctx.pair_sum[a] = (int8_t)b; ctx.fused_by[b] = (int8_t)a; @@ -10778,6 +10989,9 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (pair || prod[a].enabled || ext->agg_ops[a] == OP_COUNT || (agg_vecs[a] && group_fp_type(agg_vecs[a]->type))) sums[di].f += partials[si].f; + else if (sums_hi) + ray_i128_add128(&sums_hi[di], (uint64_t*)&sums[di].i, + part_hi[si], (uint64_t)partials[si].i); else sums[di].i = (int64_t)((uint64_t)sums[di].i + (uint64_t)partials[si].i); @@ -10790,7 +11004,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } ray_free(task_hdr); - scratch_free(part_hdr); + scratch_free(part_hdr); scratch_free(part_hi_hdr); scratch_free(part_sumsq_hdr); scratch_free(part_sum_y_hdr); scratch_free(part_sumsq_y_hdr); scratch_free(part_sumxy_hdr); } @@ -10803,6 +11017,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } ray_t* kc = col_vec_new(key_col, K > 0 ? K : 1); @@ -10812,6 +11027,7 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return NULL; } if (kc->type == RAY_SYM) @@ -10826,17 +11042,19 @@ static ray_t* exec_group_slices(ray_graph_t* g, ray_op_t* op, ray_t* tbl, scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } emit_agg_columns(&result, g, ext, agg_vecs, (uint32_t)K, n_aggs, - (double*)sums, (int64_t*)sums, + (double*)sums, (int64_t*)sums, sums_hi, NULL, NULL, NULL, NULL, counts, NULL, prod, sumsq, NULL, sum_y, sumsq_y, sumxy); scratch_free(sum_hdr); scratch_free(cnt_hdr); scratch_free(sumsq_hdr); scratch_free(sum_y_hdr); scratch_free(sumsq_y_hdr); scratch_free(sumxy_hdr); + scratch_free(sum_hi_hdr); return result; } @@ -10996,6 +11214,7 @@ typedef struct { int64_t n_scan; uint8_t key_esz; bool sp_need_sum; + bool sp_need_sum128; /* an integer AVG: carry high words */ const int64_t* match_idx; ray_t* rowsel; ray_t* match_idx_block; @@ -11066,6 +11285,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { int64_t n_scan = c->n_scan; uint8_t key_esz = c->key_esz; bool sp_need_sum = c->sp_need_sum; + bool sp_need_sum128 = c->sp_need_sum128; const int64_t* match_idx = c->match_idx; ray_t* rowsel = c->rowsel; ray_t* match_idx_block = c->match_idx_block; @@ -11077,16 +11297,23 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { : (1u << 20); const uint64_t max_dense_cap = 1u << 24; bool count_only_first = (key_types[0] == RAY_SYM); - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)cap * sizeof(uint32_t)); da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ bool dyn_ok = range_count != NULL; if (dyn_ok && sp_need_sum && !count_only_first) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, (size_t)cap * n_aggs * sizeof(da_val_t)); dyn_ok = range_sum != NULL; + if (dyn_ok && sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)cap * n_aggs * sizeof(int64_t)); + dyn_ok = range_sum_hi != NULL; + } } uint64_t max_seen = 0; @@ -11203,6 +11430,19 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memset(range_sum + (size_t)old_cap * n_aggs, 0, \ (size_t)(cap - old_cap) * n_aggs * sizeof(da_val_t)); \ } \ + if (range_sum_hi) { \ + int64_t* new_hi = (int64_t*)scratch_realloc( \ + &range_hi_hdr, \ + (size_t)old_cap * n_aggs * sizeof(int64_t), \ + (size_t)cap * n_aggs * sizeof(int64_t)); \ + if (!new_hi) { \ + dyn_ok = false; \ + goto dyn_dense_done; \ + } \ + range_sum_hi = new_hi; \ + memset(range_sum_hi + (size_t)old_cap * n_aggs, 0, \ + (size_t)(cap - old_cap) * n_aggs * sizeof(int64_t)); \ + } \ } \ have_dyn_key = true; \ if (off > max_seen) max_seen = off; \ @@ -11218,6 +11458,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (range_sum_hi) \ + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11272,7 +11516,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11287,7 +11531,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11298,16 +11542,21 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -11316,9 +11565,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (sp_need_sum && !range_sum) + if (sp_need_sum && !range_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, + (size_t)grp_count * n_aggs * sizeof(int64_t)); + } uint32_t gi = 0; for (uint64_t off = 0; off <= max_seen; off++) { @@ -11334,6 +11587,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } if (!range_sum) range_count[off] = gi + 1u; gi++; @@ -11357,6 +11614,10 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { agg_vecs[a], dyn_row, strlen_sym_strings, strlen_sym_count)); \ else if (agg_f64_mask & ((uint64_t)1 << a)) \ sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], dyn_row); \ + else if (dense_sum_hi) \ + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], \ + (uint64_t*)&sums[a].i, \ + read_col_i64(agg_ptrs[a], dyn_row, agg_types[a], 0)); \ else \ sums[a].i = wrap_add_i64( \ sums[a].i, \ @@ -11402,13 +11663,13 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { * row is non-null and the legacy count-based divisor is * correct. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); scratch_free(cnt_hdr); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); for (uint32_t k = 0; k < n_keys; k++) @@ -11417,7 +11678,7 @@ exec_group_sp_dyn_emit(const sp_dyn_ctx_t* c) { return result; } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); /* Dynamic-dense probe bailed (unbounded key or no surviving row): shared @@ -11966,6 +12227,8 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, * once n_aggs exceeded its bit width. */ bool *sc_int_null_has = (bool*)(agg_types + vla_aggs); bool sc_any_nullable = false; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool sc_need_sum128 = false; for (uint32_t a = 0; a < n_aggs; a++) { if (agg_prod[a].enabled) { /* Fused product: F64 accumulate, no source vec. */ @@ -11996,6 +12259,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, sc_int_null_sentinel[a] = 0; sc_int_null_has[a] = false; } + if (ext->agg_ops[a] == OP_AVG && !group_fp_type(agg_types[a]) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + sc_need_sum128 = true; } if (!match_idx && !rowsel && !sc_any_nullable && n_aggs > 1) { @@ -12051,8 +12317,21 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } } + /* An integer AVG needs the exact 128-bit total: take it from + * the chunk-zone metadata when the column carries it (the + * same (hi, lo) ray_avg_fn reads), otherwise skip the shortcut + * and let the parallel scan below carry the high word. */ + int64_t zone_hi = 0, zone_nn = 0; uint64_t zone_lo = 0; + bool zone_128 = false; + if (one_base_input && base_col && sc_need_sum128 && + !group_fp_type(base_type)) { + zone_128 = ray_zone_int_sum128(base_col, &zone_hi, &zone_lo, &zone_nn) + && zone_nn == nrows; + if (!zone_128) one_base_input = false; + } if (one_base_input && base_col) { - ray_t* base_sum_obj = ray_sum_fn(base_col); + ray_t* base_sum_obj = zone_128 ? ray_i64((int64_t)zone_lo) + : ray_sum_fn(base_col); if (!base_sum_obj || RAY_IS_ERR(base_sum_obj)) { scratch_free(sc_vla_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -12084,12 +12363,14 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, ray_t *sum_hdr = NULL, *cnt_hdr = NULL; double* sums_f64 = NULL; int64_t* sums_i64 = NULL; + int64_t* sums_hi = NULL; if (base_is_f64) sums_f64 = (double*)scratch_alloc(&sum_hdr, (size_t)n_aggs * sizeof(double)); else + /* [sums | high words] in one carve */ sums_i64 = (int64_t*)scratch_alloc(&sum_hdr, - (size_t)n_aggs * sizeof(int64_t)); + (size_t)2 * n_aggs * sizeof(int64_t)); int64_t* counts = (int64_t*)scratch_alloc(&cnt_hdr, sizeof(int64_t)); if ((base_is_f64 ? (sums_f64 != NULL) : (sums_i64 != NULL)) && @@ -12100,6 +12381,11 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { for (uint32_t a = 0; a < n_aggs; a++) sums_i64[a] = base_sum_i64; + if (zone_128) { + sums_hi = sums_i64 + n_aggs; + for (uint32_t a = 0; a < n_aggs; a++) + sums_hi[a] = zone_hi; + } } counts[0] = nrows; @@ -12119,7 +12405,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - sums_f64, sums_i64, + sums_f64, sums_i64, sums_hi, NULL, NULL, NULL, NULL, counts, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); @@ -12172,6 +12458,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, const size_t sc_line = 64; size_t sc_words = 1; /* count[1] */ if (need_flags & DA_NEED_SUM) sc_words += n_aggs; /* sum */ + if (sc_need_sum128) sc_words += n_aggs; /* sum_hi */ if (need_flags & DA_NEED_MIN) sc_words += n_aggs; /* min_val */ if (need_flags & DA_NEED_MAX) sc_words += n_aggs; /* max_val */ if (need_flags & DA_NEED_SUMSQ) sc_words += n_aggs; /* sumsq_f64 */ @@ -12188,6 +12475,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, if (need_flags & DA_NEED_SUM) { sc_acc[w].sum = (da_val_t*)(void*)(blk + off); off += n_aggs; } + if (sc_need_sum128) { + sc_acc[w].sum_hi = blk + off; off += n_aggs; + } if (need_flags & DA_NEED_MIN) { sc_acc[w].min_val = (da_val_t*)(void*)(blk + off); off += n_aggs; for (uint32_t a = 0; a < n_aggs; a++) { @@ -12298,6 +12588,9 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } else { if (group_fp_type(agg_types[a])) m->sum[a].f += wa->sum[a].f; + else if (m->sum_hi) + ray_i128_add128(&m->sum_hi[a], (uint64_t*)&m->sum[a].i, + wa->sum_hi[a], (uint64_t)wa->sum[a].i); else m->sum[a].i = wrap_add_i64(m->sum[a].i, wa->sum[a].i); } @@ -12363,7 +12656,7 @@ static ray_t* exec_group_run(ray_graph_t* g, ray_op_t* op, ray_t* tbl, } emit_agg_columns(&result, g, ext, agg_vecs, 1, n_aggs, - (double*)m->sum, (int64_t*)m->sum, + (double*)m->sum, (int64_t*)m->sum, m->sum_hi, (double*)m->min_val, (double*)m->max_val, (int64_t*)m->min_val, (int64_t*)m->max_val, m->count, agg_affine, agg_prod, m->sumsq_f64, m->nn_count, @@ -12727,6 +13020,8 @@ da_path:; int64_t da_int_null_sentinel[vla_aggs]; uint64_t agg_f64_mask = 0; uint64_t da_int_null_mask = 0; + /* Integer AVG divides the exact 128-bit sum: carry high words. */ + bool da_need_sum128 = false; /* Track whether any agg column can produce a null so we can * allocate per-(group, agg) non-null counts only when required. * F64 with HAS_NULLS uses NaN-skip; sentinel-typed integers @@ -12778,6 +13073,9 @@ da_path:; agg_types[a] = 0; da_int_null_sentinel[a] = 0; } + if (ext->agg_ops[a] == OP_AVG && !(agg_f64_mask & ((uint64_t)1 << a)) && + !agg_strlen[a] && group_avg_needs_hi(agg_types[a], nrows)) + da_need_sum128 = true; } ray_pool_t* da_pool = ray_pool_get(); @@ -12788,6 +13086,7 @@ da_path:; * cells must remain O(contributing rows). */ uint32_t arrays_per_agg = 0; if (need_flags & DA_NEED_SUM) arrays_per_agg += 1; + if (da_need_sum128) arrays_per_agg += 1; /* sum_hi */ if (need_flags & DA_NEED_MIN) arrays_per_agg += 1; if (need_flags & DA_NEED_MAX) arrays_per_agg += 1; if (need_flags & DA_NEED_SUMSQ) arrays_per_agg += 1; @@ -12839,6 +13138,11 @@ da_path:; total * sizeof(da_val_t)); if (!accums[w].sum) { alloc_ok = false; break; } } + if (da_need_sum128) { + accums[w].sum_hi = (int64_t*)scratch_calloc(&accums[w]._h_sum_hi, + total * sizeof(int64_t)); + if (!accums[w].sum_hi) { alloc_ok = false; break; } + } if (need_flags & DA_NEED_SUMSQ) { accums[w].sumsq_f64 = (double*)scratch_calloc(&accums[w]._h_sumsq, total * sizeof(double)); @@ -13020,6 +13324,7 @@ da_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP || agg_is_binary_agg(aop)) { if (group_fp_type(agg_types[a]) || agg_is_binary_agg(aop)) merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } else if (aop == OP_PROD) { /* Use per-(group, agg) non-null counts when @@ -13146,6 +13451,9 @@ da_path:; /* binary aggs accumulate Σx as double even * for integer x-columns — merge as float. */ merged->sum[idx].f += wa->sum[idx].f; + else if (merged->sum_hi) + ray_i128_add128(&merged->sum_hi[idx], (uint64_t*)&merged->sum[idx].i, + wa->sum_hi[idx], (uint64_t)wa->sum[idx].i); else merged->sum[idx].i = wrap_add_i64(merged->sum[idx].i, wa->sum[idx].i); } @@ -13206,6 +13514,7 @@ da_path:; da_accum_free(&accums[w]); da_val_t* da_sum = merged->sum; /* may be NULL if !DA_NEED_SUM */ + int64_t* da_sum_hi = merged->sum_hi; /* NULL unless an integer AVG */ da_val_t* da_min_val = merged->min_val; /* may be NULL if !DA_NEED_MIN */ da_val_t* da_max_val = merged->max_val; /* may be NULL if !DA_NEED_MAX */ double* da_sumsq = merged->sumsq_f64; /* may be NULL if !DA_NEED_SUMSQ */ @@ -13271,8 +13580,9 @@ da_path:; size_t dense_total = (size_t)grp_count * n_aggs; ray_t *_h_dsum = NULL, *_h_dmin = NULL, *_h_dmax = NULL; ray_t *_h_dsq = NULL, *_h_dcnt = NULL, *_h_dnn = NULL; - ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL; + ray_t *_h_dsy = NULL, *_h_dsqy = NULL, *_h_dxy = NULL, *_h_dshi = NULL; da_val_t* dense_sum = da_sum ? (da_val_t*)scratch_alloc(&_h_dsum, dense_total * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = da_sum_hi ? (int64_t*)scratch_alloc(&_h_dshi, dense_total * sizeof(int64_t)) : NULL; da_val_t* dense_min_val = da_min_val ? (da_val_t*)scratch_alloc(&_h_dmin, dense_total * sizeof(da_val_t)) : NULL; da_val_t* dense_max_val = da_max_val ? (da_val_t*)scratch_alloc(&_h_dmax, dense_total * sizeof(da_val_t)) : NULL; double* dense_sumsq = da_sumsq ? (double*)scratch_alloc(&_h_dsq, dense_total * sizeof(double)) : NULL; @@ -13294,6 +13604,7 @@ da_path:; size_t si = (size_t)s * n_aggs + a; size_t di = (size_t)gi * n_aggs + a; if (dense_sum) dense_sum[di] = da_sum[si]; + if (dense_sum_hi) dense_sum_hi[di] = da_sum_hi[si]; if (dense_min_val) dense_min_val[di] = da_min_val[si]; if (dense_max_val) dense_max_val[di] = da_max_val[si]; if (dense_sumsq) dense_sumsq[di] = da_sumsq[si]; @@ -13306,7 +13617,7 @@ da_path:; } emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, (double*)dense_min_val, (double*)dense_max_val, (int64_t*)dense_min_val, (int64_t*)dense_max_val, dense_counts, agg_affine, agg_prod, dense_sumsq, @@ -13317,6 +13628,7 @@ da_path:; scratch_free(_h_dsq); scratch_free(_h_dcnt); scratch_free(_h_dnn); scratch_free(_h_dsy); scratch_free(_h_dsqy); scratch_free(_h_dxy); + scratch_free(_h_dshi); da_accum_free(&accums[0]); scratch_free(accums_hdr); for (uint32_t a = 0; a < n_aggs; a++) @@ -13346,6 +13658,9 @@ da_path:; sp_eligible = false; } bool sp_need_sum = false; + /* An integer AVG divides the exact 128-bit sum: the scatter arrays + * and the sparse table then carry a high word per slot. */ + bool sp_need_sum128 = false; for (uint32_t a = 0; a < n_aggs && sp_eligible; a++) { uint16_t op = ext->agg_ops[a]; if (op == OP_COUNT) continue; @@ -13360,8 +13675,13 @@ da_path:; * accum_from_entry inherits the same nullable-agg gap.) */ if (agg_vecs[a] && ray_vec_may_have_nulls(agg_vecs[a])) sp_eligible = false; - else + else { sp_need_sum = true; + if (op == OP_AVG && !(agg_vecs[a] && group_fp_type(agg_vecs[a]->type)) && + !agg_strlen[a] && + group_avg_needs_hi(agg_vecs[a] ? agg_vecs[a]->type : 0, nrows)) + sp_need_sum128 = true; + } } } @@ -13441,7 +13761,8 @@ da_path:; .strlen_sym_count = strlen_sym_count, .agg_f64_mask = agg_f64_mask, .n_aggs = n_aggs, .n_keys = n_keys, .n_scan = n_scan, .key_esz = key_esz, - .sp_need_sum = sp_need_sum, .match_idx = match_idx, + .sp_need_sum = sp_need_sum, .sp_need_sum128 = sp_need_sum128, + .match_idx = match_idx, .rowsel = rowsel, .match_idx_block = match_idx_block, .vla_hdr = vla_hdr, .emit_filter = emit_filter, }; @@ -13473,13 +13794,14 @@ da_path:; ? (uint64_t)((uint64_t)max_key - (uint64_t)min_key + 1u) : 0u; if (have_key && key_range > 0 && key_range <= (1u << 26)) { - ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *cnt_hdr = NULL, *range_sum_hdr = NULL, *range_hi_hdr = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; uint32_t* range_count = (uint32_t*)scratch_calloc( &cnt_hdr, (size_t)key_range * sizeof(uint32_t)); if (!range_count) goto ht_path; da_val_t* range_sum = NULL; + int64_t* range_sum_hi = NULL; /* high words (integer AVG) */ if (sp_need_sum && key_range <= (1u << 24)) { range_sum = (da_val_t*)scratch_calloc( &range_sum_hdr, @@ -13488,6 +13810,16 @@ da_path:; scratch_free(cnt_hdr); goto ht_path; } + if (sp_need_sum128) { + range_sum_hi = (int64_t*)scratch_calloc( + &range_hi_hdr, + (size_t)key_range * n_aggs * sizeof(int64_t)); + if (!range_sum_hi) { + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); + scratch_free(cnt_hdr); + goto ht_path; + } + } } for (int64_t i = 0; i < n_scan; i++) { @@ -13511,6 +13843,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (range_sum_hi) + ray_i128_add(&range_sum_hi[(size_t)off * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13534,7 +13870,7 @@ da_path:; ray_t* result = ray_table_new((int64_t)n_keys + n_aggs); if (!result || RAY_IS_ERR(result)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13550,7 +13886,7 @@ da_path:; /* raw cell ids from key_vecs[0] — adopt its domain */ ray_sym_vec_adopt_domain(key_col, sym_domain_rep(key_vecs[0])); if (!key_col || RAY_IS_ERR(key_col)) { - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13566,11 +13902,16 @@ da_path:; ? (da_val_t*)scratch_calloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_calloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc( &_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); - scratch_free(range_sum_hdr); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_release(key_col); ray_release(result); for (uint32_t a = 0; a < n_aggs; a++) @@ -13596,6 +13937,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &range_sum[(size_t)off * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &range_sum_hi[(size_t)off * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } range_count[off] = gi + 1u; gi++; @@ -13622,6 +13967,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)(marker - 1u) * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13630,7 +13979,7 @@ da_path:; } } - scratch_free(range_sum_hdr); + scratch_free(range_sum_hdr); scratch_free(range_hi_hdr); scratch_free(cnt_hdr); ray_op_ext_t* key_ext = find_ext(g, ext->keys[0]); int64_t name_id = key_ext ? key_ext->sym : 0; @@ -13641,12 +13990,12 @@ da_path:; * emit-filter range path only runs when sp_eligible was * true. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13668,7 +14017,7 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, false, false)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13687,7 +14036,8 @@ da_path:; uint64_t expected = (uint64_t)nrows / 64u; if (expected < 4096) expected = 4096; if (expected > (1u << 20)) expected = (1u << 20); - if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum)) + if (!sparse_i64_init(&sp_ht, (uint32_t)expected, n_aggs, sp_need_sum, + sp_need_sum128)) goto ht_path; for (int64_t i = 0; i < n_scan; i++) { @@ -13714,6 +14064,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (sp_ht.sums_hi) + ray_i128_add(&sp_ht.sums_hi[(size_t)slot * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13769,15 +14123,20 @@ da_path:; } key_col->len = (int64_t)grp_count; - ray_t *_h_sum = NULL, *_h_cnt = NULL; + ray_t *_h_sum = NULL, *_h_cnt = NULL, *_h_sum_hi = NULL; da_val_t* dense_sum = sp_need_sum ? (da_val_t*)scratch_alloc(&_h_sum, (size_t)grp_count * n_aggs * sizeof(da_val_t)) : NULL; + int64_t* dense_sum_hi = (sp_need_sum && sp_need_sum128) + ? (int64_t*)scratch_alloc(&_h_sum_hi, + (size_t)grp_count * n_aggs * sizeof(int64_t)) + : NULL; int64_t* dense_count = (int64_t*)scratch_alloc(&_h_cnt, (size_t)grp_count * sizeof(int64_t)); - if ((sp_need_sum && !dense_sum) || !dense_count) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if ((sp_need_sum && !dense_sum) || !dense_count || + (sp_need_sum && sp_need_sum128 && !dense_sum_hi)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13787,14 +14146,17 @@ da_path:; if (match_idx_block) { ray_release(match_idx_block); } scratch_free(vla_hdr); return ray_error("oom", NULL); } - if (use_emit_filter && sp_need_sum) + if (use_emit_filter && sp_need_sum) { memset(dense_sum, 0, (size_t)grp_count * n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memset(dense_sum_hi, 0, (size_t)grp_count * n_aggs * sizeof(int64_t)); + } sparse_i64_ht_t heavy_ht; memset(&heavy_ht, 0, sizeof(heavy_ht)); if (use_emit_filter && grp_count > 0) { - if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + if (!sparse_i64_init(&heavy_ht, grp_count * 2u, n_aggs, false, false)) { + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&sp_ht); for (uint32_t a = 0; a < n_aggs; a++) @@ -13816,7 +14178,7 @@ da_path:; if (use_emit_filter) { int32_t hslot; if (!sparse_i64_touch(&heavy_ht, sp_ht.keys[s], n_aggs, false, &hslot)) { - scratch_free(_h_sum); scratch_free(_h_cnt); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); ray_release(key_col); ray_release(result); sparse_i64_free(&heavy_ht); sparse_i64_free(&sp_ht); @@ -13832,6 +14194,10 @@ da_path:; memcpy(&dense_sum[(size_t)gi * n_aggs], &sp_ht.sums[(size_t)s * n_aggs], (size_t)n_aggs * sizeof(da_val_t)); + if (dense_sum_hi) + memcpy(&dense_sum_hi[(size_t)gi * n_aggs], + &sp_ht.sums_hi[(size_t)s * n_aggs], + (size_t)n_aggs * sizeof(int64_t)); } gi++; } @@ -13859,6 +14225,10 @@ da_path:; strlen_sym_count)); else if (agg_f64_mask & ((uint64_t)1 << a)) sums[a].f += group_fp_at(agg_ptrs[a], agg_types[a], r); + else if (dense_sum_hi) + ray_i128_add(&dense_sum_hi[(size_t)out_gi * n_aggs + a], + (uint64_t*)&sums[a].i, + read_col_i64(agg_ptrs[a], r, agg_types[a], 0)); else sums[a].i = wrap_add_i64( sums[a].i, @@ -13876,12 +14246,12 @@ da_path:; * and is gated to null-free agg columns (sp_eligible guard at * ~line 5737), so counts[gi] is the correct divisor. */ emit_agg_columns(&result, g, ext, agg_vecs, grp_count, n_aggs, - (double*)dense_sum, (int64_t*)dense_sum, + (double*)dense_sum, (int64_t*)dense_sum, dense_sum_hi, NULL, NULL, NULL, NULL, dense_count, agg_affine, agg_prod, NULL, NULL, NULL, NULL, NULL); - scratch_free(_h_sum); + scratch_free(_h_sum); scratch_free(_h_sum_hi); scratch_free(_h_cnt); for (uint32_t a = 0; a < n_aggs; a++) if (agg_owned[a] && agg_vecs[a]) ray_release(agg_vecs[a]); @@ -13919,6 +14289,10 @@ ht_path:; uint16_t aop = ext->agg_ops[a]; if (aop == OP_SUM || aop == OP_PROD || aop == OP_AVG || aop == OP_ALL || aop == OP_ANY || aop == OP_FIRST || aop == OP_LAST) ght_need |= GHT_NEED_SUM; + /* Integer avg divides the exact 128-bit sum: carry its high words. */ + if (aop == OP_AVG && agg_vecs[a] && !group_fp_type(agg_vecs[a]->type) && + !agg_strlen[a] && group_avg_needs_hi(agg_vecs[a]->type, nrows)) + ght_need |= GHT_NEED_SUM128; if (aop == OP_STDDEV || aop == OP_STDDEV_POP || aop == OP_VAR || aop == OP_VAR_POP) { ght_need |= GHT_NEED_SUM; ght_need |= GHT_NEED_SUMSQ; } if (agg_is_binary_agg(aop)) @@ -15581,7 +15955,10 @@ sequential_fallback:; case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, gi, true); break; } v = is_f64 ? ROW_RD_F64(row, ly->off_sum, s) / nn - : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; + : (ly->need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly->off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly->off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly->off_sum, s) / nn; if (agg_affine[a].enabled) v += agg_affine[a].bias_f64; break; case OP_MIN: diff --git a/src/ops/idxop.c b/src/ops/idxop.c index 873984f55..cc9f9a36e 100644 --- a/src/ops/idxop.c +++ b/src/ops/idxop.c @@ -605,7 +605,8 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, int64_t s = (int64_t)g * csz; int64_t e = s + csz; if (e > n) e = n; int64_t mn = INT64_MAX, mx = INT64_MIN; - uint64_t sum = 0; /* wraps like the engine's int64 sum */ + uint64_t sum = 0; /* low word: wraps like the engine's int64 sum */ + int64_t hi = 0; /* high word of the exact 128-bit sum */ int64_t nn = 0; bool any_null = false; for (int64_t i = s; i < e; i++) { @@ -620,13 +621,15 @@ static ray_err_t chunk_zone_scan_int(ray_t* v, ray_index_t* ix, } if (val < mn) mn = val; if (val > mx) mx = val; - sum += (uint64_t)val; + ray_i128_add(&hi, &sum, val); nn++; } if (ix->u.chunk_zone.aggs) { int64_t* ag = (int64_t*)ray_data(ix->u.chunk_zone.aggs); ag[g] = (int64_t)sum; ag[n_chunks + g] = nn; + if (ix->u.chunk_zone.aggs->len >= 3 * (int64_t)n_chunks) + ag[2 * (int64_t)n_chunks + g] = hi; } /* Empty (all-null) chunks keep mn=INT64_MAX / mx=INT64_MIN so * the reduce path's min(mins[*]) / max(maxs[*]) ignores them. */ @@ -824,9 +827,10 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2) { ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; if (!ix->u.chunk_zone.is_f64) { - ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } - aggs->len = 2 * (int64_t)n_chunks; + aggs->len = 3 * (int64_t)n_chunks; ix->u.chunk_zone.aggs = aggs; } @@ -880,9 +884,10 @@ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2) { ix->u.chunk_zone.maxs = maxs; ix->u.chunk_zone.null_bits = nbits; if (!ix->u.chunk_zone.is_f64) { - ray_t* aggs = ray_vec_new(RAY_I64, 2 * (int64_t)n_chunks); + /* [sum low words | non-null counts | sum high words], one per chunk */ + ray_t* aggs = ray_vec_new(RAY_I64, 3 * (int64_t)n_chunks); if (!aggs || RAY_IS_ERR(aggs)) { ray_release(idx); return ray_error("oom", "chunk_zone: aggs alloc"); } - aggs->len = 2 * (int64_t)n_chunks; + aggs->len = 3 * (int64_t)n_chunks; ix->u.chunk_zone.aggs = aggs; } @@ -1118,9 +1123,11 @@ ray_t* ray_index_inline_map(uint8_t* region, int64_t region_size) { } *slots[i] = c; } + /* two layouts: [lo | nn] (2 per chunk) and [lo | nn | hi] (3 per chunk) */ if (ix->kind == RAY_IDX_CHUNK_ZONE && ix->u.chunk_zone.aggs && (ix->u.chunk_zone.aggs->type != RAY_I64 || ix->u.chunk_zone.is_f64 || - ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks)) + (ix->u.chunk_zone.aggs->len != 2 * (int64_t)ix->u.chunk_zone.n_chunks && + ix->u.chunk_zone.aggs->len != 3 * (int64_t)ix->u.chunk_zone.n_chunks))) ix->u.chunk_zone.aggs = NULL; ix->markers |= RAY_MARK_MMAP; idx->mmod = 1; diff --git a/src/ops/idxop.h b/src/ops/idxop.h index 076a77475..609895a9f 100644 --- a/src/ops/idxop.h +++ b/src/ops/idxop.h @@ -277,11 +277,40 @@ ray_t* ray_index_attach_chunk_zone(ray_t** vp, uint8_t chunk_log2); * compute an index for persistence without COWing a shared column. */ ray_t* ray_index_chunk_zone_compute(ray_t* v, uint8_t chunk_log2); +/* 128-bit two's-complement accumulation of int64 values: (hi, lo) += v. + * The engine's integer avg sums this way — exact for any column, and the + * same bits whatever the morsel split — and the chunk-zone metadata keeps + * the per-chunk (hi, lo) so it can answer the same value. */ +static inline void ray_i128_add(int64_t* hi, uint64_t* lo, int64_t v) { + uint64_t l = *lo + (uint64_t)v; + *hi += (v < 0 ? -1 : 0) + (l < (uint64_t)v ? 1 : 0); + *lo = l; +} +static inline void ray_i128_add128(int64_t* hi, uint64_t* lo, int64_t vhi, uint64_t vlo) { + uint64_t l = *lo + vlo; + *hi += vhi + (l < vlo ? 1 : 0); + *lo = l; +} +/* Double nearest the 128-bit value (hi, lo): the magnitude is converted + * (high word scaled by 2^64 plus the low word) and the sign reapplied, so + * a small negative total is not lost in 2^64 - |s|. One formula + * everywhere so every path agrees bit for bit. */ +static inline double ray_i128_to_f64(int64_t hi, uint64_t lo) { + bool neg = hi < 0; + uint64_t h = (uint64_t)hi, l = lo; + if (neg) { l = ~l + 1u; h = ~h + (l == 0 ? 1u : 0u); } + double d = (double)h * 18446744073709551616.0 + (double)l; + return neg ? -d : d; +} + /* Whole-column sum (int64 wraparound) and non-null count of an integer * column from its chunk-zone per-chunk aggregates; false when the column - * carries none for its current length. exact_f64 (optional): whether the - * sum is exactly what a double accumulation of the rows gives. */ -bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out, bool* exact_f64); + * carries none for its current length. */ +bool ray_zone_int_sum(ray_t* x, int64_t* sum_out, int64_t* nn_out); +/* The exact 128-bit sum (hi, lo) and non-null count from the same + * metadata: true when the per-chunk high words are stored, or when no + * partial sum can wrap int64 (then the wrapped sum is the exact one). */ +bool ray_zone_int_sum128(ray_t* x, int64_t* hi_out, uint64_t* lo_out, int64_t* nn_out); /* Build a RAY_IDX_DICT (codes + distinct values) for STR vector `v` WITHOUT * attaching it — standalone RAY_INDEX object (caller releases / attaches). diff --git a/src/ops/internal.h b/src/ops/internal.h index 3a271d274..f4381f88e 100644 --- a/src/ops/internal.h +++ b/src/ops/internal.h @@ -1228,6 +1228,12 @@ ray_t* desc_vec_eager(ray_t* x); /* OP_PEARSON_CORR per-group accumulators: x-side piggybacks on SUM and * SUMSQ blocks; this flag enables the y-side blocks (Σy, Σy², Σxy). */ #define GHT_NEED_PEARSON 0x10 +/* Exact integer AVG: an extra int64 block (off_sum_hi) carries the high + * word of a 128-bit two's-complement sum next to each off_sum slot, so an + * integer mean never divides a wrapped int64 (ray_i128_add / ray_i128_to_f64 + * in idxop.h). Set whenever an OP_AVG agg has a non-float input; SUM keeps + * reading off_sum alone (int64 wraparound is its contract). */ +#define GHT_NEED_SUM128 0x20 /* ── ght_layout_t — inline-or-spill, fixed-size, by-value embeddable ── * @@ -1300,6 +1306,9 @@ typedef struct { uint16_t off_sum_y; uint16_t off_sumsq_y; uint16_t off_sumxy; + /* High words of the 128-bit integer sums (GHT_NEED_SUM128); 0 when the + * layout carries none. */ + uint16_t off_sum_hi; /* Earliest contributing source row for this group. Every packed entry * carries its source row in the tail slot; partition merges retain the * minimum so output order is independent of radix partition count. */ diff --git a/src/ops/pivot.c b/src/ops/pivot.c index 1851634fa..9d25bcc30 100644 --- a/src/ops/pivot.c +++ b/src/ops/pivot.c @@ -1607,6 +1607,12 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { uint8_t need_flags = GHT_NEED_SUM; /* always need sum (used for FIRST/LAST too) */ if (agg_op == OP_MIN) need_flags |= GHT_NEED_MIN; if (agg_op == OP_MAX) need_flags |= GHT_NEED_MAX; + /* Integer avg divides the exact 128-bit sum (high words in off_sum_hi), + * like every group engine — never a wrapped int64. */ + if (agg_op == OP_AVG && (vcol->type == RAY_I64 || vcol->type == RAY_TIMESTAMP || + (vcol->type != RAY_F64 && vcol->type != RAY_F32 && + nrows >= ((int64_t)1 << 31)))) + need_flags |= GHT_NEED_SUM128; /* n_keys/n_aggs are no longer capped: ght_compute_layout spills to an * owned heap block (ly.spill_hdr) whenever n_keys exceeds GHT_INLINE @@ -2143,7 +2149,10 @@ ray_t* exec_pivot(ray_graph_t* g, ray_op_t* op, ray_t* tbl) { case OP_AVG: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } v = val_is_f64 ? ROW_RD_F64(row, ly.off_sum, s) / nn - : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; + : (ly.need_flags & GHT_NEED_SUM128) + ? ray_i128_to_f64(ROW_RD_I64(row, ly.off_sum_hi, s), + (uint64_t)ROW_RD_I64(row, ly.off_sum, s)) / nn + : (double)ROW_RD_I64(row, ly.off_sum, s) / nn; break; case OP_MIN: if (nn == 0) { v = NULL_F64; ray_vec_set_null(new_col, (int64_t)r, true); break; } diff --git a/src/ops/query.c b/src/ops/query.c index f00da25a9..028c53bc2 100644 --- a/src/ops/query.c +++ b/src/ops/query.c @@ -549,7 +549,7 @@ static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t uint16_t op = resolve_agg_opcode(el[0]->i64); bool int_col = col->type == RAY_I64 || col->type == RAY_I32 || col->type == RAY_I16 || col->type == RAY_U8; - int64_t zs, zn; bool exact = false; + int64_t zs, zn, zh; uint64_t zl; switch (op) { case OP_COUNT: break; case OP_MIN: case OP_MAX: { @@ -560,10 +560,10 @@ static ray_t* select_aggs_from_metadata(ray_t* tbl, ray_t** dict_elems, int64_t break; } case OP_SUM: - if (!int_col || !ray_zone_int_sum(col, &zs, &zn, NULL)) return NULL; + if (!int_col || !ray_zone_int_sum(col, &zs, &zn)) return NULL; break; case OP_AVG: - if (!int_col || !ray_zone_int_sum(col, &zs, &zn, &exact) || !exact) return NULL; + if (!int_col || !ray_zone_int_sum128(col, &zh, &zl, &zn)) return NULL; break; default: return NULL; } diff --git a/src/store/col.c b/src/store/col.c index c38a523ea..295a17364 100644 --- a/src/store/col.c +++ b/src/store/col.c @@ -1754,9 +1754,29 @@ static ray_t* col_mmap_impl(const char* path, struct ray_sym_domain_s* dom, ray_t* r = ray_index_attach_built(&vec, idx); if (r && !RAY_IS_ERR(r)) vec = r; } - /* The munmap size is derived from the column + index at free time - * (ray_free), so no aux slot is needed — leaving str_pool intact on STR. */ /* Attach failure → column loads unindexed; correctness unaffected. */ + /* The mapping is longer than the payload by the index region, and + * ray_free can only size it from the attached index. Once the + * column drops that index — an in-place edit of the loaded column's + * only reference detaches a mapped index without releasing it — or + * when ray_index_inline_map discarded a child that is still in the + * file, the tail would stay mapped for the life of the process. So + * a mapped column that carries an index registers its region under + * its own address with the true length, exactly as str_pool_cow does + * for a string column whose pool pointer stops leading to it; a + * string column already reaches the region through its pool. */ + if (vec->type != RAY_STR && (vec->attrs & RAY_ATTR_HAS_INDEX)) { + ray_file_map_t* m = (ray_file_map_t*)ray_sys_alloc(sizeof(*m)); + if (m) { + m->base = cm.mapped; + m->len = cm.mapped_size; + m->rc = 1; + m->next = NULL; + ray_file_map_register(vec, m); + } + /* No descriptor (oom): the free falls back to sizing from the + * index, the behaviour before this registration existed. */ + } } return vec; diff --git a/test/rfl/group/avg_exact_i128.rfl b/test/rfl/group/avg_exact_i128.rfl new file mode 100644 index 000000000..0217ee916 --- /dev/null +++ b/test/rfl/group/avg_exact_i128.rfl @@ -0,0 +1,139 @@ +;; avg of an integer column divides the EXACT 128-bit sum in every group +;; engine: keyless, dense/hash/radix keyed (v2), the float-key legacy hash +;; path, the small-table serial finishes, the slice-indexed path, pivot, +;; nullable inputs and a splayed copy — all the same bits as (avg vec). +;; SUM keeps its int64 wraparound contract next to the exact mean. +;; A column of 2^62 averages to exactly 2^62; alternating +(2^63-1-i) / +;; -(2^63-1-i) pairs sum to exactly -1 per pair, so their mean is -0.5 +;; whenever a group holds whole pairs (keys derive from (div d 2)). +(set n 70000) +(set d (til n)) +(set hc (take [4611686018427387904] n)) +(set hv (* (- (* 2 (% d 2)) 1) (- 9223372036854775807 d))) +(set p (div d 2)) +(set k3 (% p 3)) +(set f3 (as 'F64 k3)) +(set k350 (% p 350)) +(set ksp (* (% p 350) 1000003)) +(set k35k (% p 35000)) +(set T (table [c v d k3 f3 k350 ksp k35k] (list hc hv d k3 f3 k350 ksp k35k))) +(set E 4611686018427387904.0) +(set allE (fn [x] (== (count x) (sum (== x E))))) +(set allH (fn [x] (== (count x) (sum (== x -0.5))))) +;; the bare column +(avg hc) -- 4611686018427387904.0 +(avg hv) -- -0.5 +;; keyless select: one agg, a where, two aggs over one column (sum stays wrapped) +(at (select {from: T b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v)}) 'b) -- [-0.5] +(== (first (at (select {from: T b: (avg v)}) 'b)) (avg hv)) -- true +(== (first (at (select {from: T b: (avg c)}) 'b)) (avg hc)) -- true +(at (select {from: T where: (>= d 2) b: (avg v)}) 'b) -- [-0.5] +(at (select {from: T where: (>= d 2) b: (avg c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg c) s: (sum c)}) 'b) -- [4611686018427387904.0] +(at (select {from: T b: (avg v) s: (sum v)}) 'b) -- [-0.5] +(at (select {from: T b: (avg v) s: (sum v)}) 's) -- [-35000] +(at (select {from: T b: (avg c) m: (max c)}) 'b) -- [4611686018427387904.0] +;; few groups (dense), 350 groups, 35000 groups of two rows (each group's +;; own total overflows int64), sparse keys, a where, two keys +(allE (at (select {from: T by: k3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: k350 b: (avg v)}) 'b)) -- true +(count (select {from: T by: k35k b: (avg c)})) -- 35000 +(allE (at (select {from: T by: k35k b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: k35k b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: ksp b: (avg v)}) 'b)) -- true +(allE (at (select {from: T by: ksp b: (avg c) s: (sum c)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: k3 b: k350} m: (avg v)}) 'm)) -- true +;; avg exact while sum of the same column wraps (23334, 23334, 23332 rows of 2^62) +(at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 's) -- [0Nl 0Nl 0] +(allE (at (select {from: T by: k3 asc: k3 b: (avg c) s: (sum c)}) 'b)) -- true +;; float keys take the legacy hash engine (radix merge + row emit) +(allE (at (select {from: T by: f3 b: (avg c)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: f3 b: (avg v) s: (sum v)}) 'b)) -- true +(allH (at (select {from: T where: (>= d 2) by: f3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T by: {a: f3 b: (as 'F64 k350)} m: (avg v)}) 'm)) -- true +;; small tables: the serial finishes +(allH (at (select {from: T where: (< d 60) by: k3 b: (avg v)}) 'b)) -- true +(allH (at (select {from: T where: (< d 60) by: f3 b: (avg v)}) 'b)) -- true +(at (select {from: T where: (< d 60) b: (avg v)}) 'b) -- [-0.5] +;; narrow and temporal inputs agree with the bare vector avg +(set i32 (as 'I32 (* (- (* 2 (% d 2)) 1) (- 2147483647 (% d 1000))))) +(set i16 (as 'I16 (* (- (* 2 (% d 2)) 1) (- 32767 (% d 100))))) +(set u8 (as 'U8 (% d 256))) +(set bo (== (% d 3) 0)) +(set dt (as 'DATE (- 2147483647 (% d 1000)))) +(set tm (as 'TIME (- 2147483647 (% d 1000)))) +(set ts (as 'TIMESTAMP hv)) +(set TN (table [k3 f3 i32 i16 u8 bo dt tm ts] (list k3 f3 i32 i16 u8 bo dt tm ts))) +(at (select {from: TN a: (avg i32)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg i16)}) 'a) -- [-0.5] +(at (select {from: TN a: (avg ts)}) 'a) -- [-0.5] +(== (first (at (select {from: TN a: (avg u8)}) 'a)) (avg u8)) -- true +(== (first (at (select {from: TN a: (avg bo)}) 'a)) (avg bo)) -- true +(== (first (at (select {from: TN a: (avg dt)}) 'a)) (avg dt)) -- true +(== (first (at (select {from: TN a: (avg tm)}) 'a)) (avg tm)) -- true +(at (select {from: TN by: k3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i32)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg i16)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: k3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(at (select {from: TN by: f3 a: (avg ts)}) 'a) -- [-0.5 -0.5 -0.5] +(== (at (select {from: TN by: k3 asc: k3 a: (avg u8)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg u8)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg bo)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg bo)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg dt)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg dt)}) 'a)) -- [true true true] +(== (at (select {from: TN by: k3 asc: k3 a: (avg tm)}) 'a) (at (select {from: TN by: f3 asc: f3 a: (avg tm)}) 'a)) -- [true true true] +;; nulls are skipped; an all-null group is null +(set hcn (* hc (at [1 0N] (as 'I64 (== (% d 7) 0))))) +(set TNu (table [c d k3 f3] (list hcn d k3 f3))) +(at (select {from: TNu a: (avg c)}) 'a) -- [4611686018427387904.0] +(== (first (at (select {from: TNu a: (avg c)}) 'a)) (avg hcn)) -- true +(at (select {from: TNu a: (avg c) s: (sum c)}) 'a) -- [4611686018427387904.0] +(at (select {from: TNu where: (>= d 2) a: (avg c)}) 'a) -- [4611686018427387904.0] +(allE (at (select {from: TNu by: k3 a: (avg c)}) 'a)) -- true +(allE (at (select {from: TNu by: f3 a: (avg c)}) 'a)) -- true +(set TA (table [c k3 f3] (list (* hc (at [1 0N] (as 'I64 (== k3 2)))) k3 f3))) +(at (select {from: TA by: k3 asc: k3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: TA by: f3 asc: f3 a: (avg c)}) 'a) -- [4611686018427387904.0 4611686018427387904.0 0Nf] +(at (select {from: (select {from: TA where: (== k3 2)}) a: (avg c)}) 'a) -- [0Nf] +;; pivot +(set TP (table [r pc c v] (list (% p 3) (% (div d 6) 2) hc hv))) +(set P (pivot TP 'r 'pc 'v avg)) +(at P (at (cols P) 1)) -- [-0.5 -0.5 -0.5] +(at P (at (cols P) 2)) -- [-0.5 -0.5 -0.5] +(set Q (pivot TP 'r 'pc 'c avg)) +(allE (at Q (at (cols Q) 1))) -- true +(allE (at Q (at (cols Q) 2))) -- true +;; slice-indexed group (hash index on the SYM key + in-filter) +(set TS (update {from: (table [s v c] (list (as 'SYM (% p 5)) hv hc)) s: (.idx.hash s)})) +(at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg v)}) 'a) -- [-0.5 -0.5 -0.5] +(allE (at (select {from: TS where: (in s ['0 '1 '2]) by: s a: (avg c) s2: (sum c)}) 'a)) -- true +;; single-key sparse scatter path (wide I32 key, avg over strlen of a STR +;; column, a count, a where): the mean of the lengths, bit for bit +(set cid (as 'I32 (* 35000 (% (* d 2654435761) 6000)))) +(set url (concat "http://example.com/some/path/" (as 'STR (* d (% d 977))))) +(set H (table [cid url c] (list cid url hc))) +(set R (select {from: H by: cid asc: cid l: (avg (strlen url)) n: (count url) where: (!= url "")})) +(count R) -- 6000 +(== (first (at R 'l)) (avg (strlen (at (select {from: H where: (== cid 0)}) 'url)))) -- true +(== (at R 'l) (at (select {from: H by: cid asc: cid l: (avg (strlen url)) where: (!= url "")}) 'l)) -- (take [true] 6000) +(allE (at (select {from: H by: cid l: (avg c) n: (count url)}) 'l)) -- true +(allE (at (select {from: H by: cid l: (avg c) s: (sum c) where: (!= url "")}) 'l)) -- true +;; a splayed copy answers exactly what the in-memory table answers +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 +(set Tm (table [c v d k3 f3 k350] (list hc hv d k3 f3 k350))) +(.db.splayed.set "rf_test_avg_i128/" Tm) +(set Ts (.db.splayed.get "rf_test_avg_i128/")) +(at (select {from: Ts a: (avg v) b: (avg c)}) 'a) -- [-0.5] +(at (select {from: Ts a: (avg v) b: (avg c)}) 'b) -- [4611686018427387904.0] +(== (at (select {from: Ts a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm a: (avg v) b: (avg c)}) 'b)) -- [true] +(== (at (select {from: Ts a: (avg v)}) 'a) (at (select {from: Tm a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts where: (>= d 2) a: (avg v)}) 'a) (at (select {from: Tm where: (>= d 2) a: (avg v)}) 'a)) -- [true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'a)) -- [true true true] +(== (at (select {from: Ts by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: k3 asc: k3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(allH (at (select {from: Ts by: k350 a: (avg v)}) 'a)) -- true +(== (at (select {from: Ts by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b) (at (select {from: Tm by: f3 asc: f3 a: (avg v) b: (avg c)}) 'b)) -- [true true true] +(== (avg (at Ts 'c)) (first (at (select {from: Ts a: (avg c)}) 'a))) -- true +(.sys.exec "rm -rf rf_test_avg_i128") -- 0 diff --git a/test/rfl/store/splayed_zone_aggs.rfl b/test/rfl/store/splayed_zone_aggs.rfl index 4716dc305..e340e9c3b 100644 --- a/test/rfl/store/splayed_zone_aggs.rfl +++ b/test/rfl/store/splayed_zone_aggs.rfl @@ -22,9 +22,11 @@ (at (at R 's16) 0) -- (sum (at Tm 'a16)) (at (at R 'mn) 0) -- 2000.01.01 (at (at R 'c) 0) -- 200000 -;; sums past double's integer range: avg is left to the row-wise path and -;; still agrees; a float column and a filter take the usual path too +;; sums past double's integer range: the metadata keeps the exact 128-bit +;; per-chunk sums and the row-wise reduction accumulates the same way, so +;; avg agrees bit for bit; a float column and a filter take the usual path (at (select {from: T a: (avg big)}) 'a) -- (at (select {from: Tm a: (avg big)}) 'a) +(avg (at T 'big)) -- (avg (at Tm 'big)) (at (select {from: T a: (sum a16) b: (sum f)}) 'b) -- (at (select {from: Tm a: (sum a16) b: (sum f)}) 'b) (at (select {from: T a: (sum a16) where: (> a32 0)}) 'a) -- (at (select {from: Tm a: (sum a16) where: (> a32 0)}) 'a) ;; scalar forms read the same metadata @@ -44,6 +46,27 @@ (at (select {from: TX a: (min m) b: (max m)}) 'a) -- [9223372036854775807] (min (at TX 'm)) -- 9223372036854775807 (.sys.exec "rm -rf rf_test_zone_max") -- 0 +;; the exact mean of values that overflow int64 within every chunk: +;; alternating +(2^63-1-i) / -(2^63-1-i) pairs sum to exactly one per pair, +;; and a column of 2^62 averages to exactly 2^62 — from the metadata, from +;; the row-wise reduction, and from a parted copy +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 +(set hn 150000) +(set hi (til hn)) +(set hv (* (- (* 2 (% hi 2)) 1) (- 9223372036854775807 hi))) +(set hc (take [4611686018427387904] hn)) +(.db.splayed.set "rf_test_zone_huge/" (table [hv hc] (list hv hc))) +(set TH (.db.splayed.get "rf_test_zone_huge/")) +(set THm (table [hv hc] (list hv hc))) +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: THm a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'a) -- [-0.5] +(at (select {from: TH a: (avg hv) b: (avg hc)}) 'b) -- [4611686018427387904.0] +(avg hv) -- -0.5 +(avg (at TH 'hv)) -- -0.5 +(avg (at TH 'hc)) -- 4611686018427387904.0 +(at (select {from: TH a: (avg hv) where: (>= hi 0)}) 'a) -- (at (select {from: THm a: (avg hv) where: (>= hi 0)}) 'a) +(.sys.exec "rm -rf rf_test_zone_huge") -- 0 ;; an empty table keeps the planner's answers (at (select {from: (select {from: T where: (< a64 -1000000)}) c: (count a16)}) 'c) -- (at (select {from: (select {from: Tm where: (< a64 -1000000)}) c: (count a16)}) 'c) (.sys.exec "rm -rf rf_test_zone_aggs rf_test_zone_aggs.csv rf_test_zone_aggs.csv.bak") -- 0 diff --git a/test/test_agg_engine.c b/test/test_agg_engine.c index ebd94475c..3adb1a8ba 100644 --- a/test/test_agg_engine.c +++ b/test/test_agg_engine.c @@ -2103,6 +2103,101 @@ static test_result_t test_group_values_f64(void) { PASS(); } + +/* ══════════════════════════════════════════════════════════════════════ + * Exact integer AVG on the legacy engines (v2 OFF) and on v2 (ON). + * A column of 2^62 averages to exactly 2^62 and alternating + * +(2^63-1-i) / -(2^63-1-i) pairs to exactly -0.5 whenever a group holds + * whole pairs — only the 128-bit sum gives those bits (the wrapped int64 + * total is garbage, a double running sum loses the low bits). The key + * type and the row count steer the legacy ladder: an I64 key over 70000 + * rows takes the direct-array path, the same key over 60 rows the serial + * finish, an F64 key the hash/radix row layout, and a sparse I64 key + * (range far above the row count) the single-key sparse paths, which now + * hand an integer AVG to the row layout. */ +static ray_t* avg128_make(int64_t n, bool f64_key, bool sparse_key) { + ray_t* kvec = ray_vec_new(f64_key ? RAY_F64 : RAY_I64, n); kvec->len = n; + ray_t* vvec = ray_vec_new(RAY_I64, n); vvec->len = n; + ray_t* cvec = ray_vec_new(RAY_I64, n); cvec->len = n; + int64_t* vd = (int64_t*)ray_data(vvec); + int64_t* cd = (int64_t*)ray_data(cvec); + for (int64_t i = 0; i < n; i++) { + int64_t k = (i / 2) % 3; + if (sparse_key) k *= 1000000007LL; + if (f64_key) ((double*)ray_data(kvec))[i] = (double)k; + else ((int64_t*)ray_data(kvec))[i] = k; + int64_t mag = INT64_MAX - i; + vd[i] = (i & 1) ? mag : -mag; + cd[i] = (int64_t)1 << 62; + } + ray_t* tbl = ray_table_new(3); + tbl = ray_table_add_col(tbl, ray_sym_intern("k", 1), kvec); ray_release(kvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("v", 1), vvec); ray_release(vvec); + tbl = ray_table_add_col(tbl, ray_sym_intern("c", 1), cvec); ray_release(cvec); + return tbl; +} +static ray_op_t* gb_avg128(ray_graph_t* g) { + ray_op_t* k = ray_scan(g, "k"); ray_op_t* v = ray_scan(g, "v"); ray_op_t* c = ray_scan(g, "c"); + uint16_t ops[] = { OP_AVG, OP_AVG, OP_SUM, OP_COUNT }; + ray_op_t* ins[] = { v, c, c, c }; ray_op_t* keys[] = { k }; + return ray_group(g, keys, 1, ops, ins, 4); +} +/* Run gb_avg128 with the v2 flag as given; the 2nd and 3rd result columns + * are the two means. Fails unless every group is exactly -0.5 / 2^62. */ +static test_result_t avg128_check(ray_t* tbl, bool v2, const char* what) { + ray_agg_engine_v2 = v2; + ray_graph_t* g = ray_graph_new(tbl); + ray_t* r = ray_execute(g, gb_avg128(g)); + if (r && ray_is_lazy(r)) r = ray_lazy_materialize(r); + ray_agg_engine_v2 = true; /* restore default */ + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + if (!r || RAY_IS_ERR(r) || r->type != RAY_TABLE || ray_table_ncols(r) != 5) { + res = (test_result_t){ TEST_FAIL, "avg128: bad result shape" }; + } else { + ray_t* mv = ray_table_get_col_idx(r, 1); + ray_t* mc = ray_table_get_col_idx(r, 2); + int64_t ng = ray_table_nrows(r); + if (ng != 3 || !mv || !mc || mv->type != RAY_F64 || mc->type != RAY_F64) { + res = (test_result_t){ TEST_FAIL, "avg128: expected 3 groups of F64 means" }; + } else { + for (int64_t i = 0; i < ng; i++) { + double a = ((const double*)ray_data(mv))[i]; + double b = ((const double*)ray_data(mc))[i]; + if (a != -0.5 || b != 4611686018427387904.0) { + snprintf(ray_test_fail_buf, sizeof ray_test_fail_buf, + "%s: group %lld mean(v)=%.17g mean(c)=%.17g (want -0.5, 2^62)", + what, (long long)i, a, b); + res = (test_result_t){ TEST_FAIL, ray_test_fail_buf }; + break; + } + } + } + } + if (r && !RAY_IS_ERR(r)) ray_release(r); + ray_graph_free(g); + return res; +} +static test_result_t test_avg_exact_i128_engines(void) { + ray_heap_init(); (void)ray_sym_init(); + struct { int64_t n; bool f64; bool sparse; const char* name; } shapes[] = { + { HC_N, false, false, "i64 key, 70000 rows" }, + { 60, false, false, "i64 key, 60 rows" }, + { HC_N, true, false, "f64 key, 70000 rows" }, + { 60, true, false, "f64 key, 60 rows" }, + { HC_N, false, true, "sparse i64 key, 70000 rows" }, + { 60, false, true, "sparse i64 key, 60 rows" }, + }; + test_result_t res = (test_result_t){ TEST_PASS, NULL }; + for (size_t i = 0; i < sizeof(shapes) / sizeof(shapes[0]) && res.status == TEST_PASS; i++) { + ray_t* tbl = avg128_make(shapes[i].n, shapes[i].f64, shapes[i].sparse); + res = avg128_check(tbl, false, shapes[i].name); + if (res.status == TEST_PASS) res = avg128_check(tbl, true, shapes[i].name); + ray_release(tbl); + } + ray_sym_destroy(); ray_heap_destroy(); + return res; +} + const test_entry_t agg_engine_entries[] = { { "pearson_old_engine_r_vs_r2", test_pearson_old_engine_r_vs_r2, NULL, NULL }, { "diff_group_pearson_1k", test_diff_group_pearson_1k, NULL, NULL }, @@ -2179,5 +2274,6 @@ const test_entry_t agg_engine_entries[] = { { "group_keys_i_i32", test_group_keys_i_i32, NULL, NULL }, { "group_keys_multi", test_group_keys_multi, NULL, NULL }, { "agg_run_one_i64", test_agg_run_one_i64, NULL, NULL }, + { "avg_exact_i128_engines", test_avg_exact_i128_engines, NULL, NULL }, { NULL, NULL, NULL, NULL }, }; diff --git a/test/test_index.c b/test/test_index.c index d79a0e3d4..bcba719ff 100644 --- a/test/test_index.c +++ b/test/test_index.c @@ -35,6 +35,9 @@ #include "ops/rowsel.h" #include "store/col.h" #include +#include +#include +#include #include #include #include @@ -581,6 +584,67 @@ static test_result_t test_index_persistence_roundtrip(void) { PASS(); } +/* ─── Mapped column drops its own index: the whole mapping is unmapped ── + * + * A column loaded by mmap with an inline index region is longer than its + * payload. Dropping the index from the loaded column itself (the sole + * reference: an in-place edit does exactly this) used to leave the index + * tail mapped for the life of the process, because ray_free sized the + * unmap from the index it no longer had. The file is laid out so the + * region crosses into a page of its own; after the free that page must + * be gone (msync reports ENOMEM on an unmapped range). */ +static test_result_t test_index_mapped_drop_unmaps_tail(void) { + ray_heap_init(); + /* Lay the file out so the inline index region crosses into a page of + * its own whatever the page size (4 KiB on Linux, 16 KiB on Apple + * silicon): the payload ends 64 bytes short of the second page. */ + long pg = sysconf(_SC_PAGESIZE); + TEST_ASSERT_TRUE(pg >= 4096); + int64_t n = (2 * (int64_t)pg - 96) / 8; + ray_t* v = ray_vec_new(RAY_I64, n); + for (int64_t i = 0; i < n; i++) { int64_t x = i * 3; v = ray_vec_append(v, &x); } + TEST_ASSERT_FALSE(RAY_IS_ERR(v)); + ray_t* w = v; + TEST_ASSERT_FALSE(RAY_IS_ERR(ray_index_attach_chunk_zone(&w, 8))); + + char path[] = "/tmp/idx_drop_unmap_XXXXXX"; + int fd = mkstemp(path); + TEST_ASSERT_TRUE(fd >= 0); + close(fd); + TEST_ASSERT_EQ_I(ray_col_save(w, path), RAY_OK); /* writes the inline index region too */ + ray_release(w); + + struct stat st; + TEST_ASSERT_EQ_I(stat(path, &st), 0); + TEST_ASSERT_TRUE(st.st_size > 2 * pg); /* the region reaches a further page */ + size_t mapped = ((size_t)st.st_size + (size_t)pg - 1) & ~((size_t)pg - 1); + + ray_t* m = ray_col_mmap(path); + TEST_ASSERT_FALSE(RAY_IS_ERR(m)); + TEST_ASSERT_EQ_U(m->mmod, 1); + TEST_ASSERT_TRUE(m->attrs & RAY_ATTR_HAS_INDEX); + TEST_ASSERT_EQ_I((int)ray_index_payload(m->index)->kind, RAY_IDX_CHUNK_ZONE); + char* last_page = (char*)m + mapped - (size_t)pg; + TEST_ASSERT_EQ_I(msync(last_page, (size_t)pg, MS_ASYNC), 0); /* mapped while loaded */ + + /* Sole reference: the drop detaches the mapped index in place. */ + ray_t* d = m; + ray_t* r = ray_index_drop(&d); + TEST_ASSERT_FALSE(RAY_IS_ERR(r)); + TEST_ASSERT_TRUE(d == m); + TEST_ASSERT_FALSE(d->attrs & RAY_ATTR_HAS_INDEX); + int64_t* data = (int64_t*)ray_data(d); + TEST_ASSERT_EQ_I(data[n - 1], (n - 1) * 3); + + ray_release(d); + errno = 0; + int rc = msync(last_page, (size_t)pg, MS_ASYNC); + TEST_ASSERT_TRUE(rc == -1 && errno == ENOMEM); /* the tail page is unmapped */ + unlink(path); + ray_heap_destroy(); + PASS(); +} + /* ─── Slice null detection on indexed/parent vec ───────────────────── */ static test_result_t test_index_aux_helper_slice(void) { @@ -3784,6 +3848,7 @@ const test_entry_t index_entries[] = { { "index/null_readers_through_helper", test_index_null_readers_through_helper, NULL, NULL }, { "index/aux_helper_slice", test_index_aux_helper_slice, NULL, NULL }, { "index/drop_under_shared_cow", test_index_drop_under_shared_cow, NULL, NULL }, + { "index/mapped_drop_unmaps_tail", test_index_mapped_drop_unmaps_tail, NULL, NULL }, { "index/persistence_roundtrip", test_index_persistence_roundtrip, NULL, NULL }, { "index/bool_zone_and_hash", test_index_bool_zone_and_hash, NULL, NULL }, { "index/i16_zone_and_hash", test_index_i16_zone_and_hash, NULL, NULL },