Groups, merges and sets
These calls build on sorting: totals per key, distinct values, merging sorted lists, and set operations such as intersection.
Totals per key: reduce_by_key
reduce_by_key(keys, values, op) groups equal keys and combines their values. It returns the distinct keys in order
and one result per key, like SQL SELECT key, SUM(value) ... GROUP BY key. count_by_key(keys) counts the rows of
each key.
import numpy as np
import kwker
customer = np.array([3, 1, 3, 2, 1, 3])
spent = np.array([10.0, 5.0, 2.5, 8.0, 1.0, 4.0])
keys, total = kwker.reduce_by_key(customer, spent, op="sum")
print(keys, total)
keys, visits = kwker.reduce_by_key(customer, op="count")
print(keys, visits)
[1 2 3] [ 6. 8. 16.5]
[1 2 3] [2 1 3]
use kwker::{Order, Reduction};
fn main() {
let customer: [i64; 6] = [3, 1, 3, 2, 1, 3];
let spent = [10.0, 5.0, 2.5, 8.0, 1.0, 4.0];
let (keys, total) = kwker::reduce_by_key(&customer, &spent, Reduction::Sum, Order::ASCENDING);
println!("{keys:?} {total:?}");
let (keys, visits) = kwker::count_by_key(&customer, Order::ASCENDING);
println!("{keys:?} {visits:?}");
}
[1, 2, 3] [6.0, 8.0, 16.5] [1, 2, 3] [2, 1, 3]
#include <inttypes.h>
#include <stdio.h>
#include <kwker.h>
int main(void) {
const int64_t customer[] = {3, 1, 3, 2, 1, 3};
const double spent[] = {10.0, 5.0, 2.5, 8.0, 1.0, 4.0};
int64_t keys[6];
double total[6];
uint64_t visits[6];
size_t groups;
kwker_i64_reduce_by_key_f64(customer, spent, 6, KWKER_SUM, KWKER_ASCENDING, keys, total, &groups);
for (size_t g = 0; g < groups; g++) printf("customer %" PRId64 ": %g\n", keys[g], total[g]);
kwker_i64_count_by_key(customer, 6, KWKER_ASCENDING, keys, visits, &groups);
for (size_t g = 0; g < groups; g++) printf("customer %" PRId64 ": %" PRIu64 " visits\n", keys[g], visits[g]);
return 0;
}
customer 1: 6 customer 2: 8 customer 3: 16.5 customer 1: 2 visits customer 2: 1 visits customer 3: 3 visits
#include <cstdint>
#include <iostream>
#include <vector>
#include <kwker.hpp>
int main() {
std::vector<int64_t> customer{3, 1, 3, 2, 1, 3};
std::vector<double> spent{10.0, 5.0, 2.5, 8.0, 1.0, 4.0};
auto total = kwker::reduce_by_key(customer, spent, kwker::Combine::sum);
for (size_t g = 0; g < total.keys.size(); g++) std::cout << "customer " << total.keys[g] << ": " << total.values[g] << '\n';
auto visits = kwker::count_by_key(customer);
for (size_t g = 0; g < visits.keys.size(); g++) std::cout << "customer " << visits.keys[g] << ": " << visits.values[g] << " visits\n";
}
customer 1: 6 customer 2: 8 customer 3: 16.5 customer 1: 2 visits customer 2: 1 visits customer 3: 3 visits
const kwk = require("kwker");
const customer = new Int32Array([3, 1, 3, 2, 1, 3]);
const spent = new Float64Array([10.0, 5.0, 2.5, 8.0, 1.0, 4.0]);
const total = kwk.reduceByKey(customer, spent, "sum");
console.log(total.keys, total.values);
const visits = kwk.reduceByKey(customer, null, "count");
console.log(visits.keys, visits.values);
Int32Array(3) [ 1, 2, 3 ] Float64Array(3) [ 6, 8, 16.5 ]
Int32Array(3) [ 1, 2, 3 ] Float64Array(3) [ 2, 1, 3 ]
package main
import (
"fmt"
"kwker.io/go/kwker"
)
func main() {
customer := []int64{3, 1, 3, 2, 1, 3}
spent := []float64{10.0, 5.0, 2.5, 8.0, 1.0, 4.0}
keys, total := kwker.ReduceByKey(customer, spent, kwker.ReduceSum, kwker.Ascending)
fmt.Println(keys, total)
keys, visits := kwker.CountByKey(customer, kwker.Ascending)
fmt.Println(keys, visits)
}
[1 2 3] [6 8 16.5] [1 2 3] [2 1 3]
import io.kwker.Kwker;
import java.util.Arrays;
public class Example {
public static void main(String[] args) {
long[] customer = {3, 1, 3, 2, 1, 3};
double[] spent = {10.0, 5.0, 2.5, 8.0, 1.0, 4.0};
Kwker.ByKey<long[], double[]> total = Kwker.reduceByKey(customer, spent, Kwker.SUM, Kwker.ASCENDING);
System.out.println(Arrays.toString(total.keys) + " " + Arrays.toString(total.values));
Kwker.ByKey<long[], int[]> visits = Kwker.countByKey(customer, Kwker.ASCENDING);
System.out.println(Arrays.toString(visits.keys) + " " + Arrays.toString(visits.values));
}
}
[1, 2, 3] [6.0, 8.0, 16.5] [1, 2, 3] [2, 1, 3]
using Kwker;
var customer = new long[] { 3, 1, 3, 2, 1, 3 };
var spent = new[] { 10.0, 5.0, 2.5, 8.0, 1.0, 4.0 };
var (keys, total) = Sorter.ReduceByKey<long, double>(customer, spent, Sorter.Reduce.Sum);
Console.WriteLine(string.Join(" ", keys) + " | " + string.Join(" ", total));
var (customers, visits) = Sorter.CountByKey<long>(customer);
Console.WriteLine(string.Join(" ", customers) + " | " + string.Join(" ", visits));
1 2 3 | 6 8 16.5 1 2 3 | 2 1 3
The operations are sum, min, max, first, last, mean (Python and JavaScript; mean_by_key in Rust, C and C++) and
count.
Group several key columns: group_codes
group_codes(columns) gives every row a group number. Rows with the same values in all the columns share a number,
and the numbers follow the sorted order of the groups. It also returns each group's first row and its size.
import numpy as np
import kwker
country = np.array([2, 1, 2, 1, 2])
year = np.array([2024, 2023, 2024, 2024, 2023])
codes, first_row, size = kwker.group_codes([country, year])
print(codes)
print(size)
[3 0 3 1 2]
[1 1 1 2]
use kwker::{KeyColumn, Order};
fn main() {
let country = vec![2, 1, 2, 1, 2];
let year = vec![2024, 2023, 2024, 2024, 2023];
let g = kwker::group_codes(&[(&country as &dyn KeyColumn, Order::ASCENDING), (&year as &dyn KeyColumn, Order::ASCENDING)], 1);
println!("{:?}", g.codes);
println!("{:?}", g.sizes);
}
[3, 0, 3, 1, 2] [1, 1, 1, 2]
#include <inttypes.h>
#include <stdio.h>
#include <kwker.h>
int main(void) {
const int32_t country[] = {2, 1, 2, 1, 2};
const int32_t year[] = {2024, 2023, 2024, 2024, 2023};
const kwker_column cols[] = {{country, KWKER_TYPE_I32, KWKER_ASCENDING},
{year, KWKER_TYPE_I32, KWKER_ASCENDING}};
uint32_t codes[5];
uint64_t first[5], sizes[5];
size_t groups;
kwker_group_codes(cols, 2, 5, 1, codes, first, sizes, &groups);
for (int i = 0; i < 5; i++) printf(i ? " %u" : "%u", codes[i]);
printf("\n");
for (size_t g = 0; g < groups; g++) printf(g ? " %" PRIu64 : "%" PRIu64, sizes[g]);
printf("\n");
return 0;
}
3 0 3 1 2 1 1 1 2
#include <iostream>
#include <vector>
#include <kwker.hpp>
int main() {
std::vector<int> country{2, 1, 2, 1, 2};
std::vector<int> year{2024, 2023, 2024, 2024, 2023};
auto g = kwker::group_codes({kwker::Column(country), kwker::Column(year)});
for (auto c : g.codes) std::cout << c << ' ';
std::cout << '\n';
for (auto s : g.sizes) std::cout << s << ' ';
std::cout << '\n';
}
3 0 3 1 2 1 1 1 2
const kwk = require("kwker");
const country = new Int32Array([2, 1, 2, 1, 2]);
const year = new Int32Array([2024, 2023, 2024, 2024, 2023]);
const { codes, sizes } = kwk.groupCodes([country, year]);
console.log(codes);
console.log(sizes);
Uint32Array(5) [ 3, 0, 3, 1, 2 ]
Float64Array(4) [ 1, 1, 1, 2 ]
package main
import (
"fmt"
"kwker.io/go/kwker"
)
func main() {
country := []int32{2, 1, 2, 1, 2}
year := []int32{2024, 2023, 2024, 2024, 2023}
codes, _, sizes := kwker.GroupCodes(kwker.Col(country, kwker.Ascending), kwker.Col(year, kwker.Ascending))
fmt.Println(codes)
fmt.Println(sizes)
}
[3 0 3 1 2] [1 1 1 2]
import io.kwker.Kwker;
import java.util.Arrays;
public class Example {
public static void main(String[] args) {
int[] country = {2, 1, 2, 1, 2};
int[] year = {2024, 2023, 2024, 2024, 2023};
Kwker.GroupCodes g = Kwker.groupCodes(Kwker.Column.of(country), Kwker.Column.of(year));
System.out.println(Arrays.toString(g.codes));
System.out.println(Arrays.toString(g.sizes));
}
}
[3, 0, 3, 1, 2] [1, 1, 1, 2]
using Kwker;
var country = new[] { 2, 1, 2, 1, 2 };
var year = new[] { 2024, 2023, 2024, 2024, 2023 };
var (codes, _, size) = Sorter.GroupCodes(Sorter.Column.Of(country), Sorter.Column.Of(year));
Console.WriteLine(string.Join(" ", codes));
Console.WriteLine(string.Join(" ", size));
3 0 3 1 2 1 1 1 2
For pandas, Polars and pyarrow tables, kwker.frame.group_by does the whole group-by with each library's own
rules. See DataFrames, Arrow and DuckDB.
Merge sorted lists: kway_merge
kway_merge(lists) merges any number of sorted arrays into one sorted array. When values are equal, the one from the
earlier list comes first.
import numpy as np
import kwker
monday = np.array([1, 4, 9])
tuesday = np.array([2, 4, 7, 10])
wednesday = np.array([3])
print(kwker.kway_merge([monday, tuesday, wednesday]))
[ 1 2 3 4 4 7 9 10]
use kwker::Order;
fn main() {
let monday = [1, 4, 9];
let tuesday = [2, 4, 7, 10];
let wednesday = [3];
let mut merged = vec![0; 8];
kwker::kway_merge(&[&monday[..], &tuesday[..], &wednesday[..]], &mut merged, Order::ASCENDING);
println!("{merged:?}");
}
[1, 2, 3, 4, 4, 7, 9, 10]
#include <stdio.h>
#include <kwker.h>
int main(void) {
const int32_t monday[] = {1, 4, 9};
const int32_t tuesday[] = {2, 4, 7, 10};
const int32_t wednesday[] = {3};
const int32_t* runs[] = {monday, tuesday, wednesday};
const size_t lens[] = {3, 4, 1};
int32_t merged[8];
kwker_i32_kway_merge(runs, lens, 3, KWKER_ASCENDING, merged);
for (int i = 0; i < 8; i++) printf(i ? " %d" : "%d", merged[i]);
printf("\n");
return 0;
}
1 2 3 4 4 7 9 10
#include <iostream>
#include <vector>
#include <kwker.hpp>
int main() {
std::vector<int> monday{1, 4, 9}, tuesday{2, 4, 7, 10}, wednesday{3};
auto merged = kwker::kway_merge<int>({{monday.data(), monday.size()}, {tuesday.data(), tuesday.size()},
{wednesday.data(), wednesday.size()}});
for (int v : merged) std::cout << v << ' ';
std::cout << '\n';
}
1 2 3 4 4 7 9 10
const kwk = require("kwker");
const monday = new Int32Array([1, 4, 9]);
const tuesday = new Int32Array([2, 4, 7, 10]);
const wednesday = new Int32Array([3]);
console.log(kwk.kwayMerge([monday, tuesday, wednesday]));
Int32Array(8) [ 1, 2, 3, 4, 4, 7, 9, 10 ]
package main
import (
"fmt"
"kwker.io/go/kwker"
)
func main() {
monday := []int32{1, 4, 9}
tuesday := []int32{2, 4, 7, 10}
wednesday := []int32{3}
fmt.Println(kwker.KWayMerge(kwker.Ascending, monday, tuesday, wednesday))
}
[1 2 3 4 4 7 9 10]
import io.kwker.Kwker;
import java.util.Arrays;
public class Example {
public static void main(String[] args) {
int[] monday = {1, 4, 9};
int[] tuesday = {2, 4, 7, 10};
int[] wednesday = {3};
System.out.println(Arrays.toString(Kwker.kwayMerge(Kwker.ASCENDING, monday, tuesday, wednesday)));
}
}
[1, 2, 3, 4, 4, 7, 9, 10]
using Kwker;
var monday = new[] { 1, 4, 9 };
var tuesday = new[] { 2, 4, 7, 10 };
var wednesday = new[] { 3 };
Console.WriteLine(string.Join(" ", Sorter.KWayMerge(Order.Ascending, monday, tuesday, wednesday)));
1 2 3 4 4 7 9 10
Distinct values: unique
unique(a) returns the sorted distinct values of an array, like np.unique. Add return_counts=True to count each
value, return_inverse=True for the index of every element's value, and return_index=True for where each value
first occurs.
import numpy as np
import kwker
user_ids = np.array([7, 3, 7, 9, 3, 7])
ids, counts = kwker.unique(user_ids, return_counts=True)
print(ids, counts)
[3 7 9] [2 3 1]
Set operations
In Python, intersect1d, union1d, setdiff1d and setxor1d work like NumPy's functions of the same names and
take unsorted input.
import numpy as np
import kwker
subscribed = np.array([5, 1, 9, 3, 7])
active = np.array([3, 9, 4, 5])
print(kwker.intersect1d(subscribed, active)) # in both
print(kwker.setdiff1d(subscribed, active)) # subscribed but not active
print(kwker.union1d(subscribed, active)) # in either
[3 5 9]
[1 7]
[1 3 4 5 7 9]
On arrays that are already sorted, set_op (set_op_sorted in Python) skips the sorting step. The operation is
intersection, union, difference or symmetric difference. With the multiset option, repeated values count as many
times as they occur.
import numpy as np
import kwker
a = np.array([1, 2, 2, 3, 5])
b = np.array([2, 2, 2, 5, 8])
print(kwker.set_op_sorted(a, b, "intersection"))
print(kwker.set_op_sorted(a, b, "intersection", multiset=True))
[2 5]
[2 2 5]
use kwker::{Order, SetOp};
fn main() {
let a = [1, 2, 2, 3, 5];
let b = [2, 2, 2, 5, 8];
println!("{:?}", kwker::set_op(&a, &b, SetOp::Intersection, false, Order::ASCENDING));
println!("{:?}", kwker::set_op(&a, &b, SetOp::Intersection, true, Order::ASCENDING));
}
[2, 5] [2, 2, 5]
#include <stdio.h>
#include <kwker.h>
int main(void) {
const int32_t a[] = {1, 2, 2, 3, 5};
const int32_t b[] = {2, 2, 2, 5, 8};
for (int multiset = 0; multiset <= 1; multiset++) {
int32_t out[10];
size_t len;
kwker_i32_set_op(a, 5, b, 5, KWKER_ASCENDING, 0 /* intersection */, multiset, out, &len);
for (size_t i = 0; i < len; i++) printf(i ? " %d" : "%d", out[i]);
printf("\n");
}
return 0;
}
2 5 2 2 5
#include <iostream>
#include <vector>
#include <kwker.hpp>
int main() {
std::vector<int> a{1, 2, 2, 3, 5};
std::vector<int> b{2, 2, 2, 5, 8};
for (bool multiset : {false, true}) {
for (int v : kwker::set_op(a, b, kwker::SetOp::intersection, multiset)) std::cout << v << ' ';
std::cout << '\n';
}
}
2 5 2 2 5
const kwk = require("kwker");
const a = new Int32Array([1, 2, 2, 3, 5]);
const b = new Int32Array([2, 2, 2, 5, 8]);
console.log(kwk.setOp(a, b, "intersection"));
console.log(kwk.setOp(a, b, "intersection", { multiset: true }));
Int32Array(2) [ 2, 5 ]
Int32Array(3) [ 2, 2, 5 ]
package main
import (
"fmt"
"kwker.io/go/kwker"
)
func main() {
a := []int32{1, 2, 2, 3, 5}
b := []int32{2, 2, 2, 5, 8}
fmt.Println(kwker.SetOperation(a, b, kwker.Intersection, false, kwker.Ascending))
fmt.Println(kwker.SetOperation(a, b, kwker.Intersection, true, kwker.Ascending))
}
[2 5] [2 2 5]
import io.kwker.Kwker;
import java.util.Arrays;
public class Example {
public static void main(String[] args) {
int[] a = {1, 2, 2, 3, 5};
int[] b = {2, 2, 2, 5, 8};
System.out.println(Arrays.toString(Kwker.setOperation(a, b, Kwker.INTERSECTION, false, Kwker.ASCENDING)));
System.out.println(Arrays.toString(Kwker.setOperation(a, b, Kwker.INTERSECTION, true, Kwker.ASCENDING)));
}
}
[2, 5] [2, 2, 5]
using Kwker;
var a = new int[] { 1, 2, 2, 3, 5 };
var b = new int[] { 2, 2, 2, 5, 8 };
Console.WriteLine(string.Join(" ", Sorter.SetOperation(a, b, SetOp.Intersection)));
Console.WriteLine(string.Join(" ", Sorter.SetOperation(a, b, SetOp.Intersection, multiset: true)));
2 5 2 2 5
Membership tests
isin(element, test_elements) tells, for each value of element, whether it appears in test_elements. It is
NumPy's np.isin and returns a boolean array of element's shape. Use invert=True for the values that do not
appear.
import numpy as np
import kwker
tokens = np.array([12, 7, 99, 7, 3])
stopwords = np.array([7, 3])
print(kwker.isin(tokens, stopwords))
print(tokens[~kwker.isin(tokens, stopwords)])
[False True False True True] [12 99]
Notes
Related
- Order and ranking
- DataFrames, Arrow and DuckDB
- Statistics and data helpers: running totals, lags and ranks within groups
- API reference:
reduce_by_key,kway_merge