Skip to content
3 changes: 2 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -267,4 +267,5 @@ We plan to add many GPU-accelerated, concurrent data structures to `cuCollection
`cuco::experimental::roaring_bitmap` implements a Roaring bitmap following the [Roaring bitmap format specification](https://github.com/RoaringBitmap/RoaringFormatSpec).

#### Examples:
- [Host-bulk APIs](https://github.com/NVIDIA/cuCollections/blob/dev/examples/roaring_bitmap/host_bulk_example.cu) (see [live example in godbolt](https://godbolt.org/clientstate/eJy9WAtPGzkQ_itzi1RtIMmGlEcbHteUlCq6HlSUe0hQrZxdJ7GyWW9tL5BD_Pcb2_uEpdDHXZDIrj3-5uFvxuPcOpJKyXgsncHFrcNCZ7DZdiISz1Iyo87ACdKQOG1H8lQE-t1bv4xhHT59HP3dOWYRPeLJSrDZXJ3TGzWA4hXcoAX9Xn-7g_922nDy53g0HsLR6dnH07Ph-fj0BF7A8Ph4_GE8PH_3qQvDKAKzUoKgkoorGnZLVR9YQGNJO-OQxopNGRUDGCYkmNNOv9vTct5lfBmvsTiI0pDCfpAG3BOcCBbP_AlTS5J0g3R--EAmVSxiauUpQZiS3XmSHN5HCoknVegF-C-k08NHJ1msmifVKqG-VVATUHORSuWF9Ard869ooLjozptEIj5jAYmaJ9OYXVEhSVSFqMpNcaPkSiq6rC2fSiUoqY8x3jCIQxjG2pDVZPR465YTb7QamCOAP0mjhU9vyDKJKIbdTk8Eo1MY0SWyDYOhqIRUIsuAT0HNKdR3Cy4djXLpQMT5Ik0MMAw_jqWhhYEcx7iQScg0wTVFYRKCuubwst9BILBgEkgcAo8p7GxVhsFNuFBkgkunXCyJasFU8KW2xuBfnFmT3hrpYyPyKaHBZ3euVCIHnjdjap5OugFfejXZ_K1c00JaJ1wyjNpKW2MUIH-DBTDrv3Y3dxb9RHmViliauYALgRHXqZFGSFM4IUsardraZQykMkJTHkX8GrWC2fCBUdGBC-vsNZrKUyXSWHYnLDaTro1S63sc8iYRn3jbm7u7JHztaSNCoojXqKz10JT_z44HRuSbnjFtZ6u0w9LjJ9qxs-U1qmsVJH4Tc0XhvMpjQb-kDLfabv2SLDBHEoVVGjqjoz-OTv3R6V8nH06HI__sdHg2PnnvYwk9Hw3PhwdYVxWHCQVJVZEoWBsx95MIcw6LBhajGNkDv9HVOT5jDk84jywXXbR6MLD5jqTDRH2R5YqvOeUnRM3R9FtEBZKiphmNqc5lf0FXEg7g4rPbgs4h2NI0GNRq236uEgwAaOYbJfQmEXhiYL3UytECJn2JNvpX-ZI2VGZTrLQv-746bOVAAJ4HR1i40MMvKcUUM_ZgVtvUKshwi6y8yxlR5kkO8dPod_ZuOPr9XXcZrumhjh7L1RgXsoA0OWVM38ultQtugxgsMNq9Pfzah82e_ujnjQPzUokLGLhuksq5PyG4w4tWgX1XU4LABrRE28cT3D5vbCy-hvkSefwk7m4F99XzcJswbVl8hGBN8XQN6ITOWOy22lYFjUO3lYPfAY0k_REy7mx9Axkb64Gh4n_BRCxAJRcfqLaF7ylaGveepqUWs7S8MdubRpHdbXx_nb9_x44_pWtY17XZ-wFd90R6N_0MrZRtFNmuiTxqMbNJy-4ZynTi9jXEV8zt3bzKFsAGsB_MDLOlz82M25IjRLHAJxIbdOXq_lmrwfNEt-X-lKBwUePb2MFlz7BEg_ShVBQvjE9uxaVTupJ5cHuX69df-kV_Y1KcoiqTKfrd5mbWxJpR9_5R1c6EuBwMkOpErKwuTHX3Fy3VxdzmCIou516aFQEVAvb30YVjgmKhPlS1HA7o4ft69JhZh5GIMtszX0xQzNBd4cZ7qoy9INk_tDhKDZoeQY4YsLJtt89m8oGPrTI8eIfiga46pldeYqeNtad-BhbRy1lSXh5M556wOKZhA2kmK4Xn9iSdTqlwC2sqys8odt7GK9xYnmnXcybOuEehKyhOUZFgaPyASLUfzIlYP3RzWwS59hNuZMy8a9V1dR3DHUKK1hRn0EHEJXUrlmSlN7tPFI5nnX0RAUtfrPZUsCXyl0RoQe0OUvYr9XHX7vBzzd6rbLztlqrHgkswoSV2y1GoUwSPIEUY7gGGsbJvrYImWaNVa7wavCdg91AzV-I3tXcse3-oEqBeJnQreFgaYSuE4Z2OvuZyRdVbvLF0shtLxSMyw7XZncRar-XvX8atCvloDWqXVuTTVS_zm5OOnlZkNYtHA6jXmT4XF_ilzEERBz3Op-4DpVU7MsOwPlS6A_OThFrd3llOZnlfU4MTJvt1K7QkGtX00GtsGuKV2LT0eSevu3h_ND4rKly1GQdNKT9kAu1uXLVXuCnTIKBSQv1zYNv85j6pAN_AMtd8kcuLdY7-4rsAvxXNHlZ1tOZLnJNTxFZxtNxW8UzDIKvftqJhmEiUzIkZyUy4V8grG5pL_Ao9GMAmTq7pw7FUVhwZzXe5xv3KLmvIAsNGrBwMy4COUf3ed-mUbUH2uXSecRls3V_44KDKvLP-xCGbGqo6bQc7zgQrpSh_GXTiqyDY7G-nmzhtDcNJp4OAB8HGxuYudIgI5gdy6e_2oNPB0qrwn9LtQdiJyHJifkuM2KSCGQRBhIP6DEI8HNBcWzh37Xweq3RtHuuVc_fZ_P0LZpjzEQ==))
- [Host-bulk APIs using unordered indices](https://github.com/NVIDIA/cuCollections/blob/dev/examples/roaring_bitmap/host_bulk_from_indices_example.cu) (see [live example in godbolt](https://godbolt.org/clientstate/eJyNVA1rGkEQ_SvDFRqTnnoa2sAmppWmASEkIQ2lUMuxtzfq0r29636IqeS_d-7LnGlKq7h6b2bfzHs77jawaK3MtQ3Yt20g04CNwkBxvfR8iQELhE95EAY290aUz8OjuYYj-Hx78bV_KRV-zIsHI5crd48bx2D3CD1xCONo_C6E6y-zi9kUPt7c3d7cTe9nN9fwGqaXl7Or2fT-0-cBTJWCapMFgxbNGtPBU5UrKVBb7M9S1E4uJBoG04KLFfbHg6jMG871XL-SWiifIpwJL_KhybmRehkn0mW8GAi_On-elfKhdelQ0CK1O-8G3cp464Yprql4vEbhcjNYvZSyyq3rJnRTJMUM8qyCh0e1cx8W5BpU2xKvfsQLk2ex1CkVsjFueFYopHbr3MRIXMAFZnRAznCHFhIvFfW7BA77GqFkAq9zk6LBFBrOQeMQSYSMS907nOstYeBtyUJZuIndQ4EwgdITxsgPxjzlH49jd1r2DlCrZWzPkbOnzedtuW20iaJoRJ9xCCchjELYIVGFHHeQ8WPDz73LodExKQGA8hgZw02BRmZ09Fwxtq-4W5-xrpG9VnyCy1JyuHMDddo7PPxvVT89VSdVJGNctX6yr2f3e_Rc1d_YkzxX5-Wce-Vsr-EfWPkL674A2pHNSbPUTzk7LS1QaQlbrjb-XFxnQpvipafCYbp1xmMIC64sfdUP3bUT2Ff0JyUXznNFI9Q0UwuhEIhydMF6Qe5b2H9Ndvsmu6aa7qshFLl3cHYG88BrSarbU2SElHjjVG1eCRzM5_rg9IX9dKtIrigvheTBvcgQV4F_8dQ62u1VtJTJVbHiFdII7ZCUNAadN3oXfQ8RMBhR8HGu6XoVeVbQtWCebuFAr4UYjd_6EYXzwtVXdNCnihPx5s3oBPrciNXEZvFJBP0-2edocTQTmPYVz5Lq3lYy6XAKIRSBaypEfATQ0eofwWPYxukvtxen0Q0ev1fv35rBEmA=))
- [Host-bulk APIs using serialized bitmap](https://github.com/NVIDIA/cuCollections/blob/dev/examples/roaring_bitmap/host_bulk_from_serialized_example.cu) (see [live example in godbolt](https://godbolt.org/clientstate/eJy9WAtP4zgQ_itzQUIptE1heeyWx22XLqvq9mDFcg8JVpGbuK3VNM7aDtBD_Pcb23mWsOzrrkg0scfz_GY803tHUikZj6XTv7p3WOj0t9pOROJpSqbU6TtBGhKn7UieikC_exvXMWzAxw_DvzunLKInPFkKNp2pS3qn-lC8ghu0YLu3vdvBf3ttOPtzNBwN4OT84sP5xeBydH4G6zA4PR29Hw0u337swiCKwJyUIKik4oaG3VLUexbQWNLOKKSxYhNGRR8GCQlmtLPd7Wk67zq-jtdYHERpSOEwSAPuCU4Ei6f-mKkFSbpBOjt-RJMqFjG19JQgTMnuLEmOVzmFxJMq9AL8F9LJ8ZObLFbNm2qZUN8KqBGomUil8kJ6g-b5NzRQXHRnTSQRn7KARM2bacxuqJAkqrKo0k0wUHIpFV3Ujk-kEpTU1xhvWMQldGNtyUoycrwNi4nXWgzMkIE_TqO5PxF84WMkGYnYPzT06R1ZJBHFMFjysWB0AkO6QPShcxSVkEpEHfAJqBmFevTg2tFcrx2IOJ-niREEgw8jaWBiWI5iPMgkZJLgliIxCUHdcnix3UFGYJlJIHEIPKawt1NZBjfhQpExHp1wsSCqBdoIrY3hf3VhVXpjqE8NyceEBp_cmVKJ7HvelKlZOu4GfOHVaPO38kwLYZ5wydCLS62NEYB4DubArP3a3NxYtBPpVSpiafYCLgRGQKdKGiFs4YwsaLRsa5PRkcoQTXgU8VuUCgYAfSOiA1fW2FtUladKpLHsjllsNl3rpdb3GOSNIz72drf290n4ytNKhEQRr1FY67Eq_58ej5TIg54hbW-n1MPC4yfqsbfjNYprFSB-HXNF4bKKY0E_pwxDbUO_IHPMkURh1YbO8OSPk3N_eP7X2fvzwdC_OB9cjM7e-VhSL4eDy8ER1lnFYUxBUlUkCtZKrAVJhDmHRQSLU4zogd_o8hKfMafHnEcWiy5q3e_b_EfQYaKuZ7nia0z5CVEzVP0euQJJUdKUxlTnsj-nSwlHcPXJbUHnGGyp6vdrte4wFwmGAWjkGyH0LhF4g2D91MJRAyZ9iTr6N_mRNlR2U6y8L7Z9ddzKGQF4HpxgIUMLP6cUU8zog1ltU6sAwz2i8iFHRJknOYufBr-Lt4Ph72-7i3BNL3X0Wi7GmJA5pMkoo_pBTq1NcBvIYI7e7h3g1yFs9fRHP28emZeKX8Cw6yapnPljghGetwreDzUhyNgwLbkd4o1unzc351_i-QJx_Czf_Qrfl1_Ht4mnLYtPAKzJn65hOqZTFrutthVB49Bt5cwfgEaS_ggY93a-AYyN9cBA8b9AIhagEouPRNvC9xwsjXnPw1KTWVjemfCmUWSjje-v8vfviPhzsgZ1WVu9H5C1QtK72864lbSNJLs1kic1ZjZp2YqiTCfutmbxBXV7dy-zA7AJ7Aczw4T0azPjvsQIUSzwicQ2T7m6n9Zi8D7Rbbo_IUhc1Pg2dnDZMyxQIX0pFcUL_ZNrce2UpmQW3D_k8vWXftHfmBTnKMpkin63uZk1tWbVXb2q2hkRl_0-Qp2IpZWFqe7-oqm6mNscmaLJuZXmRECFgMNDNOGUIFmoL1VNhwt6eVWOXjPn0BNRpntmi3GKWXoozHhHldEXJDbKxVVquOkVxIhhVrbx9tlsPrKxVboHZyoe6KpjeuUFdtpYe-p3YOG9HCXlMGE6-YTFMfbuj0EzXiq8t8fpZEKFW2hTEX5BsfM2VmFgeSZd7xk_Y4xCV1DcoiJB1_gBkeowmBGxcezmughy6yfc0Jh914rr6jqGEUKI1gRnrIOIS-pWNMlKbzZPFIZnnX3hAeP0laEDG5j1TzkSDIG5EjZAq4C7X6npwU9vbzI02ZRDfjhmLTDnSIS61Ewo8g8xUx_IXK3cU5n97fzvNbt6olaD8C5rDKs3oEuwdkkcDKJQVwP0jSIM4YaIqUC0VQQn6ylrPWZDoAlYuOoklfhN7ThpR6Uq1usVUXe9x6USthiaFNNA02lbEfUGh7NONpxVLCJTPJuNX1Z7Tb_6O4QVIZ8st-1Si3y7amU-JGrvaUFWsnjSgfqcaenxgF_SlOjV63ziPhJa1SNTDEthBbbm1xi1vH-wMMpAUxODG6bQ6a5vQTRXMy6ssUmI07-ZXvKhRQ8s_nB0URTz6txhMs4PmUC9G08dFGbKNAiolFD_HNmJprklLJhvYkVvnlnzeynnvv5dDL-Vm72X69ya51Unh4i9sFBze2FlEvrZVWWLN7qJRMmMmJVMhZU7qxLQnOJX6EEftnBzTVeLUlhxOzaPrY3xyuZSRIFBI1ZMhmVA-6g-4l47ZQeUfa6dr5h7W6sHH93JmXXWnjhkEwNVp-1gc53gpSDKH0Wd-CYItrZ30y3ctorhptNBhkfB5ubWPnSICGZHcuHv96DTwYqp8J_SnVDYichibH5Gjdi4wjMIgggX9XWL_HBBY23uPLTzfSy-tX2sV87DJ_P3L2TISXk=))
3 changes: 2 additions & 1 deletion benchmarks/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -94,4 +94,5 @@ ConfigureBench(BLOOM_FILTER_BENCH
###################################################################################################
# - roaring_bitmap benchmarks ---------------------------------------------------------------------
ConfigureBench(ROARING_BITMAP_BENCH
roaring_bitmap/contains_bench.cu)
roaring_bitmap/build_bench.cu
roaring_bitmap/contains_bench.cu)
145 changes: 145 additions & 0 deletions benchmarks/roaring_bitmap/build_bench.cu
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
/*
* SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: Apache-2.0
*/

#include <benchmark_utils.hpp>

#include <cuco/roaring_bitmap.cuh>
#include <cuco/utility/key_generator.cuh>

#include <nvbench/nvbench.cuh>

#include <cuda/std/cstdint>
#include <thrust/device_vector.h>
#include <thrust/reverse.h>
#include <thrust/sequence.h>
#include <thrust/sort.h>
#include <thrust/tabulate.h>

using namespace cuco::benchmark;
using namespace cuco::utility;

enum class build_mode { indices, sorted_indices, sorted_unique_indices };

template <build_mode Mode, class Dist>
void roaring_bitmap_build(nvbench::state& state, nvbench::type_list<Dist>)
{
using index_type = cuda::std::uint32_t;
using bitmap_type = cuco::experimental::roaring_bitmap<index_type>;

auto const num_inputs = state.get_int64("NumInputs");
thrust::device_vector<index_type> indices(num_inputs);

[[maybe_unused]] key_generator generator{};
if constexpr (Mode == build_mode::sorted_unique_indices) {
Comment thread
sleeepyjack marked this conversation as resolved.
thrust::sequence(indices.begin(), indices.end());
} else {
generator.generate(dist_from_state<Dist>(state), indices.begin(), indices.end());
if constexpr (Mode == build_mode::sorted_indices) {
thrust::sort(indices.begin(), indices.end());
}
}

state.add_element_count(num_inputs);
state.add_global_memory_reads<index_type>(num_inputs, "InputSize");

state.exec(nvbench::exec_tag::sync | nvbench::exec_tag::timer,
[&](nvbench::launch& launch, auto& timer) {
timer.start();
if constexpr (Mode == build_mode::indices) {
[[maybe_unused]] auto bitmap = bitmap_type::from_indices(
indices.begin(), indices.end(), {}, cuda::stream_ref{launch.get_stream()});
timer.stop();
} else if constexpr (Mode == build_mode::sorted_indices) {
[[maybe_unused]] auto bitmap = bitmap_type::from_sorted_indices(
indices.begin(), indices.end(), {}, cuda::stream_ref{launch.get_stream()});
timer.stop();
} else {
[[maybe_unused]] auto bitmap = bitmap_type::from_sorted_unique_indices(
indices.begin(), indices.end(), {}, cuda::stream_ref{launch.get_stream()});
timer.stop();
}
});
}

template <class Dist>
void roaring_bitmap_from_indices(nvbench::state& state, nvbench::type_list<Dist> types)
{
roaring_bitmap_build<build_mode::indices>(state, types);
}

template <class Dist>
void roaring_bitmap_from_sorted_indices(nvbench::state& state, nvbench::type_list<Dist> types)
{
roaring_bitmap_build<build_mode::sorted_indices>(state, types);
}

template <class Dist>
void roaring_bitmap_from_sorted_unique_indices(nvbench::state& state,
nvbench::type_list<Dist> types)
{
roaring_bitmap_build<build_mode::sorted_unique_indices>(state, types);
}

void roaring_bitmap_from_indices_array_containers(nvbench::state& state)
{
using index_type = cuda::std::uint32_t;
using bitmap_type = cuco::experimental::roaring_bitmap<index_type>;

constexpr cuda::std::int64_t num_containers = 1 << 16;
auto const cardinality = state.get_int64("ContainerCardinality");
auto const num_inputs = num_containers * cardinality;
thrust::device_vector<index_type> indices(num_inputs);

thrust::tabulate(
indices.begin(), indices.end(), [cardinality] __device__(cuda::std::int64_t index) {
auto const container = static_cast<index_type>(index / cardinality);
auto const lower = static_cast<index_type>(index % cardinality);
return (container << 16) | lower;
});
thrust::reverse(indices.begin(), indices.end());

state.add_element_count(num_inputs);
state.add_global_memory_reads<index_type>(num_inputs, "InputSize");

state.exec(nvbench::exec_tag::sync | nvbench::exec_tag::timer,
[&](nvbench::launch& launch, auto& timer) {
timer.start();
[[maybe_unused]] auto bitmap = bitmap_type::from_indices(
indices.begin(), indices.end(), {}, cuda::stream_ref{launch.get_stream()});
timer.stop();
});
}

NVBENCH_BENCH_TYPES(roaring_bitmap_from_indices,
NVBENCH_TYPE_AXES(nvbench::type_list<distribution::unique>))
.set_name("roaring_bitmap_from_indices_unique")
.set_type_axes_names({"Distribution"})
.add_int64_power_of_two_axis("NumInputs", {20, 24, 28})
.add_int64_axis("Multiplicity", {1});

NVBENCH_BENCH(roaring_bitmap_from_indices_array_containers)
.set_name("roaring_bitmap_from_indices_array_containers")
.add_int64_axis("ContainerCardinality", {1, 8, 64, 512, 4096});

NVBENCH_BENCH_TYPES(roaring_bitmap_from_indices,
NVBENCH_TYPE_AXES(nvbench::type_list<distribution::uniform>))
.set_name("roaring_bitmap_from_indices_uniform")
.set_type_axes_names({"Distribution"})
.add_int64_power_of_two_axis("NumInputs", {20, 24, 28})
.add_int64_axis("Multiplicity", {2, 8, 32});

NVBENCH_BENCH_TYPES(roaring_bitmap_from_sorted_indices,
NVBENCH_TYPE_AXES(nvbench::type_list<distribution::uniform>))
.set_name("roaring_bitmap_from_sorted_indices")
.set_type_axes_names({"Distribution"})
.add_int64_power_of_two_axis("NumInputs", {20, 24, 28})
.add_int64_axis("Multiplicity", {2, 8, 32});

NVBENCH_BENCH_TYPES(roaring_bitmap_from_sorted_unique_indices,
NVBENCH_TYPE_AXES(nvbench::type_list<distribution::unique>))
.set_name("roaring_bitmap_from_sorted_unique_indices")
.set_type_axes_names({"Distribution"})
.add_int64_power_of_two_axis("NumInputs", {20, 24, 28})
.add_int64_axis("Multiplicity", {1});
5 changes: 4 additions & 1 deletion examples/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -40,4 +40,7 @@ ConfigureExample(HYPERLOGLOG_HOST_BULK_EXAMPLE "${CMAKE_CURRENT_SOURCE_DIR}/hype
ConfigureExample(HYPERLOGLOG_DEVICE_REF_EXAMPLE "${CMAKE_CURRENT_SOURCE_DIR}/hyperloglog/device_ref_example.cu")
ConfigureExample(BLOOM_FILTER_HOST_BULK_EXAMPLE "${CMAKE_CURRENT_SOURCE_DIR}/bloom_filter/host_bulk_example.cu")
ConfigureExample(BLOOM_FILTER_PERSISTING_L2_EXAMPLE "${CMAKE_CURRENT_SOURCE_DIR}/bloom_filter/persisting_l2_example.cu")
ConfigureExample(ROARING_BITMAP_HOST_BULK_EXAMPLE "${CMAKE_CURRENT_SOURCE_DIR}/roaring_bitmap/host_bulk_example.cu")
ConfigureExample(ROARING_BITMAP_HOST_BULK_FROM_INDICES_EXAMPLE
"${CMAKE_CURRENT_SOURCE_DIR}/roaring_bitmap/host_bulk_from_indices_example.cu")
ConfigureExample(ROARING_BITMAP_HOST_BULK_FROM_SERIALIZED_EXAMPLE
"${CMAKE_CURRENT_SOURCE_DIR}/roaring_bitmap/host_bulk_from_serialized_example.cu")
40 changes: 40 additions & 0 deletions examples/roaring_bitmap/host_bulk_from_indices_example.cu
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
/*
* SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: Apache-2.0
*/

#include <cuco/roaring_bitmap.cuh>

#include <cuda/std/cstdint>
#include <thrust/device_vector.h>
#include <thrust/host_vector.h>

#include <iostream>

/**
* @file host_bulk_from_indices_example.cu
* @brief Demonstrates building a roaring_bitmap from unordered indices.
*/
int main()
{
using index_type = cuda::std::uint32_t;

thrust::device_vector<index_type> indices{0x00010002, 7, 1, 0x00010000, 7, 3, 0x00010002};

auto bitmap =
cuco::experimental::roaring_bitmap<index_type>::from_indices(indices.begin(), indices.end());

thrust::device_vector<index_type> queries{1, 2, 3, 7, 0x00010000, 0x00010001, 0x00010002};
thrust::device_vector<bool> results(queries.size());
bitmap.contains(queries.begin(), queries.end(), results.begin());

thrust::host_vector<bool> expected{true, false, true, true, true, false, true};
thrust::host_vector<bool> actual = results;
bool const success = actual == expected;

std::cout << "unique indices: " << bitmap.size() << '\n';
std::cout << "serialized bytes: " << bitmap.size_bytes() << '\n';
std::cout << "success: " << std::boolalpha << success << '\n';

return success ? 0 : 1;
}
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@
#include <vector>

/**
* @file host_bulk_example.cu
* @file host_bulk_from_serialized_example.cu
* @brief Demonstrates usage of the roaring_bitmap "bulk" lookup host APIs.
*
* In this example we load two 32-bit bitmaps and one 64-bit bitmap (portable format) from the
Expand Down Expand Up @@ -94,8 +94,14 @@ bool check(std::string const& bitmap_file_path)
file.close();

// Create roaring bitmap from the file
cuco::experimental::roaring_bitmap<KeyType> roaring_bitmap(
thrust::raw_pointer_cast(buffer.data()));
auto roaring_bitmap = [&] {
auto const* data = thrust::raw_pointer_cast(buffer.data());
if constexpr (cuda::std::is_same_v<KeyType, cuda::std::uint32_t>) {
return cuco::experimental::roaring_bitmap<KeyType>::from_serialized(data);
} else {
return cuco::experimental::roaring_bitmap<KeyType>{data};
}
}();

// Generate query keys (all should be contained in the bitmap)
auto keys = generate_keys();
Expand Down
63 changes: 62 additions & 1 deletion include/cuco/detail/roaring_bitmap/roaring_bitmap.inl
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,12 @@

#pragma once

#include <cuco/detail/roaring_bitmap/roaring_bitmap_builder.cuh>

#include <cuda/std/cstddef>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include <cuda/std/utility>
#include <cuda/stream_ref>

namespace cuco::experimental {
Expand All @@ -18,6 +23,62 @@ roaring_bitmap<T, Allocator>::roaring_bitmap(cuda::std::byte const* bitmap,
{
}

template <class T, class Allocator>
roaring_bitmap<T, Allocator> roaring_bitmap<T, Allocator>::from_serialized(
cuda::std::byte const* bitmap, Allocator const& alloc, cuda::stream_ref stream)
{
static_assert(cuda::std::is_same_v<T, cuda::std::uint32_t>,
"roaring_bitmap::from_serialized currently supports only uint32_t");
return roaring_bitmap{bitmap, alloc, stream};
}

template <class T, class Allocator>
roaring_bitmap<T, Allocator>::roaring_bitmap(storage_type&& storage)
: storage_{cuda::std::move(storage)}
{
}

template <class T, class Allocator>

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The three factory bodies are identical apart from the enumerator; a detail::build_roaring_bitmap_storage<T>(first, last, order, alloc, stream) helper reduces each to one line. Related: why three public factories rather than one with a tag or enum, given the builder is already keyed on roaring_bitmap_builder_input_order?

@sleeepyjack sleeepyjack Sep 5, 2026

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I would keep the three named factories. The input ordering guarantee is part of the API, and a public enum or tag makes that precondition easier to misuse. The small wrappers are intentional; the detail builder already centralizes the actual implementation, so I do not think another free helper buys enough to justify another layer. Also, adding another specialized enum to the public API feels unnecessary.

template <class InputIt>
roaring_bitmap<T, Allocator> roaring_bitmap<T, Allocator>::from_indices(InputIt first,
InputIt last,
Allocator const& alloc,
cuda::stream_ref stream)
{
static_assert(cuda::std::is_same_v<T, cuda::std::uint32_t>,
"Building a roaring_bitmap from indices currently supports only uint32_t");
detail::roaring_bitmap_builder<InputIt, Allocator> builder{
first, last, detail::roaring_bitmap_builder_input_order::unsorted, alloc, stream};
auto storage = cuda::std::move(builder).build();
return roaring_bitmap{cuda::std::move(storage)};
}

template <class T, class Allocator>
template <class InputIt>
roaring_bitmap<T, Allocator> roaring_bitmap<T, Allocator>::from_sorted_indices(
InputIt first, InputIt last, Allocator const& alloc, cuda::stream_ref stream)
{
static_assert(cuda::std::is_same_v<T, cuda::std::uint32_t>,
"Building a roaring_bitmap from indices currently supports only uint32_t");
detail::roaring_bitmap_builder<InputIt, Allocator> builder{
first, last, detail::roaring_bitmap_builder_input_order::sorted, alloc, stream};
auto storage = cuda::std::move(builder).build();
return roaring_bitmap{cuda::std::move(storage)};
}

template <class T, class Allocator>
template <class InputIt>
roaring_bitmap<T, Allocator> roaring_bitmap<T, Allocator>::from_sorted_unique_indices(
InputIt first, InputIt last, Allocator const& alloc, cuda::stream_ref stream)
{
static_assert(cuda::std::is_same_v<T, cuda::std::uint32_t>,
"Building a roaring_bitmap from indices currently supports only uint32_t");
detail::roaring_bitmap_builder<InputIt, Allocator> builder{
first, last, detail::roaring_bitmap_builder_input_order::sorted_unique, alloc, stream};
auto storage = cuda::std::move(builder).build();
return roaring_bitmap{cuda::std::move(storage)};
}

template <class T, class Allocator>
template <class InputIt, class OutputIt>
void roaring_bitmap<T, Allocator>::contains(InputIt first,
Expand Down Expand Up @@ -74,4 +135,4 @@ typename roaring_bitmap<T, Allocator>::ref_type roaring_bitmap<T, Allocator>::re
{
return ref_type{storage_.ref()};
}
} // namespace cuco::experimental
} // namespace cuco::experimental
Loading
Loading