diff --git a/benchmarks/CMakeLists.txt b/benchmarks/CMakeLists.txt index 0e2f0c8..fd93b7b 100644 --- a/benchmarks/CMakeLists.txt +++ b/benchmarks/CMakeLists.txt @@ -82,6 +82,18 @@ target_link_libraries(nsparse_build_benchmark PRIVATE absl::flat_hash_map ) +# Read-only size accounting for a serialized disk_seismic(_sq) index: where the +# bytes are and what each candidate encoding would save. Not a google-benchmark +# target; it measures a file, not a run. +add_executable(index_size_stats index_size_stats.cpp) +target_include_directories(index_size_stats PRIVATE ${PROJECT_SOURCE_DIR}) +target_link_libraries(index_size_stats PRIVATE + ${NSPARSE_BENCH_LIB} + OpenMP::OpenMP_CXX + absl::flat_hash_set + absl::flat_hash_map +) + # Peak-RSS build-memory driver for the term-batched build. Not a # google-benchmark target: that harness reports throughput, and what matters here # is the high-water mark of a single build, one configuration per process. diff --git a/benchmarks/DISK_SEISMIC_BENCH.md b/benchmarks/DISK_SEISMIC_BENCH.md index faaef79..20cd089 100644 --- a/benchmarks/DISK_SEISMIC_BENCH.md +++ b/benchmarks/DISK_SEISMIC_BENCH.md @@ -66,6 +66,47 @@ larger on disk but only its summaries stay resident) to expose the disk benefit. - Memory split: V1 is almost all `RssAnon` (heap); V2/V3 are almost all `RssFile` (mapped, reclaimable), V3 with a small `RssAnon` for summaries. +## Where the on-disk bytes go — `index_size_stats` + +`index_size_stats ` re-parses a serialized `disk_seismic`/`disk_seismic_sq` file +and prints the section split, the inline forward index's per-array split, and what a narrower +encoding of each array would save. It only reads, so sizing a format change costs one pass +over the file instead of a rebuild. + +That is how the block padding was found. A block holds about ten documents — a ~4.5 KB +payload — and `InlineForwardIndex` was aligning each to 4096, so **a third of the file was +padding nothing ever read**: 24.3 GiB of 74.2 on `base_full` at `lambda=6000 beta=400 +alpha=0.4`, 8-bit `disk_seismic_sq`. Packed placement (`InlineLayout::kPacked`, now the +default) removes it. Same format either way — the header records the alignment and a reader +honours whatever the file declares — so a file written before the default changed still +loads, which is what let this A/B run one binary over two files: + +| | page-aligned | packed | | +|---|---|---|---| +| index file | 74.205 GiB | **49.967 GiB** | −32.7% | +| inter-block padding | 24.273 GiB | 0.035 GiB | | +| bytes per non-zero | 4.73 | 3.09 | | +| first (cold) query pass | 982,994 ms | 888,829 ms | −9.6% | +| warm batch, 6,980 queries | 1492.6 ms | 1464.3 ms | −1.9% | +| warm p50 / p90 / p99 | 0.221 / 0.268 / 0.312 ms | 0.217 / 0.264 / 0.308 ms | −1.6% | +| `VmHWM` / `RssFile` | 6.567 / 6.236 GiB | 6.042 / 5.712 GiB | −8.0% | +| recall@10 | 0.9252 | 0.9261 | k-means seed noise | + +Padding cost storage and page cache, not page faults — a page-aligned block already spanned +two pages and the padding sat in the tail of the second one. So removing it buys file size +and a smaller resident set, and touches query CPU only to the extent that blocks now share +pages. Warm latency came out marginally *better*, not worse. + +The recall difference is seed noise: the two indexes were built at different times without a +fixed `seed=`. The layout change itself is exact, which `sq_residency_bench`'s trailing +`labels_out.txt` argument is there to show — dump one build's labels, score the other against +them, expect 1.0. With a fixed seed the two layouts' label files are byte-identical. + +What is left, per the same tool, if the file needs to shrink further: component ids are 64.8% +of the inline forward index and delta coding would take 29.5 → 19.4 GiB, but decoding sits on +the critical path of scoring a block and measured a 1.5× warm-latency cost, so it was not +taken. 4-bit values would save 7.4 GiB and is lossy. `doc_id[]`/`off[]` are ~1% together. + ## Caveats - **`base_small` is a smoke test, not a benchmark** — it fits in cache, so the diff --git a/benchmarks/index_size_stats.cpp b/benchmarks/index_size_stats.cpp new file mode 100644 index 0000000..6049868 --- /dev/null +++ b/benchmarks/index_size_stats.cpp @@ -0,0 +1,344 @@ +/** + * Copyright OpenSearch Contributors + * SPDX-License-Identifier: Apache-2.0 + * + * The OpenSearch Contributors require contributions made to + * this file be licensed under the Apache-2.0 license or a + * compatible open source license. + */ + +// Where the bytes of a serialized disk_seismic / disk_seismic_sq index go, and +// what a narrower encoding of each array would save. Read-only: it re-parses an +// existing .dat with the same components the index loads, so sizing a format +// change costs one pass over the file rather than a rebuild. +// +// index_size_stats +// +// Prints the section split (summaries / inline forward index / doc directory), +// the inline forward index's per-array split, and simulated sizes for: packing +// the blocks (what a file written before that became the default still stands to +// save), delta-coded component ids, narrower doc-id and offset tables, and 4-bit +// values. + +#include +#include +#include +#include +#include +#include +#include + +#include "nsparse/inline_forward_index.h" +#include "nsparse/io/inline_forward_index_io.h" +#include "nsparse/io/seismic_invlists_writer.h" +#include "nsparse/types.h" +#include "nsparse/utils/mmap_cursor.h" +#include "nsparse/utils/mmap_file.h" +#include "nsparse/utils/scalar_quantizer.h" + +using nsparse::MmapCursor; +using nsparse::MmapFile; +using nsparse::QuantizerType; +using nsparse::SeismicInvertedListsWriter; +using nsparse::detail::BlockView; +using nsparse::detail::InlineForwardIndex; + +namespace { + +constexpr double kGiB = 1024.0 * 1024.0 * 1024.0; + +double gib(uint64_t bytes) { return static_cast(bytes) / kGiB; } + +void line(const char* label, uint64_t bytes, uint64_t total) { + std::printf(" %-34s %16llu %8.3f GiB %6.2f%%\n", label, + static_cast(bytes), gib(bytes), + total == 0 ? 0.0 + : 100.0 * static_cast(bytes) / + static_cast(total)); +} + +// Bytes a Lucene-style VByte (seven payload bits per byte, high bit continuing) +// takes. A sizing model only: nothing in the index writes this encoding. +uint64_t vbyte_len(uint64_t value) { + uint64_t bytes = 1; + while (value >= 0x80) { + value >>= 7; + ++bytes; + } + return bytes; +} + +struct Stats { + uint64_t n_blocks = 0; + // Doc slots, i.e. postings: a doc counts once per block that holds it. + uint64_t n_docs = 0; + uint64_t nnz = 0; + + // The layout as stored, by array. + uint64_t b_prefix = 0; + uint64_t b_doc_ids = 0; + uint64_t b_off = 0; + uint64_t b_comps = 0; + uint64_t b_vals = 0; + uint64_t b_intra_pad = 0; // comps -> vals alignment + uint64_t b_inter_pad = 0; // block -> block alignment + uint64_t b_dir = 0; + uint64_t b_payload = 0; // sum of the directory entries' len + + // Encodings not taken, sized against what is stored. + uint64_t c_comps_vbyte = 0; // per-doc delta + VByte + uint64_t c_comps_flagged = 0; // per-doc delta, one width bit per value + uint64_t c_docids_vbyte = 0; // delta + VByte over sorted doc ids + uint64_t c_off_vbyte = 0; // per-doc nnz as VByte + uint64_t c_off_u16 = 0; // u16 offsets where a block permits + uint64_t blocks_off_fits_u16 = 0; + + // Diagnostics. + uint64_t docs_comps_unsorted = 0; + uint64_t blocks_docids_unsorted = 0; + uint64_t vals_nonzero = 0; + uint64_t vals_le15 = 0; // would fit 4 bits as stored +}; + +// Component ids of one doc: how much delta coding would buy, two ways, and +// whether they ascend (which is what makes the deltas small). +void size_doc_comps(const nsparse::term_t* comps, uint32_t len, Stats* stats) { + if (len == 0) { + return; + } + constexpr uint32_t kNarrowMax = 0xFF; + constexpr uint64_t kFlagBits = 8; + bool ascending = true; + uint64_t vbytes = vbyte_len(comps[0]); + // One flag bit per value, then one payload byte, or two where the difference + // needs them. + uint64_t flagged = ((len + kFlagBits - 1) / kFlagBits) + 1 + + (comps[0] > kNarrowMax ? 1 : 0); + uint32_t prev = comps[0]; + for (uint32_t j = 1; j < len; ++j) { + const uint32_t cur = comps[j]; + if (cur <= prev) { + ascending = false; + } + const uint32_t gap = cur > prev ? cur - prev : 0; + vbytes += vbyte_len(gap); + flagged += gap > kNarrowMax ? 2 : 1; + prev = cur; + } + if (!ascending) { + stats->docs_comps_unsorted += 1; + } + stats->c_comps_vbyte += vbytes; + stats->c_comps_flagged += flagged; +} + +void accumulate(const BlockView& view, size_t element_size, uint64_t align, + Stats* stats) { + const uint64_t n_docs = view.n_docs; + const uint64_t total_nnz = view.offsets[n_docs]; + stats->n_blocks += 1; + stats->n_docs += n_docs; + stats->nnz += total_nnz; + + const nsparse::detail::InlineBlockOffsets layout = + nsparse::detail::inline_block_offsets(n_docs, total_nnz, element_size); + stats->b_prefix += nsparse::detail::kInlineBlockPrefixSize; + stats->b_doc_ids += n_docs * sizeof(uint32_t); + stats->b_off += (n_docs + 1) * sizeof(uint32_t); + stats->b_comps += total_nnz * sizeof(nsparse::term_t); + stats->b_vals += total_nnz * element_size; + stats->b_intra_pad += + layout.vals - (layout.comps + total_nnz * sizeof(nsparse::term_t)); + stats->b_payload += layout.end; + stats->b_inter_pad += + nsparse::detail::inline_align_up(layout.end, align) - layout.end; + + for (uint32_t i = 0; i < view.n_docs; ++i) { + stats->c_off_vbyte += vbyte_len(view.nnz(i)); + size_doc_comps(view.doc_comps(i), view.nnz(i), stats); + } + + // Doc ids: do they ascend, and what would delta + VByte cost? The sizing + // assumes a writer that sorts them within a block, which today's does not. + bool ids_sorted = true; + for (uint64_t i = 1; i < n_docs; ++i) { + if (view.doc_ids[i] <= view.doc_ids[i - 1]) { + ids_sorted = false; + break; + } + } + if (!ids_sorted) { + stats->blocks_docids_unsorted += 1; + } + std::vector ids(view.doc_ids, view.doc_ids + n_docs); + std::sort(ids.begin(), ids.end()); + uint64_t id_bytes = n_docs == 0 ? 0 : vbyte_len(ids[0]); + for (uint64_t i = 1; i < n_docs; ++i) { + id_bytes += vbyte_len(ids[i] - ids[i - 1]); + } + stats->c_docids_vbyte += id_bytes; + + // Offsets: could a u16 offset table address this block's payload? + constexpr uint64_t kU16Max = 0xFFFF; + if (layout.end <= kU16Max) { + stats->blocks_off_fits_u16 += 1; + stats->c_off_u16 += (n_docs + 1) * sizeof(uint16_t); + } else { + stats->c_off_u16 += (n_docs + 1) * sizeof(uint32_t); + } + + // Value distribution, 8-bit codes only. + if (element_size == 1) { + constexpr uint8_t kFourBitMax = 15; + for (uint64_t j = 0; j < total_nnz; ++j) { + const uint8_t value = view.vals[j]; + stats->vals_nonzero += value != 0 ? 1 : 0; + stats->vals_le15 += value <= kFourBitMax ? 1 : 0; + } + } +} + +} // namespace + +int main(int argc, char** argv) { + if (argc < 2) { + std::fprintf(stderr, "usage: %s \n", argv[0]); + return 1; + } + const std::string path = argv[1]; + + MmapFile file(path, MmapFile::AccessPattern::kScan); + MmapCursor cursor(file.data(), file.size()); + const uint64_t file_bytes = file.size(); + + const auto type_id = cursor.read_scalar(); + const auto version = cursor.read_scalar(); + const auto dimension = cursor.read_scalar(); + std::array fourcc{}; + std::memcpy(fourcc.data(), &type_id, 4); + std::printf("index: %s type=%s version=%u dim=%d %llu B (%.3f GiB)\n", + path.c_str(), fourcc.data(), version, dimension, + static_cast(file_bytes), gib(file_bytes)); + + size_t element_size = nsparse::U32; + if (std::strcmp(fourcc.data(), "DSSQ") == 0) { + const auto quantizer = cursor.read_scalar(); + const auto vmin = cursor.read_scalar(); + const auto vmax = cursor.read_scalar(); + element_size = quantizer == QuantizerType::QT_8bit ? 1 : 2; + std::printf("quantizer: %s vmin=%g vmax=%g -> element_size=%zu\n", + quantizer == QuantizerType::QT_8bit ? "8bit" : "16bit", + vmin, vmax, element_size); + } else if (std::strcmp(fourcc.data(), "DSEI") != 0) { + std::fprintf(stderr, "not a disk seismic index (got %s)\n", + fourcc.data()); + return 1; + } + + const uint64_t header_end = cursor.pos(); + const auto num_vectors = cursor.read_scalar(); + + SeismicInvertedListsWriter inv_lists; + inv_lists.mmap_deserialize(&cursor); + const uint64_t summaries_end = cursor.pos(); + + InlineForwardIndex fwd; + fwd.mmap_deserialize(&cursor); + const uint64_t fwd_end = cursor.pos(); + + std::printf("docs: %llu posting lists: %llu blocks: %llu\n", + static_cast(num_vectors), + static_cast(fwd.num_lists()), + static_cast(fwd.num_blocks())); + std::printf("\n== sections ==\n"); + line("index + quantizer header", header_end, file_bytes); + // The u64 doc count sits between that header and the summaries. + line("cluster summaries", summaries_end - header_end - sizeof(uint64_t), + file_bytes); + line("inline forward index", fwd_end - summaries_end, file_bytes); + line("doc locators + remainder", file_bytes - fwd_end, file_bytes); + + const uint64_t align = fwd.page_size(); + Stats stats; + for (uint32_t pl = 0; pl < fwd.num_lists(); ++pl) { + const uint64_t n_blocks = fwd.num_blocks_in_list(pl); + for (uint32_t block = 0; block < n_blocks; ++block) { + const BlockView view = fwd.block(pl, block); + if (view.absent()) { + continue; + } + accumulate(view, element_size, align, &stats); + } + } + stats.b_dir = stats.n_blocks * sizeof(nsparse::detail::InlineDirEntry); + + const uint64_t inline_total = + stats.b_payload + stats.b_inter_pad + stats.b_dir; + std::printf( + "\n== inline forward index: %llu blocks, %llu doc slots, %llu nnz " + "(block alignment %llu) ==\n", + static_cast(stats.n_blocks), + static_cast(stats.n_docs), + static_cast(stats.nnz), + static_cast(align)); + std::printf( + " docs/block %.1f nnz/doc %.1f copies per doc %.2f\n", + static_cast(stats.n_docs) / static_cast(stats.n_blocks), + static_cast(stats.nnz) / static_cast(stats.n_docs), + static_cast(stats.n_docs) / static_cast(num_vectors)); + line("block prefix", stats.b_prefix, inline_total); + line("doc_id[] (u32)", stats.b_doc_ids, inline_total); + line("off[] (u32)", stats.b_off, inline_total); + line("comps[] (u16)", stats.b_comps, inline_total); + line("vals[]", stats.b_vals, inline_total); + line("intra-block pad", stats.b_intra_pad, inline_total); + line("inter-block pad", stats.b_inter_pad, inline_total); + line("directory", stats.b_dir, inline_total); + line("TOTAL", inline_total, inline_total); + std::printf( + " bytes per nnz: %.3f\n", + static_cast(inline_total) / static_cast(stats.nnz)); + + std::printf("\n== encodings not taken ==\n"); + const auto saving = [&](const char* label, uint64_t now, uint64_t then) { + const int64_t delta = + static_cast(now) - static_cast(then); + std::printf( + " %-38s %8.3f -> %8.3f GiB save %8.3f GiB (%5.2f%% of file)\n", + label, gib(now), gib(then), gib(delta > 0 ? delta : 0), + 100.0 * static_cast(delta) / + static_cast(file_bytes)); + }; + // Packing leaves each block padded to kMinBlockAlign, so on average half of + // it -- which is what a page-aligned file would come down to. + saving("pack blocks (align 8, no page pad)", stats.b_inter_pad, + stats.n_blocks * (nsparse::detail::kMinBlockAlign / 2)); + saving("comps: delta + VByte", stats.b_comps, stats.c_comps_vbyte); + saving("comps: delta + width flags", stats.b_comps, stats.c_comps_flagged); + saving("doc_id: sorted delta + VByte", stats.b_doc_ids, + stats.c_docids_vbyte); + saving("off: per-doc nnz as VByte", stats.b_off, stats.c_off_vbyte); + saving("off: u16 where the block fits", stats.b_off, stats.c_off_u16); + if (element_size == 1) { + saving("vals: 4-bit codes", stats.b_vals, stats.nnz / 2); + } + + std::printf("\n== diagnostics ==\n"); + std::printf(" docs with non-ascending comps: %llu / %llu\n", + static_cast(stats.docs_comps_unsorted), + static_cast(stats.n_docs)); + std::printf(" blocks with non-ascending doc ids: %llu / %llu\n", + static_cast(stats.blocks_docids_unsorted), + static_cast(stats.n_blocks)); + std::printf(" blocks whose payload fits a u16 offset: %llu / %llu\n", + static_cast(stats.blocks_off_fits_u16), + static_cast(stats.n_blocks)); + if (element_size == 1) { + std::printf(" 8-bit codes: %.2f%% nonzero, %.2f%% <= 15\n", + 100.0 * static_cast(stats.vals_nonzero) / + static_cast(stats.nnz), + 100.0 * static_cast(stats.vals_le15) / + static_cast(stats.nnz)); + } + return 0; +} diff --git a/benchmarks/sq_residency_bench.cpp b/benchmarks/sq_residency_bench.cpp index 5fd36d5..4af41f8 100644 --- a/benchmarks/sq_residency_bench.cpp +++ b/benchmarks/sq_residency_bench.cpp @@ -18,7 +18,8 @@ // silently. // // build -// search [truth.txt] +// search [truth.txt|-] [cut] +// [k_prime] [heap_factor] [labels_out.txt] #include #include @@ -186,7 +187,8 @@ int do_build(int argc, char** argv) { int do_search(int argc, char** argv) { if (argc < 7) { std::cerr << "search " - "[truth.txt|-] [cut] [k_prime] [heap_factor]\n"; + "[truth.txt|-] [cut] [k_prime] [heap_factor] " + "[labels_out.txt]\n"; return 2; } std::string dat_path = argv[2]; @@ -204,6 +206,9 @@ int do_search(int argc, char** argv) { // cluster_score*heap_factor < heap.peek, so LARGER = prune fewer = higher // recall + slower. Sweep this to raise SEISMIC recall for iso-recall points. const float heap_factor = argc > 10 ? static_cast(std::atof(argv[10])) : 1.0F; + // Where to write the final labels, in the truth-file format, so another run + // can be scored against them. + const std::string labels_out_path = argc > 11 ? argv[11] : ""; const int io_flags = residency == "mmap" ? nsparse::IndexIoFlag::kUseMmap : 0; @@ -293,6 +298,24 @@ int do_search(int argc, char** argv) { std::cout << "recall@" << k << " " << recall_at_k(truth_path, labels, k, n_queries) << "\n"; } + // A format change that is meant to be lossless can be held to that: dump one + // build's labels in the truth-file format, then measure the other build's + // recall against them, where anything but 1.0 is a behaviour change. + if (!labels_out_path.empty()) { + std::ofstream out(labels_out_path); + if (!out.is_open()) { + throw std::runtime_error("Cannot write labels to " + + labels_out_path); + } + for (int qi = 0; qi < n_queries; ++qi) { + for (int ki = 0; ki < k; ++ki) { + out << (ki == 0 ? "" : ",") + << labels[static_cast(qi) * k + ki]; + } + out << "\n"; + } + std::cout << "labels_written " << labels_out_path << "\n"; + } print_memory("after_search"); return 0; } diff --git a/nsparse/inline_forward_index.h b/nsparse/inline_forward_index.h index afdd636..3f43b7b 100644 --- a/nsparse/inline_forward_index.h +++ b/nsparse/inline_forward_index.h @@ -39,10 +39,19 @@ namespace nsparse::detail { * Component: io/inline_forward_index_io.h. */ -// Block placement: page-aligned (padded to page_size, mmap-friendly; default) -// or packed (blocks back-to-back on the minimum kMinBlockAlign boundary that -// keeps the per-block arrays readable in place). The header page_size records -// the effective alignment. +// Block placement: page-aligned (padded to page_size) or packed (blocks +// back-to-back on the minimum kMinBlockAlign boundary that keeps the per-block +// arrays readable in place; the default). The header page_size records the +// effective alignment, and a reader honours whichever the file declares, so the +// two are the same format and either loads in a build that writes the other. +// +// Packed is the default because a block is far smaller than a page. At the +// standard SEISMIC parameters a block holds about ten documents -- a ~4.5 KB +// payload -- so page alignment spent about as many bytes on padding as on +// values: a third of a `disk_seismic_sq` file over MS MARCO, 24.3 GiB of 74.2. +// Those bytes were never read, only stored and cached, which is why packing them +// out costs no query CPU and shows up as a smaller file and a smaller resident +// set rather than as fewer page faults. enum class InlineLayout : uint8_t { kPageAligned, kPacked }; // Minimum block-start alignment. Blocks begin on this boundary even when diff --git a/nsparse/io/inline_forward_index_io.h b/nsparse/io/inline_forward_index_io.h index 0d955af..7cd56be 100644 --- a/nsparse/io/inline_forward_index_io.h +++ b/nsparse/io/inline_forward_index_io.h @@ -85,11 +85,11 @@ class InlineForwardIndex : public MmapSerializable { // Write mode. `lists`/`vectors` must outlive any serialize() call. // page_size: block alignment when page-aligned (power of two, >= header - // size); ignored when packed. + // size); ignored when packed, which is the default (see InlineLayout). InlineForwardIndex(const std::vector& lists, const SparseVectors& vectors, uint64_t page_size = kDefaultPageSize, - InlineLayout layout = InlineLayout::kPageAligned); + InlineLayout layout = InlineLayout::kPacked); // Explicit, not defaulted: Buf's move keeps the source's size()/data() // (only its owner moves), so the moved-from members are reset to stay diff --git a/tests/inline_forward_index_test.cpp b/tests/inline_forward_index_test.cpp index 4a6f918..b5f8526 100644 --- a/tests/inline_forward_index_test.cpp +++ b/tests/inline_forward_index_test.cpp @@ -271,15 +271,70 @@ TEST(InlineForwardIndex, SerializeRejectsOutOfRangeDocId) { EXPECT_THROW(writer.serialize(&buffer), std::out_of_range); } +// page_size only constrains the page-aligned layout, which has to be asked for: +// packed is the default, and there the alignment is kMinBlockAlign and page_size +// is ignored. TEST(InlineForwardIndex, RejectsInvalidPageSize) { auto vectors = sample_float_vectors(); - EXPECT_THROW(InlineForwardIndex(build_lists(kLayout, vectors), vectors, 0), + EXPECT_THROW(InlineForwardIndex(build_lists(kLayout, vectors), vectors, 0, + InlineLayout::kPageAligned), std::invalid_argument); - EXPECT_THROW( - InlineForwardIndex(build_lists(kLayout, vectors), vectors, 100), - std::invalid_argument); + EXPECT_THROW(InlineForwardIndex(build_lists(kLayout, vectors), vectors, 100, + InlineLayout::kPageAligned), + std::invalid_argument); + EXPECT_NO_THROW(InlineForwardIndex(build_lists(kLayout, vectors), vectors, + 64, InlineLayout::kPageAligned)); EXPECT_NO_THROW( - InlineForwardIndex(build_lists(kLayout, vectors), vectors, 64)); + InlineForwardIndex(build_lists(kLayout, vectors), vectors, 100)); +} + +// The two layouts are one format: the header records the alignment and a reader +// honours whatever the file declares, so a page-aligned file written before the +// default changed still loads, and loads to the same documents. +TEST(InlineForwardIndex, BothLayoutsReadBackTheSameBlocks) { + auto vectors = sample_float_vectors(); + auto lists = build_lists(kLayout, vectors); + const auto padded = + serialize_bytes(lists, vectors, InlineForwardIndex::kDefaultPageSize, + InlineLayout::kPageAligned); + const auto packed = + serialize_bytes(lists, vectors, InlineForwardIndex::kDefaultPageSize, + InlineLayout::kPacked); + auto padded_index = read_index(padded); + auto packed_index = read_index(packed); + + ASSERT_EQ(padded_index.num_blocks(), packed_index.num_blocks()); + ASSERT_EQ(padded_index.num_lists(), packed_index.num_lists()); + for (uint32_t pl = 0; pl < padded_index.num_lists(); ++pl) { + const uint64_t blocks = padded_index.num_blocks_in_list(pl); + ASSERT_EQ(blocks, packed_index.num_blocks_in_list(pl)); + for (uint32_t block = 0; block < blocks; ++block) { + verify_block(packed_index, packed_index.block(pl, block), + kLayout[pl][block], vectors); + verify_block(padded_index, padded_index.block(pl, block), + kLayout[pl][block], vectors); + } + } +} + +// What a disk index gets when it does not ask: blocks packed on kMinBlockAlign, +// not padded to a page. This is the whole of the padding change, so it is the one +// thing a future edit to the default must trip over. +TEST(InlineForwardIndex, DefaultsToThePackedLayout) { + auto vectors = sample_float_vectors(); + auto lists = build_lists(kLayout, vectors); + InlineForwardIndex writer(lists, vectors); + nsparse::BufferedIOWriter buffer; + writer.serialize(&buffer); + const auto bin = buffer.data(); + + EXPECT_EQ(parse_header(bin).page_size, kMinBlockAlign); + // And it is smaller than the same lists written page-aligned -- padding is + // all that differs. + EXPECT_LT(bin.size(), serialize_bytes(lists, vectors, + InlineForwardIndex::kDefaultPageSize, + InlineLayout::kPageAligned) + .size()); } TEST(InlineForwardIndex, SerializeOnReadModeIndexThrows) {