Make the remaining shared node lookups faster

Encode the shared nodes as quadkeys, as the comment on struct node
already said they were, with a branch-free bit interleave, so that
vertices that are near each other are near each other in the sorted list
too. Search the list with an inlined std::lower_bound instead of bsearch.

Size the Bloom filter by the number of nodes, at about 16 bits each and
at most 32MB, instead of always using 34MB, so that it can usually stay
in the cache, and set three bits for each node, chosen by a mixing hash,
within a single 64-bit word, so that each check still touches only one
cache line but has far fewer false positives than a single bit.

In the pass that marks the vertices with whether they are shared nodes,
this is about 1.7x faster with 70 thousand nodes and 1.9x faster with
3 million.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01P2nBqZisxNQfmEmon3vE9v
This commit is contained in:
Claude
2026-09-23 19:46:09 +00:00
parent 175036930d
commit e1795ebe32
6 changed files with 88 additions and 12 deletions
+5
View File
@@ -14,6 +14,11 @@
exactly on one of these new points, and means that only
vertices that come back from a prefilter, or that are created by polygon
cleaning, still need to be looked up in the list of shared nodes during tiling.
* Make the remaining lookups of shared nodes faster: sort the list of nodes
by quadkey, as its comment always said, so that nearby vertices are near
each other in the list; search it with `std::lower_bound` instead of `bsearch`;
and size the Bloom filter in front of it by the number of nodes, with three
bits for each node within one 64-bit word, so that it usually fits in the cache.
# 2.82.0
+38 -5
View File
@@ -7,6 +7,7 @@
#include <cstdio>
#include <unistd.h>
#include <cmath>
#include <cstring>
#include <limits.h>
#include <sqlite3.h>
#include <mapbox/geometry/point.hpp>
@@ -219,16 +220,48 @@ drawvec impose_tile_boundaries(const drawvec &geom, long long extent) {
return out;
}
// Where a shared node goes in the Bloom filter: three bits, all within the same
// 64-bit word, so that checking for a node only touches one cache line.
static void shared_node_bloom_bits(unsigned long long index, size_t bloom_size, size_t *word, unsigned long long *mask) {
// splitmix64 finalizer, to spread out the index, whose low bits are
// mostly zero because of the geometry scale
unsigned long long h = index;
h ^= h >> 30;
h *= 0xBF58476D1CE4E5B9ULL;
h ^= h >> 27;
h *= 0x94D049BB133111EBULL;
h ^= h >> 31;
*word = ((h >> 32) * (unsigned long long) (bloom_size / sizeof(unsigned long long))) >> 32;
*mask = (1ULL << (h & 63)) | (1ULL << ((h >> 6) & 63)) | (1ULL << ((h >> 12) & 63));
}
void add_shared_node_to_bloom(std::string &shared_nodes_bloom, unsigned long long index) {
size_t word;
unsigned long long mask, bits;
shared_node_bloom_bits(index, shared_nodes_bloom.size(), &word, &mask);
memcpy(&bits, shared_nodes_bloom.data() + word * sizeof(bits), sizeof(bits));
bits |= mask;
memcpy(&shared_nodes_bloom[0] + word * sizeof(bits), &bits, sizeof(bits));
}
// Is the vertex at world coordinates wx, wy one of the nodes in the global list of shared nodes?
bool is_shared_node(long long wx, long long wy, struct node const *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom) {
struct node n;
n.index = encode_vertex((unsigned) wx, (unsigned) wy);
size_t bloom_ix = n.index % (shared_nodes_bloom.size() * 8);
unsigned char bloom_mask = 1 << (bloom_ix & 7);
bloom_ix >>= 3;
if (shared_nodes_bloom[bloom_ix] & bloom_mask) {
if (bsearch(&n, shared_nodes_map, nodepos / sizeof(node), sizeof(node), nodecmp) != NULL) {
size_t word;
unsigned long long mask, bits;
shared_node_bloom_bits(n.index, shared_nodes_bloom.size(), &word, &mask);
memcpy(&bits, shared_nodes_bloom.data() + word * sizeof(bits), sizeof(bits));
if ((bits & mask) == mask) {
struct node const *end = shared_nodes_map + nodepos / sizeof(node);
struct node const *found = std::lower_bound(shared_nodes_map, end, n.index, [](struct node const &a, unsigned long long b) {
return a.index < b;
});
if (found != end && found->index == n.index) {
return true;
}
}
+1
View File
@@ -110,6 +110,7 @@ drawvec stairstep(drawvec &geom, int z, int detail);
bool point_within_tile(long long x, long long y, int z);
int quick_check(const long long *bbox, int z, long long buffer);
void douglas_peucker(drawvec &geom, int start, int n, double e, size_t kept, size_t retain, bool prevent_simplify_shared_nodes);
void add_shared_node_to_bloom(std::string &shared_nodes_bloom, unsigned long long index);
bool is_shared_node(long long wx, long long wy, struct node const *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom);
void restore_node_states(drawvec const &from, drawvec &to);
drawvec simplify_lines(drawvec &geom, int z, int tx, int ty, int detail, bool mark_tile_bounds, double simplification, size_t retain, drawvec const &shared_nodes, struct node *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom);
+13 -6
View File
@@ -2136,8 +2136,8 @@ std::pair<int, metadata> read_input(std::vector<source> &sources, char *fname, i
fprintf(stderr, "Merging nodes \r");
}
// This is sized once the number of nodes is known, below
std::string shared_nodes_bloom;
shared_nodes_bloom.resize(34567891); // circa 34MB, size nowhere near a power of 2
// Sort nodes that can't be simplified away; scan the list to remove duplicates
@@ -2204,11 +2204,6 @@ std::pair<int, metadata> read_input(std::vector<source> &sources, char *fname, i
fwrite_check((void *) &here, sizeof(here), 1, shared_nodes, &nodepos, "shared nodes");
written = here;
size_t bloom_ix = here.index % (shared_nodes_bloom.size() * 8);
unsigned char bloom_mask = 1 << (bloom_ix & 7);
bloom_ix >>= 3;
shared_nodes_bloom[bloom_ix] |= bloom_mask;
#if 0
unsigned wx, wy;
decode_quadkey(here.index, &wx, &wy);
@@ -2227,6 +2222,18 @@ std::pair<int, metadata> read_input(std::vector<source> &sources, char *fname, i
perror("mmap nodes");
exit(EXIT_MEMORY);
}
// Size the Bloom filter at about 16 bits per node, so that for a moderate
// number of nodes it can stay in the cache while it is being checked,
// but no more than 32MB.
size_t nnodes = nodepos / sizeof(node);
size_t bloom_size = std::min(nnodes * 2, (size_t) 32 * 1024 * 1024);
bloom_size = std::max((bloom_size + 7) / 8 * 8, (size_t) 64);
shared_nodes_bloom.resize(bloom_size);
for (size_t i = 0; i < nnodes; i++) {
add_shared_node_to_bloom(shared_nodes_bloom, shared_nodes_map[i].index);
}
}
fclose(node_out);
+14 -1
View File
@@ -218,6 +218,19 @@ void set_projection_or_exit(const char *optarg) {
}
}
// Spread the 32 bits of v out into the even bits of a 64-bit value
static inline unsigned long long spread_bits(unsigned int v) {
unsigned long long x = v;
x = (x | (x << 16)) & 0x0000FFFF0000FFFFULL;
x = (x | (x << 8)) & 0x00FF00FF00FF00FFULL;
x = (x | (x << 4)) & 0x0F0F0F0F0F0F0F0FULL;
x = (x | (x << 2)) & 0x3333333333333333ULL;
x = (x | (x << 1)) & 0x5555555555555555ULL;
return x;
}
// The same as encode_quadkey(), but faster, for the vertices of the list of shared nodes,
// so that nodes that are near each other are also near each other in the sorted list.
unsigned long long encode_vertex(unsigned int wx, unsigned int wy) {
return (((unsigned long long) wx) << 32) | wy;
return (spread_bits(wx) << 1) | spread_bits(wy);
}
+17
View File
@@ -129,6 +129,23 @@ TEST_CASE("Bit reversal", "bit reversal") {
REQUIRE(bit_reverse(0xF3D912481E6A2C48) == 0x1234567812489BCF);
}
TEST_CASE("Vertex encoding matches quadkey encoding", "[projection]") {
unsigned int values[] = {0, 1, 2, 0x7FFFFFFF, 0x80000000, 0xFFFFFFFF, 0x12345678, 0xDEADBEEF};
for (unsigned int x : values) {
for (unsigned int y : values) {
REQUIRE(encode_vertex(x, y) == encode_quadkey(x, y));
}
}
unsigned long long seed = 1;
for (size_t i = 0; i < 10000; i++) {
seed = seed * 6364136223846793005ULL + 1442695040888963407ULL;
unsigned int x = seed >> 32;
unsigned int y = seed;
REQUIRE(encode_vertex(x, y) == encode_quadkey(x, y));
}
}
TEST_CASE("line_is_too_small") {
drawvec dv;
dv.emplace_back(VT_MOVETO, 4243099709, 2683872952);