diff --git a/CHANGELOG.md b/CHANGELOG.md index 02df70e2..2889811e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,11 @@ exactly on one of these new points, and means that only vertices that come back from a prefilter, or that are created by polygon cleaning, still need to be looked up in the list of shared nodes during tiling. +* Make the remaining lookups of shared nodes faster: sort the list of nodes + by quadkey, as its comment always said, so that nearby vertices are near + each other in the list; search it with `std::lower_bound` instead of `bsearch`; + and size the Bloom filter in front of it by the number of nodes, with three + bits for each node within one 64-bit word, so that it usually fits in the cache. # 2.82.0 diff --git a/geometry.cpp b/geometry.cpp index 21eb7ba3..9ece03ed 100644 --- a/geometry.cpp +++ b/geometry.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -219,16 +220,48 @@ drawvec impose_tile_boundaries(const drawvec &geom, long long extent) { return out; } +// Where a shared node goes in the Bloom filter: three bits, all within the same +// 64-bit word, so that checking for a node only touches one cache line. +static void shared_node_bloom_bits(unsigned long long index, size_t bloom_size, size_t *word, unsigned long long *mask) { + // splitmix64 finalizer, to spread out the index, whose low bits are + // mostly zero because of the geometry scale + unsigned long long h = index; + h ^= h >> 30; + h *= 0xBF58476D1CE4E5B9ULL; + h ^= h >> 27; + h *= 0x94D049BB133111EBULL; + h ^= h >> 31; + + *word = ((h >> 32) * (unsigned long long) (bloom_size / sizeof(unsigned long long))) >> 32; + *mask = (1ULL << (h & 63)) | (1ULL << ((h >> 6) & 63)) | (1ULL << ((h >> 12) & 63)); +} + +void add_shared_node_to_bloom(std::string &shared_nodes_bloom, unsigned long long index) { + size_t word; + unsigned long long mask, bits; + shared_node_bloom_bits(index, shared_nodes_bloom.size(), &word, &mask); + + memcpy(&bits, shared_nodes_bloom.data() + word * sizeof(bits), sizeof(bits)); + bits |= mask; + memcpy(&shared_nodes_bloom[0] + word * sizeof(bits), &bits, sizeof(bits)); +} + // Is the vertex at world coordinates wx, wy one of the nodes in the global list of shared nodes? bool is_shared_node(long long wx, long long wy, struct node const *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom) { struct node n; n.index = encode_vertex((unsigned) wx, (unsigned) wy); - size_t bloom_ix = n.index % (shared_nodes_bloom.size() * 8); - unsigned char bloom_mask = 1 << (bloom_ix & 7); - bloom_ix >>= 3; - if (shared_nodes_bloom[bloom_ix] & bloom_mask) { - if (bsearch(&n, shared_nodes_map, nodepos / sizeof(node), sizeof(node), nodecmp) != NULL) { + size_t word; + unsigned long long mask, bits; + shared_node_bloom_bits(n.index, shared_nodes_bloom.size(), &word, &mask); + memcpy(&bits, shared_nodes_bloom.data() + word * sizeof(bits), sizeof(bits)); + + if ((bits & mask) == mask) { + struct node const *end = shared_nodes_map + nodepos / sizeof(node); + struct node const *found = std::lower_bound(shared_nodes_map, end, n.index, [](struct node const &a, unsigned long long b) { + return a.index < b; + }); + if (found != end && found->index == n.index) { return true; } } diff --git a/geometry.hpp b/geometry.hpp index 8efb5d5d..90bac3e8 100644 --- a/geometry.hpp +++ b/geometry.hpp @@ -110,6 +110,7 @@ drawvec stairstep(drawvec &geom, int z, int detail); bool point_within_tile(long long x, long long y, int z); int quick_check(const long long *bbox, int z, long long buffer); void douglas_peucker(drawvec &geom, int start, int n, double e, size_t kept, size_t retain, bool prevent_simplify_shared_nodes); +void add_shared_node_to_bloom(std::string &shared_nodes_bloom, unsigned long long index); bool is_shared_node(long long wx, long long wy, struct node const *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom); void restore_node_states(drawvec const &from, drawvec &to); drawvec simplify_lines(drawvec &geom, int z, int tx, int ty, int detail, bool mark_tile_bounds, double simplification, size_t retain, drawvec const &shared_nodes, struct node *shared_nodes_map, size_t nodepos, std::string const &shared_nodes_bloom); diff --git a/main.cpp b/main.cpp index 2ad9727a..e72170f0 100644 --- a/main.cpp +++ b/main.cpp @@ -2136,8 +2136,8 @@ std::pair read_input(std::vector &sources, char *fname, i fprintf(stderr, "Merging nodes \r"); } + // This is sized once the number of nodes is known, below std::string shared_nodes_bloom; - shared_nodes_bloom.resize(34567891); // circa 34MB, size nowhere near a power of 2 // Sort nodes that can't be simplified away; scan the list to remove duplicates @@ -2204,11 +2204,6 @@ std::pair read_input(std::vector &sources, char *fname, i fwrite_check((void *) &here, sizeof(here), 1, shared_nodes, &nodepos, "shared nodes"); written = here; - size_t bloom_ix = here.index % (shared_nodes_bloom.size() * 8); - unsigned char bloom_mask = 1 << (bloom_ix & 7); - bloom_ix >>= 3; - shared_nodes_bloom[bloom_ix] |= bloom_mask; - #if 0 unsigned wx, wy; decode_quadkey(here.index, &wx, &wy); @@ -2227,6 +2222,18 @@ std::pair read_input(std::vector &sources, char *fname, i perror("mmap nodes"); exit(EXIT_MEMORY); } + + // Size the Bloom filter at about 16 bits per node, so that for a moderate + // number of nodes it can stay in the cache while it is being checked, + // but no more than 32MB. + size_t nnodes = nodepos / sizeof(node); + size_t bloom_size = std::min(nnodes * 2, (size_t) 32 * 1024 * 1024); + bloom_size = std::max((bloom_size + 7) / 8 * 8, (size_t) 64); + shared_nodes_bloom.resize(bloom_size); + + for (size_t i = 0; i < nnodes; i++) { + add_shared_node_to_bloom(shared_nodes_bloom, shared_nodes_map[i].index); + } } fclose(node_out); diff --git a/projection.cpp b/projection.cpp index 9f242d5a..7a4d645b 100644 --- a/projection.cpp +++ b/projection.cpp @@ -218,6 +218,19 @@ void set_projection_or_exit(const char *optarg) { } } +// Spread the 32 bits of v out into the even bits of a 64-bit value +static inline unsigned long long spread_bits(unsigned int v) { + unsigned long long x = v; + x = (x | (x << 16)) & 0x0000FFFF0000FFFFULL; + x = (x | (x << 8)) & 0x00FF00FF00FF00FFULL; + x = (x | (x << 4)) & 0x0F0F0F0F0F0F0F0FULL; + x = (x | (x << 2)) & 0x3333333333333333ULL; + x = (x | (x << 1)) & 0x5555555555555555ULL; + return x; +} + +// The same as encode_quadkey(), but faster, for the vertices of the list of shared nodes, +// so that nodes that are near each other are also near each other in the sorted list. unsigned long long encode_vertex(unsigned int wx, unsigned int wy) { - return (((unsigned long long) wx) << 32) | wy; + return (spread_bits(wx) << 1) | spread_bits(wy); } diff --git a/unit.cpp b/unit.cpp index 7a04e20b..c10eddb9 100644 --- a/unit.cpp +++ b/unit.cpp @@ -129,6 +129,23 @@ TEST_CASE("Bit reversal", "bit reversal") { REQUIRE(bit_reverse(0xF3D912481E6A2C48) == 0x1234567812489BCF); } +TEST_CASE("Vertex encoding matches quadkey encoding", "[projection]") { + unsigned int values[] = {0, 1, 2, 0x7FFFFFFF, 0x80000000, 0xFFFFFFFF, 0x12345678, 0xDEADBEEF}; + for (unsigned int x : values) { + for (unsigned int y : values) { + REQUIRE(encode_vertex(x, y) == encode_quadkey(x, y)); + } + } + + unsigned long long seed = 1; + for (size_t i = 0; i < 10000; i++) { + seed = seed * 6364136223846793005ULL + 1442695040888963407ULL; + unsigned int x = seed >> 32; + unsigned int y = seed; + REQUIRE(encode_vertex(x, y) == encode_quadkey(x, y)); + } +} + TEST_CASE("line_is_too_small") { drawvec dv; dv.emplace_back(VT_MOVETO, 4243099709, 2683872952);