Files
tippecanoe/pool.cpp
Erica Fischer e2a7a409c7 Improve tiling speed (#195)
* Add a way to run tippecanoe single-threaded for profiling

* Do less work when the tilestats sample values list is already full

* Save a copy when retrieving the attribute key

* Fewer atomic operations

* Move string hashing from mbtiles to text

* Only do approximate attribute deduplication when writing tiles

* Feature dropping tests are sensitive to exact tile size

* All tile creators now create a string pool for the tile

* Features clipped away to nothing should not participate in that tile

* Revert "Only do approximate attribute deduplication when writing tiles"

This reverts commit c42b34b498.

* Also revert the related test changes

* Revert "Revert "Only do approximate attribute deduplication when writing tiles""

This reverts commit 18509876c3.

* Be more specific about the string hash function

* Use fnv1a instead of std::hash for everything

* Reduce the chance of hash collisions

* Stick a hash search on the front of the tree search in addpool

* Eliminate repeated hashing of the same string

* Switch instead of ifs in json parsing

* A few more cases to populate the hash in addpool

* Store the hash in the tree instead of recalculating

* Add explanatory comment for mysterious argument

* Fewer copies in attribute stringification

* Clean up ancient weirdness in JSON attribute stringification

* More serial_val cleanup

* Pass a serial_feature to rewrite instead of many broken-down arguments

* Get rid of the multiple geometries within `partial`

* Revert "Pass a serial_feature to rewrite instead of many broken-down arguments"

This reverts commit 6f4ab9b725.

* Goodbye, struct coalesce

* Revert "Features clipped away to nothing should not participate in that tile"

This reverts commit 124462fbdc.

* Migrating fields from partial to serial_feature

* Name reconciliation between serial_feature and partial

* Replace struct partial with an augmented serial_feature

* Fix some overzealous search-and-replace renaming

* Don't say struct so often

* Remove more of the former partial construction

* Commenting and cleaning up

* Trying again to avoid all these arguments to rewrite

* I swear I did this same thing before and it didn't work.

* More rewrite cleanup

* Exile --detect-shared-borders to its own file

* Add missing headers

* More commenting and cleanup

* More comments

* Sprinkle consts around

* Emplacing and std::moving

* More cleanup

* That shouldn't have worked after a std::move

* Don't need to allocate memory to compare keys

* Reduce use of the global string pool in tiling

* Another avoidable mvt_value construction

* Further reduction to explicit string pool passing

* These reverses are no longer optimizations

* These layernames can all be references

* Don't drag an unused layername string around with every feature

* Heed a compiler warning about potential buffer overflow

* Fix my confusion about which feature's string pool is relevant

* Avoid some unnecessary allocations in attribute accumulation

* Maybe faster serialization?

* Eliminate a comparison

* Do the same here

* Save a couple of allocations when parsing numbers in JSON

* Immediately assign features to layers instead of subdividing later

* Maintain tilestats for tippecanoe:retain_points_multiplier_sequence

* Crunch out more duplicate attribute values when writing out the tile

* Do tilestats for tippecanoe:retain_points_multiplier_first too

* Shell filters need to be real threads, even if nothing else does

* Simplify tippecanoe_minzoom/maxzoom representation

* Update version and changelog
2024-02-07 17:50:35 -08:00

168 lines
4.5 KiB
C++

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <limits.h>
#include <math.h>
#include "main.hpp"
#include "memfile.hpp"
#include "pool.hpp"
#include "errors.hpp"
#include "text.hpp"
inline int swizzlecmp(const char *a, int atype, unsigned long long ahash, const char *b, int btype, unsigned long long bhash) {
if (ahash == bhash) {
if (atype == btype) {
return strcmp(a, b);
} else {
return atype - btype;
}
} else {
return (int) ahash - (int) bhash;
}
}
long long addpool(struct memfile *poolfile, struct memfile *treefile, const char *s, char type, std::vector<ssize_t> &dedup) {
unsigned long long hash = fnv1a(s, type);
size_t hash_off = hash % dedup.size();
if (dedup[hash_off] >= 0 &&
poolfile->map[dedup[hash_off]] == type &&
strcmp(poolfile->map.c_str() + dedup[hash_off] + 1, s) == 0) {
// printf("hit for %s\n", s);
return dedup[hash_off];
} else {
// printf("miss for %s\n", s);
}
unsigned long *sp = &treefile->tree;
size_t depth = 0;
// In typical data, traversal depth generally stays under 2.5x
size_t max = 3 * log(treefile->off / sizeof(struct stringpool)) / log(2);
if (max < 30) {
max = 30;
}
while (*sp != 0) {
int cmp = swizzlecmp(s, type, hash,
poolfile->map.c_str() + ((struct stringpool *) (treefile->map.c_str() + *sp))->off + 1,
(poolfile->map.c_str() + ((struct stringpool *) (treefile->map.c_str() + *sp))->off)[0],
((struct stringpool *) (treefile->map.c_str() + *sp))->hash);
if (cmp < 0) {
sp = &(((struct stringpool *) (treefile->map.c_str() + *sp))->left);
} else if (cmp > 0) {
sp = &(((struct stringpool *) (treefile->map.c_str() + *sp))->right);
} else {
dedup[hash_off] = ((struct stringpool *) (treefile->map.c_str() + *sp))->off;
return ((struct stringpool *) (treefile->map.c_str() + *sp))->off;
}
depth++;
if (depth > max) {
// Search is very deep, so string is probably unique.
// Add it to the pool without adding it to the search tree.
// This might go either to memory or the file, depending on whether
// the pool is full yet.
long long off = poolfile->off;
bool in_memory = false;
if (memfile_write(poolfile, &type, 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (in_memory) {
dedup[hash_off] = off;
}
return off;
}
}
// Size of memory divided by 10 from observation of OOM errors (when supposedly
// 20% of memory is full) and onset of thrashing (when supposedly 15% of memory
// is full) on ECS.
if ((size_t) (poolfile->off + treefile->off) > memsize / CPUS / 10) {
// If the pool and search tree get to be larger than physical memory,
// then searching will start thrashing. Switch to appending strings
// to the file instead of keeping them in memory.
if (poolfile->fp == NULL) {
memfile_full(poolfile);
}
}
if (poolfile->fp != NULL) {
// We are now appending to the file, so don't try to keep tree references
// to the newly-added strings.
long long off = poolfile->off;
bool in_memory;
if (memfile_write(poolfile, &type, 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
return off;
}
// *sp is probably in the memory-mapped file, and will move if the file grows.
long long ssp;
if (sp == &treefile->tree) {
ssp = -1;
} else {
ssp = ((char *) sp) - treefile->map.c_str();
}
long long off = poolfile->off;
bool in_memory = false;
if (memfile_write(poolfile, &type, 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1, in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (in_memory) {
dedup[hash_off] = off;
}
if (off >= LONG_MAX || treefile->off >= LONG_MAX) {
// Tree or pool is bigger than 2GB
static bool warned = false;
if (!warned) {
fprintf(stderr, "Warning: string pool is very large.\n");
warned = true;
}
return off;
}
struct stringpool tsp;
tsp.left = 0;
tsp.right = 0;
tsp.off = off;
tsp.hash = hash;
long long p = treefile->off;
if (memfile_write(treefile, &tsp, sizeof(struct stringpool), in_memory) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (ssp == -1) {
treefile->tree = p;
} else {
*((long long *) (treefile->map.c_str() + ssp)) = p;
}
return off;
}