Files
tippecanoe/pool.cpp
T
Erica Fischer f2127fec97 Reduce thrashing during feature ingestion and tiling (#56)
* Remove the concept of "separate metadata"

This was an extra level of attribute indirection (features point
to metadata records which point to key and value strings) which was
intended to reduce the size of temporary storage for features with
large numbers of attributes that were also spread across large numbers
of tiles at maxzoom.

For other kinds of features, the extra indirection slowed things down
instead, and, especially when maxzoom guessing was being used, many more
features were having their metadata externalized than could actually
benefit from it.

* Shave a few bytes off temporary files by using more unsigned integers

* Flush stderr after logging progress

* Revert "Shave a few bytes off temporary files by using more unsigned integers"

This reverts commit eef29084ec.

* Limit the size of the string pools and trees to fit in memory

* Add missing #include

* Move the string pool and search tree from mmap to allocated memory

* Sort in allocated rather than mapped memory too

* Also use pread instead of mapping to read in the data to sort

* When the pool gets too big, switch to just the file, not memory

* Switch string pool from memory to disk when memory is 10% full

* Add to-memory versions of the serialization functions

* Crashy work in progress toward compression

* Fix the pointer bug that was causing the crash

* Serialize features into memory rather than straight to disk

* Compress individual features in the temporary files

* Don't need to store the length of the geometry

* Remove per-feature compression; move minzoom back into the object

* Start adding a stream compressor object

* Track file position within fwrite_check()

* Add compressed stream writer functions

* Pull the writing of the serialized feature out to the callers

* Starting toward compression again from a different point

* Hook up more compression functions

* Remove unused code from the other day

* Make enough deflate calls to flush out all the buffered data

* Start on decompression

* Tile number is uncompressed, tile content is compressed

* Work on alternating compressed and uncompressed in decompression

* Closer, but still doesn't work

* Sort of works

* Works until we get to concatenated tiles

* More attempts that don't work

* One bug down

* It made a tileset!

* Handle nonzero initial zooms

* Fix seeking within compressed feature streams

* Tests pass!

* Remove debug spew

* Oops: remember to delete the temporary files so they don't hang around

* Test that fails with the current compression code

* Properly account for bytes read while closing the compressed stream

* Limit the number of warnings about bad label points

* A little more armor when closing decompression

* This time for sure

* A different, less fragile, test that failed previously with compression

* Move feature stream compression to its own file

* Remove now-unused code to deserialize from a file

* Forgot to add the new files

* Remove a little debugging logging

* Add a couple of comments on what it means to be within decompression

* Fix indentation

* Update changelog. Remove stray debugging comment.
2023-02-14 12:47:40 -08:00

155 lines
3.9 KiB
C++

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <limits.h>
#include <math.h>
#include "main.hpp"
#include "memfile.hpp"
#include "pool.hpp"
#include "errors.hpp"
int swizzlecmp(const char *a, const char *b) {
ssize_t alen = strlen(a);
ssize_t blen = strlen(b);
if (strcmp(a, b) == 0) {
return 0;
}
long long hash1 = 0, hash2 = 0;
for (ssize_t i = alen - 1; i >= 0; i--) {
hash1 = (hash1 * 37 + a[i]) & INT_MAX;
}
for (ssize_t i = blen - 1; i >= 0; i--) {
hash2 = (hash2 * 37 + b[i]) & INT_MAX;
}
int h1 = hash1, h2 = hash2;
if (h1 == h2) {
return strcmp(a, b);
}
return h1 - h2;
}
long long addpool(struct memfile *poolfile, struct memfile *treefile, const char *s, char type) {
unsigned long *sp = &treefile->tree;
size_t depth = 0;
// In typical data, traversal depth generally stays under 2.5x
size_t max = 3 * log(treefile->off / sizeof(struct stringpool)) / log(2);
if (max < 30) {
max = 30;
}
while (*sp != 0) {
int cmp = swizzlecmp(s, poolfile->map.c_str() + ((struct stringpool *) (treefile->map.c_str() + *sp))->off + 1);
if (cmp == 0) {
cmp = type - (poolfile->map.c_str() + ((struct stringpool *) (treefile->map.c_str() + *sp))->off)[0];
}
if (cmp < 0) {
sp = &(((struct stringpool *) (treefile->map.c_str() + *sp))->left);
} else if (cmp > 0) {
sp = &(((struct stringpool *) (treefile->map.c_str() + *sp))->right);
} else {
return ((struct stringpool *) (treefile->map.c_str() + *sp))->off;
}
depth++;
if (depth > max) {
// Search is very deep, so string is probably unique.
// Add it to the pool without adding it to the search tree.
// This might go either to memory or the file, depending on whether
// the pool is full yet.
long long off = poolfile->off;
if (memfile_write(poolfile, &type, 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
return off;
}
}
// Size of memory divided by 10 from observation of OOM errors (when supposedly
// 20% of memory is full) and onset of thrashing (when supposedly 15% of memory
// is full) on ECS.
if ((size_t) (poolfile->off + treefile->off) > memsize / CPUS / 10) {
// If the pool and search tree get to be larger than physical memory,
// then searching will start thrashing. Switch to appending strings
// to the file instead of keeping them in memory.
if (poolfile->fp == NULL) {
memfile_full(poolfile);
}
}
if (poolfile->fp != NULL) {
// We are now appending to the file, so don't try to keep tree references
// to the newly-added strings.
long long off = poolfile->off;
if (memfile_write(poolfile, &type, 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
return off;
}
// *sp is probably in the memory-mapped file, and will move if the file grows.
long long ssp;
if (sp == &treefile->tree) {
ssp = -1;
} else {
ssp = ((char *) sp) - treefile->map.c_str();
}
long long off = poolfile->off;
if (memfile_write(poolfile, &type, 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (memfile_write(poolfile, (void *) s, strlen(s) + 1) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (off >= LONG_MAX || treefile->off >= LONG_MAX) {
// Tree or pool is bigger than 2GB
static bool warned = false;
if (!warned) {
fprintf(stderr, "Warning: string pool is very large.\n");
warned = true;
}
return off;
}
struct stringpool tsp;
tsp.left = 0;
tsp.right = 0;
tsp.off = off;
long long p = treefile->off;
if (memfile_write(treefile, &tsp, sizeof(struct stringpool)) < 0) {
perror("memfile write");
exit(EXIT_WRITE);
}
if (ssp == -1) {
treefile->tree = p;
} else {
*((long long *) (treefile->map.c_str() + ssp)) = p;
}
return off;
}