From 079929717e1a56ad40383f825abf5f5cf0aac119 Mon Sep 17 00:00:00 2001 From: Erica Fischer Date: Tue, 20 Dec 2022 12:54:21 -0800 Subject: [PATCH] Avoid spending gigabytes of memory on statistics for -as-needed dropping (#50) * Downsample indices and areas during tiling if they get too big * Cap the indices rather than downsampling them --- CHANGELOG.md | 4 ++++ tile.cpp | 53 ++++++++++++++++++++++++++++++++++++++++++++++------ version.hpp | 2 +- 3 files changed, 52 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c84e1ce9..cc70fc67 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,7 @@ +## 2.16.0 + +* During tiling, limit the size of the statistics that are kept for -as-needed calculations, because they can get quite large for sources with hundreds of millions of features. + ## 2.15.2 * Change tile hash function to fnv1a diff --git a/tile.cpp b/tile.cpp index 6e473bbb..890f8fdb 100644 --- a/tile.cpp +++ b/tile.cpp @@ -1829,6 +1829,27 @@ static bool line_is_too_small(drawvec const &geometry, int z, int detail) { return true; } +// Keep only a sample of 100K extents for feature dropping, +// to avoid spending lots of memory on a complete list when there are +// hundreds of millions of features. +template +void add_sample_to(std::vector &vals, T val, size_t &increment, size_t seq) { + if (seq % increment == 0) { + vals.push_back(val); + + if (vals.size() > 100000) { + std::vector tmp; + + for (size_t i = 0; i < vals.size(); i += 2) { + tmp.push_back(vals[i]); + } + + increment *= 2; + vals = tmp; + } + } +} + long long write_tile(FILE *geoms, std::atomic *geompos_in, char *metabase, char *stringpool, int z, const unsigned tx, const unsigned ty, const int detail, int min_detail, sqlite3 *outdb, const char *outdir, int buffer, const char *fname, FILE **geomfile, int minzoom, int maxzoom, double todo, std::atomic *along, long long alongminus, double gamma, int child_shards, long long *meta_off, long long *pool_off, unsigned *initial_x, unsigned *initial_y, std::atomic *running, double simplification, std::vector> *layermaps, std::vector> *layer_unmaps, size_t tiling_seg, size_t pass, unsigned long long mingap, long long minextent, double fraction, const char *prefilter, const char *postfilter, struct json_object *filter, write_tile_args *arg, atomic_strategy *strategy) { double merge_fraction = 1; double mingap_fraction = 1; @@ -1878,8 +1899,11 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta std::vector partials; std::map> layers; + std::vector indices; std::vector extents; + size_t extents_increment = 1; + double coalesced_area = 0; drawvec shared_nodes; @@ -1974,7 +1998,7 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta prefilter_jp = json_begin_file(prefilter_read_fp); } - while (1) { + for (size_t seq = 0; ; seq++) { serial_feature sf; ssize_t which_partial = -1; @@ -2018,8 +2042,18 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta } } + // Cap the indices, rather than sampling them like extents (areas), + // because choose_mingap cares about the distance between *surviving* + // features, not between *original* features, so we can't just store + // gaps rather than indices to be able to downsample them fairly. + // Hopefully the first 100K features in the tile are reasonably + // representative of the other features in the tile. + const size_t MAX_INDICES = 100000; + if (additional[A_CLUSTER_DENSEST_AS_NEEDED] || cluster_distance != 0) { - indices.push_back(sf.index); + if (indices.size() < MAX_INDICES) { + indices.push_back(sf.index); + } if ((sf.index < merge_previndex || sf.index - merge_previndex < mingap) && find_partial(partials, sf, which_partial, layer_unmaps, LLONG_MAX)) { partials[which_partial].clustered++; @@ -2040,14 +2074,18 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta continue; } } else if (additional[A_DROP_DENSEST_AS_NEEDED]) { - indices.push_back(sf.index); + if (indices.size() < MAX_INDICES) { + indices.push_back(sf.index); + } if (sf.index - merge_previndex < mingap && find_partial(partials, sf, which_partial, layer_unmaps, LLONG_MAX)) { preserve_attributes(arg->attribute_accum, sf, stringpool, pool_off, partials[which_partial]); strategy->dropped_as_needed++; continue; } } else if (additional[A_COALESCE_DENSEST_AS_NEEDED]) { - indices.push_back(sf.index); + if (indices.size() < MAX_INDICES) { + indices.push_back(sf.index); + } if (sf.index - merge_previndex < mingap && find_partial(partials, sf, which_partial, layer_unmaps, LLONG_MAX)) { partials[which_partial].geoms.push_back(sf.geometry); partials[which_partial].coalesced = true; @@ -2057,14 +2095,14 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta continue; } } else if (additional[A_DROP_SMALLEST_AS_NEEDED]) { - extents.push_back(sf.extent); + add_sample_to(extents, sf.extent, extents_increment, seq); if (sf.extent + coalesced_area <= minextent && find_partial(partials, sf, which_partial, layer_unmaps, minextent)) { preserve_attributes(arg->attribute_accum, sf, stringpool, pool_off, partials[which_partial]); strategy->dropped_as_needed++; continue; } } else if (additional[A_COALESCE_SMALLEST_AS_NEEDED]) { - extents.push_back(sf.extent); + add_sample_to(extents, sf.extent, extents_increment, seq); if (sf.extent + coalesced_area <= minextent && find_partial(partials, sf, which_partial, layer_unmaps, minextent)) { partials[which_partial].geoms.push_back(sf.geometry); partials[which_partial].coalesced = true; @@ -2109,6 +2147,9 @@ long long write_tile(FILE *geoms, std::atomic *geompos_in, char *meta if (reduced) { strategy->tiny_polygons++; } + if (sf.geometry.size() == 0) { + continue; + } } } if (sf.t == VT_POLYGON || sf.t == VT_LINE) { diff --git a/version.hpp b/version.hpp index a85bd4f5..078659fb 100644 --- a/version.hpp +++ b/version.hpp @@ -1,6 +1,6 @@ #ifndef VERSION_HPP #define VERSION_HPP -#define VERSION "v2.15.2" +#define VERSION "v2.16.0" #endif