Be more careful to retry when the feature count is exceeded (#257)

* Clip before dealing with multiplier or filters in overzoom

* Be more careful to retry when the feature count is exceeded

* Adjust the estimated total feature count for the multiplier too

* Fix the feature count estimates, I think

* Pass build info into the version string

* Report the actual max zoom of any tiles as the metadata maxzoom

* Revert unneeded renaming to make the diff more readable

* Clean up the adjustments to tile sizes and feature counts

* Update version and changelog

* Dropping a feature into a multiplier cluster still effectively drops it

* Update changelog

* Rethink the changelog description

* Don't try to truncate zooms if we are still tiling at z18
This commit is contained in:
Erica Fischer
2024-08-20 10:54:16 -07:00
committed by GitHub
parent d891ca7289
commit 40bb4ff732
16 changed files with 9166 additions and 7963 deletions
+96 -43
View File
@@ -1525,6 +1525,7 @@ bool drop_feature_unless_it_can_be_added_to_a_multiplier_cluster(layer_features
ssize_t which_serial_feature;
if (find_feature_to_accumulate_onto(layer.features, sf, which_serial_feature, layer_unmaps, LLONG_MAX, multiplier_seq)) {
strategy.dropped_as_needed++;
if (layer.multiplier_cluster_size < (size_t) retain_points_multiplier) {
// we have capacity to keep this feature as part of an existing multiplier cluster that isn't full yet
// so do that instead of dropping it
@@ -1532,7 +1533,6 @@ bool drop_feature_unless_it_can_be_added_to_a_multiplier_cluster(layer_features
return false; // converted rather than dropped
} else {
preserve_attributes(attribute_accum, sf, layer.features[which_serial_feature]);
strategy.dropped_as_needed++;
drop_rest = true;
return true; // dropped
}
@@ -1589,8 +1589,7 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
// empirical estimate from ne_10m_admin_0_countries, CPAD units, Cal fires.
// only try to make an overzoomable final tile if it seems like it might work
long long estimated_output_tile_size = 0.6693 * estimated_complexity - 3.36e+04;
if (estimated_output_tile_size < (long long) (0.9 * max_tile_size)) {
if (estimated_output_tile_size < (long long) (0.9 * max_tile_size) && 30 - z > detail) {
first_detail = 30 - z;
second_detail = detail;
trying_to_stop_early = true;
@@ -1640,6 +1639,9 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
size_t lead_features_count = 0; // of the tile so far
size_t other_multiplier_cluster_features_count = 0; // of the tile so far
bool too_many_features = false;
bool too_many_bytes = false;
std::atomic<bool> within[child_shards];
long long start_geompos[child_shards];
for (size_t i = 0; i < (size_t) child_shards; i++) {
@@ -1867,8 +1869,8 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
} else if (additional[A_DROP_DENSEST_AS_NEEDED]) {
add_sample_to(gaps, sf.gap, gaps_increment, seq);
if (sf.gap < mingap) {
can_stop_early = false;
if (drop_feature_unless_it_can_be_added_to_a_multiplier_cluster(layer, sf, layer_unmaps, multiplier_seq, strategy, drop_rest, arg->attribute_accum)) {
can_stop_early = false;
continue;
}
}
@@ -1912,8 +1914,8 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
// search here is for LLONG_MAX, not minextent, because we are dropping features, not coalescing them,
// so we shouldn't expect to find anything small that we can related this feature to.
if (minextent != 0 && sf.extent + coalesced_area <= minextent) {
can_stop_early = false;
if (drop_feature_unless_it_can_be_added_to_a_multiplier_cluster(layer, sf, layer_unmaps, multiplier_seq, strategy, drop_rest, arg->attribute_accum)) {
can_stop_early = false;
continue;
}
}
@@ -1931,11 +1933,9 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
}
} else if (additional[A_DROP_FRACTION_AS_NEEDED] || prevent[P_DYNAMIC_DROP]) {
add_sample_to(drop_sequences, drop_sequence, drop_sequences_increment, seq);
// search here is for LLONG_MAX, not minextent, because we are dropping features, not coalescing them,
// so we shouldn't expect to find anything small that we can related this feature to.
if (mindrop_sequence != 0 && drop_sequence <= mindrop_sequence) {
can_stop_early = false;
if (drop_feature_unless_it_can_be_added_to_a_multiplier_cluster(layer, sf, layer_unmaps, multiplier_seq, strategy, drop_rest, arg->attribute_accum)) {
can_stop_early = false;
continue;
}
}
@@ -1995,10 +1995,42 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
unsigned long long sfindex = sf.index;
if (sf.geometry.size() > 0) {
if (lead_features_count > max_tile_size || (lead_features_count + other_multiplier_cluster_features_count > max_tile_features && !prevent[P_FEATURE_LIMIT])) {
// There are two adjustments that we need to make
// when determining if we have hit the feature count
// or byte size limit:
//
// The max_tile_size and max_tile_features are inflated
// to account for the number of multiplier clusters features
// we are carrying around in addition to their lead features.
size_t adjusted_max_tile_size = max_tile_size;
if (lead_features_count > 0) {
adjusted_max_tile_size = adjusted_max_tile_size * (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
}
size_t adjusted_max_tile_features = max_tile_features;
if (lead_features_count > 0) {
adjusted_max_tile_features = adjusted_max_tile_features * (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
}
// The number of features in the tile, meanwhile, is inflated
// to account for the number of features that we have skipped over
// because we were already over the limit, in addition to those
// that we have actually added to the layer (as either lead or
// multiplier features).
size_t adjusted_feature_count = lead_features_count + other_multiplier_cluster_features_count;
if (kept > 0) {
adjusted_feature_count = adjusted_feature_count * (skipped + kept) / kept;
}
if (too_many_bytes || adjusted_feature_count > adjusted_max_tile_size) {
// Even being maximally conservative, each feature is still going to be
// at least one byte in the output tile, so this can't possibly work.
skipped++;
too_many_bytes = true;
} else if (too_many_features || ((adjusted_feature_count > adjusted_max_tile_features) && !prevent[P_FEATURE_LIMIT])) {
skipped++;
too_many_features = true;
} else {
kept++;
@@ -2134,13 +2166,13 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
first_time = false;
// Adjust tile size limit based on the ratio of multiplier cluster features to lead features
size_t scaled_max_tile_size = max_tile_size;
size_t adjusted_max_tile_size = max_tile_size;
if (lead_features_count > 0) {
scaled_max_tile_size *= (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
adjusted_max_tile_size = adjusted_max_tile_size * (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
}
size_t scaled_max_tile_features = max_tile_features;
size_t adjusted_max_tile_features = max_tile_features;
if (lead_features_count > 0) {
scaled_max_tile_features *= (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
adjusted_max_tile_features = adjusted_max_tile_features * (lead_features_count + other_multiplier_cluster_features_count) / lead_features_count;
}
// Operations on the features within each layer:
@@ -2387,11 +2419,11 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
}
mvt_tile tile;
size_t totalsize = 0;
size_t feature_count = 0;
for (auto layer_iterator = layers.begin(); layer_iterator != layers.end(); ++layer_iterator) {
std::vector<serial_feature> &layer_features = layer_iterator->second.features;
totalsize += layer_features.size();
feature_count += layer_features.size();
mvt_layer layer;
layer.name = layer_iterator->first;
@@ -2482,14 +2514,23 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
continue;
}
if (totalsize > 0 && tile.layers.size() > 0) {
if (totalsize > scaled_max_tile_features && !prevent[P_FEATURE_LIMIT]) {
if (totalsize > arg->feature_count_out) {
arg->feature_count_out = totalsize;
// Again, adjust the retabulated feature count to estimate
// how many total features there would have been if we hadn't
// hit the limit and started dropping early.
size_t adjusted_feature_count = feature_count;
if (kept > 0) {
adjusted_feature_count = adjusted_feature_count * (skipped + kept) / kept;
}
if (adjusted_feature_count > 0 && tile.layers.size() > 0) {
if (too_many_features || (adjusted_feature_count > adjusted_max_tile_features && !prevent[P_FEATURE_LIMIT])) {
if (adjusted_feature_count > arg->feature_count_out) {
arg->feature_count_out = adjusted_feature_count;
}
if (!quiet) {
fprintf(stderr, "tile %d/%u/%u has %zu features, >%zu \n", z, tx, ty, totalsize, scaled_max_tile_features);
fprintf(stderr, "tile %d/%u/%u has %zu (estimated %zu) features, >%zu \n", z, tx, ty, feature_count, adjusted_feature_count, adjusted_max_tile_features);
}
if (trying_to_stop_early && line_detail == first_detail) {
@@ -2515,7 +2556,7 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
line_detail++; // to keep it the same when the loop decrements it
continue;
} else if (mingap < ULONG_MAX && (additional[A_DROP_DENSEST_AS_NEEDED] || additional[A_COALESCE_DENSEST_AS_NEEDED] || additional[A_CLUSTER_DENSEST_AS_NEEDED])) {
mingap_fraction = mingap_fraction * scaled_max_tile_features / totalsize * 0.80;
mingap_fraction = mingap_fraction * adjusted_max_tile_features / adjusted_feature_count * 0.80;
unsigned long long m = choose_mingap(gaps, mingap_fraction, mingap);
if (m != mingap) {
mingap = m;
@@ -2530,7 +2571,7 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
continue;
}
} else if (additional[A_DROP_SMALLEST_AS_NEEDED] || additional[A_COALESCE_SMALLEST_AS_NEEDED]) {
minextent_fraction = minextent_fraction * scaled_max_tile_features / totalsize * 0.75;
minextent_fraction = minextent_fraction * adjusted_max_tile_features / adjusted_feature_count * 0.75;
long long m = choose_minextent(extents, minextent_fraction, minextent);
if (m != minextent) {
minextent = m;
@@ -2544,11 +2585,11 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
line_detail++;
continue;
}
} else if (totalsize > layers.size() && (additional[A_DROP_FRACTION_AS_NEEDED] || additional[A_COALESCE_FRACTION_AS_NEEDED] || prevent[P_DYNAMIC_DROP])) {
} else if (feature_count > layers.size() && (additional[A_DROP_FRACTION_AS_NEEDED] || additional[A_COALESCE_FRACTION_AS_NEEDED] || prevent[P_DYNAMIC_DROP])) {
// The 95% is a guess to avoid too many retries
// and probably actually varies based on how much duplicated metadata there is
mindrop_sequence_fraction = mindrop_sequence_fraction * scaled_max_tile_features / totalsize * 0.95;
mindrop_sequence_fraction = mindrop_sequence_fraction * adjusted_max_tile_features / adjusted_feature_count * 0.95;
unsigned long long m = choose_mindrop_sequence(drop_sequences, mindrop_sequence_fraction, mindrop_sequence);
if (m != mindrop_sequence) {
mindrop_sequence = m;
@@ -2581,26 +2622,24 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
compressed = pbf;
}
if (trying_to_stop_early && line_detail == first_detail) {
// printf("%lld %zu\n", estimated_complexity, compressed.size());
// And similarly, adjust the compressed byte size to estimate
// what it would have been if we hadn't stopped dropping features early
size_t adjusted_tile_size = compressed.size();
if (kept > 0) {
adjusted_tile_size = adjusted_tile_size * (kept + skipped) / kept;
}
if (compressed.size() > scaled_max_tile_size && !prevent[P_KILOBYTE_LIMIT]) {
// Estimate how big it really should have been compressed
// from how many features were kept vs skipped for already being
// over the threshold
double kept_adjust = (skipped + kept) / (double) kept;
if (compressed.size() > arg->tile_size_out) {
arg->tile_size_out = compressed.size() * kept_adjust;
if (too_many_bytes || (adjusted_tile_size > adjusted_max_tile_size && !prevent[P_KILOBYTE_LIMIT])) {
if (adjusted_tile_size > arg->tile_size_out) {
arg->tile_size_out = adjusted_tile_size;
}
if (!quiet) {
if (skipped > 0) {
fprintf(stderr, "tile %d/%u/%u size is %lld (probably really %lld) with detail %d, >%zu \n", z, tx, ty, (long long) compressed.size(), (long long) (compressed.size() * kept_adjust), line_detail, scaled_max_tile_size);
if (adjusted_tile_size == compressed.size()) {
fprintf(stderr, "tile %d/%u/%u size is %lld (probably really %zu) with detail %d, >%zu \n", z, tx, ty, (long long) compressed.size(), adjusted_tile_size, line_detail, adjusted_max_tile_size);
} else {
fprintf(stderr, "tile %d/%u/%u size is %lld with detail %d, >%zu \n", z, tx, ty, (long long) compressed.size(), line_detail, scaled_max_tile_size);
fprintf(stderr, "tile %d/%u/%u size is %lld with detail %d, >%zu \n", z, tx, ty, (long long) compressed.size(), line_detail, adjusted_max_tile_size);
}
}
@@ -2627,7 +2666,7 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
}
line_detail++; // to keep it the same when the loop decrements it
} else if (mingap < ULONG_MAX && (additional[A_DROP_DENSEST_AS_NEEDED] || additional[A_COALESCE_DENSEST_AS_NEEDED] || additional[A_CLUSTER_DENSEST_AS_NEEDED])) {
mingap_fraction = mingap_fraction * scaled_max_tile_size / (kept_adjust * compressed.size()) * 0.80;
mingap_fraction = mingap_fraction * adjusted_max_tile_size / adjusted_tile_size * 0.80;
unsigned long long m = choose_mingap(gaps, mingap_fraction, mingap);
if (m != mingap) {
mingap = m;
@@ -2642,7 +2681,7 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
continue;
}
} else if (additional[A_DROP_SMALLEST_AS_NEEDED] || additional[A_COALESCE_SMALLEST_AS_NEEDED]) {
minextent_fraction = minextent_fraction * scaled_max_tile_size / (kept_adjust * compressed.size()) * 0.75;
minextent_fraction = minextent_fraction * adjusted_max_tile_size / adjusted_tile_size * 0.75;
long long m = choose_minextent(extents, minextent_fraction, minextent);
if (m != minextent) {
minextent = m;
@@ -2656,8 +2695,8 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
line_detail++;
continue;
}
} else if (totalsize > layers.size() && (additional[A_DROP_FRACTION_AS_NEEDED] || additional[A_COALESCE_FRACTION_AS_NEEDED] || prevent[P_DYNAMIC_DROP])) {
mindrop_sequence_fraction = mindrop_sequence_fraction * scaled_max_tile_size / (kept_adjust * compressed.size()) * 0.75;
} else if (feature_count > layers.size() && (additional[A_DROP_FRACTION_AS_NEEDED] || additional[A_COALESCE_FRACTION_AS_NEEDED] || prevent[P_DYNAMIC_DROP])) {
mindrop_sequence_fraction = mindrop_sequence_fraction * adjusted_max_tile_size / adjusted_tile_size * 0.75;
unsigned long long m = choose_mindrop_sequence(drop_sequences, mindrop_sequence_fraction, mindrop_sequence);
if (m != mindrop_sequence) {
mindrop_sequence = m;
@@ -2682,6 +2721,11 @@ long long write_tile(decompressor *geoms, std::atomic<long long> *geompos_in, ch
exit(EXIT_PTHREAD);
}
if (skipped > 0 || too_many_bytes || too_many_features) {
fprintf(stderr, "Can't happen: writing tile even though we skipped\n");
exit(EXIT_IMPOSSIBLE);
}
if (outdb != NULL) {
mbtiles_write_tile(outdb, z, tx, ty, compressed.data(), compressed.size());
} else if (outdir != NULL) {
@@ -2802,7 +2846,6 @@ exit(EXIT_IMPOSSIBLE);
dc.begin();
}
arg->wrote_zoom = z;
long long len;
struct zxy parent(z - 1, x / 2, y / 2);
@@ -2810,6 +2853,7 @@ exit(EXIT_IMPOSSIBLE);
skip_tile(&dc, &geompos, arg->compressed);
len = 1;
} else {
arg->wrote_zoom = z;
len = write_tile(&dc, &geompos, arg->global_stringpool, z, x, y, z == arg->maxzoom ? arg->full_detail : arg->low_detail, arg->min_detail, arg->outdb, arg->outdir, arg->buffer, arg->fname, arg->geomfile, arg->geompos, arg->minzoom, arg->maxzoom, arg->todo, arg->along, geompos, arg->gamma, arg->child_shards, arg->pool_off, arg->initial_x, arg->initial_y, arg->running, arg->simplification, arg->layermaps, arg->layer_unmaps, arg->tiling_seg, arg->pass, arg->mingap, arg->minextent, arg->mindrop_sequence, arg->prefilter, arg->postfilter, arg->filter, arg, arg->strategy, arg->compressed, arg->shared_nodes_map, arg->nodepos, *(arg->shared_nodes_bloom), (*arg->unidecode_data), estimated_complexity, arg->skip_children_out);
}
@@ -2903,6 +2947,8 @@ int traverse_zooms(int *geomfd, off_t *geom_size, char *global_stringpool, std::
std::set<zxy> skip_children;
int z;
int largest_written = -1;
for (z = iz; z <= maxzoom; z++) {
std::atomic<long long> most(0);
@@ -3122,6 +3168,9 @@ int traverse_zooms(int *geomfd, off_t *geom_size, char *global_stringpool, std::
if (args[thread].wrote_zoom > z) {
z = args[thread].wrote_zoom;
}
if (args[thread].wrote_zoom > largest_written) {
largest_written = args[thread].wrote_zoom;
}
if (args[thread].still_dropping) {
extend_zooms = true;
@@ -3184,6 +3233,10 @@ int traverse_zooms(int *geomfd, off_t *geom_size, char *global_stringpool, std::
}
}
if (largest_written >= 0 && maxzoom > largest_written) {
maxzoom = largest_written;
}
for (size_t j = 0; j < TEMP_FILES; j++) {
// Can be < 0 if there is only one source file, at z0
if (geomfd[j] >= 0) {