From daf16e915cb385134e74f3e45f13227f9cc4d183 Mon Sep 17 00:00:00 2001 From: Erica Fischer Date: Mon, 17 Jul 2023 13:19:57 -0700 Subject: [PATCH] Factoring out dot-dropping for unit testing --- Makefile | 4 ++-- drop.cpp | 58 +++++++++++++++++++++++++++++++++++++++++++++++++ drop.hpp | 47 +++++++++++++++++++++++++++++++++++++++ main.cpp | 64 +----------------------------------------------------- main.hpp | 14 ------------ serial.hpp | 1 + unit.cpp | 7 ++++++ 7 files changed, 116 insertions(+), 79 deletions(-) create mode 100644 drop.cpp create mode 100644 drop.hpp diff --git a/Makefile b/Makefile index 9791de1d..efffcaf5 100644 --- a/Makefile +++ b/Makefile @@ -57,7 +57,7 @@ C = $(wildcard *.c) $(wildcard *.cpp) INCLUDES = -I/usr/local/include -I. LIBS = -L/usr/local/lib -tippecanoe: geojson.o jsonpull/jsonpull.o tile.o pool.o mbtiles.o geometry.o projection.o memfile.o mvt.o serial.o main.o text.o dirtiles.o pmtiles_file.o plugin.o read_json.o write_json.o geobuf.o flatgeobuf.o evaluator.o geocsv.o csv.o geojson-loop.o json_logger.o visvalingam.o compression.o +tippecanoe: geojson.o jsonpull/jsonpull.o tile.o pool.o mbtiles.o geometry.o projection.o memfile.o mvt.o serial.o main.o drop.o text.o dirtiles.o pmtiles_file.o plugin.o read_json.o write_json.o geobuf.o flatgeobuf.o evaluator.o geocsv.o csv.o geojson-loop.o json_logger.o visvalingam.o compression.o $(CXX) $(PG) $(LIBS) $(FINAL_FLAGS) $(CXXFLAGS) -o $@ $^ $(LDFLAGS) -lm -lz -lsqlite3 -lpthread tippecanoe-enumerate: enumerate.o @@ -72,7 +72,7 @@ tile-join: tile-join.o projection.o mbtiles.o mvt.o memfile.o dirtiles.o jsonpul tippecanoe-json-tool: jsontool.o jsonpull/jsonpull.o csv.o text.o geojson-loop.o $(CXX) $(PG) $(LIBS) $(FINAL_FLAGS) $(CXXFLAGS) -o $@ $^ $(LDFLAGS) -lm -lz -lsqlite3 -lpthread -unit: unit.o text.o +unit: unit.o text.o drop.o $(CXX) $(PG) $(LIBS) $(FINAL_FLAGS) $(CXXFLAGS) -o $@ $^ $(LDFLAGS) -lm -lz -lsqlite3 -lpthread -include $(wildcard *.d) diff --git a/drop.cpp b/drop.cpp new file mode 100644 index 00000000..36269098 --- /dev/null +++ b/drop.cpp @@ -0,0 +1,58 @@ +#include "drop.hpp" +#include "options.hpp" +#include "geometry.hpp" + +unsigned long long preserve_point_density_threshold = 0; + +int calc_feature_minzoom(struct index *ix, struct drop_state *ds, int maxzoom, double gamma) { + int feature_minzoom = 0; + + if (gamma >= 0 && (ix->t == VT_POINT || + (additional[A_LINE_DROP] && ix->t == VT_LINE) || + (additional[A_POLYGON_DROP] && ix->t == VT_POLYGON))) { + for (ssize_t i = maxzoom; i >= 0; i--) { + ds[i].seq++; + } + ssize_t chosen = maxzoom + 1; + for (ssize_t i = maxzoom; i >= 0; i--) { + if (ds[i].seq < 0) { + feature_minzoom = i + 1; + + // The feature we are pushing out + // appears in zooms i + 1 through maxzoom, + // so track where that was so we can make sure + // not to cluster something else that is *too* + // far away into it. + for (ssize_t j = i + 1; j <= maxzoom; j++) { + ds[j].previndex = ix->ix; + } + + chosen = i + 1; + break; + } else { + ds[i].seq -= ds[i].interval; + } + } + + // If this feature has been chosen only for a high zoom level, + // check whether at a low zoom level it is nevertheless too far + // from the last feature chosen for that low zoom, in which case + // we will go ahead and push it out. + + if (preserve_point_density_threshold > 0) { + for (ssize_t i = 0; i < chosen && i < maxzoom; i++) { + if (ix->ix - ds[i].previndex > ((1LL << (32 - i)) / preserve_point_density_threshold) * ((1LL << (32 - i)) / preserve_point_density_threshold)) { + feature_minzoom = i; + + for (ssize_t j = i; j <= maxzoom; j++) { + ds[j].previndex = ix->ix; + } + + break; + } + } + } + } + + return feature_minzoom; +} diff --git a/drop.hpp b/drop.hpp new file mode 100644 index 00000000..bd13b443 --- /dev/null +++ b/drop.hpp @@ -0,0 +1,47 @@ +#ifndef DROP_HPP +#define DROP_HPP + +// As features are read during the input phase, each one is represented by +// an index entry giving its geometry type, its spatial index, and the location +// in the geometry file of the rest of its data. +// +// Note that the fields are in a specific order so that `segment` and `t` will +// packed together with `seq` so that the total structure size will be only 32 bytes +// instead of 40. (Could we save a few more, perhaps, by tracking `len` instead of +// `end` and limiting the size of individual features to 32 bits?) + +struct index { + // first and last+1 byte of the feature in the geometry temp file + long long start = 0; + long long end = 0; + + // z-index or hilbert index of the feature + unsigned long long ix = 0; + + // which thread's geometry temp file this feature is in + short segment = 0; + + // geometry type + unsigned short t : 2; + + // sequence number (sometimes with gaps in numbering) of the feature in the original input file + unsigned long long seq : (64 - 18); // pack with segment and t to stay in 32 bytes + + index() + : t(0), + seq(0) { + } +}; + +struct drop_state { + double gap; + unsigned long long previndex; + double interval; + double seq; // floating point because interval is +}; + +extern unsigned long long preserve_point_density_threshold; + +int calc_feature_minzoom(struct index *ix, struct drop_state *ds, int maxzoom, double gamma); + +#endif diff --git a/main.cpp b/main.cpp index a124c8d8..61330583 100644 --- a/main.cpp +++ b/main.cpp @@ -65,6 +65,7 @@ #include "text.hpp" #include "errors.hpp" #include "read_json.hpp" +#include "drop.hpp" static int low_detail = 12; static int full_detail = -1; @@ -90,7 +91,6 @@ size_t limit_tile_feature_count = 0; size_t limit_tile_feature_count_at_maxzoom = 0; unsigned int drop_denser = 0; std::map set_attributes; -unsigned long long preserve_point_density_threshold = 0; std::vector order_by; bool order_reverse; @@ -274,13 +274,6 @@ static void insert(struct mergelist *m, struct mergelist **head, unsigned char * *head = m; } -struct drop_state { - double gap; - unsigned long long previndex; - double interval; - double seq; // floating point because interval is -}; - struct drop_densest { unsigned long long gap; size_t seq; @@ -291,61 +284,6 @@ struct drop_densest { } }; -int calc_feature_minzoom(struct index *ix, struct drop_state *ds, int maxzoom, double gamma) { - int feature_minzoom = 0; - - if (gamma >= 0 && (ix->t == VT_POINT || - (additional[A_LINE_DROP] && ix->t == VT_LINE) || - (additional[A_POLYGON_DROP] && ix->t == VT_POLYGON))) { - for (ssize_t i = maxzoom; i >= 0; i--) { - ds[i].seq++; - } - ssize_t chosen = maxzoom + 1; - for (ssize_t i = maxzoom; i >= 0; i--) { - if (ds[i].seq < 0) { - feature_minzoom = i + 1; - - // The feature we are pushing out - // appears in zooms i + 1 through maxzoom, - // so track where that was so we can make sure - // not to cluster something else that is *too* - // far away into it. - for (ssize_t j = i + 1; j <= maxzoom; j++) { - ds[j].previndex = ix->ix; - } - - chosen = i + 1; - break; - } else { - ds[i].seq -= ds[i].interval; - } - } - - // If this feature has been chosen only for a high zoom level, - // check whether at a low zoom level it is nevertheless too far - // from the last feature chosen for that low zoom, in which case - // we will go ahead and push it out. - - if (preserve_point_density_threshold > 0) { - for (ssize_t i = 0; i < chosen && i < maxzoom; i++) { - if (ix->ix - ds[i].previndex > ((1LL << (32 - i)) / preserve_point_density_threshold) * ((1LL << (32 - i)) / preserve_point_density_threshold)) { - feature_minzoom = i; - - for (ssize_t j = i; j <= maxzoom; j++) { - ds[j].previndex = ix->ix; - } - - break; - } - } - } - - // XXX manage_gap - } - - return feature_minzoom; -} - static void merge(struct mergelist *merges, size_t nmerges, unsigned char *map, FILE *indexfile, int bytes, char *geom_map, FILE *geom_out, std::atomic *geompos, long long *progress, long long *progress_max, long long *progress_reported, int maxzoom, double gamma, struct drop_state *ds) { struct mergelist *head = NULL; diff --git a/main.hpp b/main.hpp index 50f7e381..cc20166e 100644 --- a/main.hpp +++ b/main.hpp @@ -10,20 +10,6 @@ #include "json_logger.hpp" #include "serial.hpp" -struct index { - long long start = 0; - long long end = 0; - unsigned long long ix = 0; - short segment = 0; - unsigned short t : 2; - unsigned long long seq : (64 - 18); // pack with segment and t to stay in 32 bytes - - index() - : t(0), - seq(0) { - } -}; - struct clipbbox { double lon1; double lat1; diff --git a/serial.hpp b/serial.hpp index 778456d3..2f5f7962 100644 --- a/serial.hpp +++ b/serial.hpp @@ -9,6 +9,7 @@ #include #include "geometry.hpp" #include "mbtiles.hpp" +#include "drop.hpp" // for struct index #include "jsonpull/jsonpull.h" size_t fwrite_check(const void *ptr, size_t size, size_t nitems, FILE *stream, std::atomic *fpos, const char *fname); diff --git a/unit.cpp b/unit.cpp index 24c3fb10..d29d93e7 100644 --- a/unit.cpp +++ b/unit.cpp @@ -1,6 +1,7 @@ #define CATCH_CONFIG_MAIN #include "catch/catch.hpp" #include "text.hpp" +#include "drop.hpp" TEST_CASE("UTF-8 enforcement", "[utf8]") { REQUIRE(check_utf8("") == std::string("")); @@ -18,3 +19,9 @@ TEST_CASE("UTF-8 truncation", "[trunc]") { REQUIRE(truncate16("0123456789😀😬😁😂😃😄😅😆", 17) == std::string("0123456789😀😬😁")); REQUIRE(truncate16("0123456789あいうえおかきくけこさ", 16) == std::string("0123456789あいうえおか")); } + +TEST_CASE("index structure packing", "[index]") { + REQUIRE(sizeof(struct index) == 32); +} + +unsigned int additional[256] = {0};