mirror of
https://github.com/felt/tippecanoe.git
synced 2026-10-03 00:45:41 +02:00
* Generate the usage message of each tool from its long_options
The usage messages of tile-join, tippecanoe-overzoom,
tippecanoe-json-tool, tippecanoe-decode, and tippecanoe-enumerate were
hand-written lists of options that had drifted years out of date, since
nothing tied them to the options that are really accepted. Move the
option-list printing that tippecanoe already does into a shared
print_usage(), and use it in all the tools, so that the message is
derived from the same long_options table that getopt_long() gets and
can't fall behind it again.
The tables now carry section headings, as tippecanoe's does, and the
options that were only reachable by their short names (tile-join's -O,
-b, -R, and -r among them) are listed for the first time.
Also state the non-option arguments the way each tool really treats
them: tile-join takes source tilesets unless --read-from names a file to
read them from, tippecanoe-decode takes a tileset either alone or with a
zoom/x/y, tippecanoe-json-tool reads standard input when no files are
named, and tippecanoe-overzoom's two forms are the ones its argument
parsing recognizes. tippecanoe-overzoom now reports the missing -o
instead of passing NULL to fopen(), and tippecanoe-enumerate goes
through getopt_long() so that it will pick up any options added later.
The shared getopt_string() replaces the identical loop that four of the
tools each had for building the short option string, and strip_usage_headings()
the one for dropping the headings before getopt_long() sees them.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016frkRY1xXtiWjxYuCJ8vZY
* Print the usage message when tippecanoe is run with no arguments
Running `tippecanoe` with nothing at all reported the missing output
file, which is true but is not what someone who typed the bare command
needs to know. Check for the empty command line before parsing and print
the general usage message instead, and leave the specific complaint for
the case where an input file was named but an output file wasn't.
To make the message reachable from there, the options table and the
usage printing move out of main() into a usage() function, as in the
other tools.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016frkRY1xXtiWjxYuCJ8vZY
* Address review: alternation, the dead tile-join option, and --version
Four fixes from review of the generated usage messages:
* `--output` and `--output-to-directory` are one-of, not one required and
one optional, in both tippecanoe and tile-join. A `usage_required_option`
can now name an alternation that it belongs to, and the options in one
are listed together as `(--output=... | --output-to-directory=...)`,
which is what the runtime check enforces.
* tile-join's `--use-attribute-for-id` has had no implementation since
533e000 removed it; only the table entry was left behind, so the option
parsed and then exited with "Unrecognized option". Generating the usage
message from the table turned that into a documented option that doesn't
work, so remove the leftover entry too.
* `--version` was grouped under "Progress indicator", in the options table
and in the README both. Give it a heading of its own now that the
headings are something users see.
* print_usage() left `width` holding the length of the last synopsis line,
and only got away with it because every table so far begins with a
heading, which resets it. Start the option list on a line of its own
instead of depending on that.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016frkRY1xXtiWjxYuCJ8vZY
---------
Co-authored-by: Claude <noreply@anthropic.com>
523 lines
12 KiB
C++
523 lines
12 KiB
C++
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
#include <ctype.h>
|
|
#include <string.h>
|
|
#include <stdarg.h>
|
|
#include <unistd.h>
|
|
#include <string>
|
|
#include <getopt.h>
|
|
#include <vector>
|
|
#include "jsonpull/jsonpull.h"
|
|
#include "csv.hpp"
|
|
#include "text.hpp"
|
|
#include "geojson-loop.hpp"
|
|
#include "milo/dtoa_milo.h"
|
|
#include "errors.hpp"
|
|
#include "usage.hpp"
|
|
|
|
int fail = EXIT_SUCCESS;
|
|
bool wrap = false;
|
|
const char *extract = NULL;
|
|
|
|
FILE *csvfile = NULL;
|
|
std::vector<std::string> header;
|
|
std::vector<std::string> fields;
|
|
int pe = false;
|
|
|
|
std::string buffered;
|
|
int buffered_type = -1;
|
|
// 0: nothing yet
|
|
// 1: buffered a line
|
|
// 2: wrote the line and the wrapper
|
|
int buffer_state = 0;
|
|
|
|
std::vector<unsigned long> decode32(const char *s) {
|
|
std::vector<unsigned long> utf32;
|
|
|
|
while (*s != '\0') {
|
|
unsigned long b = *(s++) & 0xFF;
|
|
|
|
if (b < 0x80) {
|
|
utf32.push_back(b);
|
|
} else if ((b & 0xe0) == 0xc0) {
|
|
unsigned long c = (b & 0x1f) << 6;
|
|
unsigned long b1 = *(s++) & 0xFF;
|
|
|
|
if ((b1 & 0xc0) == 0x80) {
|
|
c |= b1 & 0x3f;
|
|
utf32.push_back(c);
|
|
} else {
|
|
s--;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else if ((b & 0xf0) == 0xe0) {
|
|
unsigned long c = (b & 0x0f) << 12;
|
|
unsigned long b1 = *(s++) & 0xFF;
|
|
|
|
if ((b1 & 0xc0) == 0x80) {
|
|
c |= (b1 & 0x3f) << 6;
|
|
unsigned long b2 = *(s++) & 0xFF;
|
|
|
|
if ((b2 & 0xc0) == 0x80) {
|
|
c |= b2 & 0x3f;
|
|
utf32.push_back(c);
|
|
} else {
|
|
s -= 2;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else {
|
|
s--;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else if ((b & 0xf8) == 0xf0) {
|
|
unsigned long c = (b & 0x07) << 18;
|
|
unsigned long b1 = *(s++) & 0xFF;
|
|
|
|
if ((b1 & 0xc0) == 0x80) {
|
|
c |= (b1 & 0x3f) << 12;
|
|
unsigned long b2 = *(s++) & 0xFF;
|
|
|
|
if ((b2 & 0xc0) == 0x80) {
|
|
c |= (b2 & 0x3f) << 6;
|
|
unsigned long b3 = *(s++) & 0xFF;
|
|
|
|
if ((b3 & 0xc0) == 0x80) {
|
|
c |= b3 & 0x3f;
|
|
|
|
utf32.push_back(c);
|
|
} else {
|
|
s -= 3;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else {
|
|
s -= 2;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else {
|
|
s -= 1;
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
} else {
|
|
utf32.push_back(0xfffd);
|
|
}
|
|
}
|
|
|
|
return utf32;
|
|
}
|
|
|
|
// This uses a really weird encoding for strings
|
|
// so that they will sort in UTF-32 order in spite of quoting
|
|
|
|
std::string sort_quote(const char *s) {
|
|
std::vector<unsigned long> utf32 = decode32(s);
|
|
std::string ret;
|
|
|
|
for (size_t i = 0; i < utf32.size(); i++) {
|
|
if (utf32[i] < 0xD800) {
|
|
char buf[8];
|
|
snprintf(buf, sizeof(buf), "\\u%04lu", utf32[i]);
|
|
ret.append(std::string(buf));
|
|
} else {
|
|
unsigned long c = utf32[i];
|
|
|
|
if (c <= 0x7f) {
|
|
ret.push_back(c);
|
|
} else if (c <= 0x7ff) {
|
|
ret.push_back(0xc0 | (c >> 6));
|
|
ret.push_back(0x80 | (c & 0x3f));
|
|
} else if (c <= 0xffff) {
|
|
ret.push_back(0xe0 | (c >> 12));
|
|
ret.push_back(0x80 | ((c >> 6) & 0x3f));
|
|
ret.push_back(0x80 | (c & 0x3f));
|
|
} else {
|
|
ret.push_back(0xf0 | (c >> 18));
|
|
ret.push_back(0x80 | ((c >> 12) & 0x3f));
|
|
ret.push_back(0x80 | ((c >> 6) & 0x3f));
|
|
ret.push_back(0x80 | (c & 0x3f));
|
|
}
|
|
}
|
|
}
|
|
|
|
return ret;
|
|
}
|
|
|
|
void out(std::string const &s, int type, json_object *properties) {
|
|
if (extract != NULL) {
|
|
std::string extracted = sort_quote("null");
|
|
bool found = false;
|
|
|
|
json_object *o = json_hash_get(properties, extract);
|
|
if (o != NULL) {
|
|
found = true;
|
|
if (o->type == JSON_STRING || o->type == JSON_NUMBER) {
|
|
extracted = sort_quote(o->value.string.string);
|
|
} else {
|
|
// Don't really know what to do about sort quoting
|
|
// for arbitrary objects
|
|
|
|
const char *out = json_stringify(o);
|
|
extracted = sort_quote(out);
|
|
free((void *) out);
|
|
}
|
|
}
|
|
|
|
if (!found) {
|
|
static bool warned = false;
|
|
if (!warned) {
|
|
fprintf(stderr, "Warning: extract key \"%s\" not found in JSON\n", extract);
|
|
warned = true;
|
|
}
|
|
}
|
|
|
|
printf("{\"%s\":%s}\n", extracted.c_str(), s.c_str());
|
|
return;
|
|
}
|
|
|
|
if (!wrap) {
|
|
printf("%s\n", s.c_str());
|
|
return;
|
|
}
|
|
|
|
if (buffer_state == 0) {
|
|
buffered = s;
|
|
buffered_type = type;
|
|
buffer_state = 1;
|
|
return;
|
|
}
|
|
|
|
if (buffer_state == 1) {
|
|
if (buffered_type == 1) {
|
|
printf("{\"type\":\"FeatureCollection\",\"features\":[\n");
|
|
} else {
|
|
printf("{\"type\":\"GeometryCollection\",\"geometries\":[\n");
|
|
}
|
|
|
|
printf("%s\n", buffered.c_str());
|
|
buffer_state = 2;
|
|
}
|
|
|
|
printf(",\n%s\n", s.c_str());
|
|
|
|
if (type != buffered_type) {
|
|
fprintf(stderr, "Error: mix of bare geometries and features\n");
|
|
exit(EXIT_IMPOSSIBLE);
|
|
}
|
|
}
|
|
|
|
std::string prev_joinkey;
|
|
|
|
void join_csv(json_object *j) {
|
|
if (header.size() == 0) {
|
|
std::string s = csv_getline(csvfile);
|
|
if (s.size() == 0) {
|
|
fprintf(stderr, "Couldn't get column header from CSV file\n");
|
|
exit(EXIT_CSV);
|
|
}
|
|
|
|
std::string err = check_utf8(s);
|
|
if (err != "") {
|
|
fprintf(stderr, "%s\n", err.c_str());
|
|
exit(EXIT_UTF8);
|
|
}
|
|
|
|
header = csv_split(s.c_str());
|
|
|
|
for (size_t i = 0; i < header.size(); i++) {
|
|
header[i] = csv_dequote(header[i]);
|
|
}
|
|
|
|
if (header.size() == 0) {
|
|
fprintf(stderr, "No columns in CSV header \"%s\"\n", s.c_str());
|
|
exit(EXIT_CSV);
|
|
}
|
|
}
|
|
|
|
json_object *properties = json_hash_get(j, "properties");
|
|
json_object *key = NULL;
|
|
|
|
if (properties != NULL) {
|
|
key = json_hash_get(properties, header[0].c_str());
|
|
}
|
|
|
|
if (key == NULL) {
|
|
static bool warned = false;
|
|
if (!warned) {
|
|
fprintf(stderr, "Warning: couldn't find CSV key \"%s\" in JSON\n", header[0].c_str());
|
|
warned = true;
|
|
}
|
|
return;
|
|
}
|
|
|
|
std::string joinkey;
|
|
if (key->type == JSON_STRING) {
|
|
joinkey = key->value.string.string;
|
|
} else if (key->type == JSON_NUMBER) {
|
|
joinkey = milo::dtoa_milo(key->value.number.number);
|
|
} else {
|
|
const char *s = json_stringify(key);
|
|
joinkey = s;
|
|
free((void *) s);
|
|
}
|
|
|
|
if (joinkey < prev_joinkey) {
|
|
fprintf(stderr, "GeoJSON file is out of sort: \"%s\" follows \"%s\"\n", joinkey.c_str(), prev_joinkey.c_str());
|
|
exit(EXIT_IMPOSSIBLE);
|
|
}
|
|
prev_joinkey = joinkey;
|
|
|
|
if (fields.size() == 0 || joinkey > fields[0]) {
|
|
std::string prevkey;
|
|
if (fields.size() > 0) {
|
|
prevkey = fields[0];
|
|
}
|
|
|
|
while (true) {
|
|
std::string s = csv_getline(csvfile);
|
|
if (s.size() == 0) {
|
|
fields.clear();
|
|
break;
|
|
}
|
|
|
|
std::string err = check_utf8(s);
|
|
if (err != "") {
|
|
fprintf(stderr, "%s\n", err.c_str());
|
|
exit(EXIT_UTF8);
|
|
}
|
|
|
|
fields = csv_split(s.c_str());
|
|
|
|
for (size_t i = 0; i < fields.size(); i++) {
|
|
fields[i] = csv_dequote(fields[i]);
|
|
}
|
|
|
|
if (fields.size() > 0 && fields[0] < prevkey) {
|
|
fprintf(stderr, "CSV file is out of sort: \"%s\" follows \"%s\"\n", fields[0].c_str(), prevkey.c_str());
|
|
exit(EXIT_CSV);
|
|
}
|
|
|
|
if (fields.size() > 0 && fields[0] >= joinkey) {
|
|
break;
|
|
}
|
|
|
|
if (fields.size() > 0) {
|
|
prevkey = fields[0];
|
|
}
|
|
}
|
|
}
|
|
|
|
if (fields.size() > 0 && joinkey == fields[0]) {
|
|
// This knows more about the structure of JSON objects than it ought to
|
|
// The 8 is to round up at least as much as SIZE_FOR in json_pull.c
|
|
properties->value.object.keys = (json_object **) realloc((void *) properties->value.object.keys, (properties->value.object.length + 8 + fields.size()) * sizeof(json_object *));
|
|
properties->value.object.values = (json_object **) realloc((void *) properties->value.object.values, (properties->value.object.length + 8 + fields.size()) * sizeof(json_object *));
|
|
if (properties->value.object.keys == NULL || properties->value.object.values == NULL) {
|
|
perror("realloc");
|
|
exit(EXIT_MEMORY);
|
|
}
|
|
|
|
for (size_t i = 1; i < fields.size(); i++) {
|
|
std::string k = header[i];
|
|
std::string v = fields[i];
|
|
json_type attr_type = JSON_STRING;
|
|
|
|
if (v.size() > 0) {
|
|
if (v[0] == '"') {
|
|
v = csv_dequote(v);
|
|
} else if (is_number(v)) {
|
|
attr_type = JSON_NUMBER;
|
|
}
|
|
} else if (pe) {
|
|
attr_type = JSON_NULL;
|
|
}
|
|
|
|
if (attr_type != JSON_NULL) {
|
|
// This knows more about the structure of JSON objects than it ought to
|
|
|
|
json_object *ko = (json_object *) malloc(sizeof(json_object));
|
|
json_object *vo = (json_object *) malloc(sizeof(json_object));
|
|
if (ko == NULL || vo == NULL) {
|
|
perror("malloc");
|
|
exit(EXIT_MEMORY);
|
|
}
|
|
|
|
ko->type = JSON_STRING;
|
|
ko->parent = properties;
|
|
ko->parser = properties->parser;
|
|
|
|
ko->value.string.string = strdup(k.c_str());
|
|
if (ko->value.string.string == NULL) {
|
|
perror("strdup");
|
|
exit(EXIT_MEMORY);
|
|
}
|
|
|
|
vo->type = attr_type;
|
|
vo->parent = properties;
|
|
vo->parser = properties->parser;
|
|
|
|
if (attr_type == JSON_STRING) {
|
|
vo->value.string.string = strdup(v.c_str());
|
|
if (vo->value.string.string == NULL) {
|
|
perror("strdup");
|
|
exit(EXIT_MEMORY);
|
|
}
|
|
} else if (attr_type == JSON_NUMBER) {
|
|
vo->value.number.number = atof(v.c_str());
|
|
vo->value.number.large_unsigned = 0;
|
|
vo->value.number.large_signed = 0;
|
|
} else {
|
|
abort();
|
|
}
|
|
|
|
properties->value.object.keys[properties->value.object.length] = ko;
|
|
properties->value.object.values[properties->value.object.length] = vo;
|
|
properties->value.object.length++;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
struct json_join_action : json_feature_action {
|
|
int add_feature(json_object *geometry, bool, json_object *, json_object *, json_object *, json_object *feature) {
|
|
if (feature != geometry) { // a real feature, not a bare geometry
|
|
if (csvfile != NULL) {
|
|
join_csv(feature);
|
|
}
|
|
|
|
char *s = json_stringify(feature);
|
|
out(s, 1, json_hash_get(feature, "properties"));
|
|
free(s);
|
|
} else {
|
|
char *s = json_stringify(geometry);
|
|
out(s, 2, NULL);
|
|
free(s);
|
|
}
|
|
|
|
return 1;
|
|
}
|
|
|
|
void check_crs(json_object *) {
|
|
}
|
|
};
|
|
|
|
void process(FILE *fp, const char *fname) {
|
|
json_pull *jp = json_begin_file(fp);
|
|
|
|
json_join_action jja;
|
|
jja.fname = fname;
|
|
parse_json(&jja, jp);
|
|
json_end(jp);
|
|
}
|
|
|
|
static const struct option long_options[] = {
|
|
{"Wrapping the output", 0, 0, 0},
|
|
{"wrap", no_argument, 0, 'w'},
|
|
|
|
{"Sorting and joining", 0, 0, 0},
|
|
{"extract", required_argument, 0, 'e'},
|
|
{"csv", required_argument, 0, 'c'},
|
|
{"empty-csv-columns-are-null", no_argument, &pe, 1},
|
|
|
|
{"", 0, 0, 0},
|
|
{"prevent", required_argument, 0, 'p'},
|
|
|
|
{0, 0, 0, 0},
|
|
};
|
|
|
|
// the options above, with the usage message headings removed
|
|
static struct option real_long_options[sizeof(long_options) / sizeof(long_options[0])];
|
|
|
|
void usage(char **argv) {
|
|
static const char *const forms[] = {
|
|
"[options] [file.json ...]",
|
|
NULL,
|
|
};
|
|
|
|
print_usage(stderr, argv[0], forms, long_options, NULL);
|
|
fprintf(stderr, "\nIf no files are named, the JSON is read from the standard input.\n");
|
|
exit(EXIT_ARGS);
|
|
}
|
|
|
|
int main(int argc, char **argv) {
|
|
const char *csv = NULL;
|
|
|
|
strip_usage_headings(long_options, real_long_options);
|
|
std::string getopt_str = getopt_string(real_long_options);
|
|
|
|
extern int optind;
|
|
int i;
|
|
|
|
while ((i = getopt_long(argc, argv, getopt_str.c_str(), real_long_options, NULL)) != -1) {
|
|
switch (i) {
|
|
case 0:
|
|
break;
|
|
|
|
case 'w':
|
|
wrap = true;
|
|
break;
|
|
|
|
case 'e':
|
|
extract = optarg;
|
|
break;
|
|
|
|
case 'c':
|
|
csv = optarg;
|
|
break;
|
|
|
|
case 'p':
|
|
if (strcmp(optarg, "e") == 0) {
|
|
pe = true;
|
|
} else {
|
|
fprintf(stderr, "%s: Unknown option for -p%s\n", argv[0], optarg);
|
|
exit(EXIT_ARGS);
|
|
}
|
|
break;
|
|
|
|
default:
|
|
usage(argv);
|
|
}
|
|
}
|
|
|
|
if (extract != NULL && wrap) {
|
|
fprintf(stderr, "%s: --wrap and --extract not supported together\n", argv[0]);
|
|
exit(EXIT_ARGS);
|
|
}
|
|
|
|
if (csv != NULL) {
|
|
csvfile = fopen(csv, "r");
|
|
if (csvfile == NULL) {
|
|
perror(csv);
|
|
exit(EXIT_OPEN);
|
|
}
|
|
}
|
|
|
|
if (optind >= argc) {
|
|
process(stdin, "standard input");
|
|
} else {
|
|
for (i = optind; i < argc; i++) {
|
|
FILE *f = fopen(argv[i], "r");
|
|
if (f == NULL) {
|
|
perror(argv[i]);
|
|
exit(EXIT_OPEN);
|
|
}
|
|
|
|
process(f, argv[i]);
|
|
fclose(f);
|
|
}
|
|
}
|
|
|
|
if (buffer_state == 1) {
|
|
printf("%s\n", buffered.c_str());
|
|
} else if (buffer_state == 2) {
|
|
printf("]}\n");
|
|
}
|
|
|
|
if (csvfile != NULL) {
|
|
if (fclose(csvfile) != 0) {
|
|
perror("close");
|
|
exit(EXIT_CLOSE);
|
|
}
|
|
}
|
|
|
|
return fail;
|
|
}
|