Files
tippecanoe/jsontool.cpp
T
Claude 7127e49c86 Replace the Grisu2 float formatter with a C++ port of rsc/fpfmt
Tippecanoe formatted every double it wrote through milo::dtoa_milo, a
vendored Grisu2. Grisu2 is fast, but it guarantees neither the shortest
digit string nor the correctly rounded one: it only guarantees that what
it prints parses back to the value it came from. In practice it prints a
digit more than necessary about 0.16% of the time, and picks a neighbor
of the correctly rounded digits about 32% of the time.

This ports Russ Cox's fpfmt (https://github.com/rsc/fpfmt) to C++ in
fpfmt/ and formats through it instead. fpfmt is both shortest and
correctly rounded, and it is faster:

  full std::string formatting     Grisu2      fpfmt   speedup
  random bit patterns          156.62 ns   66.83 ns     2.34x
  geo coordinates              124.07 ns   58.62 ns     2.12x
  short decimals                69.37 ns   49.16 ns     1.41x
  small integers                44.18 ns   38.06 ns     1.16x

  digit generation only           Grisu2      fpfmt   speedup
  random bit patterns           90.07 ns   20.81 ns     4.33x
  geo coordinates               80.64 ns   20.18 ns     4.00x
  short decimals                55.61 ns   21.90 ns     2.54x
  small integers                40.23 ns   22.50 ns     1.79x

(Intel Xeon @ 2.80GHz, g++ 13.3 -O3. `make fpfmt-bench` reproduces this,
and `./fpfmt-bench -check` reruns the correctness sweep, which is why
milo/dtoa_milo.h is kept even though nothing links it any more.)

The port is deliberately literal, so it can be diffed against fpfmt.go.
Its Short() agrees bit for bit with the Go original's on 445,640 values
covering powers of ten, small integers and reciprocals, subnormals, and
random bit patterns. Over 38.5 million values, fpfmt::dtoa always round
trips, is never longer than Grisu2's output, and is shorter 61,329 times.

Output is otherwise formatted exactly as before, including the choice
between plain and exponential notation, so 26 expected test outputs
change: some numbers lose digits (-26.170044999999999 becomes
-26.170045), and some have a corrected final digit (9.823748927348929e+55
becomes 9.823748927348928e+55). Every changed token was checked to parse
back to the identical double; none of the values themselves moved.

milo/milo.h, whose only job was to declare the C shim jsonpull calls, is
replaced by fpfmt/fpfmt.h, and the shim is renamed dtoa_shortest.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014wJRAuhMninQE4wK2TUfuZ
2026-08-31 00:22:35 +00:00

483 lines
10 KiB
C++

#include <stdio.h>
#include <stdlib.h>
#include <ctype.h>
#include <string.h>
#include <stdarg.h>
#include <unistd.h>
#include <string>
#include <getopt.h>
#include <vector>
#include "jsonpull/jsonpull.h"
#include "csv.hpp"
#include "text.hpp"
#include "geojson-loop.hpp"
#include "fpfmt/fpfmt.hpp"
#include "errors.hpp"
#include "usage.hpp"
int fail = EXIT_SUCCESS;
bool wrap = false;
const char *extract = NULL;
FILE *csvfile = NULL;
std::vector<std::string> header;
std::vector<std::string> fields;
int pe = false;
std::string buffered;
int buffered_type = -1;
// 0: nothing yet
// 1: buffered a line
// 2: wrote the line and the wrapper
int buffer_state = 0;
std::vector<unsigned long> decode32(const char *s) {
std::vector<unsigned long> utf32;
while (*s != '\0') {
unsigned long b = *(s++) & 0xFF;
if (b < 0x80) {
utf32.push_back(b);
} else if ((b & 0xe0) == 0xc0) {
unsigned long c = (b & 0x1f) << 6;
unsigned long b1 = *(s++) & 0xFF;
if ((b1 & 0xc0) == 0x80) {
c |= b1 & 0x3f;
utf32.push_back(c);
} else {
s--;
utf32.push_back(0xfffd);
}
} else if ((b & 0xf0) == 0xe0) {
unsigned long c = (b & 0x0f) << 12;
unsigned long b1 = *(s++) & 0xFF;
if ((b1 & 0xc0) == 0x80) {
c |= (b1 & 0x3f) << 6;
unsigned long b2 = *(s++) & 0xFF;
if ((b2 & 0xc0) == 0x80) {
c |= b2 & 0x3f;
utf32.push_back(c);
} else {
s -= 2;
utf32.push_back(0xfffd);
}
} else {
s--;
utf32.push_back(0xfffd);
}
} else if ((b & 0xf8) == 0xf0) {
unsigned long c = (b & 0x07) << 18;
unsigned long b1 = *(s++) & 0xFF;
if ((b1 & 0xc0) == 0x80) {
c |= (b1 & 0x3f) << 12;
unsigned long b2 = *(s++) & 0xFF;
if ((b2 & 0xc0) == 0x80) {
c |= (b2 & 0x3f) << 6;
unsigned long b3 = *(s++) & 0xFF;
if ((b3 & 0xc0) == 0x80) {
c |= b3 & 0x3f;
utf32.push_back(c);
} else {
s -= 3;
utf32.push_back(0xfffd);
}
} else {
s -= 2;
utf32.push_back(0xfffd);
}
} else {
s -= 1;
utf32.push_back(0xfffd);
}
} else {
utf32.push_back(0xfffd);
}
}
return utf32;
}
// This uses a really weird encoding for strings
// so that they will sort in UTF-32 order in spite of quoting
std::string sort_quote(const char *s) {
std::vector<unsigned long> utf32 = decode32(s);
std::string ret;
for (size_t i = 0; i < utf32.size(); i++) {
if (utf32[i] < 0xD800) {
char buf[8];
snprintf(buf, sizeof(buf), "\\u%04lu", utf32[i]);
ret.append(std::string(buf));
} else {
unsigned long c = utf32[i];
if (c <= 0x7f) {
ret.push_back(c);
} else if (c <= 0x7ff) {
ret.push_back(0xc0 | (c >> 6));
ret.push_back(0x80 | (c & 0x3f));
} else if (c <= 0xffff) {
ret.push_back(0xe0 | (c >> 12));
ret.push_back(0x80 | ((c >> 6) & 0x3f));
ret.push_back(0x80 | (c & 0x3f));
} else {
ret.push_back(0xf0 | (c >> 18));
ret.push_back(0x80 | ((c >> 12) & 0x3f));
ret.push_back(0x80 | ((c >> 6) & 0x3f));
ret.push_back(0x80 | (c & 0x3f));
}
}
}
return ret;
}
void out(std::string const &s, int type, json_object *properties) {
if (extract != NULL) {
std::string extracted = sort_quote("null");
bool found = false;
json_object *o = json_hash_get(properties, extract);
if (o != nullptr) {
found = true;
if (o->type == JSON_STRING) {
extracted = sort_quote(o->string().c_str());
} else {
// Numbers, booleans, null, and any other non-string
// values are rendered via json_stringify(); calling
// o->string() here would assert because the type-tagged
// accessor requires JSON_STRING.
extracted = sort_quote(json_stringify(o).c_str());
}
}
if (!found) {
static bool warned = false;
if (!warned) {
fprintf(stderr, "Warning: extract key \"%s\" not found in JSON\n", extract);
warned = true;
}
}
printf("{\"%s\":%s}\n", extracted.c_str(), s.c_str());
return;
}
if (!wrap) {
printf("%s\n", s.c_str());
return;
}
if (buffer_state == 0) {
buffered = s;
buffered_type = type;
buffer_state = 1;
return;
}
if (buffer_state == 1) {
if (buffered_type == 1) {
printf("{\"type\":\"FeatureCollection\",\"features\":[\n");
} else {
printf("{\"type\":\"GeometryCollection\",\"geometries\":[\n");
}
printf("%s\n", buffered.c_str());
buffer_state = 2;
}
printf(",\n%s\n", s.c_str());
if (type != buffered_type) {
fprintf(stderr, "Error: mix of bare geometries and features\n");
exit(EXIT_IMPOSSIBLE);
}
}
std::string prev_joinkey;
void join_csv(json_object *j) {
if (header.size() == 0) {
std::string s = csv_getline(csvfile);
if (s.size() == 0) {
fprintf(stderr, "Couldn't get column header from CSV file\n");
exit(EXIT_CSV);
}
std::string err = check_utf8(s);
if (err != "") {
fprintf(stderr, "%s\n", err.c_str());
exit(EXIT_UTF8);
}
header = csv_split(s.c_str());
for (size_t i = 0; i < header.size(); i++) {
header[i] = csv_dequote(header[i]);
}
if (header.size() == 0) {
fprintf(stderr, "No columns in CSV header \"%s\"\n", s.c_str());
exit(EXIT_CSV);
}
}
json_object *properties = json_hash_get(j, "properties");
json_object *key = nullptr;
if (properties != nullptr) {
key = json_hash_get(properties, header[0].c_str());
}
if (key == nullptr) {
static bool warned = false;
if (!warned) {
fprintf(stderr, "Warning: couldn't find CSV key \"%s\" in JSON\n", header[0].c_str());
warned = true;
}
return;
}
std::string joinkey;
if (key->type == JSON_STRING) {
joinkey = key->string();
} else if (key->type == JSON_NUMBER) {
joinkey = fpfmt::dtoa(key->number());
} else {
joinkey = json_stringify(key);
}
if (joinkey < prev_joinkey) {
fprintf(stderr, "GeoJSON file is out of sort: \"%s\" follows \"%s\"\n", joinkey.c_str(), prev_joinkey.c_str());
exit(EXIT_IMPOSSIBLE);
}
prev_joinkey = joinkey;
if (fields.size() == 0 || joinkey > fields[0]) {
std::string prevkey;
if (fields.size() > 0) {
prevkey = fields[0];
}
while (true) {
std::string s = csv_getline(csvfile);
if (s.size() == 0) {
fields.clear();
break;
}
std::string err = check_utf8(s);
if (err != "") {
fprintf(stderr, "%s\n", err.c_str());
exit(EXIT_UTF8);
}
fields = csv_split(s.c_str());
for (size_t i = 0; i < fields.size(); i++) {
fields[i] = csv_dequote(fields[i]);
}
if (fields.size() > 0 && fields[0] < prevkey) {
fprintf(stderr, "CSV file is out of sort: \"%s\" follows \"%s\"\n", fields[0].c_str(), prevkey.c_str());
exit(EXIT_CSV);
}
if (fields.size() > 0 && fields[0] >= joinkey) {
break;
}
if (fields.size() > 0) {
prevkey = fields[0];
}
}
}
if (fields.size() > 0 && joinkey == fields[0]) {
properties->entries().reserve(properties->entries().size() + fields.size());
for (size_t i = 1; i < fields.size(); i++) {
std::string k = header[i];
std::string v = fields[i];
json_type attr_type = JSON_STRING;
if (v.size() > 0) {
if (v[0] == '"') {
v = csv_dequote(v);
} else if (is_number(v)) {
attr_type = JSON_NUMBER;
}
} else if (pe) {
attr_type = JSON_NULL;
}
if (attr_type != JSON_NULL) {
json_object_ptr ko(new json_string(properties, properties->parser));
ko->string() = k;
json_object_ptr vo;
if (attr_type == JSON_STRING) {
vo = json_object_ptr(new json_string(properties, properties->parser));
vo->string() = v;
} else if (attr_type == JSON_NUMBER) {
vo = json_object_ptr(new json_number(properties, properties->parser));
vo->set_number(atof(v.c_str()));
} else {
abort();
}
properties->entries().push_back({std::move(ko), std::move(vo)});
}
}
}
}
struct json_join_action : json_feature_action {
int add_feature(json_object *geometry, bool, json_object *, json_object *, json_object *, json_object *feature) {
if (feature != geometry) { // a real feature, not a bare geometry
if (csvfile != NULL) {
join_csv(feature);
}
out(json_stringify(feature), 1, json_hash_get(feature, "properties"));
} else {
out(json_stringify(geometry), 2, nullptr);
}
return 1;
}
void check_crs(json_object *) {
}
};
void process(FILE *fp, const char *fname) {
json_pull_ptr jp = json_begin_file(fp);
json_join_action jja;
jja.fname = fname;
parse_json(&jja, jp);
}
static const struct option long_options[] = {
{"Wrapping the output", 0, 0, 0},
{"wrap", no_argument, 0, 'w'},
{"Sorting and joining", 0, 0, 0},
{"extract", required_argument, 0, 'e'},
{"csv", required_argument, 0, 'c'},
{"empty-csv-columns-are-null", no_argument, &pe, 1},
{"", 0, 0, 0},
{"prevent", required_argument, 0, 'p'},
{0, 0, 0, 0},
};
// the options above, with the usage message headings removed
static struct option real_long_options[sizeof(long_options) / sizeof(long_options[0])];
void usage(char **argv) {
static const char *const forms[] = {
"[options] [file.json ...]",
NULL,
};
print_usage(stderr, argv[0], forms, long_options, NULL);
fprintf(stderr, "\nIf no files are named, the JSON is read from the standard input.\n");
exit(EXIT_ARGS);
}
int main(int argc, char **argv) {
const char *csv = NULL;
strip_usage_headings(long_options, real_long_options);
std::string getopt_str = getopt_string(real_long_options);
extern int optind;
int i;
while ((i = getopt_long(argc, argv, getopt_str.c_str(), real_long_options, NULL)) != -1) {
switch (i) {
case 0:
break;
case 'w':
wrap = true;
break;
case 'e':
extract = optarg;
break;
case 'c':
csv = optarg;
break;
case 'p':
if (strcmp(optarg, "e") == 0) {
pe = true;
} else {
fprintf(stderr, "%s: Unknown option for -p%s\n", argv[0], optarg);
exit(EXIT_ARGS);
}
break;
default:
usage(argv);
}
}
if (extract != NULL && wrap) {
fprintf(stderr, "%s: --wrap and --extract not supported together\n", argv[0]);
exit(EXIT_ARGS);
}
if (csv != NULL) {
csvfile = fopen(csv, "r");
if (csvfile == NULL) {
perror(csv);
exit(EXIT_OPEN);
}
}
if (optind >= argc) {
process(stdin, "standard input");
} else {
for (i = optind; i < argc; i++) {
FILE *f = fopen(argv[i], "r");
if (f == NULL) {
perror(argv[i]);
exit(EXIT_OPEN);
}
process(f, argv[i]);
fclose(f);
}
}
if (buffer_state == 1) {
printf("%s\n", buffered.c_str());
} else if (buffer_state == 2) {
printf("]}\n");
}
if (csvfile != NULL) {
if (fclose(csvfile) != 0) {
perror("close");
exit(EXIT_CLOSE);
}
}
return fail;
}