diff --git a/jsonpull/jsonpull.cpp b/jsonpull/jsonpull.cpp index 3929e1f0..c96adcc0 100644 --- a/jsonpull/jsonpull.cpp +++ b/jsonpull/jsonpull.cpp @@ -230,7 +230,10 @@ again: if (o == nullptr) { return nullptr; } - j->container_stack.push_back({o, JSON_ITEM}); + // add_object already installed `o` in the parent (or the + // parser's root); moving the local copy into the frame + // avoids one shared_ptr atomic inc/dec pair per container. + j->container_stack.push_back({std::move(o), JSON_ITEM}); if (cb != nullptr) { cb(JSON_ARRAY, j, state); @@ -259,7 +262,9 @@ again: } } - json_object_ptr ret = f->container; + // Move the container out of the frame so pop_back doesn't + // drop the last reference; saves one atomic inc/dec. + json_object_ptr ret = std::move(f->container); j->container_stack.pop_back(); return ret; } @@ -271,7 +276,9 @@ again: if (o == nullptr) { return nullptr; } - j->container_stack.push_back({o, JSON_KEY}); + // See the [ case above: move into the frame to skip a + // shared_ptr atomic inc/dec round-trip. + j->container_stack.push_back({std::move(o), JSON_KEY}); if (cb != nullptr) { cb(JSON_HASH, j, state); @@ -300,7 +307,8 @@ again: } } - json_object_ptr ret = f->container; + // See the ] case: move out to skip an atomic refcount round-trip. + json_object_ptr ret = std::move(f->container); j->container_stack.pop_back(); return ret; } @@ -511,7 +519,11 @@ again: /////////////////////////// Strings case '"': { - std::string val; + // Reuse the parser-wide string buffer so we don't construct a + // fresh std::string (with its inevitable SSO->heap promotion + // and capacity doublings) for every JSON_STRING token. + std::string &val = j->string_buffer; + val.clear(); int surrogate = -1; while ((c = read_wrap(j)) != EOF) { @@ -632,7 +644,12 @@ again: json_object_ptr s = add_object(j, JSON_STRING); if (s != nullptr) { - s->string() = std::move(val); + // Copy (don't move) so j->string_buffer retains its + // grown capacity for the next token. The copy is a + // single right-sized allocation plus one memcpy, which + // is cheaper than the multiple capacity doublings the + // per-token std::string would otherwise incur. + s->string() = val; } return s; } diff --git a/jsonpull/jsonpull.h b/jsonpull/jsonpull.h index 4153b4b6..a6d5cc6c 100644 --- a/jsonpull/jsonpull.h +++ b/jsonpull/jsonpull.h @@ -128,15 +128,25 @@ struct json_string : json_object { struct json_array : json_object { std::vector array_value; - json_array() : json_object(JSON_ARRAY) {} - json_array(json_object *p, json_pull *pl) : json_object(JSON_ARRAY, p, pl) {} + // Coordinate-heavy GeoJSON dominates the parse workload, and every + // `[x, y]` (or `[x, y, z]`) pair would otherwise force the inner + // vector through 0 -> 1 -> 2 -> 4 growths plus the matching + // shared_ptr copies. Reserving 2 slots up front eliminates those + // reallocations for the common case and adds only a single small + // allocation for larger rings (which still grow geometrically). + json_array() : json_object(JSON_ARRAY) { array_value.reserve(2); } + json_array(json_object *p, json_pull *pl) : json_object(JSON_ARRAY, p, pl) { array_value.reserve(2); } }; struct json_hash : json_object { std::vector entries_value; - json_hash() : json_object(JSON_HASH) {} - json_hash(json_object *p, json_pull *pl) : json_object(JSON_HASH, p, pl) {} + // Most GeoJSON property hashes have a handful of keys (type, id, + // properties, geometry, plus a few attribute fields). Reserving 4 + // slots avoids the 0 -> 1 -> 2 -> 4 growth chain for the typical + // case while only modestly over-allocating for one-key hashes. + json_hash() : json_object(JSON_HASH) { entries_value.reserve(4); } + json_hash(json_object *p, json_pull *pl) : json_object(JSON_HASH, p, pl) { entries_value.reserve(4); } }; inline std::string &json_object::string() { @@ -232,7 +242,14 @@ struct json_pull { std::vector container_stack; json_object_ptr root; + // Scratch buffers reused across tokens so we don't reallocate per + // number/string. number_buffer accumulates raw digits before atof(); + // string_buffer accumulates decoded bytes before being copied into + // the final json_string. Both are cleared (capacity preserved) at + // the start of each token, so once they grow to the largest seen + // size they stop reallocating entirely. std::string number_buffer; + std::string string_buffer; }; json_pull_ptr json_begin_file(FILE *f);