mirror of
https://github.com/felt/tippecanoe.git
synced 2026-10-02 16:35:40 +02:00
Found the place that expected to be able to mask bits off the top
This commit is contained in:
@@ -822,9 +822,20 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
||||
|
||||
for (size_t a = 0; a < indexst.st_size / sizeof(struct index); a++) {
|
||||
struct index ix = indexmap[a];
|
||||
index_t which = (ix.ix << prefix) >> (2 * GLOBAL_DETAIL - splitbits);
|
||||
long long pos = sub_geompos[which];
|
||||
|
||||
// I think what is going on here is that `prefix` represents the top bits
|
||||
// of the index, which we have already sorted on, so we are shifting up
|
||||
// to mask off those bits (which previously fell off the top of the word)
|
||||
// and then shifting back down to bring the top `splitbits` of what remains
|
||||
// down to be the new partitions.
|
||||
index_t ixmask = (((__int128_t) 1) << (2 * GLOBAL_DETAIL)) - 1;
|
||||
index_t which = ((ix.ix << prefix) & ixmask) >> (2 * GLOBAL_DETAIL - splitbits);
|
||||
if ((int) which >= splits) {
|
||||
fprintf(stderr, "splits off the edge! segment %d of %d\n", (int) which, splits);
|
||||
exit(EXIT_IMPOSSIBLE);
|
||||
}
|
||||
|
||||
long long pos = sub_geompos[which];
|
||||
fwrite_check(geommap + ix.start, ix.end - ix.start, 1, geomfiles[which], &sub_geompos[which], "geom");
|
||||
|
||||
// Count this as a 25%-accomplishment, since we will copy again
|
||||
@@ -897,18 +908,13 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
||||
std::atomic<long long> indexpos(indexst.st_size);
|
||||
int bytes = sizeof(struct index);
|
||||
|
||||
int page = sysconf(_SC_PAGESIZE);
|
||||
// Don't try to sort more than 2GB at once,
|
||||
// which used to crash Macs and may still
|
||||
long long max_unit = 2LL * 1024 * 1024 * 1024;
|
||||
long long max_unit = ((2LL * 1024 * 1024 * 1024) / bytes) * bytes;
|
||||
long long unit = ((indexpos / CPUS + bytes - 1) / bytes) * bytes;
|
||||
if (unit > max_unit) {
|
||||
unit = max_unit;
|
||||
}
|
||||
unit = ((unit + page - 1) / page) * page;
|
||||
if (unit < page) {
|
||||
unit = page;
|
||||
}
|
||||
|
||||
size_t nmerges = (indexpos + unit - 1) / unit;
|
||||
struct mergelist merges[nmerges];
|
||||
@@ -975,7 +981,7 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
||||
perror("unmap geom");
|
||||
exit(EXIT_MEMORY);
|
||||
}
|
||||
} else if (indexst.st_size == sizeof(struct index) || prefix + splitbits >= 64) {
|
||||
} else if (indexst.st_size == sizeof(struct index) || prefix + splitbits >= 2 * GLOBAL_DETAIL) {
|
||||
struct index *indexmap = (struct index *) mmap(NULL, indexst.st_size, PROT_READ, MAP_PRIVATE, indexfds[i], 0);
|
||||
if (indexmap == MAP_FAILED) {
|
||||
fprintf(stderr, "fd %lld, len %lld\n", (long long) indexfds[i], (long long) indexst.st_size);
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@
|
||||
#define PROJECTION_HPP
|
||||
|
||||
#define GLOBAL_DETAIL 32
|
||||
typedef unsigned long long index_t;
|
||||
typedef __uint128_t index_t;
|
||||
|
||||
void lonlat2tile(double lon, double lat, int zoom, long long *x, long long *y);
|
||||
void epsg3857totile(double ix, double iy, int zoom, long long *x, long long *y);
|
||||
|
||||
@@ -129,203 +129,3 @@ void fqsort(std::vector<FILE *> &inputs, size_t width, int (*cmp)(const void *,
|
||||
fqsort(v2, width, cmp, out, mem);
|
||||
fclose(fp2);
|
||||
}
|
||||
|
||||
#if 0
|
||||
|
||||
struct indexed_feature {
|
||||
std::string feature;
|
||||
index_t index;
|
||||
size_t seq;
|
||||
|
||||
bool operator<(indexed_feature const &f) const {
|
||||
if (index < f.index) {
|
||||
return true;
|
||||
}
|
||||
if (index == f.index) {
|
||||
if (seq < f.seq) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
int deserialize_ulong_long(FILE *fp, unsigned long long *zigzag) {
|
||||
*zigzag = 0;
|
||||
int shift = 0;
|
||||
|
||||
while (true) {
|
||||
char c;
|
||||
if (fread(&c, sizeof(char), 1, fp) != 1) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if ((c & 0x80) == 0) {
|
||||
*zigzag |= ((unsigned long long) c) << shift;
|
||||
shift += 7;
|
||||
break;
|
||||
} else {
|
||||
*zigzag |= ((unsigned long long) (c & 0x7F)) << shift;
|
||||
shift += 7;
|
||||
}
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
void feature_sort(std::vector<FILE *> &inputs, FILE *out, size_t mem) {
|
||||
FILE *fp1, *fp2;
|
||||
size_t seq = 0;
|
||||
indexed_feature pivot;
|
||||
|
||||
if (mem > MAX_MEMORY) {
|
||||
mem = MAX_MEMORY;
|
||||
}
|
||||
|
||||
{
|
||||
// read some elements into memory to choose a pivot from
|
||||
//
|
||||
// this is in its own scope so `buf` can go out of scope
|
||||
// before trying to do any sub-sorts.
|
||||
|
||||
std::vector<indexed_feature> buf;
|
||||
size_t bufsize = 0;
|
||||
|
||||
bool read_everything = false;
|
||||
for (size_t i = 0; i < inputs.size(); i++) {
|
||||
if (bufsize > mem) {
|
||||
break;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
unsigned long long len;
|
||||
|
||||
if (deserialize_ulong_long(inputs[i], &len) == 0) {
|
||||
if (i + 1 == inputs.size()) {
|
||||
read_everything = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
indexed_feature f;
|
||||
f.feature.resize(len);
|
||||
if (fread((char *) f.feature.c_str(), len, sizeof(char), inputs[i]) != 1) {
|
||||
perror("fread");
|
||||
exit(EXIT_READ);
|
||||
}
|
||||
|
||||
const char *cp = f.feature.c_str();
|
||||
deserialize_ulong_long(&cp, &f.index);
|
||||
f.seq = seq++;
|
||||
|
||||
buf.push_back(std::move(f));
|
||||
bufsize += len;
|
||||
if (bufsize > mem) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::stable_sort(buf.begin(), buf.end());
|
||||
|
||||
// If that was everything we have to sort, we are done.
|
||||
|
||||
if (read_everything) {
|
||||
std::atomic<long long> fpos(0);
|
||||
for (auto const &f : buf) {
|
||||
serialize_ulong_long(out, f.feature.size(), &fpos, "sort output");
|
||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, out, &fpos, "sort output");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Otherwise, choose a pivot from it, make some temporary files,
|
||||
// write what we have to those files, and then partition the rest
|
||||
// of the input into them.
|
||||
|
||||
// This would be unstable if the pivot is one of several elements
|
||||
// that compare equal. Does it matter?
|
||||
|
||||
size_t pivot_off = buf.size() / 2;
|
||||
pivot = buf[pivot_off];
|
||||
|
||||
std::string t1 = "/tmp/sort1.XXXXXX";
|
||||
std::string t2 = "/tmp/sort2.XXXXXX";
|
||||
|
||||
int fd1 = mkstemp((char *) t1.c_str());
|
||||
unlink(t1.c_str());
|
||||
int fd2 = mkstemp((char *) t2.c_str());
|
||||
unlink(t2.c_str());
|
||||
|
||||
fp1 = fdopen(fd1, "w+b");
|
||||
if (fp1 == NULL) {
|
||||
perror(t1.c_str());
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
fp2 = fdopen(fd2, "w+b");
|
||||
if (fp2 == NULL) {
|
||||
perror(t2.c_str());
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
std::atomic<long long> fpos(0);
|
||||
for (size_t i = 0; i < pivot_off; i++) {
|
||||
serialize_ulong_long(fp1, buf[i].feature.size(), &fpos, "sort output");
|
||||
fwrite_check(buf[i].feature.c_str(), buf[i].feature.size(), 1, fp1, &fpos, "sort output");
|
||||
}
|
||||
|
||||
for (size_t i = pivot_off; i < buf.size(); i++) {
|
||||
serialize_ulong_long(fp2, buf[i].feature.size(), &fpos, "sort output");
|
||||
fwrite_check(buf[i].feature.c_str(), buf[i].feature.size(), 1, fp2, &fpos, "sort output");
|
||||
}
|
||||
}
|
||||
|
||||
// read the remaining input into the temporary files
|
||||
|
||||
std::atomic<long long> fpos(0);
|
||||
for (size_t i = 0; i < inputs.size(); i++) {
|
||||
while (true) {
|
||||
unsigned long long len;
|
||||
|
||||
if (deserialize_ulong_long(inputs[i], &len) == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
indexed_feature f;
|
||||
f.feature.resize(len);
|
||||
if (fread((char *) f.feature.c_str(), len, sizeof(char), inputs[i]) != 1) {
|
||||
perror("fread");
|
||||
exit(EXIT_READ);
|
||||
}
|
||||
|
||||
const char *cp = f.feature.c_str();
|
||||
deserialize_ulong_long(&cp, &f.index);
|
||||
f.seq = seq++;
|
||||
|
||||
if (f < pivot) {
|
||||
serialize_ulong_long(fp1, f.feature.size(), &fpos, "sort output");
|
||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, fp1, &fpos, "sort output");
|
||||
} else {
|
||||
serialize_ulong_long(fp2, f.feature.size(), &fpos, "sort output");
|
||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, fp2, &fpos, "sort output");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Now sort the sub-ranges into the output.
|
||||
|
||||
rewind(fp1);
|
||||
rewind(fp2);
|
||||
|
||||
std::vector<FILE *> v1;
|
||||
v1.emplace_back(fp1);
|
||||
feature_sort(v1, out, mem);
|
||||
fclose(fp1);
|
||||
|
||||
std::vector<FILE *> v2;
|
||||
v2.emplace_back(fp2);
|
||||
feature_sort(v2, out, mem);
|
||||
fclose(fp2);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user