mirror of
https://github.com/felt/tippecanoe.git
synced 2026-10-06 02:15:41 +02:00
Found the place that expected to be able to mask bits off the top
This commit is contained in:
@@ -822,9 +822,20 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
|||||||
|
|
||||||
for (size_t a = 0; a < indexst.st_size / sizeof(struct index); a++) {
|
for (size_t a = 0; a < indexst.st_size / sizeof(struct index); a++) {
|
||||||
struct index ix = indexmap[a];
|
struct index ix = indexmap[a];
|
||||||
index_t which = (ix.ix << prefix) >> (2 * GLOBAL_DETAIL - splitbits);
|
|
||||||
long long pos = sub_geompos[which];
|
|
||||||
|
|
||||||
|
// I think what is going on here is that `prefix` represents the top bits
|
||||||
|
// of the index, which we have already sorted on, so we are shifting up
|
||||||
|
// to mask off those bits (which previously fell off the top of the word)
|
||||||
|
// and then shifting back down to bring the top `splitbits` of what remains
|
||||||
|
// down to be the new partitions.
|
||||||
|
index_t ixmask = (((__int128_t) 1) << (2 * GLOBAL_DETAIL)) - 1;
|
||||||
|
index_t which = ((ix.ix << prefix) & ixmask) >> (2 * GLOBAL_DETAIL - splitbits);
|
||||||
|
if ((int) which >= splits) {
|
||||||
|
fprintf(stderr, "splits off the edge! segment %d of %d\n", (int) which, splits);
|
||||||
|
exit(EXIT_IMPOSSIBLE);
|
||||||
|
}
|
||||||
|
|
||||||
|
long long pos = sub_geompos[which];
|
||||||
fwrite_check(geommap + ix.start, ix.end - ix.start, 1, geomfiles[which], &sub_geompos[which], "geom");
|
fwrite_check(geommap + ix.start, ix.end - ix.start, 1, geomfiles[which], &sub_geompos[which], "geom");
|
||||||
|
|
||||||
// Count this as a 25%-accomplishment, since we will copy again
|
// Count this as a 25%-accomplishment, since we will copy again
|
||||||
@@ -897,18 +908,13 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
|||||||
std::atomic<long long> indexpos(indexst.st_size);
|
std::atomic<long long> indexpos(indexst.st_size);
|
||||||
int bytes = sizeof(struct index);
|
int bytes = sizeof(struct index);
|
||||||
|
|
||||||
int page = sysconf(_SC_PAGESIZE);
|
|
||||||
// Don't try to sort more than 2GB at once,
|
// Don't try to sort more than 2GB at once,
|
||||||
// which used to crash Macs and may still
|
// which used to crash Macs and may still
|
||||||
long long max_unit = 2LL * 1024 * 1024 * 1024;
|
long long max_unit = ((2LL * 1024 * 1024 * 1024) / bytes) * bytes;
|
||||||
long long unit = ((indexpos / CPUS + bytes - 1) / bytes) * bytes;
|
long long unit = ((indexpos / CPUS + bytes - 1) / bytes) * bytes;
|
||||||
if (unit > max_unit) {
|
if (unit > max_unit) {
|
||||||
unit = max_unit;
|
unit = max_unit;
|
||||||
}
|
}
|
||||||
unit = ((unit + page - 1) / page) * page;
|
|
||||||
if (unit < page) {
|
|
||||||
unit = page;
|
|
||||||
}
|
|
||||||
|
|
||||||
size_t nmerges = (indexpos + unit - 1) / unit;
|
size_t nmerges = (indexpos + unit - 1) / unit;
|
||||||
struct mergelist merges[nmerges];
|
struct mergelist merges[nmerges];
|
||||||
@@ -975,7 +981,7 @@ void radix1(int *geomfds_in, int *indexfds_in, int inputs, int prefix, int split
|
|||||||
perror("unmap geom");
|
perror("unmap geom");
|
||||||
exit(EXIT_MEMORY);
|
exit(EXIT_MEMORY);
|
||||||
}
|
}
|
||||||
} else if (indexst.st_size == sizeof(struct index) || prefix + splitbits >= 64) {
|
} else if (indexst.st_size == sizeof(struct index) || prefix + splitbits >= 2 * GLOBAL_DETAIL) {
|
||||||
struct index *indexmap = (struct index *) mmap(NULL, indexst.st_size, PROT_READ, MAP_PRIVATE, indexfds[i], 0);
|
struct index *indexmap = (struct index *) mmap(NULL, indexst.st_size, PROT_READ, MAP_PRIVATE, indexfds[i], 0);
|
||||||
if (indexmap == MAP_FAILED) {
|
if (indexmap == MAP_FAILED) {
|
||||||
fprintf(stderr, "fd %lld, len %lld\n", (long long) indexfds[i], (long long) indexst.st_size);
|
fprintf(stderr, "fd %lld, len %lld\n", (long long) indexfds[i], (long long) indexst.st_size);
|
||||||
|
|||||||
+1
-1
@@ -2,7 +2,7 @@
|
|||||||
#define PROJECTION_HPP
|
#define PROJECTION_HPP
|
||||||
|
|
||||||
#define GLOBAL_DETAIL 32
|
#define GLOBAL_DETAIL 32
|
||||||
typedef unsigned long long index_t;
|
typedef __uint128_t index_t;
|
||||||
|
|
||||||
void lonlat2tile(double lon, double lat, int zoom, long long *x, long long *y);
|
void lonlat2tile(double lon, double lat, int zoom, long long *x, long long *y);
|
||||||
void epsg3857totile(double ix, double iy, int zoom, long long *x, long long *y);
|
void epsg3857totile(double ix, double iy, int zoom, long long *x, long long *y);
|
||||||
|
|||||||
@@ -129,203 +129,3 @@ void fqsort(std::vector<FILE *> &inputs, size_t width, int (*cmp)(const void *,
|
|||||||
fqsort(v2, width, cmp, out, mem);
|
fqsort(v2, width, cmp, out, mem);
|
||||||
fclose(fp2);
|
fclose(fp2);
|
||||||
}
|
}
|
||||||
|
|
||||||
#if 0
|
|
||||||
|
|
||||||
struct indexed_feature {
|
|
||||||
std::string feature;
|
|
||||||
index_t index;
|
|
||||||
size_t seq;
|
|
||||||
|
|
||||||
bool operator<(indexed_feature const &f) const {
|
|
||||||
if (index < f.index) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
if (index == f.index) {
|
|
||||||
if (seq < f.seq) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
int deserialize_ulong_long(FILE *fp, unsigned long long *zigzag) {
|
|
||||||
*zigzag = 0;
|
|
||||||
int shift = 0;
|
|
||||||
|
|
||||||
while (true) {
|
|
||||||
char c;
|
|
||||||
if (fread(&c, sizeof(char), 1, fp) != 1) {
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
if ((c & 0x80) == 0) {
|
|
||||||
*zigzag |= ((unsigned long long) c) << shift;
|
|
||||||
shift += 7;
|
|
||||||
break;
|
|
||||||
} else {
|
|
||||||
*zigzag |= ((unsigned long long) (c & 0x7F)) << shift;
|
|
||||||
shift += 7;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
void feature_sort(std::vector<FILE *> &inputs, FILE *out, size_t mem) {
|
|
||||||
FILE *fp1, *fp2;
|
|
||||||
size_t seq = 0;
|
|
||||||
indexed_feature pivot;
|
|
||||||
|
|
||||||
if (mem > MAX_MEMORY) {
|
|
||||||
mem = MAX_MEMORY;
|
|
||||||
}
|
|
||||||
|
|
||||||
{
|
|
||||||
// read some elements into memory to choose a pivot from
|
|
||||||
//
|
|
||||||
// this is in its own scope so `buf` can go out of scope
|
|
||||||
// before trying to do any sub-sorts.
|
|
||||||
|
|
||||||
std::vector<indexed_feature> buf;
|
|
||||||
size_t bufsize = 0;
|
|
||||||
|
|
||||||
bool read_everything = false;
|
|
||||||
for (size_t i = 0; i < inputs.size(); i++) {
|
|
||||||
if (bufsize > mem) {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
while (true) {
|
|
||||||
unsigned long long len;
|
|
||||||
|
|
||||||
if (deserialize_ulong_long(inputs[i], &len) == 0) {
|
|
||||||
if (i + 1 == inputs.size()) {
|
|
||||||
read_everything = true;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
indexed_feature f;
|
|
||||||
f.feature.resize(len);
|
|
||||||
if (fread((char *) f.feature.c_str(), len, sizeof(char), inputs[i]) != 1) {
|
|
||||||
perror("fread");
|
|
||||||
exit(EXIT_READ);
|
|
||||||
}
|
|
||||||
|
|
||||||
const char *cp = f.feature.c_str();
|
|
||||||
deserialize_ulong_long(&cp, &f.index);
|
|
||||||
f.seq = seq++;
|
|
||||||
|
|
||||||
buf.push_back(std::move(f));
|
|
||||||
bufsize += len;
|
|
||||||
if (bufsize > mem) {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
std::stable_sort(buf.begin(), buf.end());
|
|
||||||
|
|
||||||
// If that was everything we have to sort, we are done.
|
|
||||||
|
|
||||||
if (read_everything) {
|
|
||||||
std::atomic<long long> fpos(0);
|
|
||||||
for (auto const &f : buf) {
|
|
||||||
serialize_ulong_long(out, f.feature.size(), &fpos, "sort output");
|
|
||||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, out, &fpos, "sort output");
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Otherwise, choose a pivot from it, make some temporary files,
|
|
||||||
// write what we have to those files, and then partition the rest
|
|
||||||
// of the input into them.
|
|
||||||
|
|
||||||
// This would be unstable if the pivot is one of several elements
|
|
||||||
// that compare equal. Does it matter?
|
|
||||||
|
|
||||||
size_t pivot_off = buf.size() / 2;
|
|
||||||
pivot = buf[pivot_off];
|
|
||||||
|
|
||||||
std::string t1 = "/tmp/sort1.XXXXXX";
|
|
||||||
std::string t2 = "/tmp/sort2.XXXXXX";
|
|
||||||
|
|
||||||
int fd1 = mkstemp((char *) t1.c_str());
|
|
||||||
unlink(t1.c_str());
|
|
||||||
int fd2 = mkstemp((char *) t2.c_str());
|
|
||||||
unlink(t2.c_str());
|
|
||||||
|
|
||||||
fp1 = fdopen(fd1, "w+b");
|
|
||||||
if (fp1 == NULL) {
|
|
||||||
perror(t1.c_str());
|
|
||||||
exit(EXIT_FAILURE);
|
|
||||||
}
|
|
||||||
fp2 = fdopen(fd2, "w+b");
|
|
||||||
if (fp2 == NULL) {
|
|
||||||
perror(t2.c_str());
|
|
||||||
exit(EXIT_FAILURE);
|
|
||||||
}
|
|
||||||
|
|
||||||
std::atomic<long long> fpos(0);
|
|
||||||
for (size_t i = 0; i < pivot_off; i++) {
|
|
||||||
serialize_ulong_long(fp1, buf[i].feature.size(), &fpos, "sort output");
|
|
||||||
fwrite_check(buf[i].feature.c_str(), buf[i].feature.size(), 1, fp1, &fpos, "sort output");
|
|
||||||
}
|
|
||||||
|
|
||||||
for (size_t i = pivot_off; i < buf.size(); i++) {
|
|
||||||
serialize_ulong_long(fp2, buf[i].feature.size(), &fpos, "sort output");
|
|
||||||
fwrite_check(buf[i].feature.c_str(), buf[i].feature.size(), 1, fp2, &fpos, "sort output");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// read the remaining input into the temporary files
|
|
||||||
|
|
||||||
std::atomic<long long> fpos(0);
|
|
||||||
for (size_t i = 0; i < inputs.size(); i++) {
|
|
||||||
while (true) {
|
|
||||||
unsigned long long len;
|
|
||||||
|
|
||||||
if (deserialize_ulong_long(inputs[i], &len) == 0) {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
indexed_feature f;
|
|
||||||
f.feature.resize(len);
|
|
||||||
if (fread((char *) f.feature.c_str(), len, sizeof(char), inputs[i]) != 1) {
|
|
||||||
perror("fread");
|
|
||||||
exit(EXIT_READ);
|
|
||||||
}
|
|
||||||
|
|
||||||
const char *cp = f.feature.c_str();
|
|
||||||
deserialize_ulong_long(&cp, &f.index);
|
|
||||||
f.seq = seq++;
|
|
||||||
|
|
||||||
if (f < pivot) {
|
|
||||||
serialize_ulong_long(fp1, f.feature.size(), &fpos, "sort output");
|
|
||||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, fp1, &fpos, "sort output");
|
|
||||||
} else {
|
|
||||||
serialize_ulong_long(fp2, f.feature.size(), &fpos, "sort output");
|
|
||||||
fwrite_check(f.feature.c_str(), f.feature.size(), 1, fp2, &fpos, "sort output");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Now sort the sub-ranges into the output.
|
|
||||||
|
|
||||||
rewind(fp1);
|
|
||||||
rewind(fp2);
|
|
||||||
|
|
||||||
std::vector<FILE *> v1;
|
|
||||||
v1.emplace_back(fp1);
|
|
||||||
feature_sort(v1, out, mem);
|
|
||||||
fclose(fp1);
|
|
||||||
|
|
||||||
std::vector<FILE *> v2;
|
|
||||||
v2.emplace_back(fp2);
|
|
||||||
feature_sort(v2, out, mem);
|
|
||||||
fclose(fp2);
|
|
||||||
}
|
|
||||||
|
|
||||||
#endif
|
|
||||||
|
|||||||
Reference in New Issue
Block a user