Preparing to sort by interestingness

This commit is contained in:
Erica Fischer
2023-02-08 15:49:36 -08:00
parent c6379d74d2
commit de607d1440
5 changed files with 70 additions and 16 deletions
+3
View File
@@ -2855,6 +2855,7 @@ int main(int argc, char **argv) {
{"order-descending-by", required_argument, 0, '~'},
{"order-smallest-first", no_argument, 0, '~'},
{"order-largest-first", no_argument, 0, '~'},
{"order-interesting-first", no_argument, 0, '~'},
{"Adding calculated attributes", 0, 0, 0},
{"calculate-feature-density", no_argument, &additional[A_CALCULATE_FEATURE_DENSITY], 1},
@@ -2999,6 +3000,8 @@ int main(int argc, char **argv) {
} else if (strcmp(opt, "order-largest-first") == 0) {
order_by.push_back(order_field(ORDER_BY_SIZE, true));
order_by_size = true;
} else if (strcmp(opt, "order-interesting-first") == 0) {
order_by.push_back(order_field(ORDER_BY_INTERESTINGNESS, true));
} else if (strcmp(opt, "simplification-at-maximum-zoom") == 0) {
maxzoom_simplification = atof_require(optarg, "Mazoom simplification");
if (maxzoom_simplification <= 0) {
+2 -1
View File
@@ -69,8 +69,9 @@ struct order_field {
extern std::vector<order_field> order_by;
// not legal UTF-8, so can't appear as a real attribute name
#define ORDER_BY_SIZE "\200size"
extern bool order_by_size;
#define ORDER_BY_SIZE "\200size"
#define ORDER_BY_INTERESTINGNESS "\200interestingness"
int mkstemp_cloexec(char *name);
FILE *fopen_oflag(const char *name, const char *mode, int oflag);
+30 -15
View File
@@ -289,7 +289,11 @@ void tilestats(std::map<std::string, layermap_entry> const &layermap1, size_t el
double yd = attribute.second.ysum / attribute.second.count;
double d = sqrt(xd * xd + yd * yd);
printf("%s %0.6f\n", attribute.first.c_str(), d);
printf("%s %0.6f", attribute.first.c_str(), d);
if (attribute.second.numeric_count != 0) {
printf(" %f %f", attribute.second.mean, sqrt(attribute.second.m2 / attribute.second.numeric_count));
}
printf("\n");
size_t val_count = attribute.second.sample_values.size();
if (val_count > max_tilestats_sample_values) {
@@ -899,22 +903,33 @@ void add_to_file_keys(std::map<std::string, type_and_string_stats> &file_keys, s
if (fka->second.sample_values.size() > max_tilestats_sample_values) {
fka->second.sample_values.pop_back();
}
}
unsigned hash = 0;
for (size_t i = 0; i < val.string.size(); i++) {
// https://en.wikipedia.org/wiki/Hash_function#Fibonacci_hashing
hash = hash * 2654435769 + val.string[i];
}
// extra multiply so even single-byte values are distributed
// around the circle rather than clumped together on one side.
hash = hash * 2654435769;
double angle = ((double) hash) / UINT_MAX * 2 * M_PI;
double x = cos(angle);
double y = sin(angle);
unsigned hash = 0;
for (size_t i = 0; i < val.string.size(); i++) {
// https://en.wikipedia.org/wiki/Hash_function#Fibonacci_hashing
hash = hash * 2654435769 + val.string[i];
}
// extra multiply so even single-byte values are distributed
// around the circle rather than clumped together on one side.
hash = hash * 2654435769;
double angle = ((double) hash) / UINT_MAX * 2 * M_PI;
double x = cos(angle);
double y = sin(angle);
fka->second.xsum += x;
fka->second.ysum += y;
fka->second.count++;
fka->second.xsum += x;
fka->second.ysum += y;
fka->second.count++;
// https://en.wikipedia.org/wiki/Algorithms_for_calculating_variance#Welford's_online_algorithm
if (val.type == mvt_double) {
double newValue = atof(val.string.c_str());
fka->second.numeric_count++;
double delta = newValue - fka->second.mean;
fka->second.mean += delta / fka->second.numeric_count;
double delta2 = newValue - fka->second.mean;
fka->second.m2 += delta * delta2;
}
fka->second.type |= (1 << val.type);
+5
View File
@@ -27,6 +27,11 @@ struct type_and_string_stats {
double xsum = 0;
double ysum = 0;
size_t count = 0;
// for mean and standard deviation
double mean = 0;
double numeric_count = 0;
double m2 = 0;
};
struct layermap_entry {
+30
View File
@@ -250,6 +250,10 @@ static int metacmp(const std::vector<long long> &keys1, const std::vector<long l
}
}
double get_interestingness(const struct coalesce *c1) {
return 0; // XXX
}
static mvt_value find_attribute_value(const struct coalesce *c1, std::string key) {
if (key == ORDER_BY_SIZE) {
mvt_value v;
@@ -257,6 +261,12 @@ static mvt_value find_attribute_value(const struct coalesce *c1, std::string key
v.numeric_value.double_value = c1->extent;
return v;
}
if (key == ORDER_BY_INTERESTINGNESS) {
mvt_value v;
v.type = mvt_double;
v.numeric_value.double_value = get_interestingness(c1);
return v;
}
const std::vector<long long> &keys1 = c1->keys;
const std::vector<long long> &values1 = c1->values;
@@ -519,6 +529,14 @@ struct partial_arg {
drawvec *shared_nodes;
};
double get_interestingness(const serial_feature *c1, const char *stringpool, long long pool_off[]) {
return 0; // XXX
}
double get_interestingness(const partial *c1, const char *stringpool, long long pool_off[]) {
return 0; // XXX
}
// THIS IS RIDICULOUS to have three almost-identical representations for features. FIX FIX FIX
static mvt_value find_attribute_value(const serial_feature *c1, std::string key, const char *stringpool, long long pool_off[]) {
@@ -528,6 +546,12 @@ static mvt_value find_attribute_value(const serial_feature *c1, std::string key,
v.numeric_value.double_value = c1->extent;
return v;
}
if (key == ORDER_BY_INTERESTINGNESS) {
mvt_value v;
v.type = mvt_double;
v.numeric_value.double_value = get_interestingness(c1, stringpool, pool_off);
return v;
}
const std::vector<long long> &keys1 = c1->keys;
const std::vector<long long> &values1 = c1->values;
@@ -559,6 +583,12 @@ static mvt_value find_attribute_value(const partial *c1, std::string key, const
v.numeric_value.double_value = c1->extent;
return v;
}
if (key == ORDER_BY_INTERESTINGNESS) {
mvt_value v;
v.type = mvt_double;
v.numeric_value.double_value = get_interestingness(c1, stringpool, pool_off);
return v;
}
const std::vector<long long> &keys1 = c1->keys;
const std::vector<long long> &values1 = c1->values;