From: Ben Pfaff Date: Tue, 31 Dec 2019 05:11:16 +0000 (+0000) Subject: new research X-Git-Url: https://pintos-os.org/cgi-bin/gitweb.cgi?p=pspp;a=commitdiff_plain;h=8b7ea7f744c5e82874cd28cb36c44a2d99e5894b new research --- diff --git a/Makefile b/Makefile index cf5aa787de..c9a1c7d551 100644 --- a/Makefile +++ b/Makefile @@ -4,8 +4,10 @@ base_ldflags := $(LDFLAGS) parse-xml.o: CFLAGS := $(shell pkg-config --cflags libxml-2.0) $(base_cflags) parse-xml: LDFLAGS := $(shell pkg-config --libs libxml-2.0) $(LDFLAGS) dump2.o: CFLAGS := $(base_cflags) -Wno-unused +dump-spo.o: CFLAGS := $(base_cflags) -Wno-unused -all: dump dump2 parse-xml +all: dump dump2 parse-xml dump-spo +dump-spo: dump-spo.o u8-mbtouc.o dump: dump.o u8-mbtouc.o dump2: dump2.o u8-mbtouc.o parse-xml: parse-xml.o diff --git a/dump-float.c b/dump-float.c new file mode 100644 index 0000000000..9e2d4bef31 --- /dev/null +++ b/dump-float.c @@ -0,0 +1,15 @@ +#include +#include + +int +main (void) +{ + union + { + uint8_t b[8]; + double d; + } + x = { .b = { 0x00, 0x00, 0x00, 0x00, 0x00, 0x4c, 0xdd, 0x40 } }; + printf ("%f\n", x.d); + return 0; +} diff --git a/dump-spo.c b/dump-spo.c new file mode 100644 index 0000000000..b48110736e --- /dev/null +++ b/dump-spo.c @@ -0,0 +1,1548 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "u8-mbtouc.h" + +static const char *filename; +static uint8_t *data; +static size_t n; + +int version; + +unsigned int pos; + +#define XSTR(x) #x +#define STR(x) XSTR(x) +#define WHERE __FILE__":" STR(__LINE__) + +static uint8_t +get_byte(void) +{ + return data[pos++]; +} + +static unsigned int +get_u32(void) +{ + uint32_t x; + memcpy(&x, &data[pos], 4); + pos += 4; + return x; +} + +static unsigned long long int +get_u64(void) +{ + uint64_t x; + memcpy(&x, &data[pos], 8); + pos += 8; + return x; +} + +static unsigned int +get_be32(void) +{ + uint32_t x; + x = (data[pos] << 24) | (data[pos + 1] << 16) | (data[pos + 2] << 8) | data[pos + 3]; + pos += 4; + return x; +} + +static unsigned int +get_u16(void) +{ + uint16_t x; + memcpy(&x, &data[pos], 2); + pos += 2; + return x; +} + +static double +get_double(void) +{ + double x; + memcpy(&x, &data[pos], 8); + pos += 8; + return x; +} + +static double __attribute__((unused)) +get_float(void) +{ + float x; + memcpy(&x, &data[pos], 4); + pos += 4; + return x; +} + +static bool +match_u32(uint32_t x) +{ + if (get_u32() == x) + return true; + pos -= 4; + return false; +} + +static void +match_u32_assert(uint32_t x, const char *where) +{ + unsigned int y = get_u32(); + if (x != y) + { + fprintf(stderr, "%s: 0x%x: expected i%u, got i%u\n", where, pos - 4, x, y); + exit(1); + } +} +#define match_u32_assert(x) match_u32_assert(x, WHERE) + +static bool __attribute__((unused)) +match_u64(uint64_t x) +{ + if (get_u64() == x) + return true; + pos -= 8; + return false; +} + +static void __attribute__((unused)) +match_u64_assert(uint64_t x, const char *where) +{ + unsigned long long int y = get_u64(); + if (x != y) + { + fprintf(stderr, "%s: 0x%x: expected u64:%lu, got u64:%llu\n", where, pos - 8, x, y); + exit(1); + } +} +#define match_u64_assert(x) match_u64_assert(x, WHERE) + +static bool __attribute__((unused)) +match_be32(uint32_t x) +{ + if (get_be32() == x) + return true; + pos -= 4; + return false; +} + +static void +match_be32_assert(uint32_t x, const char *where) +{ + unsigned int y = get_be32(); + if (x != y) + { + fprintf(stderr, "%s: 0x%x: expected be%u, got be%u\n", where, pos - 4, x, y); + exit(1); + } +} +#define match_be32_assert(x) match_be32_assert(x, WHERE) + +static bool +match_byte(uint8_t b) +{ + if (pos < n && data[pos] == b) + { + pos++; + return true; + } + else + return false; +} + +static void +match_byte_assert(uint8_t b, const char *where) +{ + if (!match_byte(b)) + { + fprintf(stderr, "%s: 0x%x: expected %02x, got %02x\n", where, pos, b, data[pos]); + exit(1); + } +} +#define match_byte_assert(b) match_byte_assert(b, WHERE) + +static bool +match_bytes(int start, const int *bytes, size_t n_bytes) +{ + for (size_t i = 0; i < n_bytes; i++) + if (bytes[i] >= 0 && data[start + i] != bytes[i]) + return false; + return true; +} + +static bool +get_bool(void) +{ + if (match_byte(0)) + return false; + match_byte_assert(1); + return true; +} + +static bool __attribute__((unused)) +is_ascii(uint8_t p) +{ + return (p >= ' ' && p < 127) || p == '\r' || p == '\n' || p == '\t'; +} + +static bool __attribute__((unused)) +all_utf8(const char *p_, size_t len) +{ + const uint8_t *p = (const uint8_t *) p_; + for (size_t ofs = 0, mblen; ofs < len; ofs += mblen) + { + ucs4_t uc; + + mblen = u8_mbtouc (&uc, p + ofs, len - ofs); + if ((uc < 32 && uc != '\n') || uc == 127 || uc == 0xfffd) + return false; + } + return true; +} + +static char * +get_string(const char *where) +{ + if (1 + /*data[pos + 1] == 0 && data[pos + 2] == 0 && data[pos + 3] == 0*/ + /*&& all_ascii(&data[pos + 4], data[pos])*/) + { + int len = data[pos] + data[pos + 1] * 256; + char *s = malloc(len + 1); + + memcpy(s, &data[pos + 4], len); + s[len] = 0; + pos += 4 + len; + return s; + } + else + { + fprintf(stderr, "%s: 0x%x: expected string\n", where, pos); + exit(1); + } +} +#define get_string() get_string(WHERE) + +static char * +get_string_be(const char *where) +{ + if (1 + /*data[pos + 1] == 0 && data[pos + 2] == 0 && data[pos + 3] == 0*/ + /*&& all_ascii(&data[pos + 4], data[pos])*/) + { + int len = data[pos + 2] * 256 + data[pos + 3]; + char *s = malloc(len + 1); + + memcpy(s, &data[pos + 4], len); + s[len] = 0; + pos += 4 + len; + return s; + } + else + { + fprintf(stderr, "%s: 0x%x: expected string\n", where, pos); + exit(1); + } +} +#define get_string_be() get_string_be(WHERE) + +static int +get_end(void) +{ + int len = get_u32(); + return pos + len; +} + +static void __attribute__((unused)) +hex_dump(FILE *stream, int ofs, int n) +{ + for (int i = 0; i < n; i++) + { + int c = data[ofs + i]; + fprintf(stream, " %02x", c); + } + putc(' ', stream); + for (int i = 0; i < n; i++) + { + int c = data[ofs + i]; + putc(c >= 32 && c < 127 ? c : '.', stream); + } + putc('\n', stream); +} + +static void __attribute__((unused)) +char_dump(FILE *stream, int ofs, int n) +{ + for (int i = 0; i < n; i++) + { + int c = data[ofs + i]; + putc(c >= 32 && c < 127 ? c : '.', stream); + } + putc('\n', stream); +} + +static char * +dump_counted_string(void) +{ + int inner_end = get_end(); + if (pos == inner_end) + return NULL; + + if (match_u32(5)) + { + match_u32_assert(0); + match_byte_assert(0x58); + } + else + match_u32_assert(0); + + char *s = NULL; + if (match_byte(0x31)) + s = get_string(); + else + match_byte_assert(0x58); + if (pos != inner_end) + { + fprintf(stderr, "inner end discrepancy\n"); + exit(1); + } + return s; +} + +static void +dump_style(FILE *stream) +{ + if (match_byte(0x58)) + return; + + match_byte_assert(0x31); + if (get_bool()) + printf (" bold=\"yes\""); + if (get_bool()) + printf (" italic=\"yes\""); + if (get_bool()) + printf (" underline=\"yes\""); + if (!get_bool()) + printf (" show=\"no\""); + char *fg = get_string(); /* foreground */ + char *bg = get_string(); /* background */ + char *font = get_string(); /* font */ + int size = get_byte() * (72. / 96.); + fprintf(stream, " fgcolor=\"%s\" bgcolor=\"%s\" font=\"%s\" size=\"%dpt\"", + fg, bg, font, size); +} + +static void +dump_style2(FILE *stream) +{ + if (match_byte(0x58)) + return; + + match_byte_assert(0x31); + uint32_t halign = get_u32(); + printf (" halign=\"%s\"", + halign == 0 ? "center" + : halign == 2 ? "left" + : halign == 4 ? "right" + : halign == 6 ? "decimal" + : halign == 0xffffffad ? "mixed" + : ""); + int valign = get_u32(); + printf (" valign=\"%s\"", + valign == 0 ? "center" + : valign == 1 ? "top" + : valign == 3 ? "bottom" + : ""); + printf (" offset=\"%gpt\"", get_double()); + int l = get_u16(); + int r = get_u16(); + int t = get_u16(); + int b = get_u16(); + printf (" margins=\"%d %d %d %d\"", l, r, t, b); +} + +static char * +dump_nested_string(FILE *stream) +{ + char *s = NULL; + + match_byte_assert (0); + match_byte_assert (0); + int outer_end = get_end(); + s = dump_counted_string(); + if (s) + fprintf(stream, " \"%s\"", s); + dump_style(stream); + match_byte_assert(0x58); + if (pos != outer_end) + { + fprintf(stderr, "outer end discrepancy\n"); + exit(1); + } + + return s; +} + +static void +dump_value_modifier(FILE *stream) +{ + if (match_byte (0x31)) + { + if (match_u32 (0)) + { + fprintf(stream, "\n"); + return; + } + + int outer_end = get_end(); + + /* This counted-string appears to be a template string, + e.g. "Design\: [:^1:]1 Within Subjects Design\: [:^1:]2". */ + char *template = dump_counted_string(); + if (template) + fprintf(stream, " template=\"%s\"", template); + + dump_style(stream); + dump_style2(stream); + if (pos != outer_end) + { + fprintf(stderr, "outer end discrepancy\n"); + exit(1); + } + fprintf(stream, "/>\n"); + } + else + { + int count = get_u32(); + fprintf(stream, "\n"); + } + } + else + match_byte_assert (0x58); +} + +static const char * +format_to_string (int type) +{ + static char tmp[16]; + switch (type) + { + case 1: return "A"; + case 2: return "AHEX"; + case 3: return "COMMA"; + case 4: return "DOLLAR"; + case 5: case 40: return "F"; + case 6: return "IB"; + case 7: return "PIBHEX"; + case 8: return "P"; + case 9: return "PIB"; + case 10: return "PK"; + case 11: return "RB"; + case 12: return "RBHEX"; + case 15: return "Z"; + case 16: return "N"; + case 17: return "E"; + case 20: return "DATE"; + case 21: return "TIME"; + case 22: return "DATETIME"; + case 23: return "ADATE"; + case 24: return "JDATE"; + case 25: return "DTIME"; + case 26: return "WKDAY"; + case 27: return "MONTH"; + case 28: return "MOYR"; + case 29: return "QYR"; + case 30: return "WKYR"; + case 31: return "PCT"; + case 32: return "DOT"; + case 33: return "CCA"; + case 34: return "CCB"; + case 35: return "CCC"; + case 36: return "CCD"; + case 37: return "CCE"; + case 38: return "EDATE"; + case 39: return "SDATE"; + default: + abort(); + sprintf(tmp, "<%d>", type); + return tmp; + } +} + +static void +dump_value(FILE *stream, int level) +{ + match_byte(0); + match_byte(0); + match_byte(0); + match_byte(0); + + for (int i = 0; i <= level; i++) + fprintf (stream, " "); + + printf ("%02x: value (%d)\n", pos, data[pos]); + if (match_byte (1)) + { + unsigned int format; + double value; + + dump_value_modifier(stream); + format = get_u32 (); + value = get_double (); + fprintf (stream, "\n", + DBL_DIG, value, format_to_string(format >> 16), (format >> 8) & 0xff, format & 0xff); + } + else if (match_byte (2)) + { + unsigned int format; + char *var, *vallab; + double value; + + dump_value_modifier (stream); + format = get_u32 (); + value = get_double (); + var = get_string (); + vallab = get_string (); + fprintf (stream, "> 16), (format >> 8) & 0xff, format & 0xff); + if (var[0]) + fprintf (stream, " variable=\"%s\"", var); + if (vallab[0]) + fprintf (stream, " label=\"%s\"", vallab); + fprintf (stream, "/>\n"); + if (!match_byte (1) && !match_byte(2)) + match_byte_assert (3); + } + else if (match_byte (3)) + { + char *text = get_string(); + dump_value_modifier(stream); + char *identifier = get_string(); + char *text_eng = get_string(); + fprintf (stream, "\n"); + if (!match_byte (0)) + match_byte_assert(1); + } + else if (match_byte (4)) + { + unsigned int format; + char *var, *vallab, *value; + + dump_value_modifier(stream); + format = get_u32 (); + vallab = get_string (); + var = get_string (); + if (!match_byte(1) && !match_byte(2)) + match_byte_assert (3); + value = get_string (); + fprintf (stream, "> 16), (format >> 8) & 0xff, format & 0xff); + if (var[0]) + fprintf (stream, " variable=\"%s\"", var); + if (vallab[0]) + fprintf (stream, " label=\"%s\"/>\n", vallab); + fprintf (stream, "/>\n"); + } + else if (match_byte (5)) + { + dump_value_modifier(stream); + char *name = get_string (); + char *label = get_string (); + fprintf (stream, "\n"); + if (!match_byte(1) && !match_byte(2)) + match_byte_assert(3); + } + else + { + printf ("else %#x\n", pos); + dump_value_modifier(stream); + + char *base = get_string(); + int x = get_u32(); + fprintf (stream, "\n"); + } +} + +static int +compare_int(const void *a_, const void *b_) +{ + const int *a = a_; + const int *b = b_; + return *a < *b ? -1 : *a > *b; +} + +static void +check_permutation(int *a, int n, const char *name) +{ + int b[n]; + memcpy(b, a, n * sizeof *a); + qsort(b, n, sizeof *b, compare_int); + for (int i = 0; i < n; i++) + if (b[i] != i) + { + fprintf(stderr, "bad %s permutation:", name); + for (int i = 0; i < n; i++) + fprintf(stderr, " %d", a[i]); + putc('\n', stderr); + exit(1); + } +} + +static void +dump_category(FILE *stream, int level, int **indexes, int *allocated_indexes, + int *n_indexes) +{ + for (int i = 0; i <= level; i++) + fprintf (stream, " "); + printf ("\n"); + dump_value (stream, level + 1); + + bool merge = get_bool(); + match_byte_assert (0); + int unindexed = get_bool(); + + int x = get_u32 (); + pos -= 4; + if (!match_u32 (0)) + match_u32_assert (2); + + int indx = get_u32(); + int n_categories = get_u32(); + if (indx == -1) + { + if (merge) + { + for (int i = 0; i <= level + 1; i++) + fprintf (stream, " "); + fprintf (stream, "\n"); + } + assert (unindexed); + } + else + { + assert (!merge); + assert (!unindexed); + assert (x == 2); + assert (n_categories == 0); + if (*n_indexes >= *allocated_indexes) + { + *allocated_indexes = *allocated_indexes ? 2 * *allocated_indexes : 16; + *indexes = realloc(*indexes, *allocated_indexes * sizeof **indexes); + } + (*indexes)[(*n_indexes)++] = indx; + } + + if (n_categories == 0) + { + for (int i = 0; i <= level + 1; i++) + fprintf (stream, " "); + fprintf (stream, "%d\n", indx); + } + for (int i = 0; i < n_categories; i++) + dump_category (stream, level + 1, indexes, allocated_indexes, n_indexes); + for (int i = 0; i <= level; i++) + fprintf (stream, " "); + printf ("\n"); +} + +static int +dump_dim(int indx) +{ + int n_categories; + + printf ("\n", indx); + dump_value (stdout, 0); + + /* This byte is usually 0 but many other values have been spotted. + No visible effect. */ + pos++; + + /* This byte can cause data to be oddly replicated. */ + if (!match_byte(0) && !match_byte(1)) + match_byte_assert(2); + + if (!match_u32(0)) + match_u32_assert(2); + + bool show_dim_label = get_bool(); + if (show_dim_label) + printf(" \n"); + + bool hide_all_labels = get_bool(); + if (hide_all_labels) + printf(" \n"); + + match_byte_assert(1); + if (!match_u32(UINT32_MAX)) + match_u32_assert(indx); + + n_categories = get_u32(); + + int *indexes = NULL; + int n_indexes = 0; + int allocated_indexes = 0; + for (int i = 0; i < n_categories; i++) + dump_category (stdout, 0, &indexes, &allocated_indexes, &n_indexes); + check_permutation(indexes, n_indexes, "categories"); + + fprintf (stdout, "\n"); + return n_indexes; +} + +int n_dims; +static int dim_n_cats[64]; +#define MAX_DIMS (sizeof dim_n_cats / sizeof *dim_n_cats) + +static void +dump_dims(void) +{ + n_dims = get_u32(); + assert(n_dims < MAX_DIMS); + for (int i = 0; i < n_dims; i++) + dim_n_cats[i] = dump_dim (i); +} + +static void +dump_data(void) +{ + /* The first three numbers add to the number of dimensions. */ + int l = get_u32(); + int r = get_u32(); + int c = n_dims - l - r; + match_u32_assert(c); + + /* The next n_dims numbers are a permutation of the dimension numbers. */ + int a[n_dims]; + for (int i = 0; i < n_dims; i++) + { + int dim = get_u32(); + a[i] = dim; + + const char *name = i < l ? "layer" : i < l + r ? "row" : "column"; + printf ("<%s dimension=\"%d\"/>\n", name, dim); + } + check_permutation(a, n_dims, "dimensions"); + + int x = get_u32(); + printf ("\n"); + for (int i = 0; i < x; i++) + { + unsigned int indx = get_u32(); + printf (" \n"); + match_u32_assert(0); + if (version == 1) + match_byte(0); + dump_value(stdout, 1); + fprintf (stdout, " \n"); + } + printf ("\n"); +} + +static void +dump_title(void) +{ + printf ("\n"); + dump_value(stdout, 0); + match_byte(1); + printf ("\n"); + + printf ("\n"); + dump_value(stdout, 0); + match_byte(1); + printf ("\n"); + + match_byte_assert(0x31); + + printf ("\n"); + dump_value(stdout, 0); + match_byte(1); + printf ("\n"); + + if (match_byte(0x31)) + { + printf ("\n"); + dump_value(stdout, 0); + printf ("\n"); + } + else + match_byte_assert(0x58); + if (match_byte(0x31)) + { + printf ("\n"); + dump_value(stdout, 0); + printf ("\n"); + } + else + match_byte_assert(0x58); + + int n_footnotes = get_u32(); + for (int i = 0; i < n_footnotes; i++) + { + printf ("\n", i); + dump_value(stdout, 0); + /* Custom footnote marker string. */ + if (match_byte (0x31)) + dump_value(stdout, 0); + else + match_byte_assert (0x58); + int n = get_u32(); + if (n >= 0) + { + /* Appears to be the number of references to a footnote. */ + printf (" \n", n); + } + else if (n == -2) + { + /* The user deleted the footnote references. */ + printf (" \n"); + } + else + assert(0); + printf ("\n"); + } +} + +static void +dump_fonts(void) +{ + match_byte(0); + for (int i = 1; i <= 8; i++) + { + printf ("