1 /* PSPP - a program for statistical analysis.
2 Copyright (C) 2009 Free Software Foundation, Inc.
4 This program is free software: you can redistribute it and/or modify
5 it under the terms of the GNU General Public License as published by
6 the Free Software Foundation, either version 3 of the License, or
7 (at your option) any later version.
9 This program is distributed in the hope that it will be useful,
10 but WITHOUT ANY WARRANTY; without even the implied warranty of
11 MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
12 GNU General Public License for more details.
14 You should have received a copy of the GNU General Public License
15 along with this program. If not, see <http://www.gnu.org/licenses/>. */
21 #include "categoricals.h"
23 #include <gl/xalloc.h>
24 #include <data/variable.h>
25 #include <data/case.h>
26 #include <data/value.h>
27 #include <libpspp/hmap.h>
31 const struct variable **vars;
34 const struct variable *wv;
46 struct hmap_node node; /* Node in hash map. */
47 union value value; /* The value being labeled. */
48 double cc; /* The total of the weights of cases with this value */
49 int index; /* A zero based integer, unique within the variable.
50 Can be used as an index into an array */
54 static struct value_node *
55 lookup_value (const struct hmap *map, const struct variable *var, const union value *val)
57 struct value_node *foo;
58 unsigned int width = var_get_width (var);
59 size_t hash = value_hash (val, width, 0);
61 HMAP_FOR_EACH_WITH_HASH (foo, struct value_node, node, hash, map)
63 if (value_equal (val, &foo->value, width))
72 categoricals_create (const struct variable **v, size_t n_vars, const struct variable *wv)
75 struct categoricals *cat = xmalloc (sizeof *cat);
82 cat->map = xmalloc (sizeof *cat->map * n_vars);
83 cat->next_index = xcalloc (sizeof *cat->next_index, n_vars);
85 for (i = 0 ; i < cat->n_vars; ++i)
87 hmap_init (&cat->map[i]);
95 categoricals_update (struct categoricals *cat, const struct ccase *c)
99 const double weight = cat->wv ? case_data (c, cat->wv)->f : 1.0;
101 for (i = 0 ; i < cat->n_vars; ++i)
103 unsigned int width = var_get_width (cat->vars[i]);
104 const union value *val = case_data (c, cat->vars[i]);
105 size_t hash = value_hash (val, width, 0);
107 struct value_node *node = lookup_value (&cat->map[i], cat->vars[i], val);
111 node = xmalloc (sizeof *node);
113 value_init (&node->value, width);
114 value_copy (&node->value, val, width);
117 hmap_insert (&cat->map[i], &node->node, hash);
119 node->index = cat->next_index[i]++ ;
126 /* Return the number of categories (distinct values) for variable N */
128 categoricals_n_count (const struct categoricals *cat, size_t n)
130 return hmap_count (&cat->map[n]);
134 /* Return the index for value VAL in the Nth variable */
136 categoricals_index (const struct categoricals *cat, size_t n, const union value *val)
138 struct value_node *vn = lookup_value (&cat->map[n], cat->vars[n], val);
147 /* Return the total number of categories */
149 categoricals_total (const struct categoricals *cat)