X-Git-Url: https://pintos-os.org/cgi-bin/gitweb.cgi?a=blobdiff_plain;f=src%2Flanguage%2Fstats%2Frank.q;h=91520f5e011e2bbe2e4c819bb2ec830a7c8ec4d4;hb=7a887989fb1da56ebd264ee3338c1472c7e40a51;hp=df2150d3b1cc2d120140675834d0fe7d6916ab82;hpb=dc78471910e82d59232ce9b137b7c4fc4992d174;p=pspp-builds.git diff --git a/src/language/stats/rank.q b/src/language/stats/rank.q index df2150d3..91520f5e 100644 --- a/src/language/stats/rank.q +++ b/src/language/stats/rank.q @@ -1,20 +1,18 @@ -/* PSPP - RANK. -*-c-*- -Copyright (C) 2005, 2006, 2007 Free Software Foundation, Inc. +/* PSPP - a program for statistical analysis. + Copyright (C) 2005, 2006, 2007 Free Software Foundation, Inc. -This program is free software; you can redistribute it and/or -modify it under the terms of the GNU General Public License as -published by the Free Software Foundation; either version 2 of the -License, or (at your option) any later version. + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. -This program is distributed in the hope that it will be useful, but -WITHOUT ANY WARRANTY; without even the implied warranty of -MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -General Public License for more details. + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. -You should have received a copy of the GNU General Public License -along with this program; if not, write to the Free Software -Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA -02110-1301, USA. */ + You should have received a copy of the GNU General Public License + along with this program. If not, see . */ #include @@ -166,10 +164,10 @@ static int k_ntiles; static struct cmd_rank cmd; -static void rank_sorted_file (struct casereader *, +static void rank_sorted_file (struct casereader *, struct casewriter *, const struct dictionary *, - const struct rank_spec *rs, + const struct rank_spec *rs, int n_rank_specs, int idx, const struct variable *rank_var); @@ -233,56 +231,65 @@ create_var_label (struct variable *dest_var, } -static bool -rank_cmd (struct dataset *ds, const struct case_ordering *sc, +static bool +rank_cmd (struct dataset *ds, const struct case_ordering *sc, const struct rank_spec *rank_specs, int n_rank_specs) { - struct case_ordering *base_ordering; + struct dictionary *d = dataset_dict (ds); bool ok = true; int i; - const int n_splits = dict_get_split_cnt (dataset_dict (ds)); - base_ordering = case_ordering_create (dataset_dict (ds)); - for (i = 0; i < n_splits ; i++) - case_ordering_add_var (base_ordering, - dict_get_split_vars (dataset_dict (ds))[i], - SRT_ASCEND); - - for (i = 0; i < n_group_vars; i++) - case_ordering_add_var (base_ordering, group_vars[i], SRT_ASCEND); for (i = 0 ; i < case_ordering_get_var_cnt (sc) ; ++i ) { - struct case_ordering *ordering; - struct casegrouper *grouper; - struct casereader *group; + /* Rank variable at index I in SC. */ + struct casegrouper *split_grouper; + struct casereader *split_group; struct casewriter *output; - struct casereader *ranked_file; - - ordering = case_ordering_clone (base_ordering); - case_ordering_add_var (ordering, - case_ordering_get_var (sc, i), - case_ordering_get_direction (sc, i)); proc_discard_output (ds); - grouper = casegrouper_create_case_ordering (sort_execute (proc_open (ds), - ordering), - base_ordering); - output = autopaging_writer_create (dict_get_next_value_idx ( - dataset_dict (ds))); - while (casegrouper_get_next_group (grouper, &group)) - rank_sorted_file (group, output, dataset_dict (ds), - rank_specs, n_rank_specs, - i, src_vars[i]); - ok = casegrouper_destroy (grouper); + split_grouper = casegrouper_create_splits (proc_open (ds), d); + output = autopaging_writer_create (dict_get_next_value_idx (d)); + + while (casegrouper_get_next_group (split_grouper, &split_group)) + { + struct case_ordering *ordering; + struct casereader *ordered; + struct casegrouper *by_grouper; + struct casereader *by_group; + int j; + + /* Sort this split group by the BY variables as primary + keys and the rank variable as secondary key. */ + ordering = case_ordering_create (d); + for (j = 0; j < n_group_vars; j++) + case_ordering_add_var (ordering, group_vars[j], SRT_ASCEND); + case_ordering_add_var (ordering, + case_ordering_get_var (sc, i), + case_ordering_get_direction (sc, i)); + ordered = sort_execute (split_group, ordering); + + /* Rank the rank variable within this split group. */ + by_grouper = casegrouper_create_vars (ordered, + group_vars, n_group_vars); + while (casegrouper_get_next_group (by_grouper, &by_group)) + { + /* Rank the rank variable within this BY group + within the split group. */ + + rank_sorted_file (by_group, output, d, rank_specs, n_rank_specs, + i, src_vars[i]); + } + ok = casegrouper_destroy (by_grouper) && ok; + } + ok = casegrouper_destroy (split_grouper); ok = proc_commit (ds) && ok; - ranked_file = casewriter_make_reader (output); - ok = proc_set_active_file_data (ds, ranked_file) && ok; + ok = (proc_set_active_file_data (ds, casewriter_make_reader (output)) + && ok); if (!ok) break; } - case_ordering_destroy (base_ordering); - return ok; + return ok; } /* Hardly a rank function !! */ @@ -300,7 +307,7 @@ rank_rank (double c, double cc, double cc_1, { double rank; - if ( c >= 1.0 ) + if ( c >= 1.0 ) { switch (cmd.ties) { @@ -461,12 +468,12 @@ rank_savage (double c, double cc, double cc_1, } static void -rank_sorted_file (struct casereader *input, +rank_sorted_file (struct casereader *input, struct casewriter *output, const struct dictionary *dict, - const struct rank_spec *rs, - int n_rank_specs, - int dest_idx, + const struct rank_spec *rs, + int n_rank_specs, + int dest_idx, const struct variable *rank_var) { struct casereader *pass1, *pass2, *pass2_1; @@ -484,13 +491,13 @@ rank_sorted_file (struct casereader *input, casereader_split (input, &pass1, &pass2); /* Pass 1: Get total group weight. */ - for (; casereader_read (pass1, &c); case_destroy (&c)) + for (; casereader_read (pass1, &c); case_destroy (&c)) w += dict_get_case_weight (dict, &c, NULL); casereader_destroy (pass1); /* Pass 2: Do ranking. */ tie_grouper = casegrouper_create_vars (pass2, &rank_var, 1); - while (casegrouper_get_next_group (tie_grouper, &pass2_1)) + while (casegrouper_get_next_group (tie_grouper, &pass2_1)) { struct casereader *pass2_2; double cc_1 = cc; @@ -502,13 +509,13 @@ rank_sorted_file (struct casereader *input, casewriter_get_taint (output)); /* Pass 2.1: Sum up weight for tied cases. */ - for (; casereader_read (pass2_1, &c); case_destroy (&c)) + for (; casereader_read (pass2_1, &c); case_destroy (&c)) tw += dict_get_case_weight (dict, &c, NULL); cc += tw; casereader_destroy (pass2_1); /* Pass 2.2: Rank tied cases. */ - while (casereader_read (pass2_2, &c)) + while (casereader_read (pass2_2, &c)) { for (i = 0; i < n_rank_specs; ++i) { @@ -519,7 +526,7 @@ rank_sorted_file (struct casereader *input, casewriter_write (output, &c); } casereader_destroy (pass2_2); - + tie_group++; } casegrouper_destroy (tie_grouper); @@ -651,7 +658,7 @@ cmd_rank (struct lexer *lexer, struct dataset *ds) rank_specs = xmalloc (sizeof (*rank_specs)); rank_specs[0].rfunc = RANK; - rank_specs[0].destvars = + rank_specs[0].destvars = xcalloc (case_ordering_get_var_cnt (sc), sizeof (struct variable *)); n_rank_specs = 1; @@ -759,8 +766,8 @@ cmd_rank (struct lexer *lexer, struct dataset *ds) msg(MW, _("FRACTION has been specified, but NORMAL and PROPORTION rank functions have not been requested. The FRACTION subcommand will be ignored.") ); /* Add a variable which we can sort by to get back the original - order */ - order = dict_create_var_assert (dataset_dict (ds), "$ORDER_", 0); + order */ + order = dict_create_var_assert (dataset_dict (ds), "$ORDER_", 0); add_transformation (ds, create_resort_key, 0, order); @@ -836,10 +843,10 @@ parse_rank_function (struct lexer *lexer, struct dictionary *dict, struct cmd_ra rank_specs[n_rank_specs - 1].rfunc = f; rank_specs[n_rank_specs - 1].destvars = NULL; - rank_specs[n_rank_specs - 1].destvars = + rank_specs[n_rank_specs - 1].destvars = xcalloc (case_ordering_get_var_cnt (sc), sizeof (struct variable *)); - + if (lex_match_id (lexer, "INTO")) { struct variable *destvar; @@ -852,7 +859,7 @@ parse_rank_function (struct lexer *lexer, struct dictionary *dict, struct cmd_ra msg(SE, _("Variable %s already exists."), lex_tokid (lexer)); return 0; } - if ( var_count >= case_ordering_get_var_cnt (sc) ) + if ( var_count >= case_ordering_get_var_cnt (sc) ) { msg(SE, _("Too many variables in INTO clause.")); return 0; @@ -948,3 +955,9 @@ rank_custom_ntiles (struct lexer *lexer, struct dataset *ds, struct cmd_rank *cm return parse_rank_function (lexer, dict, cmd, NTILES); } + +/* + Local Variables: + mode: c + End: +*/