pspp-cvs
[Top][All Lists]
Advanced

[Date Prev][Date Next][Thread Prev][Thread Next][Date Index][Thread Index]

[Pspp-cvs] pspp src/language/stats/ChangeLog src/language/...


From: Ben Pfaff
Subject: [Pspp-cvs] pspp src/language/stats/ChangeLog src/language/...
Date: Sun, 05 Aug 2007 17:20:23 +0000

CVSROOT:        /cvsroot/pspp
Module name:    pspp
Changes by:     Ben Pfaff <blp> 07/08/05 17:20:23

Modified files:
        src/language/stats: ChangeLog rank.q 
        tests          : ChangeLog 
        tests/command  : rank.sh 

Log message:
        (rank_cmd): Instead of sorting by SPLIT FILE vars, group by them.
        Fixes bug #17239.  Reviewed by John Darrington.

CVSWeb URLs:
http://cvs.savannah.gnu.org/viewcvs/pspp/src/language/stats/ChangeLog?cvsroot=pspp&r1=1.59&r2=1.60
http://cvs.savannah.gnu.org/viewcvs/pspp/src/language/stats/rank.q?cvsroot=pspp&r1=1.32&r2=1.33
http://cvs.savannah.gnu.org/viewcvs/pspp/tests/ChangeLog?cvsroot=pspp&r1=1.101&r2=1.102
http://cvs.savannah.gnu.org/viewcvs/pspp/tests/command/rank.sh?cvsroot=pspp&r1=1.4&r2=1.5

Patches:
Index: src/language/stats/ChangeLog
===================================================================
RCS file: /cvsroot/pspp/pspp/src/language/stats/ChangeLog,v
retrieving revision 1.59
retrieving revision 1.60
diff -u -b -r1.59 -r1.60
--- src/language/stats/ChangeLog        2 Aug 2007 03:07:17 -0000       1.59
+++ src/language/stats/ChangeLog        5 Aug 2007 17:20:22 -0000       1.60
@@ -1,3 +1,9 @@
+2007-08-03  Ben Pfaff  <address@hidden>
+
+       * rank.q (rank_cmd): Instead of sorting by SPLIT FILE vars, group
+       by them.  Fixes bug #17239.
+       Reviewed by John Darrington.
+
 2007-08-01  Ben Pfaff  <address@hidden>
 
        Clean up handling of median, by treating it almost like any other

Index: src/language/stats/rank.q
===================================================================
RCS file: /cvsroot/pspp/pspp/src/language/stats/rank.q,v
retrieving revision 1.32
retrieving revision 1.33
diff -u -b -r1.32 -r1.33
--- src/language/stats/rank.q   7 Jul 2007 06:14:17 -0000       1.32
+++ src/language/stats/rank.q   5 Aug 2007 17:20:22 -0000       1.33
@@ -235,50 +235,59 @@
 rank_cmd (struct dataset *ds, const struct case_ordering *sc,
          const struct rank_spec *rank_specs, int n_rank_specs)
 {
-  struct case_ordering *base_ordering;
+  struct dictionary *d = dataset_dict (ds);
   bool ok = true;
   int i;
-  const int n_splits = dict_get_split_cnt (dataset_dict (ds));
 
-  base_ordering = case_ordering_create (dataset_dict (ds));
-  for (i = 0; i < n_splits ; i++)
-    case_ordering_add_var (base_ordering,
-                           dict_get_split_vars (dataset_dict (ds))[i],
-                           SRT_ASCEND);
-
-  for (i = 0; i < n_group_vars; i++)
-    case_ordering_add_var (base_ordering, group_vars[i], SRT_ASCEND);
   for (i = 0 ; i < case_ordering_get_var_cnt (sc) ; ++i )
     {
-      struct case_ordering *ordering;
-      struct casegrouper *grouper;
-      struct casereader *group;
+      /* Rank variable at index I in SC. */
+      struct casegrouper *split_grouper;
+      struct casereader *split_group;
       struct casewriter *output;
-      struct casereader *ranked_file;
 
-      ordering = case_ordering_clone (base_ordering);
+      proc_discard_output (ds);
+      split_grouper = casegrouper_create_splits (proc_open (ds), d);
+      output = autopaging_writer_create (dict_get_next_value_idx (d));
+
+      while (casegrouper_get_next_group (split_grouper, &split_group))
+        {
+          struct case_ordering *ordering;
+          struct casereader *ordered;
+          struct casegrouper *by_grouper;
+          struct casereader *by_group;
+          int j;
+
+          /* Sort this split group by the BY variables as primary
+             keys and the rank variable as secondary key. */
+          ordering = case_ordering_create (d);
+          for (j = 0; j < n_group_vars; j++)
+            case_ordering_add_var (ordering, group_vars[j], SRT_ASCEND);
       case_ordering_add_var (ordering,
                              case_ordering_get_var (sc, i),
                              case_ordering_get_direction (sc, i));
+          ordered = sort_execute (split_group, ordering);
 
-      proc_discard_output (ds);
-      grouper = casegrouper_create_case_ordering (sort_execute (proc_open (ds),
-                                                                ordering),
-                                                  base_ordering);
-      output = autopaging_writer_create (dict_get_next_value_idx (
-                                           dataset_dict (ds)));
-      while (casegrouper_get_next_group (grouper, &group))
-        rank_sorted_file (group, output, dataset_dict (ds),
-                          rank_specs, n_rank_specs,
+          /* Rank the rank variable within this split group. */
+          by_grouper = casegrouper_create_vars (ordered,
+                                                group_vars, n_group_vars);
+          while (casegrouper_get_next_group (by_grouper, &by_group))
+            {
+              /* Rank the rank variable within this BY group
+                 within the split group. */
+
+              rank_sorted_file (by_group, output, d, rank_specs, n_rank_specs,
                           i, src_vars[i]);
-      ok = casegrouper_destroy (grouper);
+            }
+          ok = casegrouper_destroy (by_grouper) && ok;
+        }
+      ok = casegrouper_destroy (split_grouper);
       ok = proc_commit (ds) && ok;
-      ranked_file = casewriter_make_reader (output);
-      ok = proc_set_active_file_data (ds, ranked_file) && ok;
+      ok = (proc_set_active_file_data (ds, casewriter_make_reader (output))
+            && ok);
       if (!ok)
         break;
     }
-  case_ordering_destroy (base_ordering);
 
   return ok;
 }

Index: tests/ChangeLog
===================================================================
RCS file: /cvsroot/pspp/pspp/tests/ChangeLog,v
retrieving revision 1.101
retrieving revision 1.102
diff -u -b -r1.101 -r1.102
--- tests/ChangeLog     2 Aug 2007 03:07:17 -0000       1.101
+++ tests/ChangeLog     5 Aug 2007 17:20:22 -0000       1.102
@@ -1,3 +1,10 @@
+2007-08-03  Ben Pfaff  <address@hidden>
+
+       * command/rank.sh: Test RANK with noncontiguous groups of SPLIT
+       FILE variables and how they should behave differently from
+       noncontiguous groups of BY variables.  Regression test for bug
+       #17239.
+
 2007-08-01  Ben Pfaff  <address@hidden>
 
        * command/weight.sh: Update to match new output format for median

Index: tests/command/rank.sh
===================================================================
RCS file: /cvsroot/pspp/pspp/tests/command/rank.sh,v
retrieving revision 1.4
retrieving revision 1.5
diff -u -b -r1.4 -r1.5
--- tests/command/rank.sh       22 Dec 2006 04:38:23 -0000      1.4
+++ tests/command/rank.sh       5 Aug 2007 17:20:23 -0000       1.5
@@ -245,6 +245,11 @@
 NEW FILE.
 DATA LIST LIST NOTABLE /a * g1 g2 *.
 BEGIN DATA.
+2 1 2
+2 1 2
+3 1 2
+4 1 2
+5 1 2
 1 0 2
 2 0 2
 3 0 2
@@ -253,11 +258,6 @@
 6 0 2
 7 0 2
 8 0 2
-2 1 2
-2 1 2
-3 1 2
-4 1 2
-5 1 2
 6 1 2
 7 1 2
 7 1 2
@@ -274,6 +274,19 @@
   /NORMAL
   .
 
+SPLIT FILE BY g1.
+
+RANK a (D) BY g2
+  /PRINT=YES
+  /TIES=LOW
+  /MISSING=INCLUDE
+  /FRACTION=RANKIT
+  /RANK
+  /NORMAL
+  .
+
+SPLIT FILE OFF.
+
 LIST.
 
 
@@ -446,26 +459,29 @@
 Variables Created By RANK
 a into Ra(RANK of a BY g2 g1)
 a into Na(NORMAL of a using RANKIT BY g2 g1)
-       a       g1       g2        Ra     Na
--------- -------- -------- --------- ------
-    1.00      .00     2.00     8.000 1.5341 
-    2.00      .00     2.00     7.000  .8871 
-    3.00      .00     2.00     6.000  .4888 
-    4.00      .00     2.00     5.000  .1573 
-    5.00      .00     2.00     4.000 -.1573 
-    6.00      .00     2.00     3.000 -.4888 
-    7.00      .00     2.00     2.000 -.8871 
-    8.00      .00     2.00     1.000 -1.534 
-    2.00     1.00     2.00     8.000  .9674 
-    2.00     1.00     2.00     8.000  .9674 
-    3.00     1.00     2.00     7.000  .5895 
-    4.00     1.00     2.00     6.000  .2822 
-    5.00     1.00     2.00     5.000  .0000 
-    6.00     1.00     2.00     4.000 -.2822 
-    7.00     1.00     2.00     2.000 -.9674 
-    7.00     1.00     2.00     2.000 -.9674 
-    8.00     1.00     2.00     1.000 -1.593 
-    9.00     1.00     1.00     1.000  .0000 
+Variables Created By RANK
+a into RAN001(RANK of a BY g2)
+a into NOR001(NORMAL of a using RANKIT BY g2)
+       a       g1       g2        Ra     Na    RAN001 NOR001 
+-------- -------- -------- --------- ------ --------- ------ 
+    2.00     1.00     2.00     8.000  .9674     4.000  .5244  
+    2.00     1.00     2.00     8.000  .9674     4.000  .5244  
+    3.00     1.00     2.00     7.000  .5895     3.000  .0000  
+    4.00     1.00     2.00     6.000  .2822     2.000 -.5244  
+    5.00     1.00     2.00     5.000  .0000     1.000 -1.282  
+    1.00      .00     2.00     8.000 1.5341     8.000 1.5341  
+    2.00      .00     2.00     7.000  .8871     7.000  .8871  
+    3.00      .00     2.00     6.000  .4888     6.000  .4888  
+    4.00      .00     2.00     5.000  .1573     5.000  .1573  
+    5.00      .00     2.00     4.000 -.1573     4.000 -.1573  
+    6.00      .00     2.00     3.000 -.4888     3.000 -.4888  
+    7.00      .00     2.00     2.000 -.8871     2.000 -.8871  
+    8.00      .00     2.00     1.000 -1.534     1.000 -1.534  
+    6.00     1.00     2.00     4.000 -.2822     4.000 1.1503  
+    7.00     1.00     2.00     2.000 -.9674     2.000 -.3186  
+    7.00     1.00     2.00     2.000 -.9674     2.000 -.3186  
+    8.00     1.00     2.00     1.000 -1.593     1.000 -1.150  
+    9.00     1.00     1.00     1.000  .0000     1.000  .0000  
 fractional ranks ( including small ones for special case of SAVAGE ranks)
 Variables Created By RANK
 a into Pa(PROPORTION of a using TUKEY)




reply via email to

[Prev in Thread] Current Thread [Next in Thread]