<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:taxo="http://purl.org/rss/1.0/modules/taxonomy/" version="2.0">
  <channel>
    <title>topic Order criteria with cas tables and data step using multiple threads with by + last operator? in SAS Programming</title>
    <link>https://communities.sas.com/t5/SAS-Programming/Order-criteria-with-cas-tables-and-data-step-using-multiple/m-p/991791#M380379</link>
    <description>&lt;P&gt;Hello, I am carrying this doubt for a long time.&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;P&gt;If I use a code like this one, do I have to use the single option or will the order criteria be met correctly without doing so. Imagine that like in this case we have 10,000 different by-groups.&amp;nbsp;&lt;/P&gt;
&lt;P&gt;Is it like playing lottery to do it like this without the 'single' option ?&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;P&gt;Are there any cas actions that handle last or first logic? I could tweak the deduplicate action to do so, but if it comes to more complex if not first.abc and not last.abc or lag functions then this will be out of the game.&amp;nbsp; &amp;nbsp;&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;PRE&gt;&lt;CODE class=" language-sas"&gt;Proc cas;
dataStep.runCode /
code="
data &amp;amp;source_tab._c;
set spc_temptab;
by groupkey data_mod_date;
if first.groupkey then obs=0;
obs+1;
if last.groupkey then output;
dummy='dummy';
drop groupkey;
run;
";
single="yes";
run;&lt;/CODE&gt;&lt;/PRE&gt;</description>
    <pubDate>Wed, 05 Aug 2026 16:37:49 GMT</pubDate>
    <dc:creator>acordes</dc:creator>
    <dc:date>2026-08-05T16:37:49Z</dc:date>
    <item>
      <title>Order criteria with cas tables and data step using multiple threads with by + last operator?</title>
      <link>https://communities.sas.com/t5/SAS-Programming/Order-criteria-with-cas-tables-and-data-step-using-multiple/m-p/991791#M380379</link>
      <description>&lt;P&gt;Hello, I am carrying this doubt for a long time.&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;P&gt;If I use a code like this one, do I have to use the single option or will the order criteria be met correctly without doing so. Imagine that like in this case we have 10,000 different by-groups.&amp;nbsp;&lt;/P&gt;
&lt;P&gt;Is it like playing lottery to do it like this without the 'single' option ?&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;P&gt;Are there any cas actions that handle last or first logic? I could tweak the deduplicate action to do so, but if it comes to more complex if not first.abc and not last.abc or lag functions then this will be out of the game.&amp;nbsp; &amp;nbsp;&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;PRE&gt;&lt;CODE class=" language-sas"&gt;Proc cas;
dataStep.runCode /
code="
data &amp;amp;source_tab._c;
set spc_temptab;
by groupkey data_mod_date;
if first.groupkey then obs=0;
obs+1;
if last.groupkey then output;
dummy='dummy';
drop groupkey;
run;
";
single="yes";
run;&lt;/CODE&gt;&lt;/PRE&gt;</description>
      <pubDate>Wed, 05 Aug 2026 16:37:49 GMT</pubDate>
      <guid>https://communities.sas.com/t5/SAS-Programming/Order-criteria-with-cas-tables-and-data-step-using-multiple/m-p/991791#M380379</guid>
      <dc:creator>acordes</dc:creator>
      <dc:date>2026-08-05T16:37:49Z</dc:date>
    </item>
    <item>
      <title>Re: Order criteria with cas tables and data step using multiple threads with by + last operator?</title>
      <link>https://communities.sas.com/t5/SAS-Programming/Order-criteria-with-cas-tables-and-data-step-using-multiple/m-p/992272#M380421</link>
      <description>&lt;P&gt;Each unique by-group is processed by a single thread and the order of rows within a by-group is guaranteed, so you do not need single="yes" for this example.&amp;nbsp; Here's some good reading on the topic:&amp;nbsp;&amp;nbsp;&lt;A href="https://go.documentation.sas.com/doc/en/pgmsascdc/v_077/casdspgm/n1mcndtr0m6wovn1cpgwxeia8iav.htm" target="_blank"&gt;SAS Help Center&lt;/A&gt;&lt;/P&gt;
&lt;P&gt;&amp;nbsp;&lt;/P&gt;
&lt;P&gt;I tried to come up with a 10,000 by-group example to demonstrate:&lt;/P&gt;
&lt;DIV&gt;
&lt;DIV&gt;
&lt;DIV&gt;
&lt;PRE&gt;&lt;CODE class=" language-sas"&gt;options fullstimer compress=yes noquotelenmax;
cas casauto sessopts=(metrics=true timeout=7200 locale="en_US");
caslib _all_ assign;

* Generate some sample data with 10000 by groups.  

  I let GPT 5.6 come up with this part.  It took a few iterations but works
  well enough for this example, I think.  Here's my prompt:

  "Generate DATA step code that generates a dataset with 10000 by groups.  
   The relevant BY statement is:  by groupkey data_mod_date   
   Let groupkey have values of group01..group100.  Let each groupNN have 
   10-200 values of data_mod_date, but do it such that the entire dataset 
   has 1000000 rows exactly.  I do not want the values of groupkey to be 
   ordered, nor the values of data_mod_date to be ordered per groupkey.  
   I also do not want the exact same set of data_mod_date values per 
   groupkey, but pull them from a bounded range of 400 continuous values."

   The real test code starts on line 194.
;

data sample_data;
  length groupkey $10;
  format data_mod_date date9.;

  call streaminit(12345);

  /*
    Requirements:
      - groupkey values: group01-group100
      - total BY groups: exactly 10,000
      - each groupkey has 10-200 data_mod_date values
      - data_mod_date values come from a bounded 400-day range
      - date sets differ by groupkey but overlap significantly
      - rows per groupkey x data_mod_date: 1-200
      - total rows: exactly 1,000,000
      - output is not ordered by groupkey or data_mod_date
  */

  array ndates[100]       _temporary_;
  array nrows[100,200]    _temporary_;
  array group_order[100]  _temporary_;
  array date_pool[400]    _temporary_;
  array date_order[400]   _temporary_;

  target_bygroups = 10000;
  target_rows     = 1000000;

  /*
    Assign random number of dates per groupkey, 10-200,
    then adjust so total BY groups is exactly 10,000.
  */
  total_bygroups = 0;

  do grp = 1 to 100;
    ndates[grp] = rand('integer', 10, 200);
    total_bygroups + ndates[grp];
  end;

  do while (total_bygroups ne target_bygroups);

    grp = rand('integer', 1, 100);

    if total_bygroups &amp;lt; target_bygroups then do;
      if ndates[grp] &amp;lt; 200 then do;
        ndates[grp] = ndates[grp] + 1;
        total_bygroups + 1;
      end;
    end;
    else do;
      if ndates[grp] &amp;gt; 10 then do;
        ndates[grp] = ndates[grp] - 1;
        total_bygroups + (-1);
      end;
    end;

  end;

  /*
    Assign random row counts, 1-200, for each actual BY group,
    then adjust so total rows is exactly 1,000,000.
  */
  total_rows = 0;

  do grp = 1 to 100;
    do dseq = 1 to ndates[grp];
      nrows[grp,dseq] = rand('integer', 1, 200);
      total_rows + nrows[grp,dseq];
    end;
  end;

  do while (total_rows ne target_rows);

    grp  = rand('integer', 1, 100);
    dseq = rand('integer', 1, ndates[grp]);

    if total_rows &amp;lt; target_rows then do;
      if nrows[grp,dseq] &amp;lt; 200 then do;
        nrows[grp,dseq] = nrows[grp,dseq] + 1;
        total_rows + 1;
      end;
    end;
    else do;
      if nrows[grp,dseq] &amp;gt; 1 then do;
        nrows[grp,dseq] = nrows[grp,dseq] - 1;
        total_rows + (-1);
      end;
    end;

  end;

  /*
    Create bounded 400-day date pool.
  */
  do i = 1 to 400;
    date_pool[i] = '01JAN2025'd + i - 1;
  end;

  /*
    Create and shuffle groupkey output order.
  */
  do i = 1 to 100;
    group_order[i] = i;
  end;

  do i = 100 to 2 by -1;
    j = rand('integer', 1, i);

    tmp            = group_order[i];
    group_order[i] = group_order[j];
    group_order[j] = tmp;
  end;

  /*
    Generate rows in randomized groupkey order.
    For each groupkey, shuffle the 400-day pool and use the first ndates[grp]
    values, giving different but overlapping date sets across groupkeys.
  */
  id = 0;

  do gseq = 1 to 100;

    grp = group_order[gseq];
    groupkey = cats('group', put(grp, z3.));

    do i = 1 to 400;
      date_order[i] = date_pool[i];
    end;

    do i = 400 to 2 by -1;
      j = rand('integer', 1, i);

      tmp_date      = date_order[i];
      date_order[i] = date_order[j];
      date_order[j] = tmp_date;
    end;

    do dseq = 1 to ndates[grp];

      data_mod_date = date_order[dseq];

      do rownum = 1 to nrows[grp,dseq];

        id + 1;
        value1 = rand('uniform');
        value2 = rand('normal');

        output;

      end;
    end;
  end;

  drop grp gseq dseq i j rownum tmp tmp_date
       total_bygroups target_bygroups
       total_rows target_rows;
  drop id value:;
run;

/*
* Sanity checks;
proc summary data=sample_data nway missing;
  class groupkey data_mod_date;
  output out=bygroup_counts;
run;

proc univariate data=bygroup_counts;
var _freq_;
histogram;
run;
*/


* Step 1:  Run the DATA step on Compute and sort the output;
* This will be our benchmark;
proc sort data=sample_data out=sample_data_sorted;
by groupkey data_mod_date;
run;

data output_compute;
set sample_data_sorted;
by groupkey data_mod_date;
if first.groupkey then obs=0;
obs+1;
if last.groupkey then do;
   dummy='last.groupkey for '||groupkey;
   output;
end;
drop groupkey;
run;

proc sort data=output_compute; 
  by data_mod_date obs;
run;


* Step 2:  Upload the sample data to CAS and run single threaded;
data casuser.sample_data;
set sample_data;
run;

* Add thread_id to each row and save mix/max thread_id per by group;
proc cas;
dataStep.runCode /
code="
data casuser.output_cas_single (replace=yes);
set casuser.sample_data;
by groupkey data_mod_date;
thread_id = _THREADID_;
if first.groupkey then do;
   obs=0;
   call missing(min_thread_id, max_thread_id);
end;
obs+1;
min_thread_id = min(min_thread_id, thread_id);
max_thread_id = max(max_thread_id, thread_id);
if last.groupkey then do;
   dummy='last.groupkey for '||groupkey;
   output;
end;
drop groupkey;
run;"          /* fyi, the example code mistakenly has a semicolon here */
single="yes";
run;

* Confirm one thread_id was used;
proc print data=casuser.output_cas_single;
run;

* Download to compute, sort, and compare to the benchmark;
data output_cas_single;
   set casuser.output_cas_single;
   drop thread_id min_thread_id max_thread_id;
run;
proc sort; 
  by data_mod_date obs;
run;

* Should compare exactly;
proc compare base=output_compute comp=output_cas_single;
run;


* Step 3:  Run multi-threaded with 10 threads;
proc cas;
dataStep.runCode /
code="
data casuser.output_cas_multi (replace=yes);
set casuser.sample_data;
by groupkey data_mod_date;
thread_id = _THREADID_;
if first.groupkey then do;
   obs=0;
   call missing(min_thread_id, max_thread_id);
end;
obs+1;
min_thread_id = min(min_thread_id, thread_id);
max_thread_id = max(max_thread_id, thread_id);
if last.groupkey then do;
   dummy='last.groupkey for '||groupkey;
   output;
end;
drop groupkey;
run;"
single="no" nthreads=10;
run;

* Confirm one thread_id was used per by group and 10 thread_ids;
proc print data=casuser.output_cas_multi;
run;

* Download to compute, sort, and compare to the benchmark;
data output_cas_multi;
   set casuser.output_cas_multi;
   drop thread_id min_thread_id max_thread_id;
run;
proc sort; 
  by data_mod_date obs;
run;

* Should compare exactly;
proc compare base=output_compute comp=output_cas_multi;
run;&lt;/CODE&gt;&lt;/PRE&gt;
&lt;/DIV&gt;
&lt;/DIV&gt;
&lt;/DIV&gt;</description>
      <pubDate>Thu, 20 Aug 2026 13:35:19 GMT</pubDate>
      <guid>https://communities.sas.com/t5/SAS-Programming/Order-criteria-with-cas-tables-and-data-step-using-multiple/m-p/992272#M380421</guid>
      <dc:creator>DerylHollick</dc:creator>
      <dc:date>2026-08-20T13:35:19Z</dc:date>
    </item>
  </channel>
</rss>

