【问题标题】:SAS: Looping and Outputing record if date criteria is metSAS:如果满足日期条件,则循环和输出记录
【发布时间】:2017-07-02 03:33:14
【问题描述】:

我必须对包含数百万条记录的 SAS 表进行分区,并根据每月日期标准将其输出到多个 SAS 表。例如,如果有一个customer_id在年月(日期格式)201308和201408之间有效,那么应该为这条记录创建12个表。每个表都有下面的列字段,以及一个新创建的列,称为“YearMonth”,表示它处于活动状态的月份,如第一个表中应该有 201308、201309、201310 等。

以下是说明上述观点的表格。

带有一条样本记录的原始表格

Cust_ID     Eff_YM  Trm_YM   
NH000001    201308  201408

新表 201308

Cust_ID     Eff_YM  Trm_YM  YearMonth     
NH000001    201308  201408  201308

新表 201309

Cust_ID     Eff_YM  Trm_YM  YearMonth     
NH000001    201308  201408  201309

新表 201310

Cust_ID     Eff_YM  Trm_YM  YearMonth     
NH000001    201308  201408  201310

【问题讨论】:

  • 是每张表中的一条记录,还是 YearMonth 201308 的所有记录都在一张表中?

标签: loops date sas


【解决方案1】:

创建示例dataset。

data test;
infile datalines;
input Cust_ID : $10.
      Eff_YM : 8.
      Trm_YM : 8.
      ;
datalines;
NH000001    201308  201408
NH000001    201308  201312
;
run;

从dataset 中选择minimum 和maximum 句点。不同的间隔将有尽可能多的不同datasets。

proc sql noprint;
select min(Eff_YM) into: min_Eff_YM from test;
select max(Trm_YM) into: max_Trm_YM from test;
quit;

由于我们需要在data 语句中预先指定datasets 的名称,所以在这里创建名称列表。

data dataset_names(keep=period dataset_name);
length dataset_name $20.;
format min_date date9. max_date date9.;
min_date=mdy((substr(compress(&min_Eff_YM.),5,2)),1,(substr(compress(&min_Eff_YM.),1,4)));
max_date=mdy((substr(compress(&max_Trm_YM.),5,2)),1,(substr(compress(&max_Trm_YM.),1,4)));
no_of_months=intck('month',min_date,max_date);
do i=0 to no_of_months;
period=put(intnx('month',min_date,i),yymmn6.);
dataset_name=compress(cat("dataset_",period));
output;
end;
run;


proc sql noprint;
select dataset_name into :all_datsets separated by " " from dataset_names;
select count(dataset_name) into :num_datasets from dataset_names;
select period into: all_periods separated by "," from dataset_names;
quit;

使用Eff_YM 和Trm_YM 之间的间隔创建可能记录的列表

%macro chk(YYMM);
data test_all;
set test;
No_of_loop=intck('month',
                  mdy((substr(compress(Eff_YM),5,2)),1,(substr(compress(Eff_YM),1,4))),
                  mdy((substr(compress(Trm_YM),5,2)),1,(substr(compress(Trm_YM),1,4))));

do i=0 to No_of_loop;
YearMonth = put(intnx('month',mdy((substr(compress(Eff_YM),5,2)),1,(substr(compress(Eff_YM),1,4))),i),yymmn6.);
output;
end;
run;
%mend;
%chk;

根据时期名称将数据集划分为单独的数据集

%macro data_dates;

data &all_datsets.;
set test_all;
%do i=1 %to &num_datasets.;
if YearMonth=scan("&all_periods.",&i.,",") then do;
output dataset_%sysfunc(scan("&all_periods.",&i.,","));
end;
%end;
run;
%mend;
%data_dates;

【讨论】:

    【解决方案2】:
    data HAVE;
        Length CUST_ID $8;
        Input Cust_ID $ Eff_YM Trm_YM;
    datalines;
    NH000001    201308  201408
    NH000002    201301  201401
    ;
    run;
    

    获取用于构建所有可能数据集的最小和最大日期

    proc sql noprint;
    select min(Eff_YM), max(Trm_YM) into: min_Eff_YM, :max_Trm_YM
        From HAVE;
    quit;
    %Put min_EFF_YM= &min_EFF_YM;
    %Put max_TRM_YM= &max_TRM_YM;
    

    构建所有可能的数据集并创建宏变量以进行循环

    data DSNs(drop=start i);
        Start=input(put(&min_EFF_YM,6.),yymmn6.);
        Diff=intck('month',Start,input(put(&max_TRM_YM,6.),yymmn6.));
        Put DIFF=;
    
        Do i = 0 to diff;
            DSN=Cats("_",put(intnx('Month',Start,i,'b'),yymmn6.));
            Output;
        End;
    run;
    
    Proc sql noprint;
        Select count(dsn) into :cnt separated by "" from DSNs;
        Select dsn into :all1 - :all&cnt from DSNs;
    Quit;
    %Put CNT: &cnt;
    %Put ALL1: &all1;
    %Put ALL&cnt: &&all&cnt;
    

    创建数据集并插入适当的记录

    %Macro Create_Tables;
    Data %do i = 1 %to &cnt; &&all&i %end;
        ;
        set HAVE;
        %do i=0 %to 12;
            YearMonth_dt=intnx('month',input(put(EFF_YM,6.),yymmn6.),&i);
            YearMonth=input(put(YearMonth_dt,yymmn6.),6.);
            YearMonth_dsn=cats("_",put(yearmonth_dt,yymmn6.));
            %do j = 1 %to &cnt;
                %Let DSN=&&all&j;
                if YearMonth_dsn="&dsn" then output &dsn;
            %end;
        %end;
        Keep CUST_ID EFF_YM TRM_YM YEARMONTH;
    run;
    %Mend;
    %Create_Tables ;
    

    【讨论】:

      【解决方案3】:

      您的问题的解决方案很简单。从旧数据集创建一个新数据集,并从年月开始循环到年月结束。稍后创建一个包含先前创建的数据集中的唯一年月的宏列表,然后循环创建数据集。

      data have;
       input cust_id $ eff_ym :yymmn6. trm_ym :yymmn6. ;
       format eff_ym trm_ym yymmdd10.;
      datalines;
      NH000001    201308  201408
      NH000002    201301  201401
      ;
      run;
      
      data staging;
      set have;
      do i = intck('month',0,eff_ym) to intck('month',0,trm_ym);
          yearmonth=intnx('month',0,i);
          output;
      end;
      
      format yearmonth yymmdd10.;
      drop i;
      run;
      %macro splitter;
      proc sql noprint;
          select distinct yearmonth format=date9. into :yearmonth1-:yearmonth99999 
            from staging;
      quit;
      
      %do i = 1 %to &sqlobs;
          %let dsn=%sysfunc(putn(%sysfunc(inputn(&&yearmonth&i,date9.)),yymmn6.));
      
              proc append base=data_&dsn data=staging(where=(yearmonth="&&yearmonth&i"d));
              run;
      %end;
      
      %mend splitter;
      options mprint;
      %splitter
      

      【讨论】:

        猜你喜欢
        • 1970-01-01
        • 1970-01-01
        • 1970-01-01
        • 1970-01-01
        • 2021-08-22
        • 2016-08-12
        • 1970-01-01
        • 2020-07-26
        • 2016-03-18
        相关资源
        最近更新 更多