[英]SAS Proc Import CSV and missing data
所以,我試圖在 SAS 中導入一些數據集並加入它們,唯一的問題是加入它們后我收到了這個錯誤 -
proc import datafile='filepath/datasetA.csv'
out = dataA
dbms= csv
replace;
run;
proc import datafile='filepath\datasetB.csv'
out = dataB
dbms= csv
replace;
run;
/* combine them all into one dataset*/
data DataC;
set &dataA. &dataB;
run;
ERROR: Variable column_k has been defined as both character and numeric
在我嘗試加入的兩個數據集中,有問題的列看起來像這樣 -
+----------+
| column_k |
+----------+
| 0 |
| 1 |
| 5 |
| 4 |
| NA |
| NA |
| 4 |
| 3 |
| NA |
+----------+
基本上,如果可能的話,我想將該列中的 NA 數據導入為“缺失”? 我需要整個列保持數字,因為我計划對該列中的數據進行一些數學運算。
謝謝你的幫助!
如果您希望繼續使用Proc IMPORT
那么您需要確保列的類型相同。 在您的情況下,您知道column_k
應該是數字,因此DATA
步驟可以使用INPUT
函數將字符值轉換為數字。
proc import … out = dataA;
proc import … out = dataB;
data dataA;
set dataA;
_num = input(column_k, best12.);
drop column_k;
rename _num = column_k;
run;
data dataB;
set dataB;
_num = input(column_k, best12.);
drop column_k;
rename _num = column_k;
run;
data want;
set dataA dataB;
run;
在更大范圍內,列名的數據類型不匹配可能發生在處理多年導入等場景中。
假設不能重新讀取舊數據並且新數據具有不同的列類型。
對於需要數值的情況,一種方法是使用宏編寫源代碼,必要時將指定的變量從字符轉換為數字。
例子:
%enforce_num (perm.loans2015, age amount remaining, out=work.loans2015)
%enforce_num (perm.loans2016, age amount remaining, out=work.loans2016)
%enforce_num (perm.loans2017, age amount remaining, out=work.loans2017)
data loans_3yrs;
set work.loans2015-loans2017;
run;
回到你更簡單的案例:
proc import … out = dataA;
proc import … out = dataB;
%enforce_num(dataA, column_k)
%enforce_num(dataB, column_k)
data want;
set dataA dataB;
run;
宏enforce_num
會是什么樣子? 它必須:
%macro enforce_num(data, vars, out=&data);
/*
* Arguments:
* data - name of input data set
* vars - space separated list of variables that must be numeric, convert type if necessary
* out - name of output data set, default same as input data set
*
* Output:
* - Unchanged data set if data and out are the same and no conversion needed
* - Changed data set if some columns in data need conversion to numeric
* - replaces data if out is same as data
* - replaces out if out is different then data
* - the column order of the changed data set will be the same as the original data set
*/
%local dsid index index2 vars varname vartype varnames debug;
%let index2 = 0; %* number of variables determined to be requiring conversion;
%let debug = 0;
%if &debug %then %put NOTE: &SYSMACRONAME: data=%superq(data);
%let dsid = %sysfunc(open(&data));
%if &dsid %then %do;
%do index = 1 %to %sysfunc(attrn(&dsid, nvars));
%let varname = %sysfunc(varname(&dsid, &index));
%let varnames = &varnames &varname;
%if %sysfunc(indexw(&varname, &vars)) %then %do;
%if C = %sysfunc(vartype(&dsid, &index)) %then %do;
%* Data contains character variable requiring enforcement;
%let index2 = %eval(&index2+1);
%local convert&index2;
%let convert&index2 = &varname;
%let varnames = &varnames ___&index2 ; %* Variables that will be converted will be named __<#> during conversion;
%end;
%end;
%end;
%let dsid = %sysfunc(close(&dsid));
%end;
%else
%put %sysfunc(sysmsg());
%*put NOTE: &=vars;
%*put NOTE: &=varnames;
%if &index2 = 0 %then %do;
%* No columns need to be converted to numeric, copy to out if necessary;
%if &data ne &out %then %do;
data &out;
set &data;
run;
%end;
%return;
%end;
%* Some columns need to be converted to numeric;
%* Ensure the converted column is at the same position (varnum) as in the original data set;
data &out;
retain &varnames;
set &data;
%do index = 1 %to &index2;
___&index = input(&&convert&index,?? best12.);
%end;
drop
%do index = 1 %to &index2;
&&convert&index
%end;
;
rename
%do index = 1 %to &index2;
___&index = &&convert&index
%end;
;
run;
%put NOTE: ------------------------------------------------;
%put NOTE: &data has been subjected to numeric enforcement.;
%put NOTE: ------------------------------------------------;
%mend enforce_num;
proc import
是一個猜測過程,通過檢查幾行數據來工作。這是一個問題,因為 Excel 數據單元格沒有任何數據類型。 一列可以在不同的單元格中包含文本、日期、日期時間和數值。
因此,最好使用具有指定變量類型的infile
語句:
filename input 'filepath/datasetA.csv';
data dataA;
infile input truncover firstobs=2/*reads from the second line*/;
input column_k;/*here you should specify input variables. If you want to read column_k as character, use : "input column_k $100." with specified length*/
run;
filename input clear;
輸入(csv文件):
+----------+
| column_k |
+----------+
| 0 |
| 1 |
| 5 |
| 4 |
| NA |
| NA |
| 4 |
| 3 |
| NA |
+----------+
輸出(作為數據集dataA):
+----------+
| column_k |
+----------+
| 0 |
| 1 |
| 5 |
| 4 |
| . |
| . |
| 4 |
| 3 |
| . |
+----------+
聲明:本站的技術帖子網頁,遵循CC BY-SA 4.0協議,如果您需要轉載,請注明本站網址或者原文地址。任何問題請咨詢:yoyou2525@163.com.