Filter dataframe based on input vector containing column names

Viewed 353

I have a dataframe as follows

Sol_name    geo_pos     loc_pos     dol_pos    pol_pos   kol_pos

A            1            1          0          0         1
B            0            1          1          0         0
C            1            0          1          1         1
D            0            1          0          0         1

I need to create a function where the user can input column names into a vector and the dataframe will get filtered where value in any of those columns is 1

Example: If the input is col_nm = c("geo_pos","dol_pos") then the output I am looking for is

Sol_name    geo_pos     loc_pos     dol_pos    pol_pos   kol_pos

A            1            1          0          0         1
B            0            1          1          0         0
C            1            0          1          1         1

Is there any efficient way to do this?

data

df <- read.table(text="Sol_name    geo_pos     loc_pos     dol_pos    pol_pos   kol_pos
A            1            1          0          0         1
B            0            1          1          0         0
C            1            0          1          1         1
D            0            1          0          0         1",h=T)
5 Answers

We can use rowSums efficiently here to filter rows which has at least one "1" in the selected columns.

get_one_rows <- function(cols) {
    df[rowSums(df[cols] == 1) > 0, ]
}

col_nm = c("geo_pos","dol_pos")
get_one_rows(col_nm)

# Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
#1        A       1       1       0       0       1
#2        B       0       1       1       0       0
#3        C       1       0       1       1       1


col_nm = c("kol_pos")
get_one_rows(col_nm)

#  Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
#1        A       1       1       0       0       1
#3        C       1       0       1       1       1
#4        D       0       1       0       0       1

With tidverse:

df %>% filter_at(col_nm,any_vars(.==1))

#  Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
#1        A       1       1       0       0       1
#2        B       0       1       1       0       0
#3        C       1       0       1       1       1

With plyr:

library(plyr)
unique(ldply(col_nm,.fun = function(x){(df[df[x]==1,])}))

Output:

     Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
1        A       1       1       0       0       1
2        C       1       0       1       1       1
3        B       0       1       1       0       0

OR

unique(as.data.frame(do.call(rbind, lapply(col_nm, function(x) df[df[x]==1,]))))

A base R option with Reduce

df1[Reduce(`|`, df1[col_nm]),]
#  Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
#1        A       1       1       0       0       1
#2        B       0       1       1       0       0
#3        C       1       0       1       1       1

You could use pmax :

df[as.logical(do.call(pmax,df[col_nm])),]

#   Sol_name geo_pos loc_pos dol_pos pol_pos kol_pos
# 1        A       1       1       0       0       1
# 2        B       0       1       1       0       0
# 3        C       1       0       1       1       1
Related