1) janitor Use adorn_totals from the janitor package ignoring the Total column. Note that within a group_by section that dot refers to the entire data set, not just that group, unless we refer to it within a do which is why we use do.
library(janitor)
res1 <- arrests %>%
select(-Total) %>%
group_by(State) %>%
do(adorn_totals(select(., -State), "row")) %>%
ungroup
res1
giving:
# A tibble: 250 x 3
State Crime Value
<chr> <chr> <dbl>
1 Alabama Murder 13.2
2 Alabama Assault 236
3 Alabama UrbanPop 58
4 Alabama Rape 21.2
5 Alabama Total 328.
6 Alaska Murder 10
7 Alaska Assault 263
8 Alaska UrbanPop 48
9 Alaska Rape 44.5
10 Alaska Total 366.
# ... with 240 more rows
We can remove the Total rows and add a column
res1 %>% {
left <- filter(., Crime != "Total")
right <- filter(., Crime == "Total") %>% select(State, Total = Value)
left_join(left, right, by = "State")
}
2) reshape2 The reshape2 package is a forerunner of the pivot_* functions. It does have margins functionality built in which seems not to have been continued in subsequent iterations in spread/gather and pivot_*. This also works if we replace the library statement with library(data.table) .
library(reshape2)
res2 <- dcast(arrests, State + Crime ~ "Value", fun.aggregate = sum,
value.var = "Value", margins = "Crime")
res2
giving:
State Crime Value
1 Alabama Assault 236.0
2 Alabama Murder 13.2
3 Alabama Rape 21.2
4 Alabama UrbanPop 58.0
5 Alabama (all) 328.4
6 Alaska Assault 263.0
7 Alaska Murder 10.0
8 Alaska Rape 44.5
9 Alaska UrbanPop 48.0
10 Alaska (all) 365.5
...etc...
To create a Total column and remove the total rows, create a factor that identifies each row as a Value or Total row and then dcast the result to wide form filling in NAs with na.locf.
library(reshape2)
library(zoo)
fac <- factor(res$Crime == '(all)', labels = c("Value", "Total"))
dc <- dcast(res2, State + Crime ~ fac, value.var = "Value")
subset(na.locf(dc, fromLast = TRUE), Crime != '(all)')
or
left <- subset(res2, Crime != "(all)")
right <- subset(res2, Crime == "(all)", c(State, Value))
names(right) <- c("State", "Total")
merge(left, right, by = "State")
3) sqldf To use SQL add a level column which is 0 for detail records and 1 for Total records and then union the details and totals and sort.
library(sqldf)
res3 <- sqldf("select State, Crime, Value from (
select 0 as level, State, Crime, Value from arrests
union
select 1 as level, State, 'Total' as Crime, sum(Value) as Total from arrests
group by State)
order by State, level")
To remove the total rows and insert a Total column
sqldf("select State, Crime, Value, Total
from res3 a
left join (
select State, sum(Value) as Total
from res3
where Crime != 'Total'
group by State) using (State)
where Crime != 'Total'")
4) Base R This is straight forward in base R using xtabs and addmargins.
Total <- sum
tab <- addmargins(xtabs(Value ~ State + Crime, arrests), 2, FUN = Total)
DF <- as.data.frame(tab, responseName = "Value")
res3 <- DF[order(DF$State, DF$Crime == "Total"), ]
and modifying (2) we can use the following to remove the Total rows and add a Total column:
left <- subset(res3, Crime != "Total")
right <- subset(res3, Crime == "Total", c(State, Value))
names(right) <- c("State", "Total")
merge(left, right, by = "State")