Convert rows into columns in R-CodePudding

I have this sample dataset and i want to convert it into the following format:

Type <- c("AGE", "AGE", "REGION", "REGION", "REGION", "DRIVERS", "DRIVERS")
Level <- c("18-25", "26-70", "London", "Southampton", "Newcastle", "1", "2")
Estimate <- c(1.5,1,2,3,1,2,2.5)

df_before <- data.frame(Type, Level, Estimate)


     Type       Level Estimate
1     AGE       18-25      1.5
2     AGE       26-70      1.0
3  REGION      London      2.0
4  REGION Southampton      3.0
5  REGION   Newcastle      1.0
6 DRIVERS           1      2.0
7 DRIVERS           2      2.5

Basically, I would like to to transform the dataset into the following format. I have tried with the function dcast() but it seems that is not working.

    AGE Estimate_AGE      REGION Estimate_REGION DRIVERS Estimate_DRIVERS
1 18-25          1.5      London               2       1              2.0
2 26-70          1.0 Southampton               3       2              2.5
3  <NA>           NA   Newcastle               1    <NA>               NA

CodePudding user response：

df_before %>%
  group_by(Type) %>%
  mutate(id = row_number(), Estimate = as.character(Estimate))%>%
  pivot_longer(-c(Type, id)) %>%
  pivot_wider(id, names_from = c(Type, name))%>%
  type.convert(as.is = TRUE)

# A tibble: 3 x 7
     id AGE_Level AGE_Estimate REGION_Level REGION_Estimate DRIVERS_Level DRIVERS_Estimate
  <int> <chr>            <dbl> <chr>                  <int>         <int>            <dbl>
1     1 18-25              1.5 London                     2             1              2  
2     2 26-70              1   Southampton                3             2              2.5
3     3 NA                NA   Newcastle                  1            NA             NA

In data.table:

library(data.table)
setDT(df_before)

dcast(melt(df_before, 'Type'), rowid(Type, variable)~Type   variable)

Note that you will get alot of warning because of the type mismatch. You could use reshape2::melt to avoid this.

Anyway your datafram is not in a standard format.

In Base R >=4.0

transform(df_before, id = ave(Estimate, Type, FUN = seq_along)) |>
  reshape(v.names = c('Level', 'Estimate'), dir = 'wide', timevar = 'Type', sep = "_")

 id Level_AGE Estimate_AGE Level_REGION Estimate_REGION Level_DRIVERS Estimate_DRIVERS
1  1     18-25          1.5       London               2             1              2.0
2  2     26-70          1.0  Southampton               3             2              2.5
5  3      <NA>           NA    Newcastle               1          <NA>               NA

IN base R <4

reshape(transform(df_before, id = ave(Estimate, Type, FUN = seq_along)),
       v.names = c('Level', 'Estimate'), dir = 'wide', timevar = 'Type', sep = "_")

CodePudding user response：

Update:

The exact output as the desired output:

df_before %>% 
  group_by(Type) %>% 
  mutate(id = row_number()) %>% 
  pivot_wider(
    names_from = Type,
    values_from = c(Level, Estimate)
  ) %>% 
  select(AGE = Level_AGE, Estimate_AGE, REGION = Level_REGION, 
         Estimate_REGION, DRIVERS = Level_DRIVERS, Estimate_DRIVERS) %>% 
  type.convert(as.is=TRUE)

  AGE   Estimate_AGE REGION      Estimate_REGION DRIVERS Estimate_DRIVERS
  <chr>        <dbl> <chr>                 <int>   <int>            <dbl>
1 18-25          1.5 London                    2       1              2  
2 26-70          1   Southampton               3       2              2.5
3 NA            NA   Newcastle                 1      NA             NA

First answer:

Main aspect is to group by Type as already provided Onyambu's solution. After that we could use one pivot_wider:

library(dplyr)
library(tidyr)

df_before %>% 
  group_by(Type) %>% 
  mutate(id = row_number()) %>% 
  pivot_wider(
    names_from = Type,
    values_from = c(Level, Estimate)
  )

     id Level_AGE Level_REGION Level_DRIVERS Estimate_AGE Estimate_REGION Estimate_DRIVERS
  <int> <chr>     <chr>        <chr>                <dbl>           <dbl>            <dbl>
1     1 18-25     London       1                      1.5               2              2  
2     2 26-70     Southampton  2                      1                 3              2.5
3     3 NA        Newcastle    NA                    NA                 1             NA

CodePudding user response：

We can try this:

library(tidyverse)

Type <- c("AGE", "AGE", "REGION", "REGION", "REGION", "DRIVERS", "DRIVERS")
Level <- c("18-25", "26-70", "London", "Southampton", "Newcastle", "1", "2")
Estimate <- c(1.5, 1, 2, 3, 1, 2, 2.5)
df_before <- data.frame(Type, Level, Estimate)

data <-
  df_before %>% group_split(Type)

data <-
  map2(
    data, map(data, ~ unique(.$Type)),
    ~ mutate(., "{.y}" := Level, "Estimate_{.y}" := Estimate) %>%
      select(-c("Type", "Level", "Estimate"))
  )

#get the longest number of rows to be able to join the columns
max_rows <- map_dbl(data, nrow) %>%
  max()

#add rows if needed
map_if(
  data, ~ nrow(.) < max_rows,
  ~ rbind(., NA)
) %>%
  bind_cols()
#> # A tibble: 3 × 6
#>   AGE   Estimate_AGE DRIVERS Estimate_DRIVERS REGION      Estimate_REGION
#>   <chr>        <dbl> <chr>              <dbl> <chr>                 <dbl>
#> 1 18-25          1.5 1                    2   London                    2
#> 2 26-70          1   2                    2.5 Southampton               3
#> 3 <NA>          NA   <NA>                NA   Newcastle                 1

^{Created on 2021-12-07 by the reprex package (v2.0.1)}

CodePudding user response：

A solution based on tidyr::pivot_wider and purrr::map_dfc:

library(tidyverse)

Type <- c("AGE", "AGE", "REGION", "REGION", "REGION", "DRIVERS", "DRIVERS")
Level <- c("18-25", "26-70", "London", "Southampton", "Newcastle", "1", "2")
Estimate <- c(1.5,1,2,3,1,2,2.5)
df_before <- data.frame(Type, Level, Estimate)

df_before %>% 
  pivot_wider(names_from=Type, values_from=c(Level, Estimate), values_fn=list) %>% 
  map_dfc(~ c(unlist(.x), rep(NA, max(table(df_before$Type))-length(unlist(.x)))))

#> # A tibble: 3 × 6
#>   Level_AGE Level_REGION Level_DRIVERS Estimate_AGE Estimate_REGION
#>   <chr>     <chr>        <chr>                <dbl>           <dbl>
#> 1 18-25     London       1                      1.5               2
#> 2 26-70     Southampton  2                      1                 3
#> 3 <NA>      Newcastle    <NA>                  NA                 1
#> # … with 1 more variable: Estimate_DRIVERS <dbl>

Another solution, based on dplyr:: group_split and purrr::map_dfc:

library(tidyverse)

df_before %>% 
  mutate(maxn = max(table(.$Type))) %>% 
  group_by(Type) %>% group_split() %>% 
  map_dfc(
    ~ data.frame(c(.x$Level, rep(NA, .x$maxn[1] - nrow(.x))),
      c(.x$Estimate, rep(NA, .x$maxn[1] - nrow(.x)))) %>%
      set_names(c(.x$Type[1], paste0("Estimate_", .x$Type[1])))) %>% 
  type.convert(as.is=T)

#>     AGE Estimate_AGE DRIVERS Estimate_DRIVERS      REGION Estimate_REGION
#> 1 18-25          1.5       1              2.0      London               2
#> 2 26-70          1.0       2              2.5 Southampton               3
#> 3  <NA>           NA      NA               NA   Newcastle               1