Because I have messed with indices lately myself, I wanted to test this behaviour and came to the following conclusion (underlying code at the end of the comment):
library(data.table)
DT <- data.table(x = c(3, 3, 2, 2, 1, 1),
y = c(2, 1, 2, 1, 2, 1))
DTix <- copy(DT)
setindex(DTix, x)
DTixy <- copy(DT)
setindex(DTixy, x, y)
## with reduced index after assign
DTir <- copy(DTixy)
DTir[1, y := 2]
i <- data.table(x = c(2, 2, 1, 1),
y = c(2, 1, 2, 1))
iix <- copy(i)
setindex(iix, x)
iixy <- copy(i)
setindex(iixy, x, y)
iixr <- copy(i)
setindex(iixr, x, y)
iixr[1, y := 2]
## function to determine, which columns have been reordered in a data.table with x and y columns
isSortedTwoCol <- function(DT){
if(identical(DT$x, sort(DT$x))){
if(identical(setorder(copy(DT), x, y), DT)) {
sorted <- "x & y"
} else {
sorted <- "x"
}
} else if(identical(setorder(copy(DT), y), DT)){
sorted <- "y"
} else {
sorted <- " none"
}
sorted
}
## function to determine, which columns have been reordered in a data.table with x, y, and i.y columns
isSortedThreeCol <- function(DT){
if(identical(DT$x, sort(DT$x))){
if(identical(setorder(copy(DT), x, y), DT)) {
if(identical(setorder(copy(DT), x, y, i.y), DT)){
sorted <- "x & y & i.y"
} else {
sorted <- "x & y"
}
} else {
sorted <- "x"
}
} else if(identical(setorder(copy(DT), y), DT)){
if(identical(setorder(copy(DT), y, i.y), DT)){
sorted <- "y & i.y"
} else {
sorted <- "y"
}
} else if(identical(setorder(copy(DT), i.y), DT)){
sorted <- " i.y"
} else {
sorted <- "none"
}
sorted
}
cat("\n\nsubsets with lists")
cat(paste0("\nDT[1,, on = 'x']: ", isSortedTwoCol(DT[.(c(1, 2)),, on = 'x'])))
cat(paste0("\nDTix[1,, on = 'x']: ", isSortedTwoCol(DTix[.(c(1, 2)),, on = 'x'])))
cat(paste0("\nDTixy[1,, on = 'x']: ", isSortedTwoCol(DTixy[.(c(1, 2)),, on = 'x'])))
cat("\n\nsubsets with calls")
cat(paste0("\nDT[x %in% c(1,2)]: ", isSortedTwoCol(DT[x %in% (c(1, 2)),])))
cat(paste0("\nDTix[x %in% c(1,2)]: ", isSortedTwoCol(DTix[x %in% (c(1, 2)),])))
cat(paste0("\nDTixy[x %in% c(1,2)]: ", isSortedTwoCol(DTixy[x %in% (c(1, 2)),])))
cat("\n\nsubsets with data.tables")
cat(paste0("\nDT[i,, on = 'x']: ", isSortedThreeCol(DT[i,, on = 'x'])))
cat(paste0("\nDT[iix,, on = 'x']: ", isSortedThreeCol(DT[iix,, on = 'x'])))
cat(paste0("\nDTix[i,, on = 'x']: ", isSortedThreeCol(DTix[i,, on = 'x'])))
cat(paste0("\nDTix[iix,, on = 'x']: ", isSortedThreeCol(DTix[iix,, on = 'x'])))
cat(paste0("\nDT[i,, on = c('x', 'y')]: ", isSortedTwoCol(DT[i,, on = c('x', 'y')])))
cat(paste0("\nDT[iixy,, on = c('x', 'y')]: ", isSortedTwoCol(DT[iixy,, on = c('x', 'y')])))
cat(paste0("\nDTixy[i,, on = c('x', 'y')]: ", isSortedTwoCol(DTixy[i,, on = c('x', 'y')])))
cat(paste0("\nDTixy[iixy,, on = c('x', 'y')]: ", isSortedTwoCol(DTixy[iixy,, on = c('x', 'y')])))
When browsing the
data.table.Rsource code, I came upon this comment in line 452, where appropriate indices for subsetting on columncolare identified:# Can't be any index with that col as the first one because those indexes will reorder within each groupThis indicates to me that the data.table policy is as follows:
Because I have messed with indices lately myself, I wanted to test this behaviour and came to the following conclusion (underlying code at the end of the comment):
Effect of secondary indices on subsetting in i
Essentially, this means that
Here is my code: