# Compute five binary variables with 30 objects each.
# Each variable has a predefined number of 0 and 1
# Variable 1: 10 x 1 and 20 x 0; the order is randomized
var1 <- sample(c(rep(1, 10), rep(0, 20)))
# Variable 2: 15 x 0 and 15 x 1, one block each
var2 <- c(rep(0, 15), rep(1, 15))
# Variable 3: alternation of 3 x 1 and 3 x 0 up to 30 objects
var3 <- rep(c(1, 1, 1, 0, 0, 0), 5)
# Variable 4: alternation of 5 x 1 and 10 x 0 up to 30 objects
var4 <- rep(c(rep(1, 5), rep(0, 10)), 2)
# Variable 5: 16 objects with randomized distribution of 7 x 1
# and 9 x 0, followed by 4 x 0 and 10 x 1
var5.1 <- sample(c(rep(1, 7), rep(0, 9)))
var5.2 <- c(rep(0, 4), rep(1, 10))
var5 <- c(var5.1, var5.2)
# Variables 1 to 5 are put into a data frame
(dat <- data.frame(var1, var2, var3, var4, var5))
dim(dat)
# Computation of a matrix of simple matching coefficients
# (called the Sokal and Michener index in ade4)
dat.s1 <- dist.binary(dat, method = 2)
coldiss(dat.s1, diag = TRUE)
3.3.5 Q Mode: Mixed Types Including Categorical
(Qualitative Multiclass) Variables
Among the association measures that can handle nominal data correctly, one is
readily available in R: Gower’s similarity S 15 . This coefficient has been devised to
handle data containing variables of various mathematical types, each variable
receiving a treatment corresponding to its category. The final (dis)similarity between
two objects is obtained by averaging the partial (dis)similarities computed for all
variables separately. We shall use Gower’s similarity as a symmetrical index; when a
variable is declared as a factor in a data frame, the simple matching rule is applied,
i.e., for each pair of objects the similarity is 1 for that variable if the factor has
the same level in the two objects and 0 if the level is different. One function to
compute Gower’s dissimilarity is daisy() of package cluster. Avoid the use
of vegdist() (method ¼ "gower"), which is appropriate for quantitative and
presence-absence data, but not for multiclass categorical variables.
daisy() can handle data frames made of mixed-type variables, provided that
each variable is correctly defined. Optionally, the user can provide an argument
3.3 Q Mode: Computing Dissimilarity Matrices Among Objects
49
# Each variable has a predefined number of 0 and 1
# Variable 1: 10 x 1 and 20 x 0; the order is randomized
var1 <- sample(c(rep(1, 10), rep(0, 20)))
# Variable 2: 15 x 0 and 15 x 1, one block each
var2 <- c(rep(0, 15), rep(1, 15))
# Variable 3: alternation of 3 x 1 and 3 x 0 up to 30 objects
var3 <- rep(c(1, 1, 1, 0, 0, 0), 5)
# Variable 4: alternation of 5 x 1 and 10 x 0 up to 30 objects
var4 <- rep(c(rep(1, 5), rep(0, 10)), 2)
# Variable 5: 16 objects with randomized distribution of 7 x 1
# and 9 x 0, followed by 4 x 0 and 10 x 1
var5.1 <- sample(c(rep(1, 7), rep(0, 9)))
var5.2 <- c(rep(0, 4), rep(1, 10))
var5 <- c(var5.1, var5.2)
# Variables 1 to 5 are put into a data frame
(dat <- data.frame(var1, var2, var3, var4, var5))
dim(dat)
# Computation of a matrix of simple matching coefficients
# (called the Sokal and Michener index in ade4)
dat.s1 <- dist.binary(dat, method = 2)
coldiss(dat.s1, diag = TRUE)
3.3.5 Q Mode: Mixed Types Including Categorical
(Qualitative Multiclass) Variables
Among the association measures that can handle nominal data correctly, one is
readily available in R: Gower’s similarity S 15 . This coefficient has been devised to
handle data containing variables of various mathematical types, each variable
receiving a treatment corresponding to its category. The final (dis)similarity between
two objects is obtained by averaging the partial (dis)similarities computed for all
variables separately. We shall use Gower’s similarity as a symmetrical index; when a
variable is declared as a factor in a data frame, the simple matching rule is applied,
i.e., for each pair of objects the similarity is 1 for that variable if the factor has
the same level in the two objects and 0 if the level is different. One function to
compute Gower’s dissimilarity is daisy() of package cluster. Avoid the use
of vegdist() (method ¼ "gower"), which is appropriate for quantitative and
presence-absence data, but not for multiclass categorical variables.
daisy() can handle data frames made of mixed-type variables, provided that
each variable is correctly defined. Optionally, the user can provide an argument
3.3 Q Mode: Computing Dissimilarity Matrices Among Objects
49
