fix proportion ci calc and theme
Gitea Organization/civilyticsR/pipeline/head There was a failure building this commit

This commit is contained in:
2024-08-09 15:05:59 -04:00
parent 4f49ae6d5a
commit 82b8db7f84
18 changed files with 481 additions and 436 deletions
+48 -48
View File
@@ -1,48 +1,48 @@
# Join utilities
#' Test the join between two sets of identifiers
#'
#' @param x a vector of identifiers to check against y
#' @param y a vector of identifiers to check against x
#' @param distinct logical, should duplicate values of x and y be removed before testing
#'
#' @return nothing, print a summary of match statistics to the console
#' @export
#'
#' @examples
#' x <- LETTERS
#' y <- c(letters, LETTERS)
#' match_test(x, y)
match_test <- function(x, y, distinct = TRUE) {
if (distinct) {
x <- unique(x)
y <- unique(y)
cat("**** Distinct Matches ****")
cat("\n")
}
# TODO: DO not report 100% if there is even 1 mismatch
xiny <- sum(x %in% y)
total_x <- length(x)
yinx <- sum(y %in% x)
total_y <- length(y)
cat("**** Match Summary ****")
cat("\n")
cat("X in Y")
cat("\n")
cat(paste0("Of the ", total_x, " X values, ", xiny, " (",
100*round(xiny/total_x, 2), "%) were matched."))
cat("\n")
cat("********************************************")
cat("\n")
cat("Y in X")
cat("\n")
cat(paste0("Of the ", total_y, " Y values, ", yinx, " (",
100*round(yinx/total_y, 2), "%) were matched."))
cat("\n")
cat("******************************************")
}
# Join utilities
#' Test the join between two sets of identifiers
#'
#' @param x a vector of identifiers to check against y
#' @param y a vector of identifiers to check against x
#' @param distinct logical, should duplicate values of x and y be removed before testing
#'
#' @return nothing, print a summary of match statistics to the console
#' @export
#'
#' @examples
#' x <- LETTERS
#' y <- c(letters, LETTERS)
#' match_test(x, y)
match_test <- function(x, y, distinct = TRUE) {
if (distinct) {
x <- unique(x)
y <- unique(y)
cat("**** Distinct Matches ****")
cat("\n")
}
# TODO: DO not report 100% if there is even 1 mismatch
xiny <- sum(x %in% y)
total_x <- length(x)
yinx <- sum(y %in% x)
total_y <- length(y)
cat("**** Match Summary ****")
cat("\n")
cat("X in Y")
cat("\n")
cat(paste0("Of the ", total_x, " X values, ", xiny, " (",
100*round(xiny/total_x, 2), "%) were matched."))
cat("\n")
cat("********************************************")
cat("\n")
cat("Y in X")
cat("\n")
cat(paste0("Of the ", total_y, " Y values, ", yinx, " (",
100*round(yinx/total_y, 2), "%) were matched."))
cat("\n")
cat("******************************************")
}
+70
View File
@@ -0,0 +1,70 @@
# https://github.com/cran/binom/blob/master/R/binom.confint.R
# Consider importing and crediting this code ^^
# https://towardsdatascience.com/five-confidence-intervals-for-proportions-that-you-should-know-about-7ff5484c024f
# https://andrewpwheeler.com/2020/11/30/confidence-intervals-around-proportions/
#' Get a simple Clopper Pearson interval
#'
#' @param num
#' @param den
#' @param confint
#'
#' @return
#' @export
#'
#' @examples
clopper_pearson <- function(num, den, confint = 0.95) {
# Same results as binom.test in base R
quant <- (1 - confint) / 2
low <- qbeta(quant, num, den-num+1)
hi <- qbeta(1-quant, num+1, den-num)
obs <- num/den
return(c("low" = low, "observed" = obs,"high" = hi))
}
# TODO consider making a vectorized version
# z_gap_test_v <- Vectorize(z_gap_test,
# SIMPLIFY = TRUE)# we only want to return a scalar
#z_gap_test(a_prop = 0.051, a_count = 2000, b_prop = 0.11, b_count = 100)
#' Calculate a univariate z score by comparing to a population
#'
#' @param unit_prop proportion for the group we are comparing
#' @param global_prop the global proportion
#' @param unit_denom the population size for the group we are comparing
#'
#' @return a z-score
#' @export
#'
#' @examples
#' z_univariate(unit_prop = 0.13, global_prop = 0.11, unit_denom = 2500)
z_univariate <- function(unit_prop, global_prop, unit_denom) {
num <- unit_prop - global_prop
denom <- sqrt(
(global_prop * (1-global_prop))/unit_denom
)
z = num / denom
return(z)
}
waldInterval <- function(x, n, conf.level = 0.95){
p <- x/n
sd <- sqrt(p*((1-p)/n))
z <- qnorm(c( (1 - conf.level)/2, 1 - (1-conf.level)/2)) #returns the value of thresholds at which conf.level has to be cut at. for 95% CI, this is -1.96 and +1.96
ci <- p + z*sd
names(ci) <- c('lwr', 'upr')
return(ci)
}#example
#waldInterval(x = 20, n =40) #this will return 0.345 and 0.655
agresti_coull_interval <- function(num, den, conf.level) {
num <- num + 2
den <- den + 4
}
+4 -4
View File
@@ -22,14 +22,14 @@ theme_civilytics <-
theme(
line = element_line(
color = "black",
size = line_size,
linewidth = line_size,
linetype = 1,
lineend = "butt"
),
rect = element_rect(
fill = NA,
color = NA,
size = line_size,
linewidth = line_size,
linetype = 1
),
text = element_text(
@@ -46,7 +46,7 @@ theme_civilytics <-
),
axis.line = element_line(
color = "black",
size = line_size,
linewidth = line_size,
lineend = "square"
),
axis.line.x = NULL,
@@ -62,7 +62,7 @@ theme_civilytics <-
axis.text.y.right = element_text(margin = margin(l = small_size / 4),
hjust = 0),
axis.ticks = element_line(color = "black",
size = line_size),
linewidth = line_size),
axis.ticks.length = unit(half_line / 2,
"pt"),
axis.title.x = element_text(margin = margin(t = half_line / 2),
-27
View File
@@ -308,35 +308,8 @@ z_gap_test <- function(a_prop, a_count, b_prop, b_count) {
return(z)
}
# TODO consider making a vectorized version
# z_gap_test_v <- Vectorize(z_gap_test,
# SIMPLIFY = TRUE)# we only want to return a scalar
#z_gap_test(a_prop = 0.051, a_count = 2000, b_prop = 0.11, b_count = 100)
#' Calculate a univariate z score by comparing to a population
#'
#' @param unit_prop proportion for the group we are comparing
#' @param global_prop the global proportion
#' @param unit_denom the population size for the group we are comparing
#'
#' @return a z-score
#' @export
#'
#' @examples
#' z_univariate(unit_prop = 0.13, global_prop = 0.11, unit_denom = 2500)
z_univariate <- function(unit_prop, global_prop, unit_denom) {
num <- unit_prop - global_prop
denom <- sqrt(
(global_prop * (1-global_prop))/unit_denom
)
z = num / denom
return(z)
}
# TODO: Consider vectorizing
#z_univariate_v <- Vectorize(z_univariate, SIMPLIFY = TRUE)