gsub() in regex mode treats backslashes in the REPLACEMENT string as
escape sequences and silently drops them. A Windows corpus path is full of
them, so C:\Users\RUNNER\AppData\... was substituted into the view SQL
as C:UsersRUNNERAppData... and every DuckDB read failed with 'No files
found that match the pattern'.
Effect: uscogdata could not read a LOCAL corpus on Windows at all -- the
bundled fixture included, so the entire test suite failed there, and any
cog_mirror() copy was unusable. Remote https URLs were unaffected, having
no backslashes, which is part of why it stayed hidden.
The bug predates the {long_files} token; it lived in the original {url}
substitution since that code was written. Nothing ever ran on Windows
until the GitHub mirror's check matrix existed, which found it on its
first run: 14 of 15 Windows failures were this, the 15th a downstream
consequence of view registration failing.
fixed = TRUE treats pattern and replacement as literal text. The
regression test reproduces on any platform -- it is string handling, not
filesystem behaviour, so it needs no Windows runner.
98 lines
4.1 KiB
R
98 lines
4.1 KiB
R
test_that(".long_files_sql enumerates every partition the manifest lists", {
|
|
manifest <- list(files = list(long_partitions = list(
|
|
list(year = 2011L, path = "data/long/year=2011/part-0.parquet"),
|
|
list(year = 2012L, path = "data/long/year=2012/part-0.parquet")
|
|
)))
|
|
expect_equal(
|
|
uscogdata:::.long_files_sql("https://example.org/corpus/", manifest),
|
|
paste0(
|
|
"['https://example.org/corpus/data/long/year=2011/part-0.parquet',",
|
|
"'https://example.org/corpus/data/long/year=2012/part-0.parquet']"
|
|
)
|
|
)
|
|
})
|
|
|
|
test_that(".long_files_sql falls back to the glob when no partition list is present", {
|
|
# test-views.R registers views with a hand-built manifest that has no
|
|
# `files` element. That must keep working: the glob is valid for the
|
|
# local paths such a manifest is used with.
|
|
expect_equal(
|
|
uscogdata:::.long_files_sql("/tmp/corpus/", list(schema_version = 4L)),
|
|
"'/tmp/corpus/data/long/**/*.parquet'"
|
|
)
|
|
expect_equal(
|
|
uscogdata:::.long_files_sql("/tmp/corpus/", list(files = list(long_partitions = list()))),
|
|
"'/tmp/corpus/data/long/**/*.parquet'"
|
|
)
|
|
})
|
|
|
|
test_that("the enumerated list matches the bundled fixture's partition count", {
|
|
skip_if_no_corpus()
|
|
m <- jsonlite::fromJSON(
|
|
file.path(fixture_corpus_path(), "manifest.json"), simplifyVector = FALSE
|
|
)
|
|
out <- uscogdata:::.long_files_sql(fixture_corpus_path(), m)
|
|
expect_equal(
|
|
lengths(regmatches(out, gregexpr("part-0\\.parquet", out))),
|
|
length(m$files$long_partitions)
|
|
)
|
|
})
|
|
|
|
test_that("no view SQL survives rendering with an unsubstituted token", {
|
|
# Introducing {long_files} broke four test sites that had hand-rolled the
|
|
# {url} substitution -- each failed with a DuckDB parser error on the
|
|
# surviving brace. This asserts the whole SQL directory renders clean, so
|
|
# a future token cannot reintroduce that silently.
|
|
sql_dir <- system.file("sql", package = "uscogdata")
|
|
for (f in list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE)) {
|
|
rendered <- uscogdata:::.render_view_sql(
|
|
paste(readLines(f, warn = FALSE), collapse = "\n"), "/tmp/corpus/"
|
|
)
|
|
expect_false(grepl("\\{[a-z_]+\\}", rendered), label = basename(f))
|
|
}
|
|
})
|
|
|
|
test_that("registered `long` view reads through the enumerated list", {
|
|
skip_if_no_corpus()
|
|
with_fixture_corpus({
|
|
con <- uscogdata:::.ensure_session()
|
|
n <- DBI::dbGetQuery(con, "SELECT count(*) AS n FROM long")$n
|
|
expect_gt(n, 0)
|
|
yrs <- DBI::dbGetQuery(con, "SELECT DISTINCT year FROM long ORDER BY year")$year
|
|
expect_true(all(c(2011, 2012, 2019, 2020) %in% yrs))
|
|
})
|
|
})
|
|
|
|
test_that("a Windows-style corpus path survives token substitution", {
|
|
# gsub() in regex mode treats backslashes in the REPLACEMENT as escape
|
|
# sequences and silently drops them, so a Windows path went in as
|
|
# C:\Users\RUNNER\... and came out as C:UsersRUNNER..., after which every
|
|
# DuckDB read failed with "No files found that match the pattern".
|
|
#
|
|
# That made a LOCAL corpus unreadable on Windows -- the bundled fixture
|
|
# included, so the whole suite failed there -- while remote https URLs
|
|
# worked fine, having no backslashes. It went unnoticed for the life of the
|
|
# package because nothing ever ran on Windows.
|
|
#
|
|
# Reproducible on any platform: this is string handling, not a filesystem
|
|
# behaviour, so it does not need a Windows runner to catch.
|
|
win <- "C:\\Users\\RUNNER~1\\AppData\\Local\\Temp\\Rtmp123/"
|
|
|
|
out <- uscogdata:::.render_view_sql(
|
|
"FROM read_parquet('{url}data/summary_categories.parquet')", win
|
|
)
|
|
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out, fixed = TRUE))
|
|
expect_false(grepl("C:Users", out, fixed = TRUE))
|
|
|
|
# The same must hold through the {long_files} path, which embeds the url
|
|
# once per enumerated partition.
|
|
manifest <- list(files = list(long_partitions = list(
|
|
list(year = 2011L, path = "data/long/year=2011/part-0.parquet")
|
|
)))
|
|
out2 <- uscogdata:::.render_view_sql(
|
|
"FROM read_parquet({long_files}, hive_partitioning = true)", win, manifest
|
|
)
|
|
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out2, fixed = TRUE))
|
|
expect_false(grepl("C:Users", out2, fixed = TRUE))
|
|
})
|