Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
72b2cc3a26
|
||
|
|
fe9238a6ef | ||
|
|
6392a74013
|
||
|
|
de2ba0cfb9 | ||
|
|
e912a926c2 | ||
|
|
785f3af16d
|
@@ -0,0 +1,61 @@
|
||||
# Multi-platform R CMD check, running on the GitHub mirror.
|
||||
#
|
||||
# This exists because the canonical Gitea runner is Linux-only, and this
|
||||
# package hard-depends on duckdb and httr2 -- both compiled, both with real
|
||||
# platform variance -- while having never been checked on Windows or macOS.
|
||||
# A large share of the audience is on Windows.
|
||||
#
|
||||
# Gitea reads .gitea/workflows and GitHub reads .github/workflows, so this
|
||||
# file is inert on the canonical repo and coexists with the Gitea CI that
|
||||
# remains authoritative for deploys.
|
||||
#
|
||||
# The suite needs NO credentials: tests/testthat/setup.R points USCOGDATA_URL
|
||||
# at the bundled fixture corpus. That is exactly why inst/extdata/fixture_corpus
|
||||
# must never be added to .Rbuildignore.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
name: R-CMD-check
|
||||
|
||||
permissions: read-all
|
||||
|
||||
jobs:
|
||||
R-CMD-check:
|
||||
runs-on: ${{ matrix.config.os }}
|
||||
name: ${{ matrix.config.os }} (${{ matrix.config.r }})
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
config:
|
||||
- {os: macos-latest, r: 'release'}
|
||||
- {os: windows-latest, r: 'release'}
|
||||
- {os: ubuntu-latest, r: 'devel', http-user-agent: 'release'}
|
||||
- {os: ubuntu-latest, r: 'release'}
|
||||
|
||||
env:
|
||||
GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }}
|
||||
R_KEEP_PKG_SOURCE: yes
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: r-lib/actions/setup-pandoc@v2
|
||||
|
||||
- uses: r-lib/actions/setup-r@v2
|
||||
with:
|
||||
r-version: ${{ matrix.config.r }}
|
||||
http-user-agent: ${{ matrix.config.http-user-agent }}
|
||||
use-public-rspm: true
|
||||
|
||||
- uses: r-lib/actions/setup-r-dependencies@v2
|
||||
with:
|
||||
extra-packages: any::rcmdcheck
|
||||
needs: check
|
||||
|
||||
- uses: r-lib/actions/check-r-package@v2
|
||||
with:
|
||||
upload-snapshots: true
|
||||
build_args: 'c("--no-manual")'
|
||||
@@ -122,9 +122,25 @@
|
||||
#' fallback -- correct for the local temp corpora the direct-execution tests
|
||||
#' build.
|
||||
#' @noRd
|
||||
#' `fixed = TRUE` is load-bearing, not a style choice.
|
||||
#'
|
||||
#' In regex mode, `gsub()` interprets backslashes in the REPLACEMENT string as
|
||||
#' escape sequences and silently drops them. A Windows corpus path is full of
|
||||
#' them, so `C:\Users\RUNNER\AppData\...` was substituted in as
|
||||
#' `C:UsersRUNNERAppData...` and every DuckDB read failed with "No files found
|
||||
#' that match the pattern". `fixed = TRUE` treats pattern and replacement as
|
||||
#' literal text, which is what a filesystem path needs.
|
||||
#'
|
||||
#' This is why the package could not read a LOCAL corpus on Windows at all --
|
||||
#' including the test fixture, hence the entire suite, and any `cog_mirror()`
|
||||
#' copy. Remote https URLs were unaffected, having no backslashes, which is
|
||||
#' part of why it stayed hidden: the bug predates the `{long_files}` token and
|
||||
#' lived in the original `{url}` substitution, unnoticed because nothing ever
|
||||
#' ran on Windows until the mirror's check matrix existed.
|
||||
#' @noRd
|
||||
.render_view_sql <- function(sql, url, manifest = list()) {
|
||||
sql <- gsub("\\{long_files\\}", .long_files_sql(url, manifest), sql, fixed = FALSE)
|
||||
gsub("\\{url\\}", url, sql, fixed = FALSE)
|
||||
sql <- gsub("{long_files}", .long_files_sql(url, manifest), sql, fixed = TRUE)
|
||||
gsub("{url}", url, sql, fixed = TRUE)
|
||||
}
|
||||
|
||||
#' Register DuckDB views from inst/sql/ SQL files
|
||||
|
||||
Binary file not shown.
+3
-3
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 6,
|
||||
"built_at": "2026-07-31T00:47:27Z",
|
||||
"pipeline_commit": "aadb46b",
|
||||
"built_at": "2026-08-03T16:51:32Z",
|
||||
"pipeline_commit": "e7394a4",
|
||||
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
||||
"data_vintage": {
|
||||
"source_vintages": {
|
||||
@@ -105,7 +105,7 @@
|
||||
},
|
||||
{
|
||||
"path": "data/series_breaks.parquet",
|
||||
"sha256": "06dcc995ff533e57cc65fa25086cc9bf83ba592c58bf7cc99269dc2576f69944",
|
||||
"sha256": "731998516cd802f63fcf7fb66053c7a62b7be955ab0794cad4a4979cb7628b87",
|
||||
"description": "series_breaks.parquet"
|
||||
},
|
||||
{
|
||||
|
||||
@@ -62,3 +62,36 @@ test_that("registered `long` view reads through the enumerated list", {
|
||||
expect_true(all(c(2011, 2012, 2019, 2020) %in% yrs))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("a Windows-style corpus path survives token substitution", {
|
||||
# gsub() in regex mode treats backslashes in the REPLACEMENT as escape
|
||||
# sequences and silently drops them, so a Windows path went in as
|
||||
# C:\Users\RUNNER\... and came out as C:UsersRUNNER..., after which every
|
||||
# DuckDB read failed with "No files found that match the pattern".
|
||||
#
|
||||
# That made a LOCAL corpus unreadable on Windows -- the bundled fixture
|
||||
# included, so the whole suite failed there -- while remote https URLs
|
||||
# worked fine, having no backslashes. It went unnoticed for the life of the
|
||||
# package because nothing ever ran on Windows.
|
||||
#
|
||||
# Reproducible on any platform: this is string handling, not a filesystem
|
||||
# behaviour, so it does not need a Windows runner to catch.
|
||||
win <- "C:\\Users\\RUNNER~1\\AppData\\Local\\Temp\\Rtmp123/"
|
||||
|
||||
out <- uscogdata:::.render_view_sql(
|
||||
"FROM read_parquet('{url}data/summary_categories.parquet')", win
|
||||
)
|
||||
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out, fixed = TRUE))
|
||||
expect_false(grepl("C:Users", out, fixed = TRUE))
|
||||
|
||||
# The same must hold through the {long_files} path, which embeds the url
|
||||
# once per enumerated partition.
|
||||
manifest <- list(files = list(long_partitions = list(
|
||||
list(year = 2011L, path = "data/long/year=2011/part-0.parquet")
|
||||
)))
|
||||
out2 <- uscogdata:::.render_view_sql(
|
||||
"FROM read_parquet({long_files}, hive_partitioning = true)", win, manifest
|
||||
)
|
||||
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out2, fixed = TRUE))
|
||||
expect_false(grepl("C:Users", out2, fixed = TRUE))
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user