diff --git a/.Rbuildignore b/.Rbuildignore index ef4cbbf..b777082 100644 --- a/.Rbuildignore +++ b/.Rbuildignore @@ -11,3 +11,8 @@ ^\.github$ ^client_secret.*\.json$ ^.*auth-key\.json$ +^src/\.cargo$ +^src/rust/vendor$ +^src/rust/target$ +^src/Makevars$ +^src/Makevars\.win$ diff --git a/.gitignore b/.gitignore index ae95d0c..955ca56 100644 --- a/.gitignore +++ b/.gitignore @@ -13,3 +13,6 @@ local/ secrets/ /Meta/ docs +src/rust/vendor +src/Makevars +src/Makevars.win diff --git a/DESCRIPTION b/DESCRIPTION index c18da67..756c904 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -10,7 +10,7 @@ Encoding: UTF-8 Roxygen: list(markdown = TRUE) RoxygenNote: 7.3.3 Imports: - arrow, + arrow (>= 24.0.0), cli, dbplyr, dplyr, @@ -21,11 +21,14 @@ Imports: glue, googleCloudStorageR, janitor, + lubridate, + nanoarrow, readr, renv, rlang, stringr, targets, + tidyr, usethis, withr Suggests: @@ -35,6 +38,12 @@ Suggests: tibble, knitr, rmarkdown +Remotes: + apache/arrow/r Config/testthat/edition: 3 VignetteBuilder: knitr URL: https://openjusticeok.github.io/ojoutils/ +Config/rextendr/version: 0.5.0 +SystemRequirements: Cargo (Rust's package manager), rustc >= 1.65.0, xz +Depends: + R (>= 4.2) diff --git a/NAMESPACE b/NAMESPACE index af33bf0..b5c331f 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -1,5 +1,6 @@ # Generated by roxygen2: do not edit by hand +export(count_interval) export(describe_change) export(gcs_auth_bucket) export(gcs_list_objects) @@ -11,10 +12,21 @@ export(ojo_create_project) export(ojo_parse_county) export(ojo_use_template) export(tar_gcs_csv) +importFrom(arrow,as_arrow_table) importFrom(arrow,read_csv_arrow) importFrom(arrow,write_csv_arrow) +importFrom(dplyr,all_of) +importFrom(dplyr,coalesce) +importFrom(dplyr,collect) +importFrom(dplyr,compute) +importFrom(dplyr,everything) +importFrom(dplyr,filter) +importFrom(dplyr,group_by) importFrom(dplyr,if_else) +importFrom(dplyr,mutate) importFrom(dplyr,pull) +importFrom(dplyr,select) +importFrom(dplyr,ungroup) importFrom(fs,path_abs) importFrom(fs,path_wd) importFrom(gargle,token_fetch) @@ -28,16 +40,28 @@ importFrom(googleCloudStorageR,gcs_get_object) importFrom(googleCloudStorageR,gcs_global_bucket) importFrom(googleCloudStorageR,gcs_list_objects) importFrom(janitor,clean_names) +importFrom(lubridate,as_datetime) +importFrom(lubridate,floor_date) +importFrom(lubridate,period) +importFrom(nanoarrow,as_nanoarrow_array_stream) importFrom(readr,write_lines) importFrom(renv,init) importFrom(renv,install) +importFrom(rlang,":=") importFrom(rlang,abort) importFrom(rlang,as_name) +importFrom(rlang,check_required) importFrom(rlang,enexpr) importFrom(rlang,ensym) importFrom(rlang,expr) +importFrom(rlang,has_name) importFrom(rlang,is_interactive) +importFrom(rlang,sym) +importFrom(rlang,syms) importFrom(stringr,str_remove_all) importFrom(stringr,str_to_lower) importFrom(targets,tar_target_raw) +importFrom(tidyr,complete) +importFrom(tidyr,fill) importFrom(usethis,create_from_github) +useDynLib(ojoutils, .registration = TRUE) diff --git a/R/count_interval.R b/R/count_interval.R new file mode 100644 index 0000000..70b9b15 --- /dev/null +++ b/R/count_interval.R @@ -0,0 +1,244 @@ +#' Count intervals over time periods +#' +#' @description +#' Counts the number of active intervals for each time period (day, hour, etc.) +#' given start and end dates. Useful for occupancy or population counts over time. +#' +#' @param data A data frame or Arrow table containing the interval data. +#' @param start Character string. Name of the column containing interval start dates. +#' @param end Character string. Name of the column containing interval end dates. +#' @param period Character string. Time period for counting (e.g., "day", "hour", +#' "week", "month", "quarter", "year"). Defaults to "day". +#' @param date_name Character string. Name for the output date column. +#' Defaults to "date". +#' @param count_name Character string. Name for the output count column. +#' Defaults to "n". +#' @param .by Character vector. Column names to group by. Defaults to empty +#' (no grouping). +#' @param .fill Named list with "start" and "end" elements. Values to use when +#' filling NA start/end dates. If NULL (default), uses min/max of data. +#' @param .inclusive Logical vector of length 2. Whether start and end boundaries +#' are inclusive. Defaults to `c(TRUE, TRUE)`. +#' +#' @return A tibble with columns for date, count, and any grouping variables. +#' +#' @examples +#' \dontrun{ +#' # Basic usage +#' df <- data.frame( +#' start = as.Date(c("2024-01-01", "2024-01-05")), +#' end = as.Date(c("2024-01-03", "2024-01-06")) +#' ) +#' count_interval(df, start = "start", end = "end", period = "day") +#' +#' # With grouping +#' df <- data.frame( +#' start = as.Date(c("2024-01-01", "2024-01-02")), +#' end = as.Date(c("2024-01-03", "2024-01-04")), +#' ward = c("A", "B") +#' ) +#' count_interval(df, start = "start", end = "end", period = "day", .by = "ward") +#' } +#' +#' @importFrom rlang check_required abort sym syms has_name := +#' @importFrom dplyr collect select mutate everything filter group_by ungroup +#' @importFrom dplyr all_of coalesce compute +#' @importFrom lubridate as_datetime floor_date period +#' @importFrom arrow as_arrow_table +#' @importFrom nanoarrow as_nanoarrow_array_stream +#' @importFrom tidyr complete fill +#' @export +count_interval <- function( + data, + start, + end, + period = "day", + date_name = "date", + count_name = "n", + .by = character(), + .fill = list(start = NULL, end = NULL), + .inclusive = c(TRUE, TRUE) +) { + rlang::check_required(data) + rlang::check_required(start) + rlang::check_required(end) + + if (is.null(data)) { + rlang::abort("`data` must not be NULL.") + } + + # Extract names for validation regardless of input type + data_names <- if (inherits(data, "ArrowTabular")) { + data$schema$names + } else { + names(data) + } + + if (!start %in% data_names) { + rlang::abort(paste0("Column '", start, "' not found in `data`.")) + } + if (!end %in% data_names) { + rlang::abort(paste0("Column '", end, "' not found in `data`.")) + } + if (length(.by) > 0 && any(!.by %in% data_names)) { + missing <- setdiff(.by, data_names) + rlang::abort(paste0( + "Columns not found in `data`: ", paste(missing, collapse = ", "), "." + )) + } + if (date_name %in% c(start, end, count_name, .by)) { + rlang::abort( + "`date_name` must not match `start`, `end`, `count_name`, or any `.by` column." + ) + } + if (count_name %in% c(start, end, date_name, .by)) { + rlang::abort( + "`count_name` must not match `start`, `end`, `date_name`, or any `.by` column." + ) + } + if (any(.by %in% c(start, end))) { + rlang::abort("`.by` columns must not include `start` or `end`.") + } + + if (!is.character(period) || length(period) != 1L || is.na(period)) { + rlang::abort("`period` must be a single non-NA character string.") + } + + period_seq <- sub("minute", "min", period, fixed = TRUE) + period_seq <- sub("second", "sec", period_seq, fixed = TRUE) + + if ( + !is.list(.fill) || + length(.fill) != 2L || + !all(rlang::has_name(.fill, c("start", "end"))) + ) { + rlang::abort( + "`.fill` must be a named list with 'start' and 'end' elements." + ) + } + + if (length(.inclusive) != 2L || !is.logical(.inclusive)) { + rlang::abort("`.inclusive` must be a length-2 logical vector.") + } + + # Convert to arrow for processing + data <- arrow::as_arrow_table(data) + + if (nrow(data) == 0) { + empty_res <- data |> + dplyr::collect() |> + dplyr::select(dplyr::all_of(.by)) |> + dplyr::mutate( + !!rlang::sym(date_name) := lubridate::as_datetime(character()), + !!rlang::sym(count_name) := integer() + ) |> + dplyr::select(!!rlang::sym(date_name), !!rlang::sym(count_name), dplyr::everything()) + return(empty_res) + } + + schema <- data$schema + start_type <- schema$GetFieldByName(start)$type + end_type <- schema$GetFieldByName(end)$type + + # Compute fill defaults from data min/max when NULL + if (is.null(.fill$start)) { + start_vals <- data[[start]]$as_vector() + if (all(is.na(start_vals))) { + rlang::abort( + "All values in 'start' are NA and no `.fill$start` was provided." + ) + } + fill_start <- lubridate::as_datetime(min(start_vals, na.rm = TRUE)) + } else { + fill_start <- lubridate::as_datetime(.fill$start) + } + + if (is.null(.fill$end)) { + end_vals <- data[[end]]$as_vector() + if (all(is.na(end_vals))) { + rlang::abort( + "All values in 'end' are NA and no `.fill$end` was provided." + ) + } + fill_end <- lubridate::as_datetime(max(end_vals, na.rm = TRUE)) + } else { + fill_end <- lubridate::as_datetime(.fill$end) + } + + query <- data |> + dplyr::mutate( + !!rlang::sym(start) := lubridate::floor_date( + dplyr::coalesce( + !!rlang::sym(start), + arrow::as_arrow_array(fill_start)$cast(start_type) + ), + unit = period + ), + !!rlang::sym(end) := lubridate::floor_date( + dplyr::coalesce( + !!rlang::sym(end), + arrow::as_arrow_array(fill_end)$cast(end_type) + ), + unit = period + ) + ) + + tab <- dplyr::compute(query) + + end_vec <- tab[[end]]$as_vector() + + max_expected_date <- max(end_vec, na.rm = TRUE) + + tab$.end_plus_one <- end_vec + if (period == "quarter") { + lubridate::period(3, units = "months") + } else { + lubridate::period(1, units = period) + } + + stream <- nanoarrow::as_nanoarrow_array_stream(tab) + + res <- count_interval_( + stream = stream, + start = start, + end = end, + end_plus_one = ".end_plus_one", + date_name = date_name, + count_name = count_name, + by = .by, + inclusive = .inclusive + ) |> + arrow::as_arrow_table() |> + dplyr::collect() + + if (nrow(res) == 0) { + return( + dplyr::select( + res, + !!rlang::sym(date_name), + !!rlang::sym(count_name), + !!!rlang::syms(.by) + ) + ) + } + + res |> + dplyr::group_by(!!!rlang::syms(.by)) |> + tidyr::complete( + !!rlang::sym(date_name) := seq( + min(!!rlang::sym(date_name), na.rm = TRUE), + max(!!rlang::sym(date_name), na.rm = TRUE), + by = period_seq + ) + ) |> + tidyr::fill(!!rlang::sym(count_name), .direction = "down") |> + dplyr::mutate( + !!rlang::sym(count_name) := dplyr::coalesce(!!rlang::sym(count_name), 0L) + ) |> + dplyr::filter(!!rlang::sym(date_name) <= max_expected_date) |> + dplyr::ungroup() |> + dplyr::select( + !!rlang::sym(date_name), + !!rlang::sym(count_name), + !!!rlang::syms(.by) + ) +} diff --git a/R/extendr-wrappers.R b/R/extendr-wrappers.R new file mode 100644 index 0000000..ddfbca5 --- /dev/null +++ b/R/extendr-wrappers.R @@ -0,0 +1,10 @@ +# Generated by extendr: Do not edit by hand +# nolint start + +#' @usage NULL +#' @useDynLib ojoutils, .registration = TRUE +NULL + +count_interval_ <- function(stream, start, end, end_plus_one, date_name, count_name, by, inclusive) .Call(wrap__count_interval_, stream, start, end, end_plus_one, date_name, count_name, by, inclusive) + +# nolint end diff --git a/cleanup b/cleanup new file mode 100644 index 0000000..e346d71 --- /dev/null +++ b/cleanup @@ -0,0 +1 @@ +rm -f src/Makevars diff --git a/cleanup.win b/cleanup.win new file mode 100644 index 0000000..a182174 --- /dev/null +++ b/cleanup.win @@ -0,0 +1 @@ +rm -f src/Makevars.win diff --git a/configure b/configure new file mode 100755 index 0000000..c608b11 --- /dev/null +++ b/configure @@ -0,0 +1,3 @@ +#!/usr/bin/env sh +: "${R_HOME=`R RHOME`}" +"${R_HOME}/bin/Rscript" tools/config.R diff --git a/configure.win b/configure.win new file mode 100644 index 0000000..57eb255 --- /dev/null +++ b/configure.win @@ -0,0 +1,2 @@ +#!/usr/bin/env sh +"${R_HOME}/bin${R_ARCH_BIN}/Rscript.exe" tools/config.R diff --git a/man/count_interval.Rd b/man/count_interval.Rd new file mode 100644 index 0000000..5b9e0de --- /dev/null +++ b/man/count_interval.Rd @@ -0,0 +1,69 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/count_interval.R +\name{count_interval} +\alias{count_interval} +\title{Count intervals over time periods} +\usage{ +count_interval( + data, + start, + end, + period = "day", + date_name = "date", + count_name = "n", + .by = character(), + .fill = list(start = NULL, end = NULL), + .inclusive = c(TRUE, TRUE) +) +} +\arguments{ +\item{data}{A data frame or Arrow table containing the interval data.} + +\item{start}{Character string. Name of the column containing interval start dates.} + +\item{end}{Character string. Name of the column containing interval end dates.} + +\item{period}{Character string. Time period for counting (e.g., "day", "hour", +"week", "month", "quarter", "year"). Defaults to "day".} + +\item{date_name}{Character string. Name for the output date column. +Defaults to "date".} + +\item{count_name}{Character string. Name for the output count column. +Defaults to "n".} + +\item{.by}{Character vector. Column names to group by. Defaults to empty +(no grouping).} + +\item{.fill}{Named list with "start" and "end" elements. Values to use when +filling NA start/end dates. If NULL (default), uses min/max of data.} + +\item{.inclusive}{Logical vector of length 2. Whether start and end boundaries +are inclusive. Defaults to \code{c(TRUE, TRUE)}.} +} +\value{ +A tibble with columns for date, count, and any grouping variables. +} +\description{ +Counts the number of active intervals for each time period (day, hour, etc.) +given start and end dates. Useful for occupancy or population counts over time. +} +\examples{ +\dontrun{ +# Basic usage +df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-05")), + end = as.Date(c("2024-01-03", "2024-01-06")) +) +count_interval(df, start = "start", end = "end", period = "day") + +# With grouping +df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + ward = c("A", "B") +) +count_interval(df, start = "start", end = "end", period = "day", .by = "ward") +} + +} diff --git a/src/.gitignore b/src/.gitignore new file mode 100644 index 0000000..24e51fc --- /dev/null +++ b/src/.gitignore @@ -0,0 +1,8 @@ +*.o +*.so +*.dll +target +.cargo +rust/vendor +Makevars +Makevars.win diff --git a/src/Makevars.in b/src/Makevars.in new file mode 100644 index 0000000..39a09d9 --- /dev/null +++ b/src/Makevars.in @@ -0,0 +1,52 @@ +TARGET_DIR = ./rust/target +LIBDIR = $(TARGET_DIR)/@LIBDIR@ +STATLIB = $(LIBDIR)/libojoutils.a +PKG_LIBS = -L$(LIBDIR) -lojoutils + +all: $(SHLIB) rust_clean + +.PHONY: $(STATLIB) + +$(SHLIB): $(STATLIB) + +CARGOTMP = $(CURDIR)/.cargo +VENDOR_DIR = $(CURDIR)/vendor + + +# RUSTFLAGS appends --print=native-static-libs to ensure that +# the correct linkers are used. Use this for debugging if need. +# +# CRAN note: Cargo and Rustc versions are reported during +# configure via tools/msrv.R. +# +# If a vendor directory exists, it is used for offline compilation. Otherwise if +# vendor.tar.xz exists, it is unzipped and used for offline compilation. +$(STATLIB): + + if [ -d ./vendor ]; then \ + echo "=== Using offline vendor directory ==="; \ + mkdir -p $(CARGOTMP) && \ + cp rust/vendor-config.toml $(CARGOTMP)/config.toml; \ + elif [ -f ./rust/vendor.tar.xz ]; then \ + echo "=== Using offline vendor tarball ==="; \ + tar xf rust/vendor.tar.xz && \ + mkdir -p $(CARGOTMP) && \ + cp rust/vendor-config.toml $(CARGOTMP)/config.toml; \ + fi + + export CARGO_HOME=$(CARGOTMP) && \ + export PATH="$(PATH):$(HOME)/.cargo/bin" && \ + @PANIC_EXPORTS@RUSTFLAGS="$(RUSTFLAGS) --print=native-static-libs" cargo build @CRAN_FLAGS@ --lib @PROFILE@ --manifest-path=./rust/Cargo.toml --target-dir $(TARGET_DIR) @TARGET@ + + export CARGO_HOME=$(CARGOTMP) && \ + export PATH="$(PATH):$(HOME)/.cargo/bin" && \ + cargo run @CRAN_FLAGS@ --bin document --manifest-path=./rust/Cargo.toml --target-dir $(TARGET_DIR) @TARGET@ + + # Always clean up CARGOTMP + rm -Rf $(CARGOTMP); + +rust_clean: $(SHLIB) + rm -Rf $(CARGOTMP) $(VENDOR_DIR) @CLEAN_TARGET@ + +clean: + rm -Rf $(SHLIB) $(STATLIB) $(OBJECTS) $(TARGET_DIR) $(VENDOR_DIR) diff --git a/src/Makevars.win.in b/src/Makevars.win.in new file mode 100644 index 0000000..0858b9e --- /dev/null +++ b/src/Makevars.win.in @@ -0,0 +1,51 @@ +TARGET = $(subst 64,x86_64,$(subst 32,i686,$(WIN)))-pc-windows-gnu + +TARGET_DIR = ./rust/target +LIBDIR = $(TARGET_DIR)/$(TARGET)/@LIBDIR@ +STATLIB = $(LIBDIR)/libojoutils.a +PKG_LIBS = -L$(LIBDIR) -lojoutils -lws2_32 -ladvapi32 -luserenv -lbcrypt -lntdll + +all: $(SHLIB) rust_clean + +.PHONY: $(STATLIB) + +$(SHLIB): $(STATLIB) + +CARGOTMP = $(CURDIR)/.cargo +VENDOR_DIR = vendor + +$(STATLIB): + mkdir -p $(TARGET_DIR)/libgcc_mock + touch $(TARGET_DIR)/libgcc_mock/libgcc_eh.a + + # If a vendor directory exists, it is used for offline compilation. Otherwise if + # vendor.tar.xz exists, it is unzipped and used for offline compilation. + if [ -d ./vendor ]; then \ + echo "=== Using offline vendor directory ==="; \ + mkdir -p $(CARGOTMP) && \ + cp rust/vendor-config.toml $(CARGOTMP)/config.toml; \ + elif [ -f ./rust/vendor.tar.xz ]; then \ + echo "=== Using offline vendor tarball ==="; \ + tar xf rust/vendor.tar.xz && \ + mkdir -p $(CARGOTMP) && \ + cp rust/vendor-config.toml $(CARGOTMP)/config.toml; \ + fi + + # Build the project using Cargo with additional flags + export CARGO_HOME=$(CARGOTMP) && \ + export LIBRARY_PATH="$(LIBRARY_PATH);$(CURDIR)/$(TARGET_DIR)/libgcc_mock" && \ + RUSTFLAGS="$(RUSTFLAGS) --print=native-static-libs" cargo build @CRAN_FLAGS@ --target=$(TARGET) --lib @PROFILE@ --manifest-path=rust/Cargo.toml --target-dir=$(TARGET_DIR) + + # Generate wrappers + export CARGO_HOME=$(CARGOTMP) && \ + export PATH="$(PATH):$(HOME)/.cargo/bin" && \ + cargo run @CRAN_FLAGS@ --bin document --target $(TARGET) --manifest-path=./rust/Cargo.toml --target-dir $(TARGET_DIR) + + # Always clean up CARGOTMP + rm -Rf $(CARGOTMP); + +rust_clean: $(SHLIB) + rm -Rf $(CARGOTMP) $(VENDOR_DIR) @CLEAN_TARGET@ + +clean: + rm -Rf $(SHLIB) $(STATLIB) $(OBJECTS) $(TARGET_DIR) $(VENDOR_DIR) diff --git a/src/entrypoint.c b/src/entrypoint.c new file mode 100644 index 0000000..c1b52ed --- /dev/null +++ b/src/entrypoint.c @@ -0,0 +1,10 @@ +// We need to forward routine registration from C to Rust +// to avoid the linker removing the static library. + +void R_init_ojoutils_extendr(void *dll); +void register_extendr_panic_hook(void); + +void R_init_ojoutils(void *dll) { + register_extendr_panic_hook(); + R_init_ojoutils_extendr(dll); +} diff --git a/src/ojoutils-win.def b/src/ojoutils-win.def new file mode 100644 index 0000000..cc8d394 --- /dev/null +++ b/src/ojoutils-win.def @@ -0,0 +1,2 @@ +EXPORTS +R_init_ojoutils diff --git a/src/rust/Cargo.lock b/src/rust/Cargo.lock new file mode 100644 index 0000000..85461fe --- /dev/null +++ b/src/rust/Cargo.lock @@ -0,0 +1,1016 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 3 + +[[package]] +name = "ahash" +version = "0.8.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" +dependencies = [ + "cfg-if", + "const-random", + "getrandom 0.3.4", + "once_cell", + "version_check", + "zerocopy", +] + +[[package]] +name = "aho-corasick" +version = "1.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +dependencies = [ + "memchr", +] + +[[package]] +name = "android_system_properties" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" +dependencies = [ + "libc", +] + +[[package]] +name = "anyhow" +version = "1.0.102" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" + +[[package]] +name = "arrow" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d441fdda254b65f3e9025910eb2c2066b6295d9c8ed409522b8d2ace1ff8574c" +dependencies = [ + "arrow-arith", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-csv", + "arrow-data", + "arrow-ipc", + "arrow-json", + "arrow-ord", + "arrow-row", + "arrow-schema", + "arrow-select", + "arrow-string", +] + +[[package]] +name = "arrow-arith" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ced5406f8b720cc0bc3aa9cf5758f93e8593cda5490677aa194e4b4b383f9a59" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "chrono", + "num-traits", +] + +[[package]] +name = "arrow-array" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "772bd34cacdda8baec9418d80d23d0fb4d50ef0735685bd45158b83dfeb6e62d" +dependencies = [ + "ahash", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "chrono", + "half", + "hashbrown 0.16.1", + "num-complex", + "num-integer", + "num-traits", +] + +[[package]] +name = "arrow-buffer" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "898f4cf1e9598fdb77f356fdf2134feedfd0ee8d5a4e0a5f573e7d0aec16baa4" +dependencies = [ + "bytes", + "half", + "num-bigint", + "num-traits", +] + +[[package]] +name = "arrow-cast" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b0127816c96533d20fc938729f48c52d3e48f99717e7a0b5ade77d742510736d" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ord", + "arrow-schema", + "arrow-select", + "atoi", + "base64", + "chrono", + "half", + "lexical-core", + "num-traits", + "ryu", +] + +[[package]] +name = "arrow-csv" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca025bd0f38eeecb57c2153c0123b960494138e6a957bbda10da2b25415209fe" +dependencies = [ + "arrow-array", + "arrow-cast", + "arrow-schema", + "chrono", + "csv", + "csv-core", + "regex", +] + +[[package]] +name = "arrow-data" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "42d10beeab2b1c3bb0b53a00f7c944a178b622173a5c7bcabc3cb45d90238df4" +dependencies = [ + "arrow-buffer", + "arrow-schema", + "half", + "num-integer", + "num-traits", +] + +[[package]] +name = "arrow-ipc" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "609a441080e338147a84e8e6904b6da482cefb957c5cdc0f3398872f69a315d0" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", + "flatbuffers", +] + +[[package]] +name = "arrow-json" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ead0914e4861a531be48fe05858265cf854a4880b9ed12618b1d08cba9bebc8" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-data", + "arrow-schema", + "chrono", + "half", + "indexmap", + "itoa", + "lexical-core", + "memchr", + "num-traits", + "ryu", + "serde_core", + "serde_json", + "simdutf8", +] + +[[package]] +name = "arrow-ord" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "763a7ba279b20b52dad300e68cfc37c17efa65e68623169076855b3a9e941ca5" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", +] + +[[package]] +name = "arrow-row" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e14fe367802f16d7668163ff647830258e6e0aeea9a4d79aaedf273af3bdcd3e" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "half", +] + +[[package]] +name = "arrow-schema" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c30a1365d7a7dc50cc847e54154e6af49e4c4b0fddc9f607b687f29212082743" +dependencies = [ + "bitflags", +] + +[[package]] +name = "arrow-select" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78694888660a9e8ac949853db393af2a8b8fc82c19ce333132dfa2e72cc1a7fe" +dependencies = [ + "ahash", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "num-traits", +] + +[[package]] +name = "arrow-string" +version = "58.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61e04a01f8bb73ce54437514c5fd3ee2aa3e8abe4c777ee5cc55853b1652f79e" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", + "memchr", + "num-traits", + "regex", + "regex-syntax", +] + +[[package]] +name = "arrow_extendr" +version = "58.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b79e2fa8f09911d0b07ccdf8063992af259cc2f4e2012a1bd6a39ed546e3d9ab" +dependencies = [ + "anyhow", + "arrow", + "extendr-api", +] + +[[package]] +name = "atoi" +version = "2.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f28d99ec8bfea296261ca1af174f24225171fea9664ba9003cbebee704810528" +dependencies = [ + "num-traits", +] + +[[package]] +name = "autocfg" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" + +[[package]] +name = "base64" +version = "0.22.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" + +[[package]] +name = "bitflags" +version = "2.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3" + +[[package]] +name = "bumpalo" +version = "3.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" + +[[package]] +name = "bytes" +version = "1.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33" + +[[package]] +name = "cc" +version = "1.2.60" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43c5703da9466b66a946814e1adf53ea2c90f10063b86290cc9eb67ce3478a20" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "chrono" +version = "0.4.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +dependencies = [ + "iana-time-zone", + "js-sys", + "num-traits", + "wasm-bindgen", + "windows-link", +] + +[[package]] +name = "const-random" +version = "0.1.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "87e00182fe74b066627d63b85fd550ac2998d4b0bd86bfed477a0ae4c7c71359" +dependencies = [ + "const-random-macro", +] + +[[package]] +name = "const-random-macro" +version = "0.1.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9d839f2a20b0aee515dc581a6172f2321f96cab76c1a38a4c584a194955390e" +dependencies = [ + "getrandom 0.2.17", + "once_cell", + "tiny-keccak", +] + +[[package]] +name = "core-foundation-sys" +version = "0.8.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" + +[[package]] +name = "crunchy" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" + +[[package]] +name = "csv" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938" +dependencies = [ + "csv-core", + "itoa", + "ryu", + "serde_core", +] + +[[package]] +name = "csv-core" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782" +dependencies = [ + "memchr", +] + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "extendr-api" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "803569de0d273b4bf281871046a7d63a23cc12776bdb5b63de5c1e81aae30728" +dependencies = [ + "extendr-ffi", + "extendr-macros", + "once_cell", + "paste", + "readonly", +] + +[[package]] +name = "extendr-ffi" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5ba82ddd48e85202654997b81e4b1d39c0c54b5dcd7cae92705f807bf528efcf" + +[[package]] +name = "extendr-macros" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba8fad8d2a0d0651b1947042cf3a8beddc73d39cec3485b200fdfd24cb3bb6aa" +dependencies = [ + "lazy_static", + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "find-msvc-tools" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" + +[[package]] +name = "flatbuffers" +version = "25.12.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" +dependencies = [ + "bitflags", + "rustc_version", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "libc", + "wasi", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi", + "wasip2", +] + +[[package]] +name = "half" +version = "2.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b" +dependencies = [ + "cfg-if", + "crunchy", + "num-traits", + "zerocopy", +] + +[[package]] +name = "hashbrown" +version = "0.16.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" + +[[package]] +name = "hashbrown" +version = "0.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4f467dd6dccf739c208452f8014c75c18bb8301b050ad1cfb27153803edb0f51" + +[[package]] +name = "iana-time-zone" +version = "0.1.65" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" +dependencies = [ + "android_system_properties", + "core-foundation-sys", + "iana-time-zone-haiku", + "js-sys", + "log", + "wasm-bindgen", + "windows-core", +] + +[[package]] +name = "iana-time-zone-haiku" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" +dependencies = [ + "cc", +] + +[[package]] +name = "indexmap" +version = "2.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +dependencies = [ + "equivalent", + "hashbrown 0.17.0", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "js-sys" +version = "0.3.95" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2964e92d1d9dc3364cae4d718d93f227e3abb088e747d92e0395bfdedf1c12ca" +dependencies = [ + "once_cell", + "wasm-bindgen", +] + +[[package]] +name = "lazy_static" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" + +[[package]] +name = "lexical-core" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d8d125a277f807e55a77304455eb7b1cb52f2b18c143b60e766c120bd64a594" +dependencies = [ + "lexical-parse-float", + "lexical-parse-integer", + "lexical-util", + "lexical-write-float", + "lexical-write-integer", +] + +[[package]] +name = "lexical-parse-float" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52a9f232fbd6f550bc0137dcb5f99ab674071ac2d690ac69704593cb4abbea56" +dependencies = [ + "lexical-parse-integer", + "lexical-util", +] + +[[package]] +name = "lexical-parse-integer" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a7a039f8fb9c19c996cd7b2fcce303c1b2874fe1aca544edc85c4a5f8489b34" +dependencies = [ + "lexical-util", +] + +[[package]] +name = "lexical-util" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2604dd126bb14f13fb5d1bd6a66155079cb9fa655b37f875b3a742c705dbed17" + +[[package]] +name = "lexical-write-float" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "50c438c87c013188d415fbabbb1dceb44249ab81664efbd31b14ae55dabb6361" +dependencies = [ + "lexical-util", + "lexical-write-integer", +] + +[[package]] +name = "lexical-write-integer" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "409851a618475d2d5796377cad353802345cba92c867d9fbcde9cf4eac4e14df" +dependencies = [ + "lexical-util", +] + +[[package]] +name = "libc" +version = "0.2.186" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" + +[[package]] +name = "libm" +version = "0.2.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" + +[[package]] +name = "log" +version = "0.4.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" + +[[package]] +name = "memchr" +version = "2.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" + +[[package]] +name = "num-bigint" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a5e44f723f1133c9deac646763579fdb3ac745e418f2a7af9cd0c431da1f20b9" +dependencies = [ + "num-integer", + "num-traits", +] + +[[package]] +name = "num-complex" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "73f88a1307638156682bada9d7604135552957b7818057dcef22705b4d509495" +dependencies = [ + "num-traits", +] + +[[package]] +name = "num-integer" +version = "0.1.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +dependencies = [ + "num-traits", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", + "libm", +] + +[[package]] +name = "ojoutils" +version = "0.1.0" +dependencies = [ + "arrow", + "arrow_extendr", + "chrono", + "extendr-api", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "paste" +version = "1.0.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" + +[[package]] +name = "proc-macro2" +version = "1.0.106" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "readonly" +version = "0.2.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2a62d85ed81ca5305dc544bd42c8804c5060b78ffa5ad3c64b0fb6a8c13d062" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "regex" +version = "1.12.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e10754a14b9137dd7b1e3e5b0493cc9171fdd105e0ab477f51b72e7f3ac0e276" +dependencies = [ + "aho-corasick", + "memchr", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "regex-automata" +version = "0.4.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-syntax" +version = "0.8.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a" + +[[package]] +name = "rustc_version" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" +dependencies = [ + "semver", +] + +[[package]] +name = "rustversion" +version = "1.0.22" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "semver" +version = "1.0.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" + +[[package]] +name = "serde" +version = "1.0.228" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.228" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.228" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "serde_json" +version = "1.0.149" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "shlex" +version = "1.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" + +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + +[[package]] +name = "syn" +version = "2.0.117" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "tiny-keccak" +version = "2.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2c9d3793400a45f954c52e73d068316d76b6f4e36977e3fcebb13a2721e80237" +dependencies = [ + "crunchy", +] + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasip2" +version = "1.0.3+wasi-0.2.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "20064672db26d7cdc89c7798c48a0fdfac8213434a1186e5ef29fd560ae223d6" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wasm-bindgen" +version = "0.2.118" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0bf938a0bacb0469e83c1e148908bd7d5a6010354cf4fb73279b7447422e3a89" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.118" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eeff24f84126c0ec2db7a449f0c2ec963c6a49efe0698c4242929da037ca28ed" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.118" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d08065faf983b2b80a79fd87d8254c409281cf7de75fc4b773019824196c904" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.118" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd04d9e306f1907bd13c6361b5c6bfc7b3b3c095ed3f8a9246390f8dbdee129" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link", + "windows-result", + "windows-strings", +] + +[[package]] +name = "windows-implement" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "windows-interface" +version = "0.59.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link", +] + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "zerocopy" +version = "0.8.48" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eed437bf9d6692032087e337407a86f04cd8d6a16a37199ed57949d415bd68e9" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.48" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "zmij" +version = "1.0.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" diff --git a/src/rust/Cargo.toml b/src/rust/Cargo.toml new file mode 100644 index 0000000..fa899f3 --- /dev/null +++ b/src/rust/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = 'ojoutils' +publish = false +version = '0.1.0' +edition = '2021' +rust-version = '1.65' + +[lib] +crate-type = [ 'rlib', 'staticlib' ] +name = 'ojoutils' + +[[bin]] +name = 'document' +path = 'document.rs' + +[dependencies] +arrow_extendr = "58.0.1" +arrow = "58.0.1" +extendr-api = '0.9.0' +chrono = "0.4" + +[profile.release] +lto = true +codegen-units = 1 diff --git a/src/rust/document.rs b/src/rust/document.rs new file mode 100644 index 0000000..59d85d0 --- /dev/null +++ b/src/rust/document.rs @@ -0,0 +1,20 @@ +// Generated by extendr: Do not edit by hand +fn main() -> Result<(), Box> { + let wrapper_path = "../R/extendr-wrappers.R"; + let header = "\ + # Generated by extendr: Do not edit by hand\n\ + # nolint start\n\ + \n\ + #' @usage NULL\n\ + #' @useDynLib ojoutils, .registration = TRUE\n\ + NULL\n\ + \n\ + "; + let footer = "# nolint end\n"; + let wrappers = ojoutils::get_ojoutils_metadata() + .make_r_wrappers(true, "ojoutils") + .map_err(|e| format!("failed to generate wrappers: {e}"))?; + std::fs::write(wrapper_path, format!("{header}{wrappers}{footer}")) + .map_err(|e| format!("failed to write {wrapper_path}: {e}"))?; + Ok(()) +} diff --git a/src/rust/src/lib.rs b/src/rust/src/lib.rs new file mode 100644 index 0000000..524bd1e --- /dev/null +++ b/src/rust/src/lib.rs @@ -0,0 +1,368 @@ +use arrow::array::{Array, ArrayRef, Int32Array, Int64Array, RecordBatchReader, UInt32Array}; +use arrow::compute::{cast, concat, take}; +use arrow::datatypes::{DataType, Field, Schema}; +use arrow::error::ArrowError; +use arrow::ffi_stream::ArrowArrayStreamReader; +use arrow::record_batch::RecordBatch; +use arrow::row::{RowConverter, SortField}; +use arrow_extendr::{FromArrowRobj, IntoArrowRobj}; +use extendr_api::prelude::*; +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; + +fn extract_columns_from_stream( + mut reader: impl RecordBatchReader, + start_col_name: &str, + end_col_name: &str, + end_plus_one_col_name: &str, + group_col_names: &[&str], +) -> Result<(ArrayRef, ArrayRef, ArrayRef, Vec), ArrowError> { + let schema = reader.schema(); + + let start_idx = schema.index_of(start_col_name)?; + let end_idx = schema.index_of(end_col_name)?; + let end_plus_one_idx = schema.index_of(end_plus_one_col_name)?; + + let group_idxs: Vec = group_col_names + .iter() + .map(|name| schema.index_of(name)) + .collect::>()?; + + let init_state = ( + Vec::new(), + Vec::new(), + Vec::new(), + vec![Vec::new(); group_col_names.len()], + ); + + let (start_chunks, end_chunks, end_plus_one_chunks, group_chunks) = reader.try_fold( + init_state, + |(mut starts, mut ends, mut ends_plus_one, mut groups), + batch_result| + -> Result<_, ArrowError> { + let batch = batch_result?; + starts.push(batch.column(start_idx).clone()); + ends.push(batch.column(end_idx).clone()); + ends_plus_one.push(batch.column(end_plus_one_idx).clone()); + + groups + .iter_mut() + .zip(&group_idxs) + .for_each(|(bucket, &idx)| { + bucket.push(batch.column(idx).clone()); + }); + + Ok((starts, ends, ends_plus_one, groups)) + }, + )?; + + let contiguous_start = concat(&start_chunks.iter().map(|a| a.as_ref()).collect::>())?; + + let contiguous_end = concat(&end_chunks.iter().map(|a| a.as_ref()).collect::>())?; + + let contiguous_end_plus_one = concat( + &end_plus_one_chunks + .iter() + .map(|a| a.as_ref()) + .collect::>(), + )?; + + let contiguous_groups = group_chunks + .into_iter() + .map(|chunks| { + let refs: Vec<&dyn Array> = chunks.iter().map(|a| a.as_ref()).collect(); + concat(&refs) + }) + .collect::, ArrowError>>()?; + + Ok(( + contiguous_start, + contiguous_end, + contiguous_end_plus_one, + contiguous_groups, + )) +} + +fn densify_time_periods( + starts: &ArrayRef, + ends: &ArrayRef, + ends_plus_one: &ArrayRef, +) -> Result<(Vec, Vec, ArrayRef), ArrowError> { + let original_type = starts.data_type(); + + let starts_casted = cast(starts, &DataType::Int64)?; + let ends_casted = cast(ends, &DataType::Int64)?; + let ends_plus_one_casted = cast(ends_plus_one, &DataType::Int64)?; + + let starts_i64 = starts_casted.as_any().downcast_ref::().unwrap(); + let ends_i64 = ends_casted.as_any().downcast_ref::().unwrap(); + let ends_plus_one_i64 = ends_plus_one_casted + .as_any() + .downcast_ref::() + .unwrap(); + + let unique_vals_set: HashSet = starts_i64 + .iter() + .chain(ends_i64.iter()) + .chain(ends_plus_one_i64.iter()) + .flatten() + .collect(); + + let mut sorted_unique_vals: Vec = unique_vals_set.into_iter().collect(); + sorted_unique_vals.sort_unstable(); + + let val_to_idx: HashMap = sorted_unique_vals + .iter() + .enumerate() + .map(|(idx, &val)| (val, idx)) + .collect(); + + let start_indices: Vec = starts_i64 + .iter() + .map(|opt_val| { + opt_val + .and_then(|val| val_to_idx.get(&val).copied()) + .unwrap_or(usize::MAX) + }) + .collect(); + + let end_indices: Vec = ends_i64 + .iter() + .map(|opt_val| { + opt_val + .and_then(|val| val_to_idx.get(&val).copied()) + .unwrap_or(usize::MAX) + }) + .collect(); + + let unique_i64_array = std::sync::Arc::new(Int64Array::from(sorted_unique_vals)) as ArrayRef; + let unique_periods = cast(&unique_i64_array, original_type)?; + + Ok((start_indices, end_indices, unique_periods)) +} + +fn encode_group_ids( + group_cols: &[ArrayRef], + num_rows: usize, +) -> Result<(Vec, Vec, usize), ArrowError> { + if group_cols.is_empty() { + let group_ids = vec![0; num_rows]; + let first_occurrences = if num_rows > 0 { vec![0] } else { vec![] }; + let num_groups = if num_rows > 0 { 1 } else { 0 }; + return Ok((group_ids, first_occurrences, num_groups)); + } + + let sort_fields: Vec = group_cols + .iter() + .map(|col| SortField::new(col.data_type().clone())) + .collect(); + + let converter = RowConverter::new(sort_fields)?; + + let rows = converter.convert_columns(group_cols)?; + + let mut first_occurrences = Vec::new(); + let mut group_map = HashMap::new(); + + let group_ids: Vec = (0..num_rows) + .map(|i| { + let row_bytes = rows.row(i); + + *group_map.entry(row_bytes).or_insert_with(|| { + let new_id = first_occurrences.len(); + first_occurrences.push(i as u32); + new_id + }) + }) + .collect(); + + let num_groups = first_occurrences.len(); + + Ok((group_ids, first_occurrences, num_groups)) +} + +fn compute_difference_matrix( + start_indices: &[usize], + end_indices: &[usize], + group_ids: &[usize], + num_groups: usize, + num_periods: usize, + inclusive: &[bool], +) -> Vec { + let start_incl = *inclusive.get(0).unwrap(); + let end_incl = *inclusive.get(1).unwrap(); + + let mut matrix = vec![0; num_groups * num_periods]; + + // Boundary Pass + start_indices + .iter() + .zip(end_indices.iter()) + .zip(group_ids.iter()) + .for_each(|((&start_idx, &end_idx), &group_id)| { + let base_idx = group_id * num_periods; + + // Handle the start boundary (+1) + // usize::MAX is our sentinel for an R `NA` value + if start_idx != usize::MAX { + let s_idx = if start_incl { start_idx } else { start_idx + 1 }; + if s_idx < num_periods { + matrix[base_idx + s_idx] += 1; + } + } + + // Handle the end boundary (-1) + if end_idx != usize::MAX { + let e_idx = if end_incl { end_idx + 1 } else { end_idx }; + if e_idx < num_periods { + matrix[base_idx + e_idx] -= 1; + } + } + }); + + // Cumulative Sum Pass + matrix.chunks_exact_mut(num_periods).for_each(|group_row| { + let mut running_sum = 0; + + group_row.iter_mut().for_each(|cell| { + running_sum += *cell; + *cell = running_sum; + }); + }); + + matrix +} + +fn assemble_arrow_columns( + matrix: Vec, + num_periods: usize, + num_groups: usize, + first_occurrences: &[u32], + group_cols: &[ArrayRef], + unique_periods: &ArrayRef, +) -> Result<(ArrayRef, ArrayRef, Vec), ArrowError> { + let final_counts = Arc::new(Int32Array::from(matrix)) as ArrayRef; + + let date_indices: UInt32Array = (0..num_groups) + .flat_map(|_| 0..(num_periods as u32)) + .collect(); + + let final_dates = take(unique_periods, &date_indices, None)?; + + let group_indices: UInt32Array = first_occurrences + .iter() + .flat_map(|&idx| std::iter::repeat(idx).take(num_periods)) + .collect(); + + let final_groups: Result, ArrowError> = group_cols + .iter() + .map(|col| take(col, &group_indices, None)) + .collect(); + + Ok((final_dates, final_counts, final_groups?)) +} + +fn export_results( + date_array: ArrayRef, + date_name: &str, + count_array: ArrayRef, + count_name: &str, + group_arrays: Vec, + group_names: &[&str], +) -> Result { + let groups_iter = group_arrays + .into_iter() + .zip(group_names) + .map(|(arr, &name)| { + let field = Field::new(name, arr.data_type().clone(), true); + (field, arr) + }); + + let date_iter = std::iter::once({ + let field = Field::new(date_name, date_array.data_type().clone(), true); + (field, date_array) + }); + + let count_iter = std::iter::once({ + let field = Field::new(count_name, count_array.data_type().clone(), false); + (field, count_array) + }); + + let (fields, columns): (Vec, Vec) = + date_iter.chain(count_iter).chain(groups_iter).unzip(); + + let schema = Arc::new(Schema::new(fields)); + + let batch = RecordBatch::try_new(schema, columns).unwrap(); + + batch.into_arrow_robj() +} + +#[extendr] +fn count_interval_( + stream: Robj, + start: String, + end: String, + end_plus_one: String, + date_name: String, + count_name: String, + by: Vec, + inclusive: Logicals, +) -> Result { + let reader = match ArrowArrayStreamReader::from_arrow_robj(&stream) { + Ok(r) => r, + Err(e) => { + throw_r_error(format!("Failed to create Arrow stream reader: {}", e)); + } + }; + + let group_cols_str: Vec<&str> = by.iter().map(|s| s.as_str()).collect(); + + let (starts, ends, ends_plus_one, groups) = + extract_columns_from_stream(reader, &start, &end, &end_plus_one, &group_cols_str) + .expect("Stream extraction failed"); + + let num_rows = starts.len(); + + let (start_idx, end_idx, unique_periods) = + densify_time_periods(&starts, &ends, &ends_plus_one).unwrap(); + let num_periods = unique_periods.len(); + + let (group_ids, first_occurrences, num_groups) = encode_group_ids(&groups, num_rows).unwrap(); + + let inclusive_bools: Vec = inclusive.iter().map(|v| v.is_true()).collect(); + + let matrix = compute_difference_matrix( + &start_idx, + &end_idx, + &group_ids, + num_groups, + num_periods, + &inclusive_bools, + ); + + let (final_dates, final_counts, final_groups) = assemble_arrow_columns( + matrix, + num_periods, + num_groups, + &first_occurrences, + &groups, + &unique_periods, + ) + .unwrap(); + + export_results( + final_dates, + &date_name, + final_counts, + &count_name, + final_groups, + &group_cols_str, + ) +} + +// Macro to generate exports. +extendr_module! { + mod ojoutils; + fn count_interval_; +} diff --git a/src/rust/vendor-config.toml b/src/rust/vendor-config.toml new file mode 100644 index 0000000..0236928 --- /dev/null +++ b/src/rust/vendor-config.toml @@ -0,0 +1,5 @@ +[source.crates-io] +replace-with = "vendored-sources" + +[source.vendored-sources] +directory = "vendor" diff --git a/src/rust/vendor.tar.xz b/src/rust/vendor.tar.xz new file mode 100644 index 0000000..8f5eba8 Binary files /dev/null and b/src/rust/vendor.tar.xz differ diff --git a/tests/testthat/test-count_interval.R b/tests/testthat/test-count_interval.R new file mode 100644 index 0000000..db06d02 --- /dev/null +++ b/tests/testthat/test-count_interval.R @@ -0,0 +1,695 @@ +test_stays <- function() { + data.frame( + stay_start = as.POSIXct(c( + "2024-01-01", "2024-01-02", "2024-01-01", + "2024-01-03", "2024-01-05", "2024-01-10" + )), + stay_end = as.POSIXct(c( + "2024-01-03", "2024-01-04", "2024-01-02", + "2024-01-05", "2024-01-06", "2024-01-11" + )), + ward = c("A", "A", "B", "B", "A", "A"), + sex = c("M", "F", "M", "F", "M", "F") + ) +} + +test_that("works without groups", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-05")), + end = as.Date(c("2024-01-03", "2024-01-06")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_s3_class(res, "tbl_df") + expect_named(res, c("date", "n")) + expect_equal(nrow(res), 6) + expect_equal(res$n, c(1L, 1L, 1L, 0L, 1L, 1L)) +}) + +test_that("works with groups", { + df <- test_stays() + + res <- count_interval( + df, + start = "stay_start", + end = "stay_end", + period = "day", + .by = c("ward", "sex") + ) + + expect_s3_class(res, "tbl_df") + expect_named(res, c("date", "n", "ward", "sex")) +}) + +test_that("columns are ordered date, count, groups", { + df <- test_stays() + + res <- count_interval( + df, + start = "stay_start", + end = "stay_end", + period = "day", + .by = c("ward", "sex") + ) + + expect_equal(names(res), c("date", "n", "ward", "sex")) +}) + +test_that("custom date_name and count_name work", { + df <- test_stays() + + res <- count_interval( + df, + start = "stay_start", + end = "stay_end", + period = "day", + date_name = "day", + count_name = "pop", + .by = "ward" + ) + + expect_equal(names(res), c("day", "pop", "ward")) +}) + +test_that("gaps in calendar are filled with correct count", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-10")), + end = as.Date(c("2024-01-02", "2024-01-11")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + jan_7 <- res[res$date == as.Date("2024-01-07"), ] + expect_equal(nrow(jan_7), 1) + expect_equal(jan_7$n, 0L) +}) + +test_that("grouped gaps are filled correctly", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-10")), + end = as.Date(c("2024-01-02", "2024-01-11")), + grp = c("A", "A") + ) + + res <- count_interval( + df, + start = "start", + end = "end", + period = "day", + .by = "grp" + ) + + jan_7 <- res[res$date == as.Date("2024-01-07") & res$grp == "A", ] + expect_equal(jan_7$n, 0L) +}) + +test_that("inclusive boundaries include end date", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-03") + ) + + res <- count_interval( + df, + start = "start", + end = "end", + period = "day", + .inclusive = c(TRUE, TRUE) + ) + + expect_equal(res$n, c(1L, 1L, 1L)) +}) + +test_that("exclusive end boundary excludes end date", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-03") + ) + + res <- count_interval( + df, + start = "start", + end = "end", + period = "day", + .inclusive = c(TRUE, FALSE) + ) + + expect_equal(res$n, c(1L, 1L, 0L)) +}) + +test_that("NA start values are filled with global min", { + df <- data.frame( + start = as.Date(c(NA, "2024-01-03")), + end = as.Date(c("2024-01-02", "2024-01-04")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_true(all(res$n >= 0)) +}) + +test_that("NA end values are filled with global max", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-03")), + end = as.Date(c("2024-01-02", NA)) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_true(all(res$n >= 0)) +}) + +test_that("empty input returns empty tibble with correct columns", { + df <- data.frame( + start = as.Date(character()), + end = as.Date(character()) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_equal(nrow(res), 0) + expect_named(res, c("date", "n")) +}) + +test_that("missing required args errors", { + expect_error(count_interval(), "absent") + expect_error(count_interval(data.frame()), "absent") +}) + +test_that("missing columns error", { + df <- data.frame(a = 1) + expect_error( + count_interval(df, start = "missing", end = "a", period = "day"), + "not found" + ) +}) + +test_that(".by columns must exist", { + df <- data.frame(start = as.Date("2024-01-01"), end = as.Date("2024-01-02")) + expect_error( + count_interval(df, start = "start", end = "end", period = "day", .by = "nope"), + "not found" + ) +}) + +test_that("name collisions error", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-02"), + date = 1 + ) + expect_error( + count_interval(df, start = "start", end = "end", period = "day", date_name = "date", .by = "date"), + "date_name" + ) +}) + +test_that(".by cannot include start or end", { + df <- data.frame(start = as.Date("2024-01-01"), end = as.Date("2024-01-02")) + expect_error( + count_interval(df, start = "start", end = "end", period = "day", .by = "start"), + "`start` or `end`" + ) +}) + +test_that("invalid period errors", { + df <- data.frame(start = as.Date("2024-01-01"), end = as.Date("2024-01-02")) + expect_error( + count_interval(df, start = "start", end = "end", period = 123), + "period" + ) +}) + +test_that("invalid .fill errors", { + df <- data.frame(start = as.Date("2024-01-01"), end = as.Date("2024-01-02")) + expect_error( + count_interval(df, start = "start", end = "end", period = "day", .fill = "bad"), + "fill" + ) +}) + +test_that("invalid .inclusive errors", { + df <- data.frame(start = as.Date("2024-01-01"), end = as.Date("2024-01-02")) + expect_error( + count_interval(df, start = "start", end = "end", period = "day", .inclusive = TRUE), + "inclusive" + ) +}) + +test_that("minute period maps to min for seq()", { + df <- data.frame( + start = as.POSIXct(c("2024-01-01 00:00:00", "2024-01-01 00:05:00")), + end = as.POSIXct(c("2024-01-01 00:02:00", "2024-01-01 00:07:00")) + ) + + # Should not error + res <- count_interval(df, start = "start", end = "end", period = "minute") + expect_s3_class(res, "tbl_df") +}) + +test_that("second period maps to sec for seq()", { + df <- data.frame( + start = as.POSIXct("2024-01-01 00:00:00"), + end = as.POSIXct("2024-01-01 00:00:05") + ) + + res <- count_interval(df, start = "start", end = "end", period = "second") + expect_s3_class(res, "tbl_df") +}) + +test_that("overlapping stays in same group sum correctly", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + jan_2 <- res[res$date == as.Date("2024-01-02"), ] + expect_equal(jan_2$n, 2L) +}) + +test_that("hour period works correctly", { + df <- data.frame( + start = as.POSIXct(c("2024-01-01 00:00:00", "2024-01-01 02:00:00")), + end = as.POSIXct(c("2024-01-01 03:00:00", "2024-01-01 04:00:00")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "hour") + + expect_s3_class(res, "tbl_df") + expect_equal(nrow(res), 5) # 00:00 to 04:00 inclusive + expect_equal(res$n[res$date == as.POSIXct("2024-01-01 02:00:00")], 2L) +}) + +test_that("week period works correctly", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-15")), + end = as.Date(c("2024-01-14", "2024-01-28")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "week") + + expect_s3_class(res, "tbl_df") + # Should have dates floored to week start + expect_true(all(res$n >= 0)) +}) + +test_that("month period works correctly", { + df <- data.frame( + start = as.Date(c("2024-01-15", "2024-02-10")), + end = as.Date(c("2024-02-15", "2024-03-10")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "month") + + expect_s3_class(res, "tbl_df") + expect_true(all(res$n >= 0)) +}) + +test_that("quarter period works correctly", { + df <- data.frame( + start = as.Date(c("2024-01-15", "2024-04-15")), + end = as.Date(c("2024-03-15", "2024-06-15")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "quarter") + + expect_s3_class(res, "tbl_df") + expect_true(all(res$n >= 0)) +}) + +test_that("year period works correctly", { + df <- data.frame( + start = as.Date(c("2024-06-01", "2025-01-01")), + end = as.Date(c("2024-12-31", "2025-06-01")) + ) + + res <- count_interval(df, start = "start", end = "end", period = "year") + + expect_s3_class(res, "tbl_df") + expect_true(all(res$n >= 0)) +}) + +test_that("POSIXct with timezone works correctly", { + df <- data.frame( + start = as.POSIXct(c("2024-01-01 00:00:00", "2024-01-02 00:00:00"), tz = "America/New_York"), + end = as.POSIXct(c("2024-01-03 00:00:00", "2024-01-04 00:00:00"), tz = "America/New_York") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_s3_class(res, "tbl_df") + expect_equal(nrow(res), 4) +}) + +test_that("mixing Date and POSIXct columns is handled", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.POSIXct(c("2024-01-03 12:00:00", "2024-01-04 12:00:00")) + ) + + # This may or may not be supported, but shouldn't crash + expect_no_error( + count_interval(df, start = "start", end = "end", period = "day") + ) +}) + +test_that("single-day stays work with inclusive boundaries", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-01") + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", + .inclusive = c(TRUE, TRUE) + ) + + expect_equal(nrow(res), 1) + expect_equal(res$n, 1L) +}) + +test_that("single-day stays work with exclusive end boundary", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-01") + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", + .inclusive = c(TRUE, FALSE) + ) + + # With exclusive end, single-day stay should not count + expect_equal(nrow(res), 1) + expect_equal(res$n, 0L) +}) + +test_that("start > end is handled gracefully", { + df <- data.frame( + start = as.Date("2024-01-05"), + end = as.Date("2024-01-01") + ) + + # Should either error or return empty/zero results + expect_no_error( + res <- count_interval(df, start = "start", end = "end", period = "day") + ) +}) + +test_that("all NA values with no fill errors appropriately", { + df <- data.frame( + start = as.Date(c(NA, NA)), + end = as.Date(c(NA, NA)) + ) + + expect_error( + count_interval(df, start = "start", end = "end", period = "day"), + "All values" + ) +}) + +test_that("single row input works", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-03") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_equal(nrow(res), 3) + expect_true(all(res$n == 1L)) +}) + +test_that("very long date spans work correctly", { + df <- data.frame( + start = as.Date("2020-01-01"), + end = as.Date("2024-12-31") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + # Should span ~5 years of daily data + expect_true(nrow(res) >= 365 * 4) + expect_true(all(res$n == 1L)) +}) + +test_that("custom .fill$start value works", { + df <- data.frame( + start = as.Date(c(NA, "2024-01-03")), + end = as.Date(c("2024-01-02", "2024-01-04")) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", + .fill = list(start = as.Date("2024-01-01"), end = NULL) + ) + + expect_true(all(res$n >= 0)) + # The filled start should create population on 2024-01-01 + jan_1 <- res[res$date == as.Date("2024-01-01"), ] + expect_equal(nrow(jan_1), 1) +}) + +test_that("custom .fill$end value works", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-03")), + end = as.Date(c("2024-01-02", NA)) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", + .fill = list(start = NULL, end = as.Date("2024-01-05")) + ) + + expect_true(all(res$n >= 0)) + # Should extend to 2024-01-05 + jan_5 <- res[res$date == as.Date("2024-01-05"), ] + expect_equal(nrow(jan_5), 1) +}) + +test_that("explicit .fill values override data min/max", { + df <- data.frame( + start = as.Date(c(NA, "2024-01-05")), + end = as.Date(c("2024-01-04", "2024-01-06")) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", + .fill = list(start = as.Date("2024-01-01"), end = as.Date("2024-01-10")) + ) + + # With explicit fill, the NA should use the provided date + expect_true(all(res$n >= 0)) + # The filled start date should appear in results + expect_true(any(as.Date(res$date) == as.Date("2024-01-01"))) +}) + +test_that("factor grouping columns work correctly", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + group = factor(c("A", "B")) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "group" + ) + + expect_s3_class(res, "tbl_df") + expect_true("group" %in% names(res)) + # Factors should be preserved or converted to character + expect_true(is.character(res$group) || is.factor(res$group)) +}) + +test_that("groups with NA values are handled", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + group = c("A", NA) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "group" + ) + + expect_s3_class(res, "tbl_df") + # NA should be a valid group value + expect_true(any(is.na(res$group)) || nrow(res[res$group == "A", ]) > 0) +}) + +test_that("special characters in group names work", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + `group name` = c("A", "B with space"), + check.names = FALSE + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "group name" + ) + + expect_s3_class(res, "tbl_df") + expect_true("group name" %in% names(res)) +}) + +test_that("single group returns correct structure", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + group = "A" + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "group" + ) + + expect_s3_class(res, "tbl_df") + expect_true(all(res$group == "A")) +}) + +test_that("many groups work correctly", { + df <- data.frame( + start = rep(as.Date("2024-01-01"), 100), + end = rep(as.Date("2024-01-03"), 100), + group = sprintf("GROUP_%03d", 1:100) + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "group" + ) + + expect_equal(length(unique(res$group)), 100) + # Each group should have 3 days + expect_equal(nrow(res), 300) +}) + +test_that("output date column has correct type for Date input", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-03") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + # Date input produces Date output (may be converted to POSIXct internally) + expect_true(inherits(res$date, "Date") || inherits(res$date, "POSIXct")) +}) + +test_that("output date column has correct type for POSIXct input", { + df <- data.frame( + start = as.POSIXct("2024-01-01 00:00:00"), + end = as.POSIXct("2024-01-03 00:00:00") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + # POSIXct input should produce POSIXct output + expect_s3_class(res$date, "POSIXct") +}) + +test_that("output count column is integer", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-03") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_type(res$n, "integer") +}) + +test_that("row count matches expected calendar range", { + df <- data.frame( + start = as.Date("2024-01-01"), + end = as.Date("2024-01-10") + ) + + res <- count_interval(df, start = "start", end = "end", period = "day") + + expect_equal(nrow(res), 10) +}) + +test_that("complex multi-group scenario produces accurate counts", { + # Create a scenario with 3 groups and overlapping stays + df <- data.frame( + start = as.Date(c( + "2024-01-01", "2024-01-02", # Group A overlaps + "2024-01-01", "2024-01-05", # Group B separate stays + "2024-01-03" # Group C single stay + )), + end = as.Date(c( + "2024-01-03", "2024-01-04", # Group A: overlap on 01-02, 01-03 + "2024-01-02", "2024-01-06", # Group B: gap between stays + "2024-01-05" # Group C: spans 01-03 to 01-05 + )), + ward = c("A", "A", "B", "B", "C") + ) + + res <- count_interval( + df, start = "start", end = "end", period = "day", .by = "ward" + ) + + # Group A: days 1-4 with 2 overlapping on 2-3 + ward_a <- res[res$ward == "A", ] + expect_equal(ward_a$n[ward_a$date == as.POSIXct("2024-01-01")], 1L) + expect_equal(ward_a$n[ward_a$date == as.POSIXct("2024-01-02")], 2L) + expect_equal(ward_a$n[ward_a$date == as.POSIXct("2024-01-03")], 2L) + expect_equal(ward_a$n[ward_a$date == as.POSIXct("2024-01-04")], 1L) + + # Group B: days 1-2 and 5-6 (gap on 3-4) + ward_b <- res[res$ward == "B", ] + jan_3_b <- ward_b[ward_b$date == as.POSIXct("2024-01-03"), ] + expect_equal(nrow(jan_3_b), 1) + expect_equal(jan_3_b$n, 0L) + + # Group C: days 3-5 + ward_c <- res[res$ward == "C", ] + expect_equal(ward_c$n[ward_c$date == as.POSIXct("2024-01-03")], 1L) + expect_equal(ward_c$n[ward_c$date == as.POSIXct("2024-01-04")], 1L) + expect_equal(ward_c$n[ward_c$date == as.POSIXct("2024-01-05")], 1L) +}) + +test_that("NULL data errors", { + expect_error( + count_interval(NULL, start = "start", end = "end", period = "day"), + "NULL" + ) +}) + +test_that("non-data-frame input errors or handles gracefully", { + expect_error( + count_interval(list(start = 1, end = 2), start = "start", end = "end", period = "day") + ) +}) + +test_that("character columns instead of Date errors appropriately", { + df <- data.frame( + start = c("2024-01-01", "2024-01-02"), + end = c("2024-01-03", "2024-01-04") + ) + + # Should error or handle gracefully + expect_error( + count_interval(df, start = "start", end = "end", period = "day") + ) +}) + +test_that("count_name collision with .by columns errors", { + df <- data.frame( + start = as.Date(c("2024-01-01", "2024-01-02")), + end = as.Date(c("2024-01-03", "2024-01-04")), + ward = c("A", "B") + ) + + expect_error( + count_interval( + df, start = "start", end = "end", period = "day", + count_name = "ward", .by = "ward" + ), + "count_name" + ) +}) diff --git a/tools/config.R b/tools/config.R new file mode 100644 index 0000000..06d57cd --- /dev/null +++ b/tools/config.R @@ -0,0 +1,112 @@ +# Note: Any variables prefixed with `.` are used for text +# replacement in the Makevars.in and Makevars.win.in + +# check the packages MSRV first +source("tools/msrv.R") + +# check DEBUG and NOT_CRAN environment variables +env_debug <- Sys.getenv("DEBUG") +env_not_cran <- Sys.getenv("NOT_CRAN") + +# check if the vendored zip file exists +vendor_exists <- file.exists("src/rust/vendor.tar.xz") + +is_not_cran <- env_not_cran != "" +is_debug <- env_debug != "" + +if (is_debug) { + # if we have DEBUG then we set not cran to true + # CRAN is always release build + is_not_cran <- TRUE + message("Creating DEBUG build.") +} + +if (!is_not_cran) { + message("Building for CRAN.") +} + +# we set cran flags only if NOT_CRAN is empty and if +# the vendored crates are present. +.cran_flags <- ifelse( + !is_not_cran && vendor_exists, + "-j 2 --offline", + "" +) + +# when DEBUG env var is present we use `--debug` build +.profile <- ifelse(is_debug, "", "--release") +.clean_targets <- ifelse(is_debug, "", "$(TARGET_DIR)") + +# We specify this target when building for webR +webr_target <- "wasm32-unknown-emscripten" + +# here we check if the platform we are building for is webr +is_wasm <- identical(R.version$platform, webr_target) + +# print to terminal to inform we are building for webr +if (is_wasm) { + message("Building for WebR") +} + +# we check if we are making a debug build or not +# if so, the LIBDIR environment variable becomes: +# LIBDIR = $(TARGET_DIR)/{wasm32-unknown-emscripten}/debug +# this will be used to fill out the LIBDIR env var for Makevars.in +target_libpath <- if (is_wasm) "wasm32-unknown-emscripten" else NULL +cfg <- if (is_debug) "debug" else "release" + +# used to replace @LIBDIR@ +.libdir <- paste(c(target_libpath, cfg), collapse = "/") + +# use this to replace @TARGET@ +# we specify the target _only_ on webR +# there may be use cases later where this can be adapted or expanded +.target <- ifelse(is_wasm, paste0("--target=", webr_target), "") + +# add panic exports only for WASM builds +.panic_exports <- ifelse( + is_wasm, + "CARGO_PROFILE_DEV_PANIC=\"abort\" CARGO_PROFILE_RELEASE_PANIC=\"abort\" ", + "" +) + +# read in the Makevars.in file checking +is_windows <- .Platform[["OS.type"]] == "windows" + +# if windows we replace in the Makevars.win.in +mv_fp <- ifelse( + is_windows, + "src/Makevars.win.in", + "src/Makevars.in" +) + +# set the output file +mv_ofp <- ifelse( + is_windows, + "src/Makevars.win", + "src/Makevars" +) + +# delete the existing Makevars{.win/.wasm} +if (file.exists(mv_ofp)) { + message("Cleaning previous `", mv_ofp, "`.") + invisible(file.remove(mv_ofp)) +} + +# read as a single string +mv_txt <- readLines(mv_fp) + +# replace placeholder values +new_txt <- gsub("@CRAN_FLAGS@", .cran_flags, mv_txt) |> + gsub("@PROFILE@", .profile, x = _) |> + gsub("@CLEAN_TARGET@", .clean_targets, x = _) |> + gsub("@LIBDIR@", .libdir, x = _) |> + gsub("@TARGET@", .target, x = _) |> + gsub("@PANIC_EXPORTS@", .panic_exports, x = _) + +message("Writing `", mv_ofp, "`.") +con <- file(mv_ofp, open = "wb") +writeLines(new_txt, con, sep = "\n") +close(con) + +message("`tools/config.R` has finished.") diff --git a/tools/msrv.R b/tools/msrv.R new file mode 100644 index 0000000..59a61ab --- /dev/null +++ b/tools/msrv.R @@ -0,0 +1,116 @@ +# read the DESCRIPTION file +desc <- read.dcf("DESCRIPTION") + +if (!"SystemRequirements" %in% colnames(desc)) { + fmt <- c( + "`SystemRequirements` not found in `DESCRIPTION`.", + "Please specify `SystemRequirements: Cargo (Rust's package manager), rustc`" + ) + stop(paste(fmt, collapse = "\n")) +} + +# extract system requirements +sysreqs <- desc[, "SystemRequirements"] + +# check that cargo and rustc is found +if (!grepl("cargo", sysreqs, ignore.case = TRUE)) { + stop("You must specify `Cargo (Rust's package manager)` in your `SystemRequirements`") +} + +if (!grepl("rustc", sysreqs, ignore.case = TRUE)) { + stop("You must specify `Cargo (Rust's package manager), rustc` in your `SystemRequirements`") +} + +# split into parts +parts <- strsplit(sysreqs, ", ")[[1]] + +# identify which is the rustc +rustc_ver <- parts[grepl("rustc", parts)] + +# perform checks for the presence of rustc and cargo on the OS +no_cargo_msg <- c( + "----------------------- [CARGO NOT FOUND]--------------------------", + "The 'cargo' command was not found on the PATH. Please install Cargo", + "from: https://www.rust-lang.org/tools/install", + "", + "Alternatively, you may install Cargo from your OS package manager:", + " - Debian/Ubuntu: apt-get install cargo", + " - Fedora/CentOS: dnf install cargo", + " - macOS: brew install rust", + "-------------------------------------------------------------------" +) + +no_rustc_msg <- c( + "----------------------- [RUST NOT FOUND]---------------------------", + "The 'rustc' compiler was not found on the PATH. Please install", + paste(rustc_ver, "or higher from:"), + "https://www.rust-lang.org/tools/install", + "", + "Alternatively, you may install Rust from your OS package manager:", + " - Debian/Ubuntu: apt-get install rustc", + " - Fedora/CentOS: dnf install rustc", + " - macOS: brew install rust", + "-------------------------------------------------------------------" +) + +# Add {user}/.cargo/bin to path before checking +new_path <- paste0( + Sys.getenv("PATH"), + ":", + paste0(Sys.getenv("HOME"), "/.cargo/bin") +) + +# set the path with the new path +Sys.setenv("PATH" = new_path) + +# check for rustc installation +rustc_version <- tryCatch( + system("rustc --version", intern = TRUE), + error = function(e) { + stop(paste(no_rustc_msg, collapse = "\n")) + } +) + +# check for cargo installation +cargo_version <- tryCatch( + system("cargo --version", intern = TRUE), + error = function(e) { + stop(paste(no_cargo_msg, collapse = "\n")) + } +) + +# helper function to extract versions +extract_semver <- function(ver) { + if (grepl("\\d+\\.\\d+(\\.\\d+)?", ver)) { + sub(".*?(\\d+\\.\\d+(\\.\\d+)?).*", "\\1", ver) + } else { + NA + } +} + +# get the MSRV +msrv <- extract_semver(rustc_ver) + +# extract current version +current_rust_version <- extract_semver(rustc_version) + +# perform check +if (!is.na(msrv)) { + # -1 when current version is later + # 0 when they are the same + # 1 when MSRV is newer than current + is_msrv <- utils::compareVersion(msrv, current_rust_version) + if (is_msrv == 1) { + fmt <- paste0( + "\n------------------ [UNSUPPORTED RUST VERSION]------------------\n", + "- Minimum supported Rust version is %s.\n", + "- Installed Rust version is %s.\n", + "---------------------------------------------------------------" + ) + stop(sprintf(fmt, msrv, current_rust_version)) + } +} + +# print the versions +versions_fmt <- "Using %s\nUsing %s" +message(sprintf(versions_fmt, cargo_version, rustc_version))