diff --git a/DESCRIPTION b/DESCRIPTION
index 32ad97a..66988e1 100644
--- a/DESCRIPTION
+++ b/DESCRIPTION
@@ -7,7 +7,6 @@ Description: What the package does (one paragraph).
License: MIT + file LICENSE
Encoding: UTF-8
Roxygen: list(markdown = TRUE)
-RoxygenNote: 7.3.2
URL: https://birdnet-team.github.io/birdnetTools/, https://github.com/birdnet-team/birdnetTools
BugReports: https://github.com/birdnet-team/birdnetTools/issues
Imports:
@@ -28,13 +27,16 @@ Imports:
shinyFiles,
shinyWidgets,
stringr,
+ tidyr,
tuneR
Suggests:
knitr,
rmarkdown,
- testthat (>= 3.0.0)
+ testthat (>= 3.0.0),
+ withr
Config/testthat/edition: 3
VignetteBuilder: knitr
Depends:
R (>= 4.1.0)
LazyData: true
+Config/roxygen2/version: 8.0.0
diff --git a/NAMESPACE b/NAMESPACE
index 8981889..2908a56 100644
--- a/NAMESPACE
+++ b/NAMESPACE
@@ -3,7 +3,9 @@
export(birdnet_add_datetime)
export(birdnet_calc_threshold)
export(birdnet_combine)
+export(birdnet_detection_history)
export(birdnet_filter)
+export(birdnet_get_effort)
export(birdnet_heatmap)
export(birdnet_launch_validation)
export(birdnet_subsample)
@@ -18,6 +20,7 @@ importFrom(bslib,sidebar)
importFrom(cli,cli_alert_success)
importFrom(cli,cli_alert_warning)
importFrom(cli,cli_li)
+importFrom(dplyr,.data)
importFrom(dplyr,bind_rows)
importFrom(dplyr,mutate)
importFrom(dplyr,select)
diff --git a/R/birdnet_detection_history.R b/R/birdnet_detection_history.R
new file mode 100644
index 0000000..184203e
--- /dev/null
+++ b/R/birdnet_detection_history.R
@@ -0,0 +1,222 @@
+#' Generate Detection History Matrix, Effort Matrix, and Summary for Occupancy Modeling
+#'
+#' Summarizes BirdNET detection data across specified survey intervals (occasions),
+#' filters sites based on minimal detection persistence thresholds, and aligns them
+#' with operational effort data. Returns a zero-filled site-by-occasion binary matrix,
+#' an identical matching matrix documenting sampling effort intensity for modeling
+#' detection probability covariates, and a detailed long-format data frame summary.
+#'
+#' @details
+#' The function groups continuous temporal data into distinct survey blocks using
+#' `lubridate::floor_date()`. Detections are cross-referenced against your
+#' `effort_data`: occasions where monitoring effort occurred but no target
+#' species were detected are explicitly zero-filled. If an ARU was not operational
+#' during a specific time block, it is preserved as an `NA` value in the detection
+#' history to ensure structural integrity for missing-visit designs.
+#'
+#' Values greater than 0 in the final detection matrix are collapsed to `1` to format
+#' the output for binary presence/absence occupancy models (e.g., `spOccupancy`, `unmarked`).
+#'
+#' @param data A data frame containing BirdNET detections, including column matches
+#' for filepaths and prediction confidence scores.
+#' @param effort_data A data frame containing monitoring operational effort,
+#' requiring at least `site` and `date` columns to indicate the active
+#' monitoring windows and locations of each ARU device. Users can generate
+#' this via [birdnet_get_effort()], which derives effort data from a directory
+#' of audio files by defining a site-date combination as "active" if at least
+#' one recording exists. If an `n_files` column is present, file counts will
+#' be aggregated per survey occasion block.
+#' @param survey_interval A character string specifying the temporal unit for
+#' grouping survey occasions (e.g., `"1 day"`, `"1 week"`, `"7 days"`).
+#' Passed directly to \code{\link[lubridate:floor_date]{lubridate::floor_date()}}.
+#' @param i An integer specifying the path hierarchy index for extracting site IDs.
+#' Passed directly to \code{\link{birdnet_add_site}}. Defaults to `-2`.
+#' @param min_unique_days An integer specifying the threshold of unique calendar days
+#' a site must possess raw detections on to be kept. Sites with detections spanning fewer
+#' than `min_unique_days` are dropped early from compilation. Defaults to `1`.
+#'
+#' @return A named `list` containing three components:
+#' \describe{
+#' \item{detection_history}{A numeric base R `matrix` where rows represent
+#' unique sites (assigned as row names), columns represent chronological temporal
+#' occasions, and cells indicate binary occupancy integers (`1`, `0`,
+#' or `NA` for missing effort).}
+#' \item{effort_matrix}{A numeric base R `matrix` matching the exact dimensions and
+#' sorting order of `detection_history`. If `n_files` was present in the effort data,
+#' cells represent total file counts per site-occasion. Otherwise, cells contain binary
+#' integers indicating presence (`1`) or absence (`0`) of operational effort.}
+#' \item{detection_summary}{A data frame in long format containing the underlying
+#' aggregated metrics per site/occasion, including detection counts (`n_detections`),
+#' maximum verification confidence (`max_conf`), and the file path of the
+#' highest confidence detection (`max_conf_audio`).}
+#' }
+#'
+#' @importFrom dplyr .data
+#' @export
+birdnet_detection_history <- function(data,
+ effort_data,
+ survey_interval,
+ i = -2,
+ min_unique_days = 1) {
+
+
+ # argument check ----------------------------------------------------------
+
+ # 1. Check data is a data frame with required columns
+ checkmate::assert_data_frame(data)
+
+ cols <- birdnet_detect_columns(data)
+ required_cols <- c("confidence", "filepath")
+ missing_cols <- required_cols[is.na(cols[required_cols])]
+
+ if (length(missing_cols) > 0) {
+ rlang::abort(
+ paste0(
+ "The input data is missing required BirdNET columns: ",
+ paste(missing_cols, collapse = ", "),
+ ". Please provide a valid BirdNET output data frame."
+ )
+ )
+ }
+
+
+ # 2. Check effort_data is a data frame with required columns
+ checkmate::assert_data_frame(effort_data)
+
+ effort_cols <- colnames(effort_data)
+ required_effort_cols <- c("site", "date")
+ missing_effort_cols <- setdiff(required_effort_cols, effort_cols)
+
+ if (length(missing_effort_cols) > 0) {
+ rlang::abort(
+ paste0(
+ "The input effort data is missing required columns: ",
+ paste(missing_effort_cols, collapse = ", "),
+ ". Please provide a valid effort data frame."
+ )
+ )
+ }
+
+ # 3. Check survey_interval is a character string following lubridate units
+ checkmate::assert_string(survey_interval, min.chars = 1)
+ if (!stringr::str_detect(survey_interval, "^\\d*\\s*(day|week|month|year|hour|minute)s?$")) {
+ rlang::abort(
+ paste0(
+ "`survey_interval` must be a valid lubridate unit string (e.g., '1 day', '2 weeks'). ",
+ "You provided: '", survey_interval, "'."
+ )
+ )
+ }
+
+ # 4. Check i is an integer
+ checkmate::assert_int(i, tol = 0)
+
+ # 5. Check min_unique_days is a positive integer
+ checkmate::assert_int(min_unique_days, lower = 1, tol = 0)
+
+
+
+
+
+ # main function -----------------------------------------------------------
+
+ cols <- birdnet_detect_columns(data)
+
+ # 1. Summarize detections by site and occasion block
+ detections_summarized <- data |>
+ birdnet_add_site(i = i) |>
+ birdnet_add_datetime() |>
+ # filter to only include sites with detections from more than n days
+ dplyr::group_by(.data$site) |>
+ dplyr::filter(dplyr::n_distinct(.data$date) >= min_unique_days) |>
+ dplyr::ungroup() |>
+ # group detections into survey occasions based on the specified interval
+ dplyr::mutate(occasion = lubridate::floor_date(x = .data$date,
+ unit = survey_interval)) |>
+ dplyr::group_by(.data$site, .data$occasion) |>
+ dplyr::summarise(n_detections = dplyr::n(),
+ max_conf = max(.data[[cols$confidence]], na.rm = TRUE),
+ max_conf_audio = .data[[cols$filepath]][which.max(.data[[cols$confidence]])],
+ .groups = "drop")
+
+
+
+ # 2. Process effort
+ baseline_effort <- effort_data |>
+ dplyr::mutate(occasion = lubridate::floor_date(x = .data$date,
+ unit = survey_interval))
+ if ("n_files" %in% names(baseline_effort)) {
+ # if n_files exists, aggregate the total file counts per site/occasion
+ baseline_effort <- baseline_effort |>
+ dplyr::group_by(.data$site, .data$occasion) |>
+ dplyr::summarise(n_files = sum(.data$n_files, na.rm = TRUE), .groups = "drop")
+ } else {
+ # if n_files is missing, simply keep unique combinations of site and occasion
+ baseline_effort <- baseline_effort |>
+ dplyr::distinct(.data$site, .data$occasion)
+ }
+
+
+
+ # 3. join detections, fill zeros, and pivot wide
+ detections_zero_filled <- baseline_effort |>
+ # left join ensures we only evaluate occasions where the devices were running
+ dplyr::left_join(detections_summarized, by = c("site", "occasion")) |>
+
+ # differentiate true zeros from missing effort
+ dplyr::mutate(n_detections = tidyr::replace_na(.data$n_detections, 0),
+ max_conf = tidyr::replace_na(.data$max_conf, 0),
+ max_conf_audio = tidyr::replace_na(.data$max_conf_audio, "none"))
+
+
+
+ # 4. Creating matrix structure: pivot to wide format with sites as rows and occasions as columns
+
+ # isolate matrix structure and shape wide for modeling packages (e.g., unmarked and spOccupancy)
+ detection_history_df <- detections_zero_filled |>
+ dplyr::select("site", "occasion", "n_detections") |>
+ # manipulate n_detections column to make it 1 if it's larger than 0,
+ # otherwise 0 (for occupancy modeling)
+ dplyr::mutate(n_detections = ifelse(.data$n_detections > 0, 1, 0)) |>
+ dplyr::arrange(.data$occasion, .data$site) |>
+ tidyr::pivot_wider(id_cols = "site",
+ names_from = "occasion",
+ values_from = "n_detections",
+ values_fill = NA)
+
+ detection_history <- as.matrix(detection_history_df[, -1])
+ rownames(detection_history) <- detection_history_df$site
+
+
+
+ # 5. Create the effort matrix for detection probability purpose
+ baseline_effort <- baseline_effort |>
+ dplyr::arrange(.data$occasion, .data$site)
+
+ if ("n_files" %in% names(baseline_effort)) {
+ # if n_files exists, we can use the file counts as a measure of effort
+ wide_effort <- baseline_effort |>
+ tidyr::pivot_wider(id_cols = "site",
+ names_from = "occasion",
+ values_from = "n_files",
+ values_fill = 0)
+ } else {
+ # if n_files is missing, we can only indicate presence (1) or absence (0) of effort
+ wide_effort <- baseline_effort |>
+ tidyr::pivot_wider(id_cols = "site",
+ names_from = "occasion",
+ values_from = "occasion",
+ values_fn = \(x) 1,
+ values_fill = 0)
+ }
+
+ # Strip the site column to create a clean matrix, keeping site names as rownames
+ effort_matrix <- as.matrix(wide_effort[, -1])
+ rownames(effort_matrix) <- wide_effort$site
+
+
+ return(list("detection_history" = detection_history,
+ "effort_matrix" = effort_matrix,
+ "detection_summary" = detections_zero_filled))
+}
+
diff --git a/R/birdnet_get_effort.R b/R/birdnet_get_effort.R
new file mode 100644
index 0000000..e06d8e3
--- /dev/null
+++ b/R/birdnet_get_effort.R
@@ -0,0 +1,68 @@
+#' Calculate recording effort by site and date
+#'
+#' Scans a directory for all audio files, extracts site and datetime
+#' metadata from their paths/filenames, and returns a unique timeline
+#' of recording effort with file counts.
+#'
+#' The function identifies audio files matching common extensions, automatically
+#' detects site names via [birdnet_add_site()], parses dates via
+#' [birdnet_add_datetime()], and reduces the output to a unique combination
+#' of sites and dates, summarizing total files recorded.
+#'
+#' @param path A character string specifying the path to the directory
+#' containing the audio files.
+#' @param i An integer specifying the index of the path element to extract
+#' as the site identifier when split by slashes. Defaults to `-2`, which
+#' corresponds to the immediate parent directory of the file, passed directly
+#' to [birdnet_add_site()]. Negative values count from the right-hand side.
+#'
+#' @return A tibble (data frame) with three columns:
+#' \describe{
+#' \item{site}{The extracted site identifier.}
+#' \item{date}{The date on which recording effort occurred.}
+#' \item{n_files}{Integer. The total number of audio files recorded at that
+#' site on that specific date.}
+#' }
+#'
+#' @examples
+#' \dontrun{
+#' effort_df <- birdnet_get_effort("path/to/audio/storage", i = -2)
+#' head(effort_df)
+#' }
+#'
+#' @export
+birdnet_get_effort <- function(path, i = -2) {
+
+
+ # argument check ----------------------------------------------------------
+
+ # 1. Check path is a single, valid directory path string
+ checkmate::assert_string(path, min.chars = 1)
+ checkmate::assert_directory_exists(path, access = "r")
+
+
+ # 2. Check i is an integer
+ checkmate::assert_int(i, tol = 0)
+
+
+ # main function -----------------------------------------------------------
+
+ # find files and build the effort dataframe
+ effort <- path |>
+ # list the file names
+ list.files(recursive = TRUE,
+ full.names = TRUE,
+ pattern = "\\.(wav|mp3|m4a|flac|ogg|wma)$",
+ ignore.case = TRUE) |>
+ # convert to tibble for processing, extract time and location
+ (\(x) dplyr::as_tibble(data.frame(filepath = x, stringsAsFactors = FALSE)))() |>
+ birdnet_add_datetime() |>
+ birdnet_add_site(i = i) |>
+
+ # keep only the relevant columns and unique rows
+ dplyr::summarise(n_files = dplyr::n(), .by = c("site", "date"))
+
+
+ return(effort)
+}
+
diff --git a/R/data_documentation.R b/R/data_documentation.R
index bc8a880..bf89ea3 100644
--- a/R/data_documentation.R
+++ b/R/data_documentation.R
@@ -32,3 +32,34 @@
#' @source
"example_jprf_2023"
+
+
+#' Example monitoring effort table from John Prince Research Forest
+#'
+#' A sample operational effort table mapping the active recording history of
+#' Autonomous Recording Units (ARUs) deployed across 5 sites in John Prince
+#' Research Forest, British Columbia, Canada, during May–June 2023. This data
+#' documents the baseline monitoring effort, where a given location and date
+#' combination is associated with an active ARU device if one or more audio files
+#' were successfully recorded.
+#'
+#' This dataset acts as the operational counterpart to `example_jprf_2023` and is
+#' useful for demonstrating workflow alignment between species detections and true
+#' field effort, specifically for zero-filling non-detections in occupancy modeling.
+#'
+#' @details
+#' This dataset was generated directly using the [birdnet_get_effort()] function.
+#' For more details on the generation parameters, data constraints, and internal
+#' file processing pipelines, please refer to the function documentation.
+#'
+#' @format ## `effort_jprf_2023`
+#' A data frame with rows and columns detailing active recording days. Key columns include:
+#' \describe{
+#' \item{site}{Character string indicating the unique identifier for the ARU deployment location}
+#' \item{date}{Date object representing the calendar day of monitoring effort}
+#' \item{n_files}{Integer representing the total number of audio files recorded at that location on that day}
+#' }
+#'
+#' @source
+"effort_jprf_2023"
+
diff --git a/R/utils_column_editing.R b/R/utils_column_editing.R
index 56fb7d1..e7723b8 100644
--- a/R/utils_column_editing.R
+++ b/R/utils_column_editing.R
@@ -51,7 +51,7 @@ birdnet_add_datetime <- function(
# parase the column name to the datetime format
dplyr::mutate(
- datetime = basename(data[[cols$filepath]]) |>
+ datetime = basename(.data[[cols$filepath]]) |>
stringr::str_extract("\\d{8}.\\d{6}") |>
lubridate::parse_date_time(
orders = c("ymd_HMS", "ymd-HMS", "ymdHMS", "ymd HM"),
@@ -102,6 +102,65 @@ birdnet_drop_datetime <- function(data) {
+#' Add site column from BirdNET output filenames
+#'
+#' Extracts a specific directory level from the file path column (automatically
+#' detected) to act as a site identifier, then adds a `site` column to the
+#' input data frame.
+#'
+#' The function uses [birdnet_detect_columns] to find the column containing
+#' file paths based on common name patterns. By default, it looks at the
+#' immediate parent folder of the file.
+#'
+#' @param data A data frame containing BirdNET output.
+#' @param i An integer specifying the index of the path element to extract
+#' when split by slashes. Defaults to `-2`, which corresponds to the immediate
+#' parent directory of the file (e.g., extracting "Site-A" from
+#' "path/to/Site-A/audio.wav"). Negative values count from the right-hand side.
+#'
+#' @return A data frame with an additional column:
+#' \describe{
+#' \item{site}{The extracted directory or folder name used as the site identifier.}
+#' }
+#'
+#' @examples
+#' \dontrun{
+#' combined_data <- birdnet_combine("path/to/BirdNET/output")
+#' data_with_site <- birdnet_add_site(combined_data, i = -2)
+#' }
+#'
+#' @keywords internal
+birdnet_add_site <- function(data,
+ i = -2) {
+
+ # argument check ----------------------------------------------------------
+
+ # detect columns
+ cols <- birdnet_detect_columns(data)
+
+
+ # ensure filepath column was actually found
+ if (is.null(cols$filepath) || !(cols$filepath %in% colnames(data))) {
+ stop("Could not automatically detect a valid file path column in the data.")
+ }
+
+ # main function -----------------------------------------------------------
+
+ # extract site safely using tidy evaluation data masking
+ data_with_site <- data |>
+ dplyr::mutate(
+ site = stringr::str_split_i(.data[[cols$filepath]], "[/\\\\]", i = i)
+ )
+
+ return(data_with_site)
+}
+
+
+
+
+
+
+
#' Clean and standardize column names
diff --git a/data-raw/effort_jprf_2023.R b/data-raw/effort_jprf_2023.R
new file mode 100644
index 0000000..ce67cbb
--- /dev/null
+++ b/data-raw/effort_jprf_2023.R
@@ -0,0 +1,11 @@
+## code to prepare `effort_jprf_2023` dataset goes here
+
+
+effort_jprf_2023 <- birdnet_get_effort("D:/Audio/2023_passerine") %>%
+ # remove the "_1" suffix from site names, which indicates the recording session at each site
+ mutate(site = str_replace(site, "_1$", "")) %>%
+ # filter to the main breeding season (May 1 to August 31)
+ filter(date >= "2023-05-01" & date <= "2023-08-31")
+
+
+usethis::use_data(effort_jprf_2023, overwrite = TRUE)
diff --git a/data-raw/example_jprf_2023.R b/data-raw/example_jprf_2023.R
index 9168129..7da2db3 100644
--- a/data-raw/example_jprf_2023.R
+++ b/data-raw/example_jprf_2023.R
@@ -2,7 +2,20 @@
library(tidyverse)
library(here)
+library(birdnetTools)
-example_jprf_2023 <- read_csv(here("data-raw", "example_jprf_2023.csv"))
+example_jprf_2023 <- read_csv(here("data-raw", "example_jprf_2023.csv")) %>%
+ birdnet_filter(species = c("Swainson's Thrush",
+ "American Robin",
+ "Pacific-slope Flycatcher",
+ "Pacific Wren",
+ "Varied Thrush",
+ "American Crow",
+ "Yellow-rumped Warbler",
+ "White-throated Sparrow",
+ "Olive-sided Flycatcher",
+ "Wilson's Warbler",
+ "Orange-crowned Warbler",
+ "Red-breasted Nuthatch"))
usethis::use_data(example_jprf_2023, overwrite = TRUE)
diff --git a/data/effort_jprf_2023.rda b/data/effort_jprf_2023.rda
new file mode 100644
index 0000000..d94bb71
Binary files /dev/null and b/data/effort_jprf_2023.rda differ
diff --git a/data/example_jprf_2023.rda b/data/example_jprf_2023.rda
index d19cbe3..ea342b9 100644
Binary files a/data/example_jprf_2023.rda and b/data/example_jprf_2023.rda differ
diff --git a/man/birdnetTools-package.Rd b/man/birdnetTools-package.Rd
index 2fc1952..9befaf9 100644
--- a/man/birdnetTools-package.Rd
+++ b/man/birdnetTools-package.Rd
@@ -22,5 +22,10 @@ Useful links:
\author{
\strong{Maintainer}: Sunny Tseng \email{sunnyyctseng@gmail.com} (\href{https://orcid.org/0000-0002-8621-2244}{ORCID})
+Authors:
+\itemize{
+ \item Sunny Tseng \email{sunnyyctseng@gmail.com} (\href{https://orcid.org/0000-0002-8621-2244}{ORCID})
+}
+
}
\keyword{internal}
diff --git a/man/birdnet_add_site.Rd b/man/birdnet_add_site.Rd
new file mode 100644
index 0000000..1d3ce38
--- /dev/null
+++ b/man/birdnet_add_site.Rd
@@ -0,0 +1,40 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/utils_column_editing.R
+\name{birdnet_add_site}
+\alias{birdnet_add_site}
+\title{Add site column from BirdNET output filenames}
+\usage{
+birdnet_add_site(data, i = -2)
+}
+\arguments{
+\item{data}{A data frame containing BirdNET output.}
+
+\item{i}{An integer specifying the index of the path element to extract
+when split by slashes. Defaults to \code{-2}, which corresponds to the immediate
+parent directory of the file (e.g., extracting "Site-A" from
+"path/to/Site-A/audio.wav"). Negative values count from the right-hand side.}
+}
+\value{
+A data frame with an additional column:
+\describe{
+\item{site}{The extracted directory or folder name used as the site identifier.}
+}
+}
+\description{
+Extracts a specific directory level from the file path column (automatically
+detected) to act as a site identifier, then adds a \code{site} column to the
+input data frame.
+}
+\details{
+The function uses \link{birdnet_detect_columns} to find the column containing
+file paths based on common name patterns. By default, it looks at the
+immediate parent folder of the file.
+}
+\examples{
+\dontrun{
+combined_data <- birdnet_combine("path/to/BirdNET/output")
+data_with_site <- birdnet_add_site(combined_data, i = -2)
+}
+
+}
+\keyword{internal}
diff --git a/man/birdnet_detection_history.Rd b/man/birdnet_detection_history.Rd
new file mode 100644
index 0000000..799016d
--- /dev/null
+++ b/man/birdnet_detection_history.Rd
@@ -0,0 +1,72 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/birdnet_detection_history.R
+\name{birdnet_detection_history}
+\alias{birdnet_detection_history}
+\title{Generate Detection History Matrix, Effort Matrix, and Summary for Occupancy Modeling}
+\usage{
+birdnet_detection_history(
+ data,
+ effort_data,
+ survey_interval,
+ i = -2,
+ min_unique_days = 1
+)
+}
+\arguments{
+\item{data}{A data frame containing BirdNET detections, including column matches
+for filepaths and prediction confidence scores.}
+
+\item{effort_data}{A data frame containing monitoring operational effort,
+requiring at least \code{site} and \code{date} columns to indicate the active
+monitoring windows and locations of each ARU device. Users can generate
+this via \code{\link[=birdnet_get_effort]{birdnet_get_effort()}}, which derives effort data from a directory
+of audio files by defining a site-date combination as "active" if at least
+one recording exists. If an \code{n_files} column is present, file counts will
+be aggregated per survey occasion block.}
+
+\item{survey_interval}{A character string specifying the temporal unit for
+grouping survey occasions (e.g., \code{"1 day"}, \code{"1 week"}, \code{"7 days"}).
+Passed directly to \code{\link[lubridate:floor_date]{lubridate::floor_date()}}.}
+
+\item{i}{An integer specifying the path hierarchy index for extracting site IDs.
+Passed directly to \code{\link{birdnet_add_site}}. Defaults to \code{-2}.}
+
+\item{min_unique_days}{An integer specifying the threshold of unique calendar days
+a site must possess raw detections on to be kept. Sites with detections spanning fewer
+than \code{min_unique_days} are dropped early from compilation. Defaults to \code{1}.}
+}
+\value{
+A named \code{list} containing three components:
+\describe{
+\item{detection_history}{A numeric base R \code{matrix} where rows represent
+unique sites (assigned as row names), columns represent chronological temporal
+occasions, and cells indicate binary occupancy integers (\code{1}, \code{0},
+or \code{NA} for missing effort).}
+\item{effort_matrix}{A numeric base R \code{matrix} matching the exact dimensions and
+sorting order of \code{detection_history}. If \code{n_files} was present in the effort data,
+cells represent total file counts per site-occasion. Otherwise, cells contain binary
+integers indicating presence (\code{1}) or absence (\code{0}) of operational effort.}
+\item{detection_summary}{A data frame in long format containing the underlying
+aggregated metrics per site/occasion, including detection counts (\code{n_detections}),
+maximum verification confidence (\code{max_conf}), and the file path of the
+highest confidence detection (\code{max_conf_audio}).}
+}
+}
+\description{
+Summarizes BirdNET detection data across specified survey intervals (occasions),
+filters sites based on minimal detection persistence thresholds, and aligns them
+with operational effort data. Returns a zero-filled site-by-occasion binary matrix,
+an identical matching matrix documenting sampling effort intensity for modeling
+detection probability covariates, and a detailed long-format data frame summary.
+}
+\details{
+The function groups continuous temporal data into distinct survey blocks using
+\code{lubridate::floor_date()}. Detections are cross-referenced against your
+\code{effort_data}: occasions where monitoring effort occurred but no target
+species were detected are explicitly zero-filled. If an ARU was not operational
+during a specific time block, it is preserved as an \code{NA} value in the detection
+history to ensure structural integrity for missing-visit designs.
+
+Values greater than 0 in the final detection matrix are collapsed to \code{1} to format
+the output for binary presence/absence occupancy models (e.g., \code{spOccupancy}, \code{unmarked}).
+}
diff --git a/man/birdnet_get_effort.Rd b/man/birdnet_get_effort.Rd
new file mode 100644
index 0000000..090ce8c
--- /dev/null
+++ b/man/birdnet_get_effort.Rd
@@ -0,0 +1,44 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/birdnet_get_effort.R
+\name{birdnet_get_effort}
+\alias{birdnet_get_effort}
+\title{Calculate recording effort by site and date}
+\usage{
+birdnet_get_effort(path, i = -2)
+}
+\arguments{
+\item{path}{A character string specifying the path to the directory
+containing the audio files.}
+
+\item{i}{An integer specifying the index of the path element to extract
+as the site identifier when split by slashes. Defaults to \code{-2}, which
+corresponds to the immediate parent directory of the file, passed directly
+to \code{\link[=birdnet_add_site]{birdnet_add_site()}}. Negative values count from the right-hand side.}
+}
+\value{
+A tibble (data frame) with three columns:
+\describe{
+\item{site}{The extracted site identifier.}
+\item{date}{The date on which recording effort occurred.}
+\item{n_files}{Integer. The total number of audio files recorded at that
+site on that specific date.}
+}
+}
+\description{
+Scans a directory for all audio files, extracts site and datetime
+metadata from their paths/filenames, and returns a unique timeline
+of recording effort with file counts.
+}
+\details{
+The function identifies audio files matching common extensions, automatically
+detects site names via \code{\link[=birdnet_add_site]{birdnet_add_site()}}, parses dates via
+\code{\link[=birdnet_add_datetime]{birdnet_add_datetime()}}, and reduces the output to a unique combination
+of sites and dates, summarizing total files recorded.
+}
+\examples{
+\dontrun{
+effort_df <- birdnet_get_effort("path/to/audio/storage", i = -2)
+head(effort_df)
+}
+
+}
diff --git a/man/effort_jprf_2023.Rd b/man/effort_jprf_2023.Rd
new file mode 100644
index 0000000..8ec7e11
--- /dev/null
+++ b/man/effort_jprf_2023.Rd
@@ -0,0 +1,41 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/data_documentation.R
+\docType{data}
+\name{effort_jprf_2023}
+\alias{effort_jprf_2023}
+\title{Example monitoring effort table from John Prince Research Forest}
+\format{
+\subsection{\code{effort_jprf_2023}}{
+
+A data frame with rows and columns detailing active recording days. Key columns include:
+\describe{
+\item{site}{Character string indicating the unique identifier for the ARU deployment location}
+\item{date}{Date object representing the calendar day of monitoring effort}
+\item{n_files}{Integer representing the total number of audio files recorded at that location on that day}
+}
+}
+}
+\source{
+\url{https://sunnytseng.ca/}
+}
+\usage{
+effort_jprf_2023
+}
+\description{
+A sample operational effort table mapping the active recording history of
+Autonomous Recording Units (ARUs) deployed across 5 sites in John Prince
+Research Forest, British Columbia, Canada, during May–June 2023. This data
+documents the baseline monitoring effort, where a given location and date
+combination is associated with an active ARU device if one or more audio files
+were successfully recorded.
+}
+\details{
+This dataset acts as the operational counterpart to \code{example_jprf_2023} and is
+useful for demonstrating workflow alignment between species detections and true
+field effort, specifically for zero-filling non-detections in occupancy modeling.
+
+This dataset was generated directly using the \code{\link[=birdnet_get_effort]{birdnet_get_effort()}} function.
+For more details on the generation parameters, data constraints, and internal
+file processing pipelines, please refer to the function documentation.
+}
+\keyword{datasets}
diff --git a/tests/testthat/test-birdnet_detection_history.R b/tests/testthat/test-birdnet_detection_history.R
new file mode 100644
index 0000000..e84706e
--- /dev/null
+++ b/tests/testthat/test-birdnet_detection_history.R
@@ -0,0 +1,185 @@
+test_that("birdnet_detection_history returns correct structures and shapes", {
+ # --- Setup Mock Data with realistic ARU filenames ---
+ mock_data <- data.frame(
+ filepath = c(
+ "project/site-A/site-A_20260601_221000.wav",
+ "project/site-A/site-A_20260602_053000.wav",
+ "project/site-B/site-B_20260601_120000.wav"
+ ),
+ confidence = c(0.8, 0.9, 0.75),
+ stringsAsFactors = FALSE
+ )
+
+ # Ensure the date column matches what birdnet_add_datetime() would extract
+ mock_data$date <- as.Date(c("2026-06-01", "2026-06-02", "2026-06-01"))
+
+ # Effort data covering both sites over a 3-day span
+ mock_effort <- data.frame(
+ site = rep(c("site-A", "site-B"), each = 3),
+ date = rep(as.Date(c("2026-06-01", "2026-06-02", "2026-06-03")), 2),
+ stringsAsFactors = FALSE
+ )
+
+ # --- Run Function ---
+ res <- birdnet_detection_history(
+ data = mock_data,
+ effort_data = mock_effort,
+ survey_interval = "1 day",
+ min_unique_days = 1,
+ i = -2
+ )
+
+ # --- Assertions ---
+ expect_type(res, "list")
+ expect_named(res, c("detection_history", "effort_matrix", "detection_summary"))
+ expect_equal(dim(res$detection_history), c(2, 3))
+ expect_equal(unname(res$detection_history["site-A", ]), c(1, 1, 0))
+ expect_equal(unname(res$detection_history["site-B", ]), c(1, 0, 0))
+})
+
+test_that("birdnet_detection_history returns correct structures and shapes", {
+ # --- Setup Mock Data ---
+ # Detections spanning 2 distinct days for site-A, 1 day for site-B
+ mock_data <- data.frame(
+ filepath = c(
+ "project/site-A/site-A_20260601_221000.wav",
+ "project/site-A/site-A_20260602_053000.wav",
+ "project/site-B/site-B_20260601_120000.wav"
+ ),
+ confidence = c(0.8, 0.9, 0.75),
+ stringsAsFactors = FALSE
+ )
+ mock_data$date <- as.Date(c("2026-06-01", "2026-06-02", "2026-06-01"))
+
+ # Effort data covering both sites over a 3-day span
+ mock_effort <- data.frame(
+ site = rep(c("site-A", "site-B"), each = 3),
+ date = rep(as.Date(c("2026-06-01", "2026-06-02", "2026-06-03")), 2),
+ stringsAsFactors = FALSE
+ )
+
+ # Mock internal column detection by returning expected mapping
+ # (If birdnet_detect_columns is an exported internal helper, we ensure it maps nicely)
+
+ # --- Run Function ---
+ res <- birdnet_detection_history(
+ data = mock_data,
+ effort_data = mock_effort,
+ survey_interval = "1 day",
+ min_unique_days = 1
+ )
+
+ # --- Assertions ---
+ # Check general structure
+ expect_type(res, "list")
+ expect_named(res, c("detection_history", "effort_matrix", "detection_summary"))
+
+ # Check matrices
+ expect_true(is.matrix(res$detection_history))
+ expect_true(is.matrix(res$effort_matrix))
+ expect_equal(dim(res$detection_history), c(2, 3)) # 2 sites, 3 occasions
+ expect_equal(rownames(res$detection_history), c("site-A", "site-B"))
+
+ # Check binary conversion and zero-filling
+ # site-A has detections on day 1 and 2, effort but no detection on day 3
+ expect_equal(unname(res$detection_history["site-A", ]), c(1, 1, 0))
+ # site-B has detection on day 1, effort but no detection on days 2 and 3
+ expect_equal(unname(res$detection_history["site-B", ]), c(1, 0, 0))
+
+ # Check effort matrix default logic (no n_files)
+ expect_equal(unname(res$effort_matrix["site-A", ]), c(1, 1, 1))
+})
+
+
+
+
+
+test_that("birdnet_detection_history filters based on min_unique_days", {
+ mock_data <- data.frame(
+ filepath = c(
+ "project/site-A/site-A_20260601_221000.wav",
+ "project/site-A/site-A_20260602_053000.wav",
+ "project/site-B/site-B_20260601_120000.wav"
+ ),
+ confidence = c(0.8, 0.9, 0.75),
+ stringsAsFactors = FALSE
+ )
+ mock_data$date <- as.Date(c("2026-06-01", "2026-06-02", "2026-06-01"))
+
+ mock_effort <- data.frame(
+ site = rep(c("site-A", "site-B"), each = 2),
+ date = rep(as.Date(c("2026-06-01", "2026-06-02")), 2),
+ stringsAsFactors = FALSE
+ )
+
+ # Require at least 2 unique detection days to keep a site
+ res <- birdnet_detection_history(
+ data = mock_data,
+ effort_data = mock_effort,
+ survey_interval = "1 day",
+ min_unique_days = 2
+ )
+
+ # site-B only has 1 detection day, so it should be dropped completely from summaries,
+ # resulting in 0s across all active effort sessions in the history matrix.
+ expect_equal(unname(res$detection_history["site-B", ]), c(0, 0))
+ expect_equal(unname(res$detection_history["site-A", ]), c(1, 1))
+})
+
+test_that("birdnet_detection_history handles continuous file-count effort tracking", {
+ mock_data <- data.frame(
+ filepath = c(
+ "project/site-A/site-A_20260601_221000.wav",
+ "project/site-A/site-A_20260602_053000.wav",
+ "project/site-B/site-B_20260601_120000.wav"
+ ),
+ confidence = c(0.8, 0.9, 0.75),
+ stringsAsFactors = FALSE
+ )
+ mock_data$date <- as.Date(c("2026-06-01", "2026-06-02", "2026-06-01"))
+
+ # Effort contains file count variables
+ mock_effort_files <- data.frame(
+ site = c("site-A", "site-A"),
+ date = as.Date(c("2026-06-01", "2026-06-02")),
+ n_files = c(10, 12),
+ stringsAsFactors = FALSE
+ )
+
+ res <- birdnet_detection_history(
+ data = mock_data,
+ effort_data = mock_effort_files,
+ survey_interval = "1 day"
+ )
+
+ # Effort matrix should capture continuous quantitative file metrics instead of binary tags
+ expect_equal(unname(res$effort_matrix["site-A", ]), c(10, 12))
+})
+
+test_that("birdnet_detection_history throws custom checkmate/rlang errors", {
+ bad_data <- data.frame(wrong_col = c(1, 2, 3))
+ good_effort <- data.frame(site = "site-A", date = as.Date("2026-06-01"))
+
+ # Test invalid data input error
+ expect_error(
+ birdnet_detection_history(bad_data, good_effort, "1 day"),
+ regexp = "missing required BirdNET columns"
+ )
+
+ # Test invalid interval string logic
+ good_data <- data.frame(
+ filepath = "path/site-A/f1.wav", confidence = 0.9,
+ site = "site-A", date = as.Date("2026-06-01")
+ )
+ expect_error(
+ birdnet_detection_history(good_data, good_effort, "bad_interval"),
+ regexp = "must be a valid lubridate unit string"
+ )
+
+ # Test bounds parameters constraints
+ expect_error(
+ birdnet_detection_history(good_data, good_effort, "1 day", min_unique_days = 0),
+ regexp = "Element 1 is not >= 1"
+ )
+})
+
diff --git a/tests/testthat/test-birdnet_get_effort.R b/tests/testthat/test-birdnet_get_effort.R
new file mode 100644
index 0000000..eec4ff9
--- /dev/null
+++ b/tests/testthat/test-birdnet_get_effort.R
@@ -0,0 +1,61 @@
+test_that("birdnet_get_effort scans directories and aggregates file counts correctly", {
+ # Create a temporary directory that self-destructs after this test block
+ tmp_dir <- withr::local_tempdir()
+
+ # Set up fake site directories
+ site_a_dir <- file.path(tmp_dir, "Site-A")
+ site_b_dir <- file.path(tmp_dir, "Site-B")
+ dir.create(site_a_dir)
+ dir.create(site_b_dir)
+
+ # Create dummy audio files with realistic datetime stamps in the names
+ # Site-A: 2 files on June 1st, 1 file on June 2nd
+ file.create(file.path(site_a_dir, "Site-A_20260601_060000.wav"))
+ file.create(file.path(site_a_dir, "Site-A_20260601_180000.wav"))
+ file.create(file.path(site_a_dir, "Site-A_20260602_060000.WAV")) # test case-insensitivity
+
+ # Site-B: 1 file on June 1st, 1 non-audio file (should be ignored)
+ file.create(file.path(site_b_dir, "Site-B_20260601_120000.mp3"))
+ file.create(file.path(site_b_dir, "summary_report.txt"))
+
+ # --- Run the function ---
+ # We use i = -2 because the immediate parent of the file will be "Site-A" or "Site-B"
+ res <- birdnet_get_effort(path = tmp_dir, i = -2)
+
+ # --- Assertions ---
+ expect_s3_class(res, "data.frame")
+ expect_named(res, c("site", "date", "n_files"))
+
+ # Check that we have exactly 3 unique site-date effort combinations
+ expect_equal(nrow(res), 3)
+
+ # Verify specific aggregations
+ site_a_efforts <- res[res$site == "Site-A", ]
+ # Depending on how birdnet_add_datetime extracts it, we check the counts:
+ # June 1st should have 2 files
+ expect_equal(site_a_efforts$n_files[site_a_efforts$date == as.Date("2026-06-01")], 2)
+ # June 2nd should have 1 file (even with uppercase .WAV)
+ expect_equal(site_a_efforts$n_files[site_a_efforts$date == as.Date("2026-06-02")], 1)
+
+ # Site-B should only count the .mp3, ignoring the .txt file
+ site_b_efforts <- res[res$site == "Site-B", ]
+ expect_equal(nrow(site_b_efforts), 1)
+ expect_equal(site_b_efforts$n_files, 1)
+})
+
+test_that("birdnet_get_effort throws custom error if directory does not exist", {
+ expect_error(
+ birdnet_get_effort("this/path/does/not/exist/at/all"))
+})
+
+test_that("birdnet_get_effort handles an empty directory gracefully", {
+ empty_dir <- withr::local_tempdir()
+
+ res <- birdnet_get_effort(empty_dir)
+
+ # It should return a 0-row data frame with the correct columns
+ expect_s3_class(res, "data.frame")
+ expect_equal(nrow(res), 0)
+ expect_named(res, c("site", "date", "n_files"))
+})
+
diff --git a/tests/testthat/test-utils_column_editing.R b/tests/testthat/test-utils_column_editing.R
index abb8b89..c9a8c39 100644
--- a/tests/testthat/test-utils_column_editing.R
+++ b/tests/testthat/test-utils_column_editing.R
@@ -96,3 +96,114 @@ test_that("birdnet_detect_columns identifies correct columns or returns NA", {
expect_true(all(vapply(detected2, function(x) is.na(x), logical(1))))
})
+test_that("birdnet_add_site extracts the correct site from file paths", {
+ # Setup dummy data with standard column names that birdnet_detect_columns would find
+ # Assuming birdnet_detect_columns looks for 'filepath'
+ mock_data <- data.frame(
+ filepath = c("project/site-A/audio1.wav", "project/site-B/audio2.wav"),
+ species = c("Cardinalis cardinalis", "Cyanocitta cristata"),
+ stringsAsFactors = FALSE
+ )
+
+ # Test standard behavior (i = -2, immediate parent folder)
+ res_default <- birdnet_add_site(mock_data, i = -2)
+ expect_s3_class(res_default, "data.frame")
+ expect_true("site" %in% colnames(res_default))
+ expect_equal(res_default$site, c("site-A", "site-B"))
+
+ # Test alternative index (i = -3, grandfather folder)
+ res_parent <- birdnet_add_site(mock_data, i = -3)
+ expect_equal(res_parent$site, c("project", "project"))
+})
+
+test_that("birdnet_add_site handles both forward and backward slashes", {
+ # BirdNET users might be on Windows or Unix
+ mixed_paths <- data.frame(
+ filepath = c("windows\\style-site\\file.wav", "unix/style-site/file.wav"),
+ stringsAsFactors = FALSE
+ )
+
+ res <- birdnet_add_site(mixed_paths, i = -2)
+ expect_equal(res$site, c("style-site", "style-site"))
+})
+
+test_that("birdnet_add_site throws an error when filepath column is missing or undetectable", {
+ # Data with completely unrelated columns
+ bad_data <- data.frame(
+ id = c(1, 2),
+ confidence = c(0.8, 0.9)
+ )
+
+ # Expect an error containing our specific error message
+ expect_error(
+ birdnet_add_site(bad_data),
+ regexp = "Could not automatically detect a valid file path column"
+ )
+})
+
+test_that("birdnet_add_site handles NA values in filepath gracefully", {
+ missing_path_data <- data.frame(
+ filepath = c("project/site-A/audio1.wav", NA),
+ stringsAsFactors = FALSE
+ )
+
+ res <- birdnet_add_site(missing_path_data, i = -2)
+ expect_equal(res$site, c("site-A", NA_character_))
+})
+
+
+test_that("birdnet_add_site extracts the correct site from file paths", {
+ # Setup dummy data with standard column names that birdnet_detect_columns would find
+ # Assuming birdnet_detect_columns looks for 'filepath'
+ mock_data <- data.frame(
+ filepath = c("project/site-A/audio1.wav", "project/site-B/audio2.wav"),
+ species = c("Cardinalis cardinalis", "Cyanocitta cristata"),
+ stringsAsFactors = FALSE
+ )
+
+ # Test standard behavior (i = -2, immediate parent folder)
+ res_default <- birdnet_add_site(mock_data, i = -2)
+ expect_s3_class(res_default, "data.frame")
+ expect_true("site" %in% colnames(res_default))
+ expect_equal(res_default$site, c("site-A", "site-B"))
+
+ # Test alternative index (i = -3, grandfather folder)
+ res_parent <- birdnet_add_site(mock_data, i = -3)
+ expect_equal(res_parent$site, c("project", "project"))
+})
+
+test_that("birdnet_add_site handles both forward and backward slashes", {
+ # BirdNET users might be on Windows or Unix
+ mixed_paths <- data.frame(
+ filepath = c("windows\\style-site\\file.wav", "unix/style-site/file.wav"),
+ stringsAsFactors = FALSE
+ )
+
+ res <- birdnet_add_site(mixed_paths, i = -2)
+ expect_equal(res$site, c("style-site", "style-site"))
+})
+
+test_that("birdnet_add_site throws an error when filepath column is missing or undetectable", {
+ # Data with completely unrelated columns
+ bad_data <- data.frame(
+ id = c(1, 2),
+ confidence = c(0.8, 0.9)
+ )
+
+ # Expect an error containing our specific error message
+ expect_error(
+ birdnet_add_site(bad_data),
+ regexp = "Could not automatically detect a valid file path column"
+ )
+})
+
+test_that("birdnet_add_site handles NA values in filepath gracefully", {
+ missing_path_data <- data.frame(
+ filepath = c("project/site-A/audio1.wav", NA),
+ stringsAsFactors = FALSE
+ )
+
+ res <- birdnet_add_site(missing_path_data, i = -2)
+ expect_equal(res$site, c("site-A", NA_character_))
+})
+