Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
46 commits
Select commit Hold shift + click to select a range
1067563
get rid of misnamed test nlcd files
mhweber Mar 24, 2026
28ddcc9
Lake watersheds (#89)
mhweber May 11, 2026
30ebad8
Merge branch 'master' into develop
mhweber May 12, 2026
39938c8
A few updates to clear warnings with merged PR for new lakecat waters…
mhweber May 13, 2026
6193c79
typo in CRAN comments
mhweber May 13, 2026
40723a1
fixed a couple warnings
mhweber May 13, 2026
2b86804
A few final fixes for latest update
mhweber May 14, 2026
8cb6d3a
version update in Description
mhweber May 14, 2026
194d571
fix to LakeCat.Rmd from master branch
mhweber May 22, 2026
6edce42
pkgdown updates
mhweber May 22, 2026
90dd1c9
updated functions to validate user supplied comids (#91)
mhweber Jun 4, 2026
42fcf52
add retry and throttle to the get_params functions
mhweber Jun 23, 2026
478dc08
update citation and vignette
mhweber Jul 21, 2026
6e36b0c
add leaflet providers file
mhweber Jul 29, 2026
a9dfd05
Tune up some functions and address a couple PRs (#94)
mhweber Aug 3, 2026
fcbf0ae
Delete LICENSE.md
mhweber Aug 18, 2026
0df566b
Add MIT License to the project
mhweber Aug 18, 2026
2892501
migrate nhdplusTools to hydrogeofetch (#97)
dblodgett-usgs Sep 10, 2026
0a9cd13
Merge branch 'develop' of https://github.com/USEPA/StreamCatTools int…
mhweber Sep 14, 2026
b76cfc0
Added count_metrics and update roxygen2
mhweber Sep 14, 2026
184c0e4
replace sc_get_comid with more performant version of the function
mhweber Sep 18, 2026
91a1ad4
A few fixes for checks
mhweber Sep 18, 2026
e73ac03
more check cleanups
mhweber Sep 18, 2026
27956fb
more clean-up
mhweber Sep 18, 2026
3775e7e
Update license in description
mhweber Sep 18, 2026
f3f1933
couple updates to sc_get_comid
mhweber Sep 21, 2026
5617a3e
added in some error handling
mhweber Sep 21, 2026
b073b59
A few fixes
mhweber Sep 21, 2026
4e5f786
some more fine-tuning
mhweber Sep 21, 2026
ab48dbf
test error handling
mhweber Sep 21, 2026
b1ed31f
a few more adjustments
mhweber Sep 21, 2026
ac28313
a few more minor updates
mhweber Sep 22, 2026
52be9c5
some more house cleaning
mhweber Sep 22, 2026
07684c7
getting checks to pass
mhweber Sep 22, 2026
cd3d8c2
Updated cran-comments
mhweber Sep 24, 2026
66a402c
added revdepcheck
mhweber Sep 24, 2026
b8f7206
updated version number in Description
mhweber Sep 24, 2026
0327900
last adjustments for CRAN
mhweber Sep 24, 2026
e84c0a9
update fullname
mhweber Sep 24, 2026
4aee9ec
pkgdown files do not need to be in develop
mhweber Sep 25, 2026
aa9790b
update license
mhweber Sep 25, 2026
fd4baaf
reconcile with master
mhweber Sep 25, 2026
375429d
reconcile description
mhweber Sep 25, 2026
7ed471c
Revert "last adjustments for CRAN"
mhweber Sep 25, 2026
3b4ddc0
a couple tweaks
mhweber Sep 25, 2026
0e8cfa0
updated dontrun
mhweber Sep 25, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .Rbuildignore
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
^docs$
^pkgdown$
.github
^LICENSE\.md$
^\.github$
^cran-comments\.md$
^CRAN-SUBMISSION$
^revdep$
44 changes: 39 additions & 5 deletions .github/workflows/coverage.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -18,19 +18,55 @@ jobs:
runs-on: ubuntu-latest
env:
GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }}
# Force CRAN for pak/renv resolution to avoid RSPM snapshot mismatch
RENV_CONFIG_REPOS_OVERRIDE: https://cran.rstudio.com

steps:
- uses: actions/checkout@v4

- uses: r-lib/actions/setup-r@v2
with:
use-public-rspm: true
# disable public RSPM to avoid using RSPM snapshots during CI
use-public-rspm: false

- name: Install system dependencies
# xml2, curl, openssl and many packages need system headers to build from source
run: |
sudo apt-get update
sudo apt-get install -y libxml2-dev libssl-dev libcurl4-openssl-dev

- name: Ensure pak is installed and current
# Install/upgrade pak from CRAN to ensure exported API is available
run: |
Rscript -e 'if (!requireNamespace("pak", quietly = TRUE)) install.packages("pak", repos = "https://cran.rstudio.com")'
Rscript -e 'message("pak version: ", as.character(packageVersion("pak"))); message("pak exports:"); print(head(getNamespaceExports("pak"), 200))'

- name: Debug pak resolution (safe checks)
# Do not call non-exported pak internals like pak::resolve; instead use exported APIs and print diagnostics
run: |
Rscript -e '
if (!requireNamespace("pak", quietly = TRUE)) {
install.packages("pak", repos = "https://cran.rstudio.com")
}
message("pak version: ", as.character(packageVersion("pak")))
message("sessionInfo:")
print(sessionInfo())
message("Trying a safe pak operation (pkg_install with upgrade=FALSE)...")
tryCatch({
# Use exported pak API to verify installer works. Use upgrade=FALSE to avoid changing runner state excessively.
pak::pkg_install(c("covr", "xml2"), upgrade = FALSE, ask = FALSE)
message("pak::pkg_install completed (or packages were already installed).")
}, error = function(e) {
message("pak::pkg_install failed: ", conditionMessage(e))
stop(e)
})
'

- uses: r-lib/actions/setup-r-dependencies@v2
with:
extra-packages: any::covr, any::xml2
needs: coverage
cache-version: 2 # Increment this to bust the cache
cache-version: 3 # Bump cache-version to avoid reusing an older pak installation

- name: Test coverage
run: |
Expand Down Expand Up @@ -71,6 +107,4 @@ jobs:
path: ./cobertura.xml
minimum_coverage: 10
show_missing: true
link_missing_lines: true


link_missing_lines: true
22 changes: 13 additions & 9 deletions DESCRIPTION
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
Package: StreamCatTools
Type: Package
Title: StreamCatTools: Tools for Working with StreamCat and LakeCat Data
Version: 0.11.0
Date: 2026-05-15
Version: 0.12.0
Date: 2026-09-23
Authors@R: c(person(given = "Marc",
family = "Weber",
role = c("aut", "cre"),
Expand Down Expand Up @@ -37,15 +37,17 @@ Authors@R: c(person(given = "Marc",
email = "bousquin.justin@epa.gov"),
person(given = "Zachary",
family = "Smith",
role = "ctb"))
role = "ctb"))
Author: Marc Weber [aut, cre], Ryan Hill [aut], Selia Markley [aut], Travis Hudson [aut], Allen Brookes [aut], David Rebhuhn [ctb], Michael Dumelle [ctb], Justin Bousquin [ctb], Zachary Smith [ctb]
Maintainer: Marc Weber <weber.marc@epa.gov>
Description: Tools for using the 'StreamCat' and 'LakeCat' API and
interacting with the 'StreamCat' and 'LakeCat' database.
Convenience functions in the package wrap the API for 'StreamCat'
on <https://api.epa.gov/StreamCat/streams/metrics>.
Depends: R (>= 4.1.0)
Imports:
sf,
nhdplusTools,
hydrogeofetch,
jsonlite,
httr2,
curl (>= 6.0.0),
Expand All @@ -54,8 +56,11 @@ Imports:
cowplot,
tigris,
ggplot2,
dplyr,
tidyr,
stringr,
tibble,
Suggests:
dplyr,
mapview,
testthat,
knitr,
Expand All @@ -64,8 +69,6 @@ Suggests:
xml2,
magrittr,
readr,
tidyr,
stringr,
purrr,
lifecycle,
tidyselect,
Expand All @@ -79,6 +82,7 @@ URL: https://usepa.github.io/StreamCatTools/, https://github.com/USEPA/StreamCat
BugReports: https://github.com/USEPA/StreamCatTools/issues
VignetteBuilder: knitr
LazyData: true
License: MIT+ file LICENSE
License: MIT + file LICENSE
NeedsCompilation: no
Config/roxygen2/version: 8.0.0
Config/roxygen2/version: 8.1.0

9 changes: 9 additions & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,12 @@
# StreamCatTools 0.12.0

- Replaced `sc_get_comid` with a new REST service implementation using
the EPA NHDPlus NP21 simplified catchments layer.
- Added a `count_metrics` helper function
- Added MIT license for package
- Refactored `sc_get_data` and `lc_get_data` slightly for improved
functionality (based on https://github.com/pauldzy/StreamCatTools)

# StreamCatTools 0.11.0

- Adds new `lc_get_watershed` function to return a lake watershed as an `sf`
Expand Down
90 changes: 90 additions & 0 deletions R/count_metrics.R
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
# count_metrics.R
# Deduplicate the StreamCat / LakeCat "over 600 / over 300" headline figures into:
# (1) distinct conceptual indicators -> unique base metric names
# (2) AOI expansion -> Cat / Ws / (RipBuf, Cat/WsRp100 etc.)
# (3) year expansion -> NLCD vintages and other multi-year layers
# so you can see how a raw column count inflates past the number of distinct variables.


# ---- helper: turn a variable_info tibble into the three counts -------------
summarize_metrics <- function(vi, label) {
if (!"metric" %in% names(vi)) {
stop("`vi` must contain a `metric` column.", call. = FALSE)
}

vi <- vi
aoi <- if ("aoi" %in% names(vi)) vi$aoi else rep("NA", nrow(vi))
year <- if ("year" %in% names(vi)) vi$year else rep("NA", nrow(vi))

vi <- vi |>
dplyr::mutate(
metric = tolower(stringr::str_trim(metric)),
aoi = ifelse(is.na(aoi) | aoi == "", "NA", aoi),
year = ifelse(is.na(year) | year == "", "NA", year)
)

# collapse to one row per base metric, unioning aoi + year tokens across rows
per_metric <- vi |>
dplyr::group_by(metric) |>
dplyr::summarise(
aoi_tokens = list(sort(unique(stringr::str_split(paste(aoi, collapse = ","), "\\s*,\\s*")[[1]]))),
year_tokens = list(sort(unique(stringr::str_split(paste(year, collapse = ","), "\\s*,\\s*")[[1]]))),
.groups = "drop"
) |>
dplyr::mutate(
n_aoi = lengths(aoi_tokens),
n_year = pmax(lengths(year_tokens), 1L), # NA-only -> counts as 1
n_columns = n_aoi * n_year # physical columns for this metric
)

cat(sprintf("\n==== %s ====\n", label))
cat(sprintf("Distinct conceptual indicators (base metric names): %d\n",
nrow(per_metric)))
cat(sprintf("Metrics offered in >1 AOI: %d\n",
sum(per_metric$n_aoi > 1)))
cat(sprintf("Metrics with multiple years: %d\n",
sum(per_metric$n_year > 1)))
cat(sprintf("Fully-expanded physical columns (metric x AOI x yr): %d\n",
sum(per_metric$n_columns)))
invisible(per_metric)
}

# ---- Option A: use your own package (idiomatic) ---------------------------
# library(StreamCatTools)
# sc_vi <- sc_get_params(param = "variable_info") # StreamCat metadata tibble
# lc_vi <- lc_get_params(param = "variable_info") # LakeCat metadata tibble
# summarize_metrics(sc_vi, "StreamCat")
# summarize_metrics(lc_vi, "LakeCat")

# ---- Option B: standalone, no package dependency (clean-room check) --------
if (sys.nframe() == 0L) {
library(jsonlite)

fetch_vi <- function(base) {
# variable_info is exposed through the metrics endpoint's parameter listing;
# the R package pulls it here. Adjust the query if the schema shifts.
url <- paste0(base, "?variable_info=variable_info")
fromJSON(url)$items
}

sc_vi <- fetch_vi("https://api.epa.gov/StreamCat/streams/metrics")
lc_vi <- fetch_vi("https://api.epa.gov/StreamCat/lakes/metrics")

sc <- summarize_metrics(sc_vi, "StreamCat")
lc <- summarize_metrics(lc_vi, "LakeCat")

# ---- LakeCat overlap check: how many LakeCat metrics are borrowed from StreamCat
# The LakeCat tables carry an `inStreamCat` flag per metric. If present in vi,
# this shows how much of LakeCat's total is NOT independent of StreamCat:
if ("instreamcat" %in% tolower(names(lc_vi))) {
flag <- lc_vi[[which(tolower(names(lc_vi)) == "instreamcat")]]
cat(sprintf("\nLakeCat metrics flagged inStreamCat (shared w/ StreamCat): %d of %d rows\n",
sum(flag %in% c(1, "1", TRUE, "Yes", "yes")), length(flag)))
}

# ---- category breakdown (Natural vs Anthropogenic vs derived/special) ------
sc_by_cat <- sc_vi |>
distinct(metric = tolower(metric), category) |>
count(category, sort = TRUE, name = "n_distinct_metrics")
print(sc_by_cat)
}
9 changes: 9 additions & 0 deletions R/helper-api.R
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
skip_if_api_unavailable <- function(x, label = "API") {
if (is.null(x) ||
(is.data.frame(x) && nrow(x) == 0L) ||
(is.list(x) && length(x) == 0L) ||
(is.character(x) && length(x) == 0L)) {
testthat::skip(paste(label, "is unavailable; upstream service returned no data."))
}
invisible(x)
}
53 changes: 32 additions & 21 deletions R/lc_get_comid.R
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
#' @param crsys The epsg code if using a raw data frame
#'
#' @param buffer The amount of buffer to use to extend search for a waterbody
#' (simply passed to nhdplusTools::get_waterbodies)
#' (simply passed to hydrogeofetch::get_waterbodies)
#'
#' @return A new sf data frame with a populated 'COMID' column
#'
Expand Down Expand Up @@ -52,27 +52,38 @@ lc_get_comid <- function(dd = NULL, xcoord = NULL,
} else {
dd <- sf::st_as_sf(dd, coords = c(xcoord, ycoord), crs = crsys, remove = FALSE)
}


output <- do.call(rbind, lapply(1:nrow(dd), function(i){
if (is.null(buffer)){
wb <- nhdplusTools::get_waterbodies(dd[i,])
} else {
wb <- nhdplusTools::get_waterbodies(dd[i,], buffer=buffer)

output <- vapply(seq_len(nrow(dd)), function(i) {
res <- tryCatch({
if (is.null(buffer)) {
hydrogeofetch::get_waterbodies(dd[i, ])
} else {
hydrogeofetch::get_waterbodies(dd[i, ], buffer = buffer)
}
}, error = function(e) NULL)

if (is.null(res) || !inherits(res, "sf") || nrow(res) == 0L) {
return(NA_character_)
}

comids <- tryCatch({
unique(as.character(dplyr::pull(res, "comid")))
}, error = function(e) character(0))

if (length(comids) == 0L || all(is.na(comids))) {
return(NA_character_)
}
if (!is.null(wb)){
comid <- wb |>
dplyr::pull(comid)
if (length(comid)==0L) comid <- NA else comid <- comid
return(comid)
}
}))
output <- as.data.frame(output)
names(output)[1] <- 'COMID'
if (any(is.na(output$COMID))){
missing <- which(is.na(output$COMID))
message(paste0('Row number ', as.character(missing), ' came back with no corresponding COMIDS because the site(s) were outside the boundary of any NHDPlus Waterbody features. Any NA values in this list of COMIDs will be dropped by default in lc_get_data()'))

paste(comids, collapse = ",")
}, character(1))

output_df <- data.frame(COMID = output, stringsAsFactors = FALSE)

if (any(is.na(output_df$COMID))) {
missing <- which(is.na(output_df$COMID))
message(paste0('Row number ', paste(as.character(missing), collapse = ", "), ' came back with no corresponding COMIDS because the site(s) were outside the boundary of any NHDPlus Waterbody features. Any NA values in this list of COMIDs will be dropped by default in lc_get_data()'))
}
comids <- paste(output$COMID, collapse=',')

comids <- paste(output_df$COMID, collapse = ',')
return(comids)
}
Loading
Loading