diff --git a/DESCRIPTION b/DESCRIPTION index 10e202d..491d995 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -2,7 +2,7 @@ Package: metajam Type: Package Title: Easily Download Data and Metadata from 'DataONE' Version: 0.3.2 -Date: 2026-06-23 +Date: 2026-08-31 Authors@R: c( person("Julien", "Brun", email = "julien.brun@alumni.duke.edu", @@ -45,7 +45,7 @@ Description: A set of tools to foster the development of reproducible analytical License: Apache License (== 2.0) Encoding: UTF-8 Language: en-US -RoxygenNote: 7.3.3 +RoxygenNote: 8.1.0 SystemRequirements: Mac OSX: redland (>= 1.0.14) ; Linux: librdf0 (>= 1.0.14), librdf0-dev (>= 1.0.14) URL: https://nceas.github.io/metajam/, https://github.com/NCEAS/metajam diff --git a/R/check_version.R b/R/check_version.R index 4e07bcf..2f4e24b 100644 --- a/R/check_version.R +++ b/R/check_version.R @@ -12,7 +12,7 @@ #' @export #' #' @examples -#' \donttest{ +#' \dontrun{ #' # Most data URLs and identifiers work #' check_version("https://cn.dataone.org/cn/v2/resolve/urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570") #' check_version("doi:10.18739/A2J09W56F") diff --git a/R/download_ISO_data.R b/R/download_ISO_data.R index e97e691..4caab65 100644 --- a/R/download_ISO_data.R +++ b/R/download_ISO_data.R @@ -110,7 +110,7 @@ ISO_type <- metadata2 %>% filter(name == "doc.children.MD_Metadata.children.meta pid <- data_id data_sys <- suppressMessages(dataone::getSystemMetadata(d1c@mn, pid)) - data_name <- data_sys@fileName %|||% ifelse(exists("entity_data"), entity_data$physical$objectName %|||% entity_data$entityName, NA) %|||% data_id + data_name <- data_sys@fileName %|||% data_id data_name <- gsub("[^a-zA-Z0-9. -]+", "_", data_name) #remove special characters & replace with _ data_extension <- gsub("(.*\\.)([^.]*$)", "\\2", data_name) data_name <- gsub("\\.[^.]*$", "", data_name) #remove extension diff --git a/R/download_d1_data.R b/R/download_d1_data.R index 83a4793..78b8e69 100644 --- a/R/download_d1_data.R +++ b/R/download_d1_data.R @@ -22,7 +22,7 @@ #' @seealso [read_d1_files()] [download_d1_data_pkg()] #' #' @examples -#' \dontest{ +#' \donttest{ #' download_d1_data("urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570", path = tempdir()) #' download_d1_data( #' "https://cn.dataone.org/cn/v2/resolve/urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570", @@ -31,7 +31,6 @@ #' } download_d1_data <- function(data_url, path) { - # TODO: add meta_doi to explicitly specify doi # Silence visible bindings note entity_data <- eml <- dir_name <- NULL diff --git a/R/download_d1_data_pkg.R b/R/download_d1_data_pkg.R index b471a30..91b9c43 100644 --- a/R/download_d1_data_pkg.R +++ b/R/download_d1_data_pkg.R @@ -17,7 +17,7 @@ #' @examples #' \donttest{ #' download_d1_data_pkg("doi:10.18739/A2CJ87M3J", tempdir()) -#' download_d1_data_pkg("https://doi.org/10.18739/A2CJ87M3J, tempdir()) +#' download_d1_data_pkg("https://doi.org/10.18739/A2CJ87M3J", tempdir()) #' } download_d1_data_pkg <- function(meta_obj, path) { diff --git a/R/tabularize_eml.R b/R/tabularize_eml.R index 29b0dbe..37769b7 100644 --- a/R/tabularize_eml.R +++ b/R/tabularize_eml.R @@ -16,7 +16,10 @@ #' @export #' #' @examples -#' eml <- system.file("extdata", "test_data", "SoilMois2012_2017__full_metadata.xml", package = "metajam") +#' eml <- system.file("extdata", +#' "test_data", +#' "SoilMois2012_2017__full_metadata.xml", +#' package = "metajam") #' tabularize_eml(eml) tabularize_eml <- function(eml, full = FALSE) { diff --git a/R/utils.R b/R/utils.R index 3fbed7b..9852296 100644 --- a/R/utils.R +++ b/R/utils.R @@ -1,6 +1,6 @@ `%|||%` <- function (x, y) { #based on the purrr/rlang op-null-default - if (is.null(x) || is.na(x)) { + if (is.null(x) || length(x) == 0 || is.na(x)) { y } else { diff --git a/cran-comments.md b/cran-comments.md index 858617d..08adc82 100644 --- a/cran-comments.md +++ b/cran-comments.md @@ -2,4 +2,8 @@ 0 errors | 0 warnings | 1 note -* This is a new release. +* Archived on 2025-12-04 as requires archived package 'dataone' + +This is a new release + +Addresses previous submission comments about /dontrun and using temporary folders diff --git a/man/check_version.Rd b/man/check_version.Rd index c658cd1..ba3111a 100644 --- a/man/check_version.Rd +++ b/man/check_version.Rd @@ -21,10 +21,10 @@ This function takes an identifier and checks to see if it has been obsoleted. \dontrun{ # Most data URLs and identifiers work check_version("https://cn.dataone.org/cn/v2/resolve/urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570") -check_version("doi:10.18739/A2ZF6M") +check_version("doi:10.18739/A2J09W56F") # Specify a formatType (data, metadata, or resource) -check_version("doi:10.18739/A2ZF6M", formatType = "metadata") +check_version("doi:10.18739/A2J09W56F", formatType = "metadata") # Returns a warning if the identifier has been obsoleted check_version("doi:10.18739/A2HF7Z", formatType = "metadata") diff --git a/man/download_d1_data.Rd b/man/download_d1_data.Rd index 0a63093..b02e065 100644 --- a/man/download_d1_data.Rd +++ b/man/download_d1_data.Rd @@ -18,11 +18,11 @@ download_d1_data(data_url, path) Downloads a data object from DataONE along with metadata. } \examples{ -\dontrun{ -download_d1_data("urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570", path = file.path(".")) +\donttest{ +download_d1_data("urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570", path = tempdir()) download_d1_data( "https://cn.dataone.org/cn/v2/resolve/urn:uuid:a2834e3e-f453-4c2b-8343-99477662b570", - path = file.path(".") + path = tempdir() ) } } diff --git a/man/download_d1_data_pkg.Rd b/man/download_d1_data_pkg.Rd index 34d7355..fedfbe3 100644 --- a/man/download_d1_data_pkg.Rd +++ b/man/download_d1_data_pkg.Rd @@ -18,9 +18,9 @@ download_d1_data_pkg(meta_obj, path) Downloads all the data objects of a data package from DataONE along with metadata. } \examples{ -\dontrun{ -download_d1_data_pkg("doi:10.18739/A2028W", ".") -download_d1_data_pkg("https://doi.org/10.18739/A2028W", ".") +\donttest{ +download_d1_data_pkg("doi:10.18739/A2CJ87M3J", tempdir()) +download_d1_data_pkg("https://doi.org/10.18739/A2CJ87M3J", tempdir()) } } \seealso{ diff --git a/man/tabularize_eml.Rd b/man/tabularize_eml.Rd index 14b198c..a4361ae 100644 --- a/man/tabularize_eml.Rd +++ b/man/tabularize_eml.Rd @@ -19,7 +19,9 @@ If \code{full = TRUE} is specified, the full set of metadata fields are returned This function takes a path to an EML (.xml) metadata file and returns a data frame. } \examples{ - eml <- system.file("extdata", "test_data", "SoilMois2012_2017__full_metadata.xml", - package = "metajam") - tabularize_eml(eml) +eml <- system.file("extdata", + "test_data", + "SoilMois2012_2017__full_metadata.xml", + package = "metajam") +tabularize_eml(eml) } diff --git a/vignettes/use02_dataset-single-dataone.Rmd b/vignettes/use02_dataset-single-dataone.Rmd index da6bdba..c6ec0a3 100644 --- a/vignettes/use02_dataset-single-dataone.Rmd +++ b/vignettes/use02_dataset-single-dataone.Rmd @@ -18,9 +18,9 @@ knitr::opts_chunk$set(collapse = TRUE, comment = "#>") This vignette aims to showcase a use case using the 2 main functions of `metajam` - `download_d1_data` and `read_d1_files` to download one dataset from the DataOne data repository. -## Note on data url provenance when using download_d1_data.R +## Note on data url provenance when using `download_d1_data()` -There are two parameters required to run the download_d1_data.R function in metajam. One is the data url for the dataset you'd like to download.You can retrieve this by navigating to the data package of interest, right-clicking on the download data button, and selecting Copy Link Address. +There are two parameters required to run the `download_d1_data()` function in metajam. One is the data url for the dataset you'd like to download.You can retrieve this by navigating to the data package of interest, right-clicking on the download data button, and selecting "Copy Link Address". For several DataOne member nodes (Arctic Data Center, Environmental Data Initiative, and The Knowledge Network for Biocomplexity), metajam users can retrieve the data url from either the 'home' site of the member node or the from the DataOne instance of that same data package. For example, if you wanted to download this dataset: @@ -40,22 +40,21 @@ We have not tested metajam's compatibility with the home sites of all DataOne me We include two examples, one downloading a dataset with metadata in eml (ecological metadata format) and the other downloading a dataset with metadata in ISO (International Organization for Standardization) format. -## Example 1: eml +## Example 1: EML metadata For the first example, we are using Diatom Community Data from Coweeta LTER, 2005-2019: Kelsey J. Solomon, Rebecca J. Bixby, and Catherine M. Pringle. Environmental Data Initiative. . -## Libraries and constants +### Libraries and constants ```{r libraries, warning=FALSE} # devtools::install_github("NCEAS/metajam") library(metajam) - ``` ```{r constants} # Directory to save the data set -path_folder <- "Data_coweeta" +path_folder <- file.path(tempdir(),"Data_coweeta") # URL to download the dataset from DataONE data_url <- "https://cn.dataone.org/cn/v2/resolve/https%3A%2F%2Fpasta.lternet.edu%2Fpackage%2Fdata%2Feml%2Fedi%2F858%2F1%2F15ad768241d2eeed9f0ba159c2ab8fd5" @@ -63,7 +62,7 @@ data_url <- "https://cn.dataone.org/cn/v2/resolve/https%3A%2F%2Fpasta.lternet.ed ``` -## Download the dataset +### Download the dataset ```{r download, eval=FALSE} @@ -73,62 +72,55 @@ dir.create(path_folder, showWarnings = FALSE) # Download the dataset and associated metdata data_folder <- metajam::download_d1_data(data_url, path_folder) - - +data_folder ``` At this point, you should have the data and the metadata downloaded inside your main directory; `Data_coweeta` in this example. `metajam` organize the files as follow: - Each dataset is stored a sub-directory named after the package DOI and the file name - Inside this sub-directory, you will find - - the data: `my_data.csv` + - the data: `CWT_Hemlock_Diatom_Data.csv` - the raw EML with the naming convention _file name_ + `__full_metadata.xml`: `my_data__full_metadata.xml` - the package level metadata summary with the naming convention _file name_ + `__summary_metadata.csv`: `my_data__summary_metadata.csv` - If relevant, the attribute level metadata with the naming convention _file name_ + `__attribute_metadata.csv`: `my_data__attribute_metadata.csv` - If relevant, the factor level metadata with the naming convention _file name_ + `__attribute_factor_metadata.csv`: my_data`__attribute_factor_metadata.csv` - -```{r, out.width="90%", echo=FALSE, fig.align="center", fig.cap="Local file structure of a dataset downloaded by metajam"} -knitr::include_graphics("../man/figures/metajam_v1_folder.png") -``` - -## Read the data and metadata in your R environment +### Read the data and metadata in your R environment ```{r read_data, eval=FALSE} # Read all the datasets and their associated metadata in as a named list coweeta_diatom <- metajam::read_d1_files(data_folder) - ``` -## Structure of the named list object +### Structure of the named list object You have now loaded in your R environment one named list object that contains the data `coweeta_diatom$data`, the general (summary) metadata `coweeta_diatom$summary_metadata` - such as title, creators, dates, locations - and the attribute level metadata information `coweeta_diatom$attribute_metadata`, allowing user to get more information, such as units and definitions of your attributes. -## Example 2: iso + + +## Example 2: ISO metadata For the second example, we are using Marine bird survey observation and density data from Northern Gulf of Alaska LTER cruises, 2018. Kathy Kuletz, Daniel Cushing, and Elizabeth Labunski. Research Workspace. -## Libraries and constants +### Libraries and constants ```{r libraries-2, warning=FALSE} # devtools::install_github("NCEAS/metajam") library(metajam) - ``` ```{r constants-2} # Directory to save the data set -path_folder <- "Data_alaska" +path_folder <- file.path(tempdir(), "Data_alaska") # URL to download the dataset from DataONE data_url <- "https://cn.dataone.org/cn/v2/resolve/4139539e-94e7-49cc-9c7a-5f879e438b16" - ``` -## Download the dataset +### Download the dataset ```{r download-2, eval=FALSE} @@ -138,8 +130,6 @@ dir.create(path_folder, showWarnings = FALSE) # Download the dataset and associated metdata data_folder <- metajam::download_d1_data(data_url, path_folder) - - ``` At this point, you should have the data and the metadata downloaded inside your main directory; `Data_alaska` in this example. `metajam` organize the files as follow: @@ -151,24 +141,3 @@ At this point, you should have the data and the metadata downloaded inside your - the package level metadata summary with the naming convention _file name_ + `__summary_metadata.csv`: `my_data__summary_metadata.csv` -```{r, out.width="90%", echo=FALSE, fig.align="center", fig.cap="Local file structure of a dataset downloaded by metajam"} -knitr::include_graphics("../man/figures/metajam_v1_folder.png") -``` - - -## Read the data and metadata in your R environment - -```{r read_data-2, eval=FALSE} -# Read all the datasets and their associated metadata in as a named list -coweeta_diatom <- metajam::read_d1_files(data_folder) - -``` - -## Structure of the named list object - -You have now loaded in your R environment one named list object that contains the data `coweeta_diatom$data`, the general (summary) metadata `coweeta_diatom$summary_metadata` - such as title, creators, dates, locations - and the attribute level metadata information `coweeta_diatom$attribute_metadata`, allowing user to get more information, such as units and definitions of your attributes. - - -```{r, out.width="90%", echo=FALSE, fig.align="center", fig.cap="Structure of the named list object containing tabular metadata and data as loaded by metajam"} -knitr::include_graphics("../man/figures/metajam_v1_named_list.png") -``` diff --git a/vignettes/use03_dataset-batch-processing.Rmd b/vignettes/use03_dataset-batch-processing.Rmd index 2f4ecf2..46d1c98 100644 --- a/vignettes/use03_dataset-batch-processing.Rmd +++ b/vignettes/use03_dataset-batch-processing.Rmd @@ -40,7 +40,7 @@ library(stringr) ```{r constants} # Download the data from DataONE on your local machine -data_folder <- "Data_SEC" +data_folder <- file.path(tempdir(), "Data_SEC") # Ammonium to Ammoniacal-nitrogen conversion. We will use this conversion later. coeff_conv_NH4_to_NH4N <- 0.7764676534