diff --git a/DESCRIPTION b/DESCRIPTION
index fc6bbf4..00b7439 100644
--- a/DESCRIPTION
+++ b/DESCRIPTION
@@ -1,6 +1,6 @@
Package: fscontext
Title: File System Contextualisation and Record Set Reconstruction
-Version: 0.2.0
+Version: 0.2.001
Language: en-GB
Authors@R:
person(given = "Daniel",
@@ -34,7 +34,6 @@ Imports:
digest,
fs,
progress,
- here,
dplyr,
utils,
rlang,
@@ -47,7 +46,8 @@ Imports:
tidyr,
magrittr,
stringr,
- labelled
+ labelled,
+ jsonlite
Suggests:
knitr,
rmarkdown,
diff --git a/NAMESPACE b/NAMESPACE
index 874e9d7..f8bd2e5 100644
--- a/NAMESPACE
+++ b/NAMESPACE
@@ -2,7 +2,6 @@
export(add_snapshot_context)
export(as_character)
-export(as_recordset_df)
export(as_value_key)
export(classify_operational_file_type)
export(compile_rulebook)
@@ -10,7 +9,6 @@ export(construct_structural_paths)
export(context_roots)
export(coverage_roots)
export(coverage_rules_path)
-export(create_record_set)
export(derive_record_set)
export(derive_structural_groups)
export(detect_generated_artifacts)
@@ -19,8 +17,10 @@ export(invert_contextual_grouping)
export(invert_value_key)
export(is.prelabelled)
export(observe_universe)
+export(observe_wacz)
export(prelabel)
export(quick_signature)
+export(quick_signature_text)
export(read_snapshot)
export(recordset_df)
export(refine)
@@ -35,16 +35,21 @@ export(summarise_duplicates)
export(summarise_observed_activity)
export(summarize_duplicates)
export(summarize_observed_activity)
+export(wacz_to_recordset_df)
importFrom(dataset,"provenance<-")
importFrom(dataset,as_character)
+importFrom(dataset,as_dataset_df)
importFrom(dataset,as_value_key)
+importFrom(dataset,defined)
importFrom(dataset,dublincore)
+importFrom(dataset,identifier)
importFrom(dataset,invert_value_key)
importFrom(dataset,is.prelabelled)
importFrom(dataset,n_triple)
importFrom(dataset,n_triples)
importFrom(dataset,prelabel)
importFrom(dataset,provenance)
+importFrom(dataset,subject)
importFrom(dataset,subject_create)
importFrom(digest,digest)
importFrom(dplyr,across)
@@ -64,6 +69,7 @@ importFrom(dplyr,mutate)
importFrom(dplyr,n)
importFrom(dplyr,n_distinct)
importFrom(dplyr,relocate)
+importFrom(dplyr,rename)
importFrom(dplyr,row_number)
importFrom(dplyr,select)
importFrom(dplyr,semi_join)
@@ -73,7 +79,14 @@ importFrom(fs,dir_create)
importFrom(fs,dir_exists)
importFrom(fs,file_info)
importFrom(fs,path)
+importFrom(fs,path_abs)
+importFrom(fs,path_ext)
+importFrom(fs,path_file)
+importFrom(fs,path_rel)
importFrom(glue,glue)
+importFrom(jsonlite,fromJSON)
+importFrom(jsonlite,read_json)
+importFrom(jsonlite,stream_in)
importFrom(labelled,labelled)
importFrom(magrittr,"%>%")
importFrom(purrr,imap_dfr)
@@ -98,9 +111,11 @@ importFrom(tibble,tibble)
importFrom(tidyr,pivot_wider)
importFrom(tidyr,unnest)
importFrom(tools,file_ext)
+importFrom(tools,file_path_sans_ext)
importFrom(utils,flush.console)
importFrom(utils,globalVariables)
importFrom(utils,head)
importFrom(utils,person)
importFrom(utils,setTxtProgressBar)
importFrom(utils,txtProgressBar)
+importFrom(utils,unzip)
diff --git a/NEWS.md b/NEWS.md
index db82345..6ac65de 100644
--- a/NEWS.md
+++ b/NEWS.md
@@ -1,3 +1,9 @@
# fscontext 0.2.0
- Initial CRAN submission.
+- Added support for observing ZIP archives using `scan_storage()`.
+- Added structural aggregation profiles (`folder-depth-1` to `folder-depth-4`
+ and `wacz`) in `derive_structural_groups()`.
+- Added a vignette illustrating structural aggregation metadata, candidate
+ Record Sets, and RiC-inspired contextual reconstruction using directories,
+ ZIP archives, and WACZ packages.
diff --git a/R/add_structural_groups.R b/R/add_structural_groups.R
index 1c57fb8..9b8e8c8 100644
--- a/R/add_structural_groups.R
+++ b/R/add_structural_groups.R
@@ -50,7 +50,7 @@
#'
#' Future versions of the package may replace or extend this logic
#' with more explicit provenance-aware Record Set construction workflows
-#' (for example via `create_record_set()`).
+#' (for example via `record_set_projection()`).
#'
#' @seealso [derive_structural_groups()]
#'
diff --git a/R/as_recordset_df.R b/R/as_recordset_df.R
deleted file mode 100644
index 7f9904a..0000000
--- a/R/as_recordset_df.R
+++ /dev/null
@@ -1,254 +0,0 @@
-#' Coerce a contextual Record Set projection into a semantically enriched
-#' `recordset_df`
-#'
-#' Converts a contextual Record Set projection created with
-#' [create_record_set()] into a semantically enriched `recordset_df`
-#' object.
-#'
-#' The function adds lightweight dataset-level semantics and publication
-#' metadata while preserving tidyverse compatibility.
-#'
-#' This staged design deliberately separates:
-#'
-#' - operational contextualisation (`create_record_set()`)
-#'
-#' from:
-#'
-#' - semantic stabilisation and publication (`as_recordset_df()`)
-#'
-#' The resulting object aligns with the package philosophy of:
-#'
-#' - observational acquisition,
-#' - contextual enrichment,
-#' - deferred semantic interpretation.
-#'
-#' `as_recordset_df()` is conceptually aligned with:
-#'
-#' - RiC-O Record Set projections,
-#' - contextual research workspaces,
-#' - analytical Heritage Digital Twin layers,
-#' - and semantically enriched reconstruction corpora.
-#'
-#' The function builds on the `dataset_df` framework and therefore
-#' inherits:
-#'
-#' - tibble semantics,
-#' - lightweight dataset metadata,
-#' - publication-oriented enrichment,
-#' - and future linked-data extensibility.
-#'
-#' The function also acts as a lightweight semantic alignment layer
-#' between:
-#'
-#' - operational resource-oriented contextualisation
-#'
-#' and:
-#'
-#' - semantically stabilised record set member representations.
-#'
-#' Operational columns are mapped into the opinionated
-#' `recordset_df` vocabulary:
-#'
-#' - `resource_id` → `member_id`
-#' - `locator_path` → `member_path`
-#' - `resource_type` → `member_type`
-#'
-#' by default, although alternative mappings may be supplied.
-#'
-#' @param x A tibble or `data.frame`, typically created with
-#' [create_record_set()].
-#'
-#' @param title Human-readable title of the contextual Record Set.
-#'
-#' @param creator Creator metadata passed to
-#' [dataset::dublincore()].
-#'
-#' @param member_id Character scalar giving the column name in `x`
-#' that should be mapped to `member_id` in the resulting
-#' `recordset_df`.
-#'
-#' Defaults to `"resource_id"`.
-#'
-#' @param member_path Optional character scalar giving the column name
-#' in `x` that should be mapped to `member_path`.
-#'
-#' Defaults to `"locator_path"`.
-#'
-#' @param member_type Optional character scalar giving the column name
-#' in `x` that should be mapped to `member_type`.
-#'
-#' Defaults to `"resource_type"`.
-#'
-#' @param description Optional textual description documenting the
-#' contextual scope, construction logic, provenance assumptions,
-#' or analytical purpose of the Record Set.
-#'
-#' @param publisher Optional publisher metadata passed to
-#' [dataset::dublincore()].
-#'
-#' @param subject Optional subject metadata for future semantic
-#' enrichment.
-#'
-#' @return
-#' A semantically enriched `recordset_df` object inheriting from:
-#'
-#' - `recordset_df`
-#' - `dataset_df`
-#' - `tbl_df`
-#'
-#' @details
-#' The function creates lightweight semantic metadata but intentionally
-#' avoids:
-#'
-#' - authoritative archival description,
-#' - full RiC-O graph construction,
-#' - provenance reasoning,
-#' - or ontology-complete archival modelling.
-#'
-#' This lightweight semantic layer is intended for:
-#'
-#' - analytical reconstruction,
-#' - contextual reporting,
-#' - HDTO-like analytical workspaces,
-#' - and iterative semantic enrichment workflows.
-#'
-#' @examples
-#' toy_record_set <- tibble::tibble(
-#' structural_group = c(
-#' "_packages/eviota",
-#' "_packages/eviota",
-#' "_packages/iotables"
-#' ),
-#' path_id = c(
-#' "l480::R/import.R",
-#' "l480::data-raw/build.R",
-#' "l480::R/cube.R"
-#' ),
-#' rel_root_path = c(
-#' "R/import.R",
-#' "data-raw/build.R",
-#' "R/cube.R"
-#' )
-#' )
-#'
-#' toy_record_set <- toy_record_set |>
-#' create_record_set(
-#' record_set_id = "structural_group",
-#' resource_id = "path_id",
-#' locator_path = "rel_root_path",
-#' construction_rule =
-#' "filtered_project_roots|structural_group",
-#' resource_type = "file"
-#' ) |>
-#' as_recordset_df(
-#' title = "Toy reconstruction workspace",
-#' creator = person("Daniel", "Antal"),
-#' description =
-#' "Contextual reconstruction record set"
-#' )
-#'
-#' @importFrom tibble as_tibble
-#' @importFrom dataset dublincore
-#' @importFrom utils person
-#'
-#' @export
-as_recordset_df <- function(
- x,
- title,
- creator,
- member_id = "resource_id",
- member_path = "locator_path",
- member_type = "resource_type",
- description = NULL,
- publisher = NULL,
- subject = NULL
-) {
- stopifnot(is.data.frame(x))
-
- x <- tibble::as_tibble(x)
-
- # --------------------------------------------------------------
- # Semantic alignment
- # --------------------------------------------------------------
-
- if (!member_id %in% names(x)) {
- stop(
- "Column not found: ",
- member_id,
- call. = FALSE
- )
- }
-
- names(x)[names(x) == member_id] <-
- "member_id"
-
- if (
- !is.null(member_path) &&
- member_path %in% names(x)
- ) {
- names(x)[names(x) == member_path] <-
- "member_path"
- }
-
- if (
- !is.null(member_type) &&
- member_type %in% names(x)
- ) {
- names(x)[names(x) == member_type] <-
- "member_type"
- }
-
- # --------------------------------------------------------------
- # Semantic metadata
- # --------------------------------------------------------------
-
-
- # Create the record set construction rules as a description
-
- construction_rule <-
- attr(x, "construction_rule")
-
- full_description <- description
-
- if (!is.null(construction_rule)) {
- rule_text <- paste(
- "Construction rule:",
- construction_rule
- )
-
- if (is.null(full_description)) {
- full_description <- rule_text
- } else {
- full_description <- paste(
- full_description,
- "",
- rule_text,
- sep = "\n"
- )
- }
- }
-
- bibentry <- dataset::dublincore(
- title = title,
- description = full_description,
- creator = creator,
- publisher = publisher
- )
-
- # --------------------------------------------------------------
- # Construct semantically enriched recordset_df
- # --------------------------------------------------------------
-
- out <- do.call(
- recordset_df,
- c(
- as.list(x),
- list(
- dataset_bibentry = bibentry,
- dataset_subject = subject
- )
- )
- )
-
- out
-}
diff --git a/R/create_record_set.R b/R/create_record_set.R
deleted file mode 100644
index d384149..0000000
--- a/R/create_record_set.R
+++ /dev/null
@@ -1,352 +0,0 @@
-#' Create a contextual Record Set projection from observational resources
-#'
-#' @description
-#' Constructs a lightweight contextual Record Set projection from an
-#' observational resource table.
-#'
-#' `Record Set` is a contextual aggregation concept defined by the
-#' International Council on Archives (ICA) Records in Contexts
-#' standard (RiC).
-#'
-#' In operational terms, a Record Set may represent:
-#'
-#' - a project workspace;
-#' - a synchronized cloud folder;
-#' - a repository inventory;
-#' - a digitisation batch;
-#' - a web archive collection;
-#' - a reconstruction corpus;
-#' - or another contextual grouping of related digital resources.
-#'
-#' The function is designed as an operational bridge between:
-#'
-#' - filesystem observations;
-#' - web archive inventories;
-#' - digitised heritage collections;
-#' - repository inventories;
-#' - and later semantically enriched Record Set representations.
-#'
-#' The returned object is intentionally a plain tibble rather than a
-#' semantically enriched `recordset_df`.
-#'
-#' This allows:
-#'
-#' - efficient tidyverse workflows;
-#' - exploratory analytical pipelines;
-#' - lightweight contextual reconstruction;
-#' - deferred semantic stabilisation;
-#' - provenance-aware iterative enrichment.
-#'
-#' More information:
-#'
-#' - ICA Records in Contexts overview:
-#' \url{https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/}
-#'
-#' - RiC-O ontology repository:
-#' \url{https://github.com/ica-egad/ric-o}
-#'
-#' In RiC-aligned operational terminology:
-#'
-#' - rows typically represent observed or derived Record Resources,
-#' Instantiations, or other operational resource proxies;
-#'
-#' - the resulting tibble represents a contextual Record Set projection
-#' constructed from deterministic operational rules;
-#'
-#' - the function does not create authoritative archival arrangement,
-#' fonds hierarchy, or curatorial description;
-#'
-#' - Record Set semantics remain analytical and operational unless
-#' later stabilised through curatorial or semantic workflows.
-#'
-#' Typical use cases include:
-#'
-#' - grouping filesystem observations into project-level Record Sets;
-#' - constructing analytical corpora from repository structures;
-#' - creating candidate archival aggregations;
-#' - preparing Heritage Digital Twin analytical spaces;
-#' - contextualising WARC/WACZ collections;
-#' - constructing enrichment workspaces for knowledge-graph workflows.
-#'
-#' The function deliberately separates:
-#'
-#' - operational contextualisation (`create_record_set()`)
-#'
-#' from:
-#'
-#' - semantic publication and metadata enrichment
-#' (`as_recordset_df()`).
-#'
-#' This mirrors the package philosophy used throughout the observational
-#' pipeline:
-#'
-#' - observe first;
-#' - contextualise second;
-#' - interpret later.
-#'
-#' @param x A `data.frame` or tibble containing observational or derived
-#' resource rows.
-#'
-#' @param record_set_id Character scalar or existing column name defining
-#' the contextual Record Set membership of each resource.
-#'
-#' Typical examples include:
-#'
-#' - structural filesystem groupings;
-#' - repository roots;
-#' - WARC collection identifiers;
-#' - digitisation batches;
-#' - curatorial aggregation identifiers.
-#'
-#' @param resource_id Character scalar or existing column name defining
-#' the operational identity of each resource within the Record Set.
-#'
-#' In many filesystem workflows, `resource_id` will often correspond
-#' to what users informally think of as a "file" or "file name".
-#'
-#' However, the identifier intentionally represents an operational or
-#' contextual resource approximation rather than an authoritative or
-#' permanent file identity.
-#'
-#' This distinction matters because digital resources frequently evolve
-#' over time:
-#'
-#' - filenames and paths may change;
-#' - synchronized copies may diverge;
-#' - local and cloud versions may coexist;
-#' - files may be copied, renamed, or reorganised;
-#' - multiple observations may refer to evolving versions of the
-#' same underlying resource.
-#'
-#' For example:
-#'
-#' - the same digital resource ("file") may exist in multiple locations;
-#' - a synchronized cloud copy may differ from a local working copy;
-#' - a renamed file may still represent the continuation of the same
-#' evolving digital resource.
-#'
-#' Typical examples include:
-#'
-#' - `storage_path_id`
-#' (storage-scoped filesystem resource approximation);
-#'
-#' - URI identifiers;
-#'
-#' - WARC record identifiers;
-#'
-#' - repository-relative identifiers;
-#'
-#' - IIIF resource identifiers.
-#'
-#' @param construction_rule Character description documenting the
-#' deterministic operational rule used to construct the contextual
-#' Record Set projection.
-#'
-#' Examples:
-#'
-#' - `"filtered_project_roots|structural_group"`
-#' - `"warc_collection|domain_partition"`
-#' - `"iiif_manifest|folder_batch"`
-#'
-#' The construction rule is stored as lightweight provenance metadata
-#' attached to the resulting tibble.
-#'
-#' @param locator_path Optional character scalar or existing column name
-#' providing a human-readable operational locator associated with the
-#' resource.
-#'
-#' Examples include:
-#'
-#' - filesystem paths;
-#' - repository-relative paths;
-#' - URIs;
-#' - WARC locators;
-#' - IIIF resource paths.
-#'
-#' @param resource_title Optional character scalar or existing column name
-#' containing a human-readable resource title or label.
-#'
-#' @param resource_type Optional character scalar or existing column name
-#' describing the operational resource type.
-#'
-#' Examples:
-#'
-#' - `"file"`
-#' - `"warc_record"`
-#' - `"iiif_canvas"`
-#' - `"rdf_resource"`
-#' - `"digitised_page"`
-#'
-#' @return
-#' A tibble representing a contextual operational Record Set projection.
-#'
-#' The resulting tibble contains:
-#'
-#' - `record_set_id`
-#' - `resource_id`
-#' - optional contextual resource variables
-#'
-#' together with lightweight provenance attributes:
-#'
-#' - `construction_rule`
-#' - `created_by`
-#' - `record_set_created_at`
-#'
-#' @details
-#' The function intentionally performs only lightweight contextual
-#' projection and validation.
-#'
-#' It does not:
-#'
-#' - infer authoritative documentary hierarchy;
-#' - enforce archival arrangement;
-#' - construct RiC-complete semantic graphs;
-#' - perform provenance reasoning;
-#' - stabilise resource identity across time.
-#'
-#' Semantic enrichment and publication-oriented metadata are intended
-#' to be added later via `as_recordset_df()`.
-#'
-#' This staged architecture supports:
-#'
-#' - efficient analytical workflows;
-#' - iterative reconstruction;
-#' - provenance-aware contextualisation;
-#' - future alignment with RiC-O and RiC-CM.
-#'
-#' @examples
-#' toy_record_set <- tibble::tibble(
-#' structural_group = c(
-#' "heritage_collection",
-#' "heritage_collection",
-#' "digitisation_batch"
-#' ),
-#' storage_path_id = c(
-#' "laptop01::scans/photo_001.tif",
-#' "laptop01::ocr/photo_001.txt",
-#' "archive01::reports/summary.qmd"
-#' ),
-#' rel_root_path = c(
-#' "scans/photo_001.tif",
-#' "ocr/photo_001.txt",
-#' "reports/summary.qmd"
-#' )
-#' )
-#'
-#' toy_record_set <- create_record_set(
-#' toy_record_set,
-#' record_set_id = "structural_group",
-#' resource_id = "storage_path_id",
-#' locator_path = "rel_root_path",
-#' construction_rule =
-#' "filtered_project_roots|structural_group",
-#' resource_type = "file"
-#' )
-#'
-#' @importFrom dplyr mutate select all_of
-#' @importFrom tibble tibble as_tibble
-#' @importFrom rlang .data
-#' @importFrom glue glue
-#' @importFrom stats setNames
-#'
-#' @export
-
-create_record_set <- function(
- x,
- record_set_id,
- resource_id,
- construction_rule,
- locator_path = NULL,
- resource_title = NULL,
- resource_type = NULL
-) {
- stopifnot(is.data.frame(x))
-
- resolve_value <- function(data, value) {
- if (is.null(value)) {
- return(NULL)
- }
-
- if (
- is.character(value) &&
- length(value) == 1 &&
- value %in% names(data)
- ) {
- return(data[[value]])
- }
-
- rep(value, nrow(data))
- }
-
- out <- tibble::as_tibble(x)
-
- # --------------------------------------------------------------
- # Contextual record set projection
- # --------------------------------------------------------------
-
- out$record_set_id <-
- resolve_value(out, record_set_id)
-
- out$resource_id <-
- resolve_value(out, resource_id)
-
- out$locator_path <-
- resolve_value(out, locator_path)
-
- out$resource_title <-
- resolve_value(out, resource_title)
-
- out$resource_type <-
- resolve_value(out, resource_type)
-
- # --------------------------------------------------------------
- # Validation
- # --------------------------------------------------------------
-
- required_cols <- c(
- "record_set_id",
- "resource_id"
- )
-
- missing_required_cols <- setdiff(
- required_cols,
- names(out)
- )
-
- missing_required_values <- required_cols[
- required_cols %in% names(out) &
- vapply(
- out[required_cols[required_cols %in% names(out)]],
- function(col) all(is.na(col)),
- logical(1)
- )
- ]
-
- missing_cols <- c(
- missing_required_cols,
- missing_required_values
- )
-
- if (length(missing_cols) > 0) {
- stop(
- "Missing required values for: ",
- paste(missing_cols, collapse = ", "),
- call. = FALSE
- )
- }
-
- # --------------------------------------------------------------
- # Lightweight provenance
- # --------------------------------------------------------------
-
- attr(out, "construction_rule") <-
- construction_rule
-
- attr(out, "created_by") <-
- "create_record_set"
-
- attr(out, "record_set_created_at") <-
- Sys.time()
-
- out
-}
diff --git a/R/derive_group_path.R b/R/derive_group_path.R
index bd6a763..7fdb85f 100644
--- a/R/derive_group_path.R
+++ b/R/derive_group_path.R
@@ -15,7 +15,7 @@
#'
#' Future versions of the package are expected to replace or absorb
#' this functionality into higher-level Record Set construction logic
-#' (for example via `create_record_set()`), where grouping rules will
+#' (for example via `record_set_projection()`), where grouping rules will
#' be explicitly contextualised and provenance-aware.
#' @param rel_path Character vector of relative file paths.
#' @param repo_root Optional. Currently unused.
diff --git a/R/derive_structural_groups.R b/R/derive_structural_groups.R
index 408ee1d..6d90fc6 100644
--- a/R/derive_structural_groups.R
+++ b/R/derive_structural_groups.R
@@ -1,124 +1,148 @@
-#' Derive structural grouping heuristics from relative paths
+#' Derive structural aggregation metadata from relative paths
#'
-#' Derives lightweight structural grouping heuristics from relative
-#' filesystem paths.
+#' @description
+#' Derives lightweight structural aggregation metadata from observed
+#' relative filesystem paths.
#'
-#' The function extracts shallow structural patterns commonly found in
-#' software projects, research workflows, and digital working environments.
+#' The function identifies recurring structural patterns in directory
+#' hierarchies and creates candidate aggregations that can support
+#' exploratory analysis, navigation, contextual reconstruction, and
+#' later semantic interpretation.
#'
-#' It assigns:
-#'
-#' - `structural_group`:
-#' grouping heuristic derived from the first path component
-#' (e.g. `innolab25`, `_packages`, `_markdown`)
-#'
-#' - `component`:
-#' immediate structural subdivision within the grouping,
-#' if present
-#' (e.g. `eviota`, `filmledgerimport`, `iotables`)
-#'
-#' These derived structures support:
-#'
-#' - exploratory grouping of filesystem observations
-#' - navigation of large observational snapshots
-#' - reconstruction of operational project environments
-#' - identification of candidate documentary aggregations
-#'
-#' The function performs deterministic structural projection only.
-#' It does not validate repository semantics, documentary structure,
-#' or authoritative Record Set boundaries.
+#' The resulting groupings are derived solely from path structure.
+#' They are analytical projections rather than authoritative Record
+#' Sets, provenance assertions, or documentary relationships.
#'
#' @param rel_path Character vector of relative filesystem paths.
#'
-#' @return A `data.frame` with columns:
+#' @param profile Character scalar specifying the structural
+#' aggregation strategy. Available profiles are:
+#' \describe{
+#' \item{"folder-depth-1"}{Group by the first directory level.}
+#' \item{"folder-depth-2"}{Group by the first two directory levels
+#' (default).}
+#' \item{"folder-depth-3"}{Group by the first three directory levels.}
+#' \item{"folder-depth-4"}{Group by the first four directory levels.}
+#' \item{"wacz"}{Use the first path component as the structural group
+#' and the second component as the structural subdivision, matching
+#' the standard organisation of WACZ archives.}
+#' }
+#'
+#' @return
+#' A `data.frame` with two columns:
#' \describe{
#' \item{structural_group}{
-#' Filesystem-based structural grouping heuristic derived from
-#' the first path component.
+#' Candidate structural aggregation derived from the selected path
+#' profile.
#' }
#' \item{component}{
-#' Immediate structural subdivision within the grouping,
-#' if present.
+#' Immediate structural subdivision within the aggregation, when
+#' present.
#' }
#' }
#'
#' @details
-#' This function provides a lightweight structural interpretation layer
-#' on top of observational filesystem data.
-#'
-#' In RiC-aligned operational terms:
-#'
-#' - rows in observational snapshots represent filesystem
-#' Instantiations
-#'
-#' - `rel_path` acts as an operational locator associated with
-#' observed filesystem occurrences
-#'
-#' - the derived structural groupings provide analytical heuristics
-#' that may later support Record Set construction
-#'
-#' The derived groupings are operational analytical projections,
-#' not authoritative RiC Record Sets.
-#'
-#' The function is intended for analytical, navigational,
-#' and exploratory reconstruction workflows.
-#'
-#' Future versions of the package may replace or extend this logic with
-#' more explicit provenance-aware Record Set construction workflows
-#' (for example via `create_record_set()`).
-
+#' Structural aggregation metadata provides a lightweight abstraction
+#' of observed directory organisation. It can increase the
+#' informativeness of filesystem observations by exposing recurring
+#' organisational patterns without asserting semantic meaning.
+#'
+#' Within the fscontext workflow:
+#'
+#' * filesystem observations provide evidence about observed resources;
+#' * relative paths provide structural organisation;
+#' * structural aggregations expose candidate groups that may later
+#' support contextual reconstruction, Record Set construction,
+#' semantic stabilisation, or other downstream analyses.
+#'
+#' Future versions may introduce additional aggregation profiles based
+#' on repository structure, provenance, temporal patterns, or other
+#' observational evidence.
#' @examples
-#' data("fscontextdemo_snapshot_02")
-#'
-#' example_paths <- c(
-#' "_packages/fscontextdemo/R/derive_fsdemo_country_data.R",
-#' "_packages/fscontextdemo/tests/testthat/test-country-data.R",
-#' "_packages/fscontextdemo/data-raw/create_fsdemo_country_data.R",
-#' "_packages/fscontextdemo/docs/index.html"
+#' rel_path <- c(
+#' "_packages/demo/R/file.R",
+#' "_packages/demo/tests/testthat/test-file.R",
+#' "_packages/demo/data/input.csv"
#' )
#'
-#' data.frame(
-#' rel_path = example_paths,
-#' derive_structural_groups(example_paths)
+#' derive_structural_groups(rel_path)
+#'
+#' derive_structural_groups(
+#' rel_path,
+#' profile = "folder-depth-1"
#' )
#'
+#' derive_structural_groups(
+#' c(
+#' "archive/data.warc.gz",
+#' "indexes/index.cdx",
+#' "pages/pages.jsonl"
+#' ),
+#' profile = "wacz"
+#' )
#' @importFrom dplyr bind_rows
#' @export
-derive_structural_groups <- function(rel_path) {
+derive_structural_groups <- function(
+ rel_path,
+ profile = "folder-depth-2"
+) {
+ if (is.null(rel_path) || !is.character(rel_path)) {
+ stop(
+ "rel_path must be a character vector",
+ call. = FALSE
+ )
+ }
+
+ rel_path <- gsub("\\\\", "/", rel_path)
+
parts <- strsplit(rel_path, "/", fixed = TRUE)
+ depth <- switch(profile,
+ "folder-depth-1" = 1L,
+ "folder-depth-2" = 2L,
+ "folder-depth-3" = 3L,
+ "folder-depth-4" = 4L,
+ "wacz" = NA_integer_,
+ stop(
+ "Unknown profile: ",
+ profile,
+ call. = FALSE
+ )
+ )
+
res <- lapply(parts, function(p) {
p <- p[nzchar(p)]
- if (length(p) == 0) {
+ if (length(p) == 0 || all(is.na(p))) {
return(list(
structural_group = NA_character_,
component = NA_character_
))
}
- if (length(p) == 1) {
- return(list(
- structural_group = p[1],
- component = NA_character_
- ))
- }
+ if (profile == "wacz") {
+ structural_group <- p[1]
- if (length(p) == 2) {
- return(list(
- structural_group = paste(p[1], p[2], sep = "/"),
- component = NA_character_
- ))
- }
+ component <- if (length(p) > 1) {
+ p[2]
+ } else {
+ NA_character_
+ }
+ } else {
+ group_depth <- min(depth, length(p))
- structural_group <- paste(
- p[1],
- p[2],
- sep = "/"
- )
+ structural_group <- paste(
+ p[seq_len(group_depth)],
+ collapse = "/"
+ )
- component <- p[3]
+ component <- if (length(p) > group_depth) {
+ p[group_depth + 1]
+ } else {
+ NA_character_
+ }
+ }
list(
structural_group = structural_group,
diff --git a/R/globals.R b/R/globals.R
index f39c9fa..07b00f5 100644
--- a/R/globals.R
+++ b/R/globals.R
@@ -2,6 +2,8 @@
utils::globalVariables(
c(
".",
+ "id",
+ "ts",
"storage_id",
"root",
"dir_path",
@@ -26,6 +28,11 @@ utils::globalVariables(
"avg_size_unit",
"matched_rule",
"refine_id",
+ "resource_locator",
+ "hasText",
+ "text",
+ "mime",
+ "favIconUrl",
"tmp_observed_unit",
"fsdemo_country_data"
)
diff --git a/R/observe_wacz.R b/R/observe_wacz.R
new file mode 100644
index 0000000..c9c1e56
--- /dev/null
+++ b/R/observe_wacz.R
@@ -0,0 +1,267 @@
+#' Observe a WACZ web archive
+#'
+#' @description
+#' Creates an observational data frame from a WACZ web archive.
+#'
+#' The function extracts structural metadata from the archive,
+#' combines page-level information with WARC index metadata, and returns
+#' one observational row for each archived web page.
+#'
+#' The resulting object represents observations only. It intentionally
+#' avoids making semantic assertions about Records, Record Parts,
+#' Instantiations, or other archival entities. Such interpretation can
+#' be added later with [wacz_to_recordset_df()] or downstream semantic
+#' enrichment workflows.
+#'
+#' @param wacz
+#' Path to a `.wacz` archive.
+#'
+#' @return
+#' A tibble containing observations extracted from the archive.
+#'
+#' The returned object carries two attributes:
+#'
+#' * `datapackage`, containing the parsed `datapackage.json`
+#' metadata supplied by the WACZ archive;
+#' * `wacz`, containing the normalized path to the source archive.
+#'
+#' Typical variables include:
+#'
+#' * page identifiers;
+#' * resource locators (URLs);
+#' * page titles;
+#' * timestamps;
+#' * extracted text;
+#' * text signatures;
+#' * MIME types;
+#' * WARC digests;
+#' * archive offsets;
+#' * version counts.
+#'
+#' @details
+#' The function performs the following steps:
+#'
+#' * extracts the WACZ archive into a temporary directory;
+#' * reads the archive `datapackage.json`;
+#' * parses page metadata from `pages/pages.jsonl`;
+#' * parses WARC index metadata from `indexes/index.cdx`;
+#' * collapses multiple archived versions of the same resource;
+#' * joins page observations with archive metadata.
+#'
+#' The resulting observations preserve the evidence contained in the
+#' archive without interpreting its archival semantics.
+#'
+#' @references
+#' The WACZ format specification:
+#' \url{https://specs.webrecorder.net/wacz/1.1.1/}
+#'
+#' @seealso
+#' [wacz_to_recordset_df()]
+#'
+#' @examples
+#' wacz <- system.file("testdata", "fscontext_020.wacz", package = "fscontext")
+#'
+#' observe_wacz(wacz)
+#'
+#' @export
+
+observe_wacz <- function(wacz) {
+ tmp <- tempfile("wacz")
+
+ extract_storage(
+ archive = wacz,
+ exdir = tmp
+ )
+
+ datapackage <- read_datapackage(tmp)
+
+ pages <- read_pages_jsonl(tmp)
+
+ cdx <- read_cdx(tmp) |>
+ collapse_cdx_versions()
+
+ observations <- match_pages_to_cdx(
+ pages,
+ cdx
+ ) |>
+ dplyr::mutate(
+ archive = basename(wacz),
+ full_path = normalizePath(wacz)
+ )
+
+ attr(observations, "datapackage") <- datapackage
+ attr(observations, "wacz") <- normalizePath(wacz)
+
+ observations
+}
+
+#' @keywords internal
+#' @importFrom jsonlite stream_in
+#' @importFrom dplyr filter mutate rename
+#' @importFrom tibble as_tibble
+#' @noRd
+
+read_pages_jsonl <- function(path) {
+ pages_file <- file.path(path, "pages", "pages.jsonl")
+
+ if (!file.exists(pages_file)) {
+ stop(
+ "Cannot find 'pages/pages.jsonl' in ",
+ path,
+ call. = FALSE
+ )
+ }
+
+ pages <- suppressWarnings(
+ jsonlite::stream_in(
+ file(pages_file),
+ verbose = FALSE
+ )
+ ) |>
+ tibble::as_tibble() |>
+ dplyr::rename(
+ page_id = id,
+ resource_locator = url,
+ timestamp = ts,
+ has_text = hasText,
+ favicon = favIconUrl
+ ) |>
+ dplyr::filter(!is.na(resource_locator)) |>
+ dplyr::mutate(
+ text_length = nchar(text),
+ quick_sig_text = quick_signature_text(text)
+ )
+}
+
+#' @keywords internal
+#' @importFrom jsonlite fromJSON
+#' @importFrom tibble as_tibble tibble
+#' @importFrom dplyr mutate rename
+#' @importFrom purrr map_dfr
+#' @noRd
+
+read_cdx <- function(path) {
+ cdx_file <- file.path(path, "indexes", "index.cdx")
+
+ if (!file.exists(cdx_file)) {
+ stop(
+ "Cannot find 'indexes/index.cdx' in ",
+ path,
+ call. = FALSE
+ )
+ }
+
+ lines <- readLines(
+ cdx_file,
+ warn = FALSE,
+ encoding = "UTF-8"
+ )
+
+ lines <- lines[nzchar(lines)]
+
+ parsed <- purrr::map_dfr(
+ lines,
+ function(line) {
+ parts <- strsplit(
+ line,
+ " ",
+ fixed = TRUE
+ )[[1]]
+
+ if (length(parts) < 3) {
+ return(NULL)
+ }
+
+ urlkey <- parts[1]
+ timestamp <- parts[2]
+ json <- paste(
+ parts[-c(1, 2)],
+ collapse = " "
+ )
+
+ meta <- jsonlite::fromJSON(json)
+
+ tibble::tibble(
+ urlkey = urlkey,
+ cdx_timestamp = timestamp,
+ resource_locator = meta$url,
+ digest = meta$digest %||% NA_character_,
+ mime = meta$mime %||% NA_character_,
+ offset = meta$offset %||% NA_real_,
+ length = meta$length %||% NA_real_,
+ record_digest = meta$recordDigest %||% NA_character_,
+ status = meta$status %||% NA_integer_,
+ warc_filename = meta$filename %||% NA_character_
+ )
+ }
+ )
+
+ parsed
+}
+
+#' Collapse repeated WACZ observations of the same archived resource
+#'
+#' @keywords internal
+#' @noRd
+
+collapse_cdx_versions <- function(cdx) {
+ version_count <-
+ cdx |>
+ dplyr::count(
+ resource_locator,
+ name = "n_versions"
+ )
+
+ cdx |>
+ dplyr::filter(
+ mime == "text/html"
+ ) |>
+ dplyr::group_by(resource_locator) |>
+ dplyr::slice(1) |>
+ dplyr::ungroup() |>
+ dplyr::left_join(
+ version_count,
+ by = "resource_locator"
+ )
+}
+
+
+#' Match page observations to archived payload metadata
+#'
+#' @keywords internal
+#' @noRd
+
+match_pages_to_cdx <- function(
+ pages,
+ cdx
+) {
+ dplyr::left_join(
+ pages,
+ cdx,
+ by = "resource_locator"
+ )
+}
+
+#' @keywords internal
+#' @importFrom jsonlite read_json
+#' @importFrom tibble as_tibble
+#' @noRd
+
+read_datapackage <- function(path) {
+ datapackage_file <- file.path(path, "datapackage.json")
+
+ if (!file.exists(datapackage_file)) {
+ stop(
+ "Cannot find 'datapackage.json' in ",
+ path,
+ call. = FALSE
+ )
+ }
+
+ dp <- jsonlite::read_json(
+ datapackage_file,
+ simplifyVector = TRUE
+ )
+
+ dp
+}
diff --git a/R/quick_signature.R b/R/quick_signature.R
index d0453d4..118ab2b 100644
--- a/R/quick_signature.R
+++ b/R/quick_signature.R
@@ -1,53 +1,47 @@
-#' Compute a fast content signature for a file
+#' Compute a fast operational signature for a file
#'
-#' Generates a lightweight content signature based on hashing selected
-#' byte regions of a file. This provides a fast approximation for detecting
-#' identical or differing file instances without computing a full file hash.
+#' Generates a lightweight content signature by hashing sampled byte
+#' regions from a file. The signature provides a fast operational
+#' approximation for detecting identical or differing file instances
+#' without computing a full cryptographic hash.
#'
-#' The function is designed for performance and is suitable for use in
-#' large-scale filesystem observations, where full hashing would be
-#' computationally expensive.
+#' The function is designed for large-scale observational workflows
+#' where complete file hashing would be unnecessarily expensive.
#'
#' @param path Character. Path to the file.
-#' @param n Integer. Number of bytes to read from selected regions
+#' @param n Integer. Number of bytes sampled from selected regions
#' (default: 1024).
#'
-#' @return Character. A signature string representing sampled file content.
+#' @return Character. A lightweight operational signature.
#'
#' @details
-#' The signature is constructed from hashed byte segments:
+#' The signature is constructed by hashing sampled byte regions:
#'
-#' - small files: hash of full content
-#' - medium files: hash of first and last segments
-#' - large files: hash of first, middle, and last segments
+#' * small files: full file content
+#' * medium files: beginning and end
+#' * large files: beginning, middle and end
#'
-#' The function provides a fast operational signal for probable
-#' content equivalence:
+#' The resulting signature is intended as a fast observational aid:
#'
-#' - identical signatures strongly suggest identical content
-#' - different signatures indicate content differences
-#' - collisions are possible but unlikely in practice
+#' * identical signatures suggest identical file content;
+#' * differing signatures indicate differing file content;
+#' * collisions are possible but unlikely in operational use.
#'
#' Missing or inaccessible files return `NA_character_`.
#'
-#' In RiC-aligned operational terms, the signature supports later
-#' interpretation of observed filesystem Instantiations:
+#' The signature does not establish authoritative identity or provenance.
+#' It provides lightweight observational evidence that may support later
+#' contextual reconstruction, duplicate detection, version analysis,
+#' or Record Set construction.
#'
-#' - identifying likely identical Instantiations
-#' - distinguishing likely versions or derivations
-#' - detecting distributed or duplicated work
-#' - supporting later Record Set construction and reconciliation
+#' Unlike [quick_signature_text()], this function operates on the
+#' binary representation of a file rather than its textual content.
#'
-#' The function does not establish authoritative identity or provenance.
-#' It provides observational evidence that may later support analytical
-#' or curatorial interpretation.
+#' @seealso
+#' [quick_signature_text()],
+#' [scan_storage()],
+#' [summarise_duplicates()]
#'
-#' This function is typically used in conjunction with:
-#'
-#' - [scan_storage()] for generating observational snapshots
-#' - [summarise_duplicates()] for detecting duplicate and versioned files
-#'
-#' @seealso [summarise_duplicates()]
#' @importFrom fs file_info
#' @importFrom digest digest
#' @export
@@ -114,3 +108,104 @@ quick_signature <- function(path, n = 1024) {
digest::digest(last, algo = "xxhash32")
)
}
+
+
+#' Compute a fast operational signature for text
+#'
+#' Generates a lightweight content signature by hashing sampled character
+#' regions from one or more text strings. The signature provides a fast
+#' approximation for detecting identical or differing textual content
+#' without comparing complete strings.
+#'
+#' The function is intended for observational workflows where textual
+#' representations have already been extracted from digital resources,
+#' such as HTML pages, OCR output, PDFs, or office documents.
+#'
+#' @param x Character vector.
+#' @param n Integer. Number of characters sampled from selected regions
+#' (default: 1024).
+#'
+#' @return Character vector of operational signatures that summarises the
+#' observed textual representation of a resource.
+#'
+#' @details
+#' The signature is constructed by hashing sampled character regions:
+#'
+#' * short texts: complete text;
+#' * medium texts: beginning and end;
+#' * long texts: beginning, middle and end.
+#'
+#' The resulting signature is intended as a fast observational aid:
+#'
+#' * identical signatures suggest identical textual content;
+#' * differing signatures indicate differing textual content;
+#' * collisions are possible but unlikely in operational use.
+#'
+#' Missing values return `NA_character_`.
+#' Empty strings return `"empty"`.
+#'
+#' Unlike [quick_signature()], this function operates on extracted text
+#' rather than binary file content. Consequently, different file formats
+#' (for example DOCX, PDF and HTML) containing the same textual content
+#' may produce identical text signatures while retaining different file
+#' signatures.
+#'
+#' The function provides lightweight observational evidence that may
+#' support duplicate detection, content reconciliation, semantic
+#' stabilisation, or later contextual reconstruction.
+#'
+#' @seealso
+#' [quick_signature()],
+#' [observe_wacz()]
+#'
+#' @importFrom digest digest
+#' @export
+quick_signature_text <- function(x, n = 1024) {
+ if (length(x) == 0) {
+ return(character())
+ }
+
+ vapply(
+ x,
+ function(text) {
+ if (is.na(text)) {
+ return(NA_character_)
+ }
+
+ text <- enc2utf8(text)
+
+ if (!nzchar(text)) {
+ return("empty")
+ }
+
+ n_chars <- nchar(text, type = "chars")
+
+ if (n_chars <= n) {
+ return(digest::digest(text, algo = "xxhash32"))
+ }
+
+ first <- substr(text, 1, n)
+
+ if (n_chars <= 3 * n) {
+ last <- substr(text, n_chars - n + 1, n_chars)
+
+ return(paste0(
+ digest::digest(first, algo = "xxhash32"),
+ "_",
+ digest::digest(last, algo = "xxhash32")
+ ))
+ }
+
+ middle_start <- floor(n_chars / 2)
+ middle <- substr(text, middle_start, middle_start + n - 1)
+ last <- substr(text, n_chars - n + 1, n_chars)
+
+ paste0(
+ digest::digest(first, algo = "xxhash32"), "_",
+ digest::digest(middle, algo = "xxhash32"), "_",
+ digest::digest(last, algo = "xxhash32")
+ )
+ },
+ character(1)
+ )
+}
diff --git a/R/record_set_projection.R b/R/record_set_projection.R
new file mode 100644
index 0000000..283c4f9
--- /dev/null
+++ b/R/record_set_projection.R
@@ -0,0 +1,186 @@
+#' Project observational resources into a Record Set layout
+#'
+#' @description
+#' Internal helper that projects observational resources into the
+#' canonical tabular layout used by `recordset_df()`.
+#'
+#' The function standardises contextual variables produced during
+#' observational reconstruction by creating the operational columns
+#' expected by the Record Set data model.
+#'
+#' It performs only lightweight projection and validation. No semantic
+#' metadata, RiC assertions, or provenance graph are created.
+#'
+#' @param x A data frame containing observational or derived resources.
+#'
+#' @param record_set_identifier Character scalar or existing column name
+#' defining the contextual Record Set to which each resource belongs.
+#'
+#' @param resource_id Character scalar or existing column name defining
+#' the identifier of each resource.
+#'
+#' @param construction_rule Optional description of the deterministic
+#' rule used to derive the Record Set projection. Stored as an attribute
+#' of the returned tibble.
+#'
+#' @param locator_path Optional character scalar or existing column name
+#' giving a human-readable locator for each resource.
+#'
+#' @param resource_title Optional character scalar or existing column
+#' name containing a human-readable resource title.
+#'
+#' @param resource_type Optional character scalar or existing column
+#' name describing the operational resource type.
+#'
+#' @return
+#' A tibble containing the canonical operational Record Set variables,
+#' including:
+#'
+#' \describe{
+#' \item{record_set_identifier}{Contextual Record Set identifier.}
+#' \item{resource_id}{Operational resource identifier.}
+#' \item{locator_path}{Optional resource locator.}
+#' \item{resource_title}{Optional resource label.}
+#' \item{resource_type}{Optional resource type.}
+#' }
+#'
+#' The returned object also retains the `construction_rule` attribute
+#' when supplied.
+#'
+#' @details
+#' This function is an internal step in the fscontext reconstruction
+#' pipeline:
+#'
+#' \preformatted{
+#' filesystem observations
+#' ↓
+#' derive_record_set()
+#' ↓
+#' record_set_projection()
+#' ↓
+#' recordset_df()
+#' }
+#'
+#' `record_set_projection()` standardises an operational reconstruction.
+#' Semantic metadata, provenance and RiC-oriented annotations are added
+#' later by `recordset_df()`.
+#'
+#' Typical use:
+#'
+#' \preformatted{
+#' toy_record_set <- tibble::tibble(
+#' structural_group = c("heritage_collection",
+#' "heritage_collection"),
+#' storage_path_id = c("a", "b"),
+#' rel_root_path = c("a.html", "b.html")
+#' )
+#'
+#' toy_record_set <- record_set_projection(
+#' toy_record_set,
+#' record_set_identifier = "structural_group",
+#' resource_id = "storage_path_id",
+#' locator_path = "rel_root_path"
+#' )
+#' }
+#'
+#' @keywords internal
+#' @noRd
+#' @importFrom dplyr mutate select all_of
+#' @importFrom tibble tibble as_tibble
+#' @importFrom rlang .data
+#' @importFrom glue glue
+#' @importFrom stats setNames
+
+record_set_projection <- function(
+ x,
+ record_set_identifier,
+ resource_id,
+ construction_rule,
+ locator_path = NULL,
+ resource_title = NULL,
+ resource_type = NULL
+) {
+ stopifnot(is.data.frame(x))
+
+ resolve_value <- function(data, value) {
+ if (is.null(value)) {
+ return(NULL)
+ }
+
+ if (
+ is.character(value) &&
+ length(value) == 1 &&
+ value %in% names(data)
+ ) {
+ return(data[[value]])
+ }
+
+ rep(value, nrow(data))
+ }
+
+ out <- tibble::as_tibble(x)
+
+ # --------------------------------------------------------------
+ # Contextual record set projection
+ # --------------------------------------------------------------
+
+ out$record_set_identifier <-
+ resolve_value(out, record_set_identifier)
+
+ out$resource_id <-
+ resolve_value(out, resource_id)
+
+ out$locator_path <-
+ resolve_value(out, locator_path)
+
+ out$resource_title <-
+ resolve_value(out, resource_title)
+
+ out$resource_type <-
+ resolve_value(out, resource_type)
+
+ # --------------------------------------------------------------
+ # Validation
+ # --------------------------------------------------------------
+
+ required_cols <- c(
+ "record_set_identifier",
+ "resource_id"
+ )
+
+ missing_required_cols <- setdiff(
+ required_cols,
+ names(out)
+ )
+
+ missing_required_values <- required_cols[
+ required_cols %in% names(out) &
+ vapply(
+ out[required_cols[required_cols %in% names(out)]],
+ function(col) all(is.na(col)),
+ logical(1)
+ )
+ ]
+
+ missing_cols <- c(
+ missing_required_cols,
+ missing_required_values
+ )
+
+ if (length(missing_cols) > 0) {
+ stop(
+ "Missing required values for: ",
+ paste(missing_cols, collapse = ", "),
+ call. = FALSE
+ )
+ }
+
+ # --------------------------------------------------------------
+ # Lightweight provenance
+ # --------------------------------------------------------------
+
+ attr(out, "construction_rule") <- construction_rule
+
+
+ out
+}
diff --git a/R/recordset_df.R b/R/recordset_df.R
index 72b8bc8..abe5f77 100644
--- a/R/recordset_df.R
+++ b/R/recordset_df.R
@@ -1,255 +1,251 @@
-#' @title Create a provenance-aware Record Set data frame
+#' @title Create a semantically annotated Record Set
#'
#' @description
-#' Construct a `recordset_df`, a provenance-aware contextual dataset
-#' representing members of a Record Set.
+#' Create a `recordset_df`, a lightweight extension of
+#' `dataset::dataset_df` for representing archival Record Sets.
#'
-#' `Record Set` is a contextual aggregation concept defined by the
-#' International Council on Archives (ICA) Records in Contexts
-#' standard (RiC). In operational terms, a Record Set may represent:
+#' A `recordset_df` stores dataset-level metadata together with
+#' optional Record and Record Part identifiers using lightweight
+#' conventions inspired by the Records in Contexts (RiC) model.
+#' It supports provenance-aware archival, curatorial and semantic
+#' enrichment workflows while remaining compatible with ordinary
+#' data frames.
#'
-#' - a project workspace;
-#' - a research corpus;
-#' - a synchronized working environment;
-#' - a digital collection;
-#' - a reconstruction context;
-#' - or another contextual grouping of related digital resources.
+#' See the **"Working with Record Sets"** vignette for a complete
+#' workflow starting from filesystem observations.
#'
-#' The Records in Contexts (RiC) standard provides a flexible and
-#' provenance-aware approach for describing evolving digital records,
-#' their relationships, and their contextual environments.
+#' @param x A `data.frame` or `dataset_df`.
#'
-#' Unlike rigid hierarchical archival models, RiC allows records and
-#' digital resources to participate in multiple overlapping contextual
-#' groupings while preserving provenance and contextual relationships.
+#' @param title Character scalar giving the title of the Record Set.
#'
-#' More information:
+#' @param creator A `utils::person()` object describing the creator of
+#' the Record Set metadata.
#'
-#' - ICA Records in Contexts overview:
-#' \url{https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/}
+#' @param description Optional description of the Record Set.
#'
-#' - RiC-O ontology repository:
-#' \url{https://github.com/ica-egad/ric-o}
+#' @param record_set_identifier Optional identifier of the Record Set.
#'
-#' A `recordset_df` extends the
-#' \code{\link[dataset:dataset_df]{dataset_df}} class with lightweight
-#' contextual Record Set semantics suitable for:
+#' @param record_identifier Name of the column containing Record
+#' identifiers. The selected column is annotated as
+#' `rico:Identifier` and labelled "Record Identifier".
#'
-#' - filesystem observations;
-#' - synchronized cloud folders;
-#' - web archive members;
-#' - digital surrogate collections;
-#' - curation batches;
-#' - Digital Twin workspaces;
-#' - provenance-aware research collections;
-#' - contextual digital preservation workflows.
+#' @param record_part_identifier Name of the column containing Record
+#' Part identifiers. The selected column is annotated as
+#' `rico:Identifier` and labelled "Record Part Identifier".
#'
-#' The class is designed to work together with:
+#' @param record_subject Subject term describing the Record Set.
+#' Defaults to `"Record Set"`.
#'
-#' - [read_snapshot()]
-#' - [snapshot_to_reconstruction_context()]
-#' - [snapshot_to_recordset_df()]
-#'
-#' while preserving the distinction between:
-#'
-#' - observed filesystem evidence;
-#' - contextual grouping of related resources;
-#' - later analytical interpretation;
-#' - and archival or semantic enrichment workflows.
-#'
-#' In operational terms:
-#'
-#' - `record_set_id`
-#' identifies a contextual grouping of related digital resources
-#' (similar to a project workspace, collection, or reconstruction
-#' environment);
-#'
-#' - `member_id`
-#' identifies one observed or asserted member within that grouping.
-#'
-#' The resulting object inherits from:
-#'
-#' - `recordset_df`
-#' - `dataset_df`
-#' - `tbl_df`
-#' - `tbl`
-#' - `data.frame`
-#'
-#' @param ...
-#' Vectors (columns) to include in the record set.
-#'
-#' @param identifier
-#' A named vector of URI prefixes used to generate row identifiers.
-#'
-#' Defaults to:
-#'
-#' `c(member = "http://example.com/recordset#member")`
-#'
-#' @param var_labels
-#' Optional named list of human-readable variable labels.
-#'
-#' @param units
-#' Optional named list of measurement units.
-#'
-#' @param concepts
-#' Optional named list of semantic concept URIs.
-#'
-#' @param dataset_bibentry
-#' Optional bibliographic metadata created with
-#' \code{dataset::dublincore()} or
-#' \code{dataset::datacite()}.
-#'
-#' @param dataset_subject
-#' Optional dataset subject metadata.
+#' @param ... Reserved for future extensions.
#'
#' @return
-#' A `recordset_df` object.
-#'
-#' @details
-#' The constructor requires at minimum the columns:
-#'
-#' - `record_set_id`
-#' - `member_id`
-#'
-#' Validation and class assignment are delegated to
-#' \code{\link{new_recordset_df}}.
-#'
-#' The constructor is intentionally lightweight and does not:
-#'
-#' - infer authoritative archival hierarchy;
-#' - reconcile duplicate identities;
-#' - infer canonical resources;
-#' - construct ontology-complete provenance graphs;
-#' - or replace curatorial or archival interpretation.
-#'
-#' Instead, it provides a stable contextual preservation layer for
-#' provenance-aware reconstruction and human-in-the-loop workflows.
+#' A `recordset_df`, which inherits from `dataset_df`, `tbl_df`,
+#' `tbl` and `data.frame`.
#'
#' @examples
-#' toy_recordset <- recordset_df(
-#' record_set_id = c(
-#' "heritage_digitisation",
-#' "heritage_digitisation",
-#' "heritage_digitisation"
-#' ),
-#' member_id = c(
-#' "inst_001",
-#' "inst_002",
-#' "inst_003"
-#' ),
-#' member_path = c(
-#' "scans/photo_001.tif",
-#' "ocr/photo_001.txt",
-#' "reports/collection_summary.qmd"
-#' ),
-#' member_type = c(
-#' "file",
-#' "file",
-#' "file"
-#' ),
-#' source_type = c(
-#' "filesystem",
-#' "filesystem",
-#' "filesystem"
+#' x <- data.frame(
+#' resource_locator = c(
+#' "https://example.org/1", "https://example.org/2"
#' ),
-#' identifier = c(
-#' member =
-#' "https://example.org/recordset/heritage#member"
-#' ),
-#' var_labels = list(
-#' record_set_id = "Record set identifier",
-#' member_id = "Member identifier",
-#' member_path = "Member path"
-#' ),
-#' concepts = list(
-#' record_set_id =
-#' "https://www.ica.org/standards/RiC/ontology#RecordSet",
-#' member_id =
-#' "https://www.ica.org/standards/RiC/ontology#Instantiation"
-#' ),
-#' dataset_bibentry = dataset::dublincore(
-#' title = "Toy Heritage Digitisation Record Set",
-#' creator = person("Jane", "Doe"),
-#' publisher = "fscontext"
-#' )
+#' filename = c("a.html", "b.html"),
+#' stringsAsFactors = FALSE
+#' )
+#'
+#' rs <- recordset_df(
+#' x,
+#' title = "Demo Record Set",
+#' creator = utils::person("Joe", "Doe", role = "aut"),
+#' record_identifier = "resource_locator",
+#' record_part_identifier = "filename"
#' )
#'
-#' toy_recordset
+#' rs
+#'
+#' @references
+#' International Council on Archives Expert Group on Archival
+#' Description (2023). Records in Contexts (RiC).
+#' https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/
#'
+#' @seealso
+#' [dataset::dataset_df()], [observe_wacz()],
+#' [wacz_to_recordset_df()]
+#'
+#' @importFrom dataset as_dataset_df defined dublincore subject identifier
+#' @importFrom utils person
#' @export
recordset_df <- function(
- ...,
- identifier = c(
- member = "http://example.com/recordset#member"
- ),
- var_labels = NULL,
- units = NULL,
- concepts = NULL,
- dataset_bibentry = NULL,
- dataset_subject = NULL
+ x,
+ title = NULL,
+ creator = utils::person("Jane", "Doe"),
+ description = NULL,
+ record_set_identifier = NULL,
+ record_identifier = NULL,
+ record_part_identifier = NULL,
+ record_subject = "Record Set",
+ ...
) {
- x <- dataset::dataset_df(
- ...,
- identifier = identifier,
- var_labels = var_labels,
- units = units,
- concepts = concepts,
- dataset_bibentry = dataset_bibentry,
- dataset_subject = dataset_subject
+ if (is.null(creator)) {
+ rs_creator <- "Untitled Record Set"
+ } else if (!inherits(creator, "person")) {
+ rs_creator <- person(creator)
+ } else {
+ rs_creator <- creator
+ }
+
+ if (is.null(description) || is.na(description)) {
+ rs_description <- NULL
+ } else {
+ rs_description <- description
+ }
+
+ ds_bibentry <- dataset::dublincore(
+ title = if (is.null(title)) "Untitled Record Set" else title,
+ identifier = record_set_identifier,
+ creator = creator,
+ description = description
)
- new_recordset_df(x)
+ ds_bibentry
+
+ y <- dataset::as_dataset_df(x)
+
+ attr(y, "dataset_bibentry") <- ds_bibentry
+
+ dataset::subject(y) <-
+ dataset::subject_create(term = record_subject)
+
+
+ new_recordset_df(
+ y,
+ record_identifier = record_identifier,
+ record_part_identifier = record_part_identifier
+ )
}
-#' Internal constructor for `recordset_df`
+#' Internal constructor for recordset_df
#'
#' @description
-#' Low-level internal constructor for creating `recordset_df` objects.
+#' Low-level constructor for recordset_df objects.
#'
-#' This function:
+#' This function extends an existing dataset_df with lightweight
+#' Record Set semantics by:
#'
-#' - validates required columns
-#' - assigns the `recordset_df` class
-#' - preserves existing classes
+#' * assigning the recordset_df class;
+#' * optionally assigning a Record Set identifier and provenance;
+#' * optionally declaring Record and Record Part identifier columns as
+#' rico:Identifier using [dataset::defined()].
#'
-#' Unlike \code{\link{recordset_df}}, this constructor does not create
-#' semantic metadata structures or perform user-facing coercion.
+#' Unlike [recordset_df()], this constructor assumes that dataset-level
+#' metadata have already been created and performs no coercion from
+#' ordinary data.frame objects.
#'
-#' @param x
-#' A data.frame or tibble containing at minimum:
+#' @param x A dataset_df object.
#'
-#' - `record_set_id`
-#' - `member_id`
+#' @param record_identifier Optional name of the column containing
+#' Record identifiers.
+#'
+#' @param record_part_identifier Optional name of the column containing
+#' Record Part identifiers.
#'
#' @return
-#' A `recordset_df` object.
+#' A recordset_df object inheriting from dataset_df.
#'
+#' @importFrom dataset identifier
#' @keywords internal
-new_recordset_df <- function(x) {
- stopifnot(is.data.frame(x))
+new_recordset_df <- function(
+ x,
+ record_identifier = NULL,
+ record_part_identifier = NULL
+) {
+ if (!inherits(x, "dataset_df")) {
+ stop("`x` must inherit from dataset_df.", call. = FALSE)
+ }
- required_cols <- c(
- "record_set_id",
- "member_id"
- )
+ record_set_identifier <- dataset::identifier(x)
- missing_cols <- setdiff(
- required_cols,
- names(x)
- )
+ dataset::provenance(x) <- recordset_provenance(record_set_identifier)
+
+ if (!is.null(record_identifier)) {
+ if (!record_identifier %in% names(x)) {
+ stop("Column not found: ", record_identifier, call. = FALSE)
+ }
- if (length(missing_cols) > 0) {
- stop(
- "Missing required columns: ",
- paste(missing_cols, collapse = ", "),
- call. = FALSE
+ if (anyDuplicated(x[[record_identifier]])) {
+ warning("Record identifiers are not unique.", call. = FALSE)
+ }
+
+ x[[record_identifier]] <- dataset::defined(
+ x[[record_identifier]],
+ label = "Record Identifier",
+ concept = "rico:Identifier"
)
}
- class(x) <- unique(c(
- "recordset_df",
- class(x)
- ))
+ if (!is.null(record_part_identifier)) {
+ if (!record_part_identifier %in% names(x)) {
+ stop("Column not found: ", record_part_identifier, call. = FALSE)
+ }
+
+ if (anyDuplicated(x[[record_part_identifier]])) {
+ warning("Record part identifiers are not unique.", call. = FALSE)
+ }
+
+ x[[record_part_identifier]] <- dataset::defined(
+ x[[record_part_identifier]],
+ label = "Record Part Identifier",
+ concept = "rico:Identifier"
+ )
+ }
+
+ class(x) <- unique(c("recordset_df", class(x)))
+
+
+ attr(x, "prov") <-
+ recordset_provenance(record_set_identifier = record_set_identifier)
x
}
+
+#' Default internal provenance constructor
+#' A wrapper around [dataset::n_triples()] and [dataset::n_triple()].
+#' @keywords internal
+#' @importFrom dataset n_triples n_triple
+#' @noRd
+recordset_provenance <- function(
+ record_set_identifier = NULL
+) {
+ if (is.null(record_set_identifier)) {
+ record_set_identifier <- "http://example.com/recordset"
+ }
+
+
+ activity <- paste0(record_set_identifier, "/activity")
+
+ dataset::n_triples(c(
+ dataset::n_triple(
+ record_set_identifier,
+ "http://www.w3.org/1999/02/22-rdf-syntax-ns#type",
+ "http://www.w3.org/ns/prov#Entity"
+ ),
+ dataset::n_triple(
+ activity,
+ "http://www.w3.org/1999/02/22-rdf-syntax-ns#type",
+ "http://www.w3.org/ns/prov#Activity"
+ ),
+ dataset::n_triple(
+ "https://fscontext.dataobservatory.eu/software/fscontext",
+ "http://www.w3.org/1999/02/22-rdf-syntax-ns#type",
+ "http://www.w3.org/ns/prov#SoftwareAgent"
+ ),
+ dataset::n_triple(
+ activity,
+ "http://www.w3.org/ns/prov#wasAssociatedWith",
+ "https://fscontext.dataobservatory.eu/software/fscontext"
+ ),
+ dataset::n_triple(
+ record_set_identifier,
+ "http://www.w3.org/ns/prov#wasGeneratedBy",
+ activity
+ )
+ ))
+}
diff --git a/R/save_scan.R b/R/save_scan.R
index 38484d9..1f7d5a3 100644
--- a/R/save_scan.R
+++ b/R/save_scan.R
@@ -28,26 +28,26 @@
#'
#' @importFrom fs dir_exists dir_create
#' @examples
-#' \dontrun{
-#' root <- tempfile()
-#' dir.create(root)
+#' tmp_dir <- tempfile()
+#' dir.create(tmp_dir)
#'
-#' dir.create(file.path(root, "R"))
-#' dir.create(file.path(root, "data"))
+#' dir.create(file.path(tmp_dir, "R"))
+#' dir.create(file.path(tmp_dir, "data"))
#'
-#' file.create(file.path(root, "R", "a.R"))
-#' file.create(file.path(root, "R", "b.R"))
-#' file.create(file.path(root, "data", "c.csv"))
+#' file.create(file.path(tmp_dir, "R", "a.R"))
+#' file.create(file.path(tmp_dir, "R", "b.R"))
+#' file.create(file.path(tmp_dir, "data", "c.csv"))
#'
-#' scan_storage(
-#' root = root,
-#' storage_id = "test-storage",
-#' path = tmp
+#' scan <- scan_storage(
+#' root = tmp_dir,
+#' storage_id = "test-storage"
#' )
#'
-#' save_scan(scan, "test-storage")
-#' }
-#'
+#' save_scan(
+#' df = scan,
+#' storage_id = "test-storage",
+#' path = tmp_dir
+#' )
#' @export
save_scan <- function(df,
storage_id,
diff --git a/R/scan_storage.R b/R/scan_storage.R
index 7975030..1483d40 100644
--- a/R/scan_storage.R
+++ b/R/scan_storage.R
@@ -96,11 +96,213 @@
#'
#' @export
scan_storage <- function(root,
- storage_id = "l480-1-ssd",
- person_id = "antaldaniel",
+ storage_id = "local-storage",
+ person_id = "local-user",
scan_time = Sys.time(),
compute_signature = TRUE,
- max_signature_size = 200 * 1024 * 1024) {
+ max_signature_size =
+ 200 * 1024 * 1024) {
+ root <- fs::path_abs(root)
+
+ if (fs::dir_exists(root)) {
+ return(
+ scan_directory_storage(
+ root = root,
+ storage_id = storage_id,
+ person_id = person_id,
+ scan_time = scan_time,
+ compute_signature = compute_signature,
+ max_signature_size = max_signature_size
+ )
+ )
+ }
+
+ ext <- tolower(fs::path_ext(root))
+
+ if (ext %in% c("zip", "wacz")) {
+ return(
+ scan_zip_storage(
+ root = root,
+ storage_id = storage_id,
+ person_id = person_id,
+ scan_time = scan_time,
+ compute_signature = compute_signature,
+ max_signature_size = max_signature_size
+ )
+ )
+ }
+
+ stop(
+ "scan_storage(): unsupported storage type: ",
+ root,
+ call. = FALSE
+ )
+}
+
+
+#' Extract a ZIP-based storage container
+#'
+#' Internal helper that extracts a ZIP-compatible archive
+#' into a temporary directory and returns the extraction path.
+#'
+#' The function currently supports ZIP-based containers,
+#' including `.zip` and `.wacz` files.
+#'
+#' The extracted directory is intended for immediate use by
+#' `scan_directory_storage()` and should be considered temporary.
+#'
+#' @param archive Character. Path to a ZIP-compatible archive.
+#'
+#' @param exdir Character. Extraction directory.
+#'
+#' @details
+#' The function returns a character scalar of the extraction directory.
+#'
+#' @keywords internal
+#' @noRd
+#' @importFrom utils unzip
+
+extract_storage <- function(
+ archive,
+ exdir
+) {
+ archive <- fs::path_abs(archive)
+
+ if (!fs::file_exists(archive)) {
+ stop(
+ "extract_storage(): archive does not exist: ",
+ archive,
+ call. = FALSE
+ )
+ }
+
+ fs::dir_create(exdir)
+
+ utils::unzip(zipfile = archive, exdir = exdir)
+
+ exdir
+}
+
+
+#' Observe a ZIP-based storage container
+#'
+#' Internal helper used by [scan_storage()] for ZIP-compatible
+#' archive containers.
+#'
+#' The archive is extracted into a temporary directory and then
+#' observed using `scan_directory_storage()`. This preserves the
+#' existing filesystem observation workflow while allowing
+#' archive-backed storage roots to be analysed.
+#'
+#' Supported formats currently include:
+#'
+#' - `.zip`
+#' - `.wacz`
+#'
+#' Archive-specific interpretation is deliberately excluded.
+#' The function observes the extracted file hierarchy and records
+#' the archive container as contextual provenance.
+#'
+#' @inheritParams scan_storage
+#'
+#' @details
+#' Returns a filesystem observation data frame equivalent to that returned
+#' by `scan_directory_storage()`, with additional archive
+#' provenance columns.
+#'
+#' @keywords internal
+#' @noRd
+#' @importFrom fs path_abs path_ext path dir_create
+#' @importFrom utils unzip
+#' @importFrom tools file_path_sans_ext
+
+scan_zip_storage <- function(
+ root,
+ storage_id,
+ person_id,
+ scan_time = Sys.time(),
+ compute_signature = TRUE,
+ max_signature_size = 200 * 1024 * 1024
+) {
+ archive_path <- fs::path_abs(root)
+ archive_id <- tools::file_path_sans_ext(basename(archive_path))
+ extract_root <- fs::path(tempdir(), archive_id)
+
+ if (fs::dir_exists(extract_root)) {
+ fs::dir_delete(extract_root)
+ }
+
+ # ---------------------------------------------------------
+ # Determine observational root.
+ #
+ # Many ZIP tools (including Windows Explorer) create:
+ #
+ # archive.zip
+ # └─ folder/
+ # ├─ DESCRIPTION
+ # ├─ NAMESPACE
+ # └─ ...
+ #
+ # while direct filesystem scans start at:
+ #
+ # folder/
+ # ├─ DESCRIPTION
+ # ├─ NAMESPACE
+ # └─ ...
+ #
+ # To preserve observational equivalence, descend into the
+ # single extracted top-level directory when present.
+ # ---------------------------------------------------------
+
+
+ zip_manifest <- utils::unzip(archive_path, list = TRUE)
+ zip_paths <- gsub("\\\\", "/", zip_manifest$Name)
+ zip_paths <- zip_paths[nzchar(zip_paths)]
+
+ top_parts <- sub("/.*$", "", zip_paths)
+ top_parts <- unique(top_parts[nzchar(top_parts)])
+
+ extract_storage(archive = archive_path, exdir = extract_root)
+
+ archive_scan_root <- extract_root
+
+ if (length(top_parts) == 1) {
+ candidate_root <- fs::path(extract_root, top_parts[1])
+ if (fs::dir_exists(candidate_root)) {
+ archive_scan_root <- candidate_root
+ }
+ }
+
+ out <- scan_directory_storage(
+ root = archive_scan_root,
+ storage_id = storage_id,
+ person_id = person_id,
+ scan_time = scan_time,
+ compute_signature = compute_signature,
+ max_signature_size = max_signature_size
+ )
+
+ out$container_file <- basename(archive_path)
+ out$container_type <- tolower(fs::path_ext(archive_path))
+
+ attr(out, "archive_root") <- archive_path
+ attr(out, "extracted_root") <- extract_root
+ attr(out, "archive_scan_root") <- archive_scan_root
+
+ out
+}
+
+#' @keywords internal
+#' @noRd
+#' @importFrom fs path_abs dir_exists file_info path_file path_rel path_ext
+scan_directory_storage <- function(
+ root,
+ storage_id,
+ person_id,
+ scan_time = Sys.time(),
+ compute_signature = TRUE,
+ max_signature_size = 200 * 1024 * 1024
+) {
start_time <- Sys.time()
message("Starting scan_storage() on: ", root)
diff --git a/R/snapshot_storage.R b/R/snapshot_storage.R
index 2f9d3f7..e7142a9 100644
--- a/R/snapshot_storage.R
+++ b/R/snapshot_storage.R
@@ -64,7 +64,7 @@ snapshot_storage <- function(
person_id = "user",
scan_time = Sys.time(),
label = NULL,
- path = here::here("data-raw", "snapshots"),
+ path = tempdir(),
compute_signature = TRUE,
max_signature_size = 200 * 1024 * 1024
) {
diff --git a/R/snapshot_to_reconstruction_context.R b/R/snapshot_to_reconstruction_context.R
index f92f783..e82ad90 100644
--- a/R/snapshot_to_reconstruction_context.R
+++ b/R/snapshot_to_reconstruction_context.R
@@ -61,7 +61,7 @@
#' - `observation_id`;
#' - `structural_group`;
#' - `component`;
-#' - `record_set_id`;
+#' - `record_set_identifier`;
#' - `resource_id`;
#' - `locator_path`.
#'
@@ -93,8 +93,7 @@
#' [snapshot_to_recordset_df()],
#' [subset_snapshot()],
#' [add_snapshot_context()],
-#' [add_structural_groups()],
-#' [create_record_set()]
+#' [add_structural_groups()].
#'
#' @examples
#' data("fscontextdemo_snapshot_01")
@@ -261,8 +260,8 @@ snapshot_to_reconstruction_context <- function(
contextual_snapshot |>
- create_record_set(
- record_set_id = "structural_group",
+ record_set_projection(
+ record_set_identifier = "structural_group",
resource_id = "inst_id",
locator_path = "rel_root_path",
construction_rule = c(
diff --git a/R/snapshot_to_recordset_df.R b/R/snapshot_to_recordset_df.R
index 1d99ca1..7d618fb 100644
--- a/R/snapshot_to_recordset_df.R
+++ b/R/snapshot_to_recordset_df.R
@@ -1,7 +1,7 @@
#' Create a contextual Record Set dataset
#'
#' @description
-#' Creates a provenance-aware `recordset_df` from observational
+#' Creates a provenance-aware [recordset_df()] object from observational
#' filesystem snapshots and contextual reconstruction workflows.
#'
#' The function preserves observed filesystem resources while adding:
@@ -32,12 +32,12 @@
#' @param roots Character vector of contextual root paths used
#' for observational selection.
#'
-#' @param record_set_id Character scalar giving the asserted
+#' @param record_set_identifier Character scalar giving the asserted
#' identifier of the resulting Record Set.
#'
#' @param record_set_title Optional human-readable title.
#'
-#' @param person A [utils::person()] object describing the creator
+#' @param creator A [utils::person()] object describing the creator
#' of the semantic Record Set assertion.
#'
#' @param exclude_patterns Character vector of exclusion patterns
@@ -77,70 +77,57 @@
snapshot_to_recordset_df <- function(
snapshot_files,
roots,
- record_set_id,
+ record_set_identifier,
record_set_title = NULL,
- person = utils::person("Jane", "Doe"),
+ creator = utils::person("Jane", "Doe", role = "aut"),
exclude_patterns = c("\\\\.Rcheck")
) {
- stopifnot(is.character(snapshot_files))
- stopifnot(is.character(roots))
- stopifnot(length(record_set_id) == 1)
+ if (!is.character(snapshot_files)) {
+ stop("`snapshot_files` must be a character vector.", call. = FALSE)
+ }
+
+ if (!all(file.exists(snapshot_files))) {
+ stop("All `snapshot_files` must exist.", call. = FALSE)
+ }
+
+ if (!is.character(roots)) {
+ stop("`roots` must be a character vector.", call. = FALSE)
+ }
+
+ if (
+ !is.character(record_set_identifier) ||
+ length(record_set_identifier) != 1L ||
+ is.na(record_set_identifier)
+ ) {
+ stop("`record_set_identifier` must be a character scalar.", call. = FALSE)
+ }
if (is.null(record_set_title)) {
record_set_title <- paste0(
- "The ",
- record_set_id,
- " filesystem record set"
+ "The ", record_set_identifier, " filesystem record set"
)
}
+
# ------------------------------------------------------------
# Contextual observational reconstruction
# ------------------------------------------------------------
- recordset_df <- snapshot_to_reconstruction_context(
+ rs_tbl_df <- snapshot_to_reconstruction_context(
snapshot_files = snapshot_files,
roots = roots,
exclude_patterns = exclude_patterns
)
# ------------------------------------------------------------
- # Human-defined Record Set assertion
+ # Create recorddataset_df
# ------------------------------------------------------------
- recordset_df$record_set_id <- record_set_id
-
- # ------------------------------------------------------------
- # Create dataset_df
- # ------------------------------------------------------------
-
- recordset_df <- dataset::dataset_df(
- recordset_df,
- identifier = c(
- obs = paste0(
- "https://fscontext.example.org/recordset/",
- record_set_id,
- "#"
- )
- ),
- dataset_bibentry = dataset::dublincore(
- title = record_set_title,
- creator = person,
- description = paste(
- "Filesystem-derived contextual record set",
- "created from observational filesystem snapshots."
- )
- ),
- dataset_subject = dataset::subject_create(
- term = "Record Set",
- valueURI = "https://www.ica.org/standards/RiC/ontology#RecordSet",
- subjectScheme = "RiC-O"
- )
- )
-
- class(recordset_df) <- c(
- "recordset_df",
- class(recordset_df)
+ rs_df <- recordset_df(
+ x = rs_tbl_df,
+ title = record_set_title,
+ record_set_identifier = record_set_identifier,
+ creator = creator
)
# ------------------------------------------------------------
@@ -149,7 +136,7 @@ snapshot_to_recordset_df <- function(
recordset_uri <- paste0(
"https://fscontext.example.org/recordset/",
- record_set_id
+ record_set_identifier
)
activity_uri <- paste0(
@@ -161,8 +148,8 @@ snapshot_to_recordset_df <- function(
"https://fscontext.example.org/software/snapshot_to_recordset_df"
)
- given_name <- paste(person$given, collapse = "_")
- family_name <- paste(person$family, collapse = "_")
+ given_name <- paste(creator$given, collapse = "_")
+ family_name <- paste(creator$family, collapse = "_")
person_uri <- paste0(
"https://fscontext.example.org/agent/",
@@ -184,7 +171,7 @@ snapshot_to_recordset_df <- function(
dataset::n_triple(
person_uri,
"http://www.w3.org/2000/01/rdf-schema#label",
- paste(person$given, person$family)
+ paste(creator$given, creator$family)
),
dataset::n_triple(
software_agent_uri,
@@ -296,7 +283,7 @@ snapshot_to_recordset_df <- function(
# Attach provenance graph
# ------------------------------------------------------------
- dataset::provenance(recordset_df) <- dataset::n_triples(c(
+ dataset::provenance(rs_df) <- dataset::n_triples(c(
agent_triples,
typing_triples,
time_triples,
@@ -306,5 +293,5 @@ snapshot_to_recordset_df <- function(
derivation_triples
))
- recordset_df
+ rs_df
}
diff --git a/R/wacz_to_recordset_df.R b/R/wacz_to_recordset_df.R
new file mode 100644
index 0000000..b2d686f
--- /dev/null
+++ b/R/wacz_to_recordset_df.R
@@ -0,0 +1,189 @@
+#' Create a Record Set dataset from a WACZ observation
+#'
+#' @description
+#' Converts a `wacz_observation` created with [observe_wacz()] into a
+#' semantically enriched `dataset_df` representing a Record Set.
+#'
+#' The function preserves the original observations while attaching
+#' dataset-level metadata and lightweight Records in Contexts (RiC)
+#' semantics. Selected identifier columns may be declared as
+#' `rico:Identifier` values, allowing downstream workflows to distinguish
+#' identifiers intended to refer to Records or Record Parts without
+#' requiring a complete RiC-O implementation.
+#'
+#' The function intentionally performs only lightweight semantic
+#' enrichment. It does not infer Records, Record Parts, Instantiations,
+#' or other archival entities, nor does it reconcile identities or build
+#' provenance graphs. Such interpretation is expected to occur in later
+#' human-guided curation or semantic stabilisation workflows.
+#'
+#' @param wacz_observation
+#' A `wacz_observation` object created with [observe_wacz()].
+#'
+#' @param record_set_id
+#' Optional identifier for the resulting Record Set. If `NULL`, the
+#' basename of the WACZ archive (without extension) is used.
+#'
+#' @param record_set_title
+#' Optional human-readable title for the Record Set. If omitted, a title
+#' is constructed automatically.
+#'
+#' @param record_identifier
+#' Name of the column whose values identify Records represented in the
+#' Record Set. The selected column is annotated as
+#' `rico:Identifier` using [dataset::defined()]. Set to `NULL` to skip
+#' annotation.
+#'
+#' @param record_part_identifier
+#' Optional name of a column whose values identify Record Parts. The
+#' selected column is annotated as `rico:Identifier`.
+#'
+#' @param person
+#' A [utils::person()] object describing the creator of the resulting
+#' dataset metadata.
+#'
+#' @return
+#' A `dataset_df` object enriched with:
+#'
+#' * Dublin Core dataset metadata;
+#' * a RiC Record Set subject;
+#' * optional semantic annotations for Record and Record Part identifiers;
+#' * the original `datapackage` and `wacz` attributes.
+#'
+#' @details
+#' This function occupies the boundary between observational data and
+#' semantic interpretation.
+#'
+#' `observe_wacz()` records observations extracted from a WACZ archive.
+#' `wacz_to_recordset_df()` adds curatorial assertions describing how
+#' particular observed identifiers should be interpreted within a Record
+#' Set, while deliberately avoiding stronger ontological commitments such
+#' as identity reconciliation or Record construction.
+#'
+#' The resulting object is intended for reproducible archival,
+#' curatorial, and semantic enrichment workflows.
+#'
+#' @references
+#' International Council on Archives Expert Group on Archival Description
+#' (2023). Records in Contexts (RiC).
+#'
+#'
+#' @seealso
+#' [observe_wacz()], [dataset::dataset_df()], [dataset::defined()]
+#'
+#' @export
+
+wacz_to_recordset_df <- function(
+ wacz_observation,
+ record_set_id = NULL,
+ record_set_title = NULL,
+ record_identifier = "resource_locator",
+ record_part_identifier = NULL,
+ person = utils::person("Jane", "Doe")
+) {
+ wacz <- attr(wacz_observation, "wacz")
+
+ if (
+ is.null(wacz) ||
+ !is.character(wacz) ||
+ length(wacz) != 1 ||
+ !grepl("\\.wacz$", wacz, ignore.case = TRUE)
+ ) {
+ stop(
+ "`wacz_observation` must be created with observe_wacz().",
+ call. = FALSE
+ )
+ }
+
+ if (is.null(record_set_id)) {
+ record_set_id <- tools::file_path_sans_ext(
+ basename(wacz)
+ )
+ }
+
+ if (is.null(record_set_title)) {
+ record_set_title <- paste(
+ "WACZ Record Set:",
+ record_set_id
+ )
+ }
+
+ if (!is.null(record_identifier) && !record_identifier %in% names(wacz_observation)) {
+ stop(
+ "`record_identifier` must name a column in `wacz_observation`.",
+ call. = FALSE
+ )
+ }
+
+ if (!is.null(record_part_identifier) &&
+ !record_part_identifier %in% names(wacz_observation)) {
+ stop(
+ "`record_part_identifier` must name a column in `wacz_observation`.",
+ call. = FALSE
+ )
+ }
+
+
+ recordset_df <- dataset::dataset_df(
+ wacz_observation,
+ dataset_bibentry = dataset::dublincore(
+ title = record_set_title,
+ creator = person,
+ description = paste(
+ "Record Set created from the WACZ web archive",
+ basename(wacz)
+ )
+ )
+ )
+
+ dataset::subject(recordset_df) <- dataset::subject_create(
+ term = "Record Set",
+ valueURI = "https://www.ica.org/standards/RiC/ontology#RecordSet",
+ subjectScheme = "RiC-O"
+ )
+
+
+ if (!is.null(record_identifier) && (record_identifier %in% names(recordset_df))) {
+ tmp <- recordset_df[[record_identifier]]
+
+ if (anyDuplicated(tmp)) {
+ warning(
+ "Record identifiers are not unique.",
+ call. = FALSE
+ )
+ }
+
+ recordset_df[[record_identifier]] <- dataset::defined(tmp,
+ label = "Record Identifier",
+ concept = "rico:Identifier"
+ )
+ }
+
+ if (!is.null(record_part_identifier) &&
+ (record_part_identifier %in% names(recordset_df))
+ ) {
+ tmp <- recordset_df[[record_part_identifier]]
+
+ if (anyDuplicated(tmp)) {
+ warning(
+ "Record part identifiers are not unique.",
+ call. = FALSE
+ )
+ }
+
+ recordset_df[[record_part_identifier]] <- dataset::defined(
+ tmp,
+ label = "Record Part Identifier",
+ concept = "rico:Identifier"
+ )
+ }
+
+ if (!is.null(record_set_id) && !is.na(record_set_id)) {
+ dataset::identifier(recordset_df) <- record_set_id
+ }
+
+ attr(recordset_df, "datapackage") <- attr(wacz_observation, "datapackage")
+ attr(recordset_df, "wacz") <- wacz
+
+ recordset_df
+}
diff --git a/README.Rmd b/README.Rmd
index d0be4e5..83eb912 100644
--- a/README.Rmd
+++ b/README.Rmd
@@ -24,9 +24,9 @@ knitr::opts_chunk$set(
[](https://lifecycle.r-lib.org/articles/stages.html#experimental)
[](https://www.repostatus.org/#wip)
-[](https://github.com/dataobservatory-eu/fscontext)
+[](https://github.com/dataobservatory-eu/fscontext/tree/devel)
[](https://dataobservatory.eu/)
-[](https://app.codecov.io/gh/dataobservatory-eu/fscontext)
+[](https://app.codecov.io/gh/dataobservatory-eu/fscontext)
@@ -49,16 +49,37 @@ pak::pak("dataobservatory-eu/fscontext")
## Getting started
-The package includes two introductory vignettes:
+The package includes four introductory vignettes that follow the typical
+`fscontext` workflow.
-- [Introduction to fscontext](https://fscontext.dataobservatory.eu/articles/intro.html) introduces file system observations, contextualisation, and Record Set construction.
-- [Prelabelled values and semantic stabilisation](https://fscontext.dataobservatory.eu/articles/prelabelled.html) demonstrates progressive semantic enrichment and refinement workflows.
+- [Introduction to
+ fscontext](https://fscontext.dataobservatory.eu/articles/intro.html)
+ introduces filesystem observations, reproducible snapshots, and contextual
+ reconstruction.
-These vignettes provide a guided introduction to the observational, contextual, and semantic layers of the package.
+- [Working with Record
+ Sets](https://fscontext.dataobservatory.eu/articles/recordset_df.html)
+ demonstrates how observational data can be transformed into provenance-aware
+ `recordset_df` objects inspired by the Records in Contexts (RiC) conceptual
+ model.
-## Contextualisation
+- [Prelabelled values and semantic
+ stabilisation](https://fscontext.dataobservatory.eu/articles/prelabelled.html)
+ introduces lightweight semantic enrichment, rulebooks, and human-in-the-loop
+ refinement workflows.
-Many digital collections contain valuable contextual information but little
+- [Organising Evidence with Structural
+ Aggregations](https://fscontext.dataobservatory.eu/articles/structural_aggregations.html)
+ demonstrates how structural aggregation metadata can identify potentially
+ informative objects and candidate Record Sets across folders, ZIP archives,
+ and WACZ packages.
+
+Together these vignettes introduce the observational, contextual, documentary,
+and semantic layers of the package.
+
+## Context before semantics
+
+Many digital collections contain valuable contextual information but little
documentation explaining how files, datasets, reports, source code, inventories,
or digital surrogates relate to one another.
@@ -77,14 +98,16 @@ as evidence from which contextual structures can be reconstructed.
```
Filesystem observations
- ↓
-Contextualisation
- ↓
-Record Sets
- ↓
-Semantic stabilisation
- ↓
-Knowledge systems
+ ↓
+ Snapshots
+ ↓
+Contextual reconstruction
+ ↓
+ Record Sets
+ ↓
+ Semantic stabilisation
+ ↓
+ Knowledge systems
```
Rather than replacing archival description or provenance models, `fscontext`
@@ -93,25 +116,30 @@ focuses on the earlier task of contextual reconstruction.
The package is inspired by the archival conceptual model [Records in
Contexts](https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/)
(RiC), developed by the *International Council on Archives*. Rather than
-implementing `RiC-CM` or `RiC-O` directly, `fscontext` focuses on the earlier task
-of contextual reconstruction: deriving contextual relationships and candidate
-`Record Sets` from filesystem observations, repository structures, inventories,
-and other digital traces.
+implementing `RiC-CM` or `RiC-O` directly, `fscontext` focuses on the earlier
+task of contextual reconstruction: deriving contextual relationships and
+candidate `Record Sets` from filesystem observations, repository structures,
+inventories, and other digital traces.
For more information, see:
-- [RiC-CM 1.0](https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-conceptual-model/)
+- [RiC-CM
+ 1.0](https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-conceptual-model/)
(Records in Contexts Conceptual Model)
-- [RiC-O 1.1](https://www.ica.org/standards/RiC/RiC-O_1-1.html) (Records in Contexts Ontology)
+- [RiC-O 1.1](https://www.ica.org/standards/RiC/RiC-O_1-1.html) (Records in
+ Contexts Ontology)
## A reproducible example
-The package includes two example filesystem snapshots derived from the companion repository `fscontextdemo`.
+The package includes two example filesystem snapshots derived from the companion
+repository `fscontextdemo`.
The demonstration repository is available at:
-It contains a small but realistic digital work environment including source code, datasets, generated artefacts, documentation, tests, package metadata, semantic enrichment examples.
+It contains a small but realistic digital work environment including source
+code, datasets, generated artefacts, documentation, tests, package metadata,
+semantic enrichment examples.
The snapshots, `fscontextdemo_snapshot_01` and `fscontextdemo_snapshot_02`,
capture the repository at different points in time, allowing reconstruction and
@@ -156,24 +184,31 @@ See the package vignettes for complete end-to-end examples.
The package separates three complementary analytical layers:
| Layer | Purpose |
-|---------------------|------------------------------------------------------|
-| observational | reproducible observations of digital resources and their filesystem context |
-| contextual | grouping observations into Record Sets, projects, collections, and reconstruction workspaces |
-| analytical | reconstruction, temporal comparison, activity analysis, and semantic stabilisation |
-
-The framework intentionally separates observational evidence, contextual
-abstraction, semantic interpretation, and analytical reconstruction. In
-RiC-inspired terms, filesystem observations represent observed digital resources
-and their associated instantiations at a particular point in time. These
-observations may later be aggregated into contextual `Record Sets`, while
-preserving the distinction between the observed resource itself and the
-contextual structures derived from it.
+|----|----|
+| Observation | Observe filesystems and related digital environments as reproducible snapshots. |
+| Context | Derive contextual identifiers, structural aggregations, and candidate Record Sets from observations. |
+| Record Sets | Create lightweight documentary objects using recordset_df, inspired by RiC. |
+| Semantic stabilisation | Support progressive semantic enrichment through prelabelled values, rulebooks, and human review. |
+| Analysis | Compare snapshots, detect duplicates, reconstruct activity, and analyse evolving digital work environments. |
+
+In RiC-inspired terms, filesystem observations represent observed digital
+resources and their associated instantiations at a particular point in time.
+These observations may later be aggregated into contextual `Record Sets` while
+preserving the distinction between the observed resources themselves and the
+contextual structures derived from them. The framework intentionally separates
+observation, contextual organisation, semantic stabilisation, and
+domain-specific interpretation. This allows the same observational evidence to
+support different analytical perspectives—including archival description,
+business process reconstruction, software development, digital forensics,
+historical research, and other forms of contextual analysis—without conflating
+the evidence with its interpretation.
## What this package does not do
-The package does not modify files. It is not intended to replace version control
-systems, reconstruct file contents, infer authoritative archival hierarchy, or
-perform ontology-complete provenance modelling.
+`fscontext` does not attempt to replace archival description, provenance
+ontologies, or knowledge graph platforms. Instead, it provides a reproducible
+observational and contextual layer that can support those systems by making
+digital working environments easier to understand, review, and reconstruct.
## Notes
diff --git a/README.md b/README.md
index 8ca180c..0cb888e 100644
--- a/README.md
+++ b/README.md
@@ -9,9 +9,9 @@
[](https://lifecycle.r-lib.org/articles/stages.html#experimental)
[](https://www.repostatus.org/#wip)
-[](https://github.com/dataobservatory-eu/fscontext)
+[](https://github.com/dataobservatory-eu/fscontext/tree/devel)
[](https://dataobservatory.eu/)
-[](https://app.codecov.io/gh/dataobservatory-eu/fscontext)
+[](https://app.codecov.io/gh/dataobservatory-eu/fscontext)
@@ -33,20 +33,35 @@ reconstruction-oriented analysis.
## Getting started
-The package includes two introductory vignettes:
+The package includes four introductory vignettes that follow the typical
+`fscontext` workflow.
- [Introduction to
fscontext](https://fscontext.dataobservatory.eu/articles/intro.html)
- introduces file system observations, contextualisation, and Record Set
- construction.
+ introduces filesystem observations, reproducible snapshots, and
+ contextual reconstruction.
+
+- [Working with Record
+ Sets](https://fscontext.dataobservatory.eu/articles/recordset_df.html)
+ demonstrates how observational data can be transformed into
+ provenance-aware `recordset_df` objects inspired by the Records in
+ Contexts (RiC) conceptual model.
+
- [Prelabelled values and semantic
stabilisation](https://fscontext.dataobservatory.eu/articles/prelabelled.html)
- demonstrates progressive semantic enrichment and refinement workflows.
+ introduces lightweight semantic enrichment, rulebooks, and
+ human-in-the-loop refinement workflows.
+
+- [Organising Evidence with Structural
+ Aggregations](https://fscontext.dataobservatory.eu/articles/structural_aggregations.html)
+ demonstrates how structural aggregation metadata can identify
+ potentially informative objects and candidate Record Sets across
+ folders, ZIP archives, and WACZ packages.
-These vignettes provide a guided introduction to the observational,
-contextual, and semantic layers of the package.
+Together these vignettes introduce the observational, contextual,
+documentary, and semantic layers of the package.
-## Contextualisation
+## Context before semantics
Many digital collections contain valuable contextual information but
little documentation explaining how files, datasets, reports, source
@@ -68,14 +83,16 @@ treated as evidence from which contextual structures can be
reconstructed.
Filesystem observations
- ↓
- Contextualisation
- ↓
- Record Sets
- ↓
- Semantic stabilisation
- ↓
- Knowledge systems
+ ↓
+ Snapshots
+ ↓
+ Contextual reconstruction
+ ↓
+ Record Sets
+ ↓
+ Semantic stabilisation
+ ↓
+ Knowledge systems
Rather than replacing archival description or provenance models,
`fscontext` focuses on the earlier task of contextual reconstruction.
@@ -188,23 +205,32 @@ The package separates three complementary analytical layers:
| Layer | Purpose |
|----|----|
-| observational | reproducible observations of digital resources and their filesystem context |
-| contextual | grouping observations into Record Sets, projects, collections, and reconstruction workspaces |
-| analytical | reconstruction, temporal comparison, activity analysis, and semantic stabilisation |
-
-The framework intentionally separates observational evidence, contextual
-abstraction, semantic interpretation, and analytical reconstruction. In
-RiC-inspired terms, filesystem observations represent observed digital
-resources and their associated instantiations at a particular point in
-time. These observations may later be aggregated into contextual
-`Record Sets`, while preserving the distinction between the observed
-resource itself and the contextual structures derived from it.
+| Observation | Observe filesystems and related digital environments as reproducible snapshots. |
+| Context | Derive contextual identifiers, structural aggregations, and candidate Record Sets from observations. |
+| Record Sets | Create lightweight documentary objects using recordset_df, inspired by RiC. |
+| Semantic stabilisation | Support progressive semantic enrichment through prelabelled values, rulebooks, and human review. |
+| Analysis | Compare snapshots, detect duplicates, reconstruct activity, and analyse evolving digital work environments. |
+
+In RiC-inspired terms, filesystem observations represent observed
+digital resources and their associated instantiations at a particular
+point in time. These observations may later be aggregated into
+contextual `Record Sets` while preserving the distinction between the
+observed resources themselves and the contextual structures derived from
+them. The framework intentionally separates observation, contextual
+organisation, semantic stabilisation, and domain-specific
+interpretation. This allows the same observational evidence to support
+different analytical perspectives—including archival description,
+business process reconstruction, software development, digital
+forensics, historical research, and other forms of contextual
+analysis—without conflating the evidence with its interpretation.
## What this package does not do
-The package does not modify files. It is not intended to replace version
-control systems, reconstruct file contents, infer authoritative archival
-hierarchy, or perform ontology-complete provenance modelling.
+`fscontext` does not attempt to replace archival description, provenance
+ontologies, or knowledge graph platforms. Instead, it provides a
+reproducible observational and contextual layer that can support those
+systems by making digital working environments easier to understand,
+review, and reconstruct.
## Notes
diff --git a/_pkgdown.yml b/_pkgdown.yml
index 0ad092b..3da00ed 100644
--- a/_pkgdown.yml
+++ b/_pkgdown.yml
@@ -21,66 +21,77 @@ authors:
articles:
+
- title: "Getting Started"
- desc: >
- Learn how to create filesystem snapshots, explore structural
- organisation, construct contextual Record Sets, and analyse
- digital working environments with fscontext.
contents:
- intro
+
+ - title: "Semantic Stabilisation"
+ contents:
- prelabelled
+ - title: "Working with Record Sets"
+ contents:
+ - recordset_df
+
+ - title: "Structural Aggregation"
+ contents:
+ - structural_aggregations
+
reference:
- - title: "Filesystem Snapshots"
+ - title: "Observation"
desc: >
- Create, store and retrieve filesystem observations.
+ Observe digital resources and create reproducible filesystem
+ snapshots.
contents:
- scan_storage
- snapshot_storage
- - read_snapshot
- save_scan
+ - read_snapshot
- subset_snapshot
-
- - title: "Record Sets"
- desc: >
- Construct contextual Record Sets from filesystem observations
- and create semantically enriched recordset_df objects inspired
- by the Records in Contexts model.
- contents:
- - construct_structural_paths
- - create_record_set
- - snapshot_to_reconstruction_context
- - snapshot_to_recordset_df
- - recordset_df
- - as_recordset_df
-
+ - observe_universe
+ - observe_wacz
+ - quick_signature
+ - quick_signature_text
+
- title: "Contextualisation"
+ desc: >
+ Construct contextual groupings and enrich observations with
+ structural context.
contents:
- context_roots
- coverage_roots
- derive_record_set
- invert_contextual_grouping
+ - add_snapshot_context
+ - add_structural_groups
+ - construct_structural_paths
- - title: "Observational Analysis"
+ - title: "Analysis"
desc: >
- Analyse filesystem observations, identify duplicates,
- generated artefacts and operational noise before
- contextual reconstruction.
+ Analyse observations before semantic interpretation.
contents:
- - observe_universe
- - quick_signature
- - detect_generated_artifacts
- - exclude_operational_noise
- - classify_operational_file_type
- - derive_structural_groups
-
+ - summarise_duplicates
+ - summarise_observed_activity
+ - detect_generated_artifacts
+ - exclude_operational_noise
+ - classify_operational_file_type
+ - derive_structural_groups
+
+ - title: "Record Sets"
+ desc: >
+ Create contextual Record Sets and semantically enriched
+ recordset_df objects.
+ contents:
+ - snapshot_to_reconstruction_context
+ - snapshot_to_recordset_df
+ - wacz_to_recordset_df
+ - recordset_df
- title: "Semantic Stabilisation"
desc: >
- Support human-in-the-loop semantic stabilisation through
- prelabelled values, rulebooks, semantic refinement and
- controlled vocabulary workflows.
+ Prepare observations for human review and semantic enrichment.
contents:
- prelabel
- is.prelabelled
@@ -92,26 +103,9 @@ reference:
- compile_rulebook
- coverage_rules_path
- - title: "Activity & Change Analysis"
- desc: >
- Reconstruct activity, change and event patterns from filesystem
- observations.
- contents:
- - summarise_duplicates
- - summarise_observed_activity
-
- - title: "Context Enrichment"
- desc: >
- Add contextual identifiers and structural grouping
- variables to filesystem observations.
- contents:
- - add_snapshot_context
- - add_structural_groups
-
- - title: "Example Snapshots"
+ - title: "Example Data"
desc: >
- Reproducible filesystem snapshots and contextual reconstruction
- examples used throughout the package.
+ Reproducible snapshots used throughout the documentation.
contents:
- fscontextdemo_snapshot_01
- - fscontextdemo_snapshot_02
+ - fscontextdemo_snapshot_02
\ No newline at end of file
diff --git a/codemeta.json b/codemeta.json
index 4020290..8e5ea06 100644
--- a/codemeta.json
+++ b/codemeta.json
@@ -2,12 +2,12 @@
"@context": "https://doi.org/10.5063/schema/codemeta-2.0",
"@type": "SoftwareSourceCode",
"identifier": "fscontext",
- "description": " Provides a provenance-aware framework for contextual reconstruction from file systems and related digital resource collections. The package creates reproducible snapshots of file-level metadata, paths, repository context, and optional content signatures. It supports contextual grouping, structural abstraction, temporal analysis, semantic stabilization, duplicate and reuse detection, and lightweight workflow reconstruction from filesystem observations. The framework deliberately separates observational evidence, contextual abstraction, semantic interpretation, and analytical reconstruction, enabling reproducible and inspectable workflows. It is designed to support future alignment with archival and contextual knowledge representation models, including RiC-CM, RiC-O, and PROV-O.",
- "name": "fscontext: Filesystem Contextualisation and Record Set Reconstruction",
+ "description": " Provides a provenance-aware framework for contextual reconstruction from file systems and related digital resource collections. The package creates reproducible snapshots of file-level metadata, paths, repository context, and optional content signatures. It supports contextual grouping, structural abstraction, temporal analysis, semantic stabilization, duplicate and reuse detection, and lightweight workflow reconstruction from file system observations. The framework deliberately separates observational evidence, contextual abstraction, semantic interpretation, and analytical reconstruction, enabling reproducible workflows that can be inspected by reviewers. It is designed to support future alignment with archival and contextual knowledge representation models, including the World Wide Web Consortium Provenance Ontology (PROV-O): Lebo et al. (2013) and Records in Contexts developed by the International Council on Archives Expert Group on Archival Description (EGAD) .",
+ "name": "fscontext: File System Contextualisation and Record Set Reconstruction",
"codeRepository": "https://fscontext.dataobservatory.eu/",
"issueTracker": "https://github.com/dataobservatory-eu/fscontext/issues",
- "license": "file LICENSE",
- "version": "0.2.0",
+ "license": "https://spdx.org/licenses/GPL-3.0",
+ "version": "0.2.1",
"programmingLanguage": {
"@type": "ComputerLanguage",
"name": "R",
@@ -121,18 +121,6 @@
"sameAs": "https://CRAN.R-project.org/package=progress"
},
"4": {
- "@type": "SoftwareApplication",
- "identifier": "here",
- "name": "here",
- "provider": {
- "@id": "https://cran.r-project.org",
- "@type": "Organization",
- "name": "Comprehensive R Archive Network (CRAN)",
- "url": "https://cran.r-project.org"
- },
- "sameAs": "https://CRAN.R-project.org/package=here"
- },
- "5": {
"@type": "SoftwareApplication",
"identifier": "dplyr",
"name": "dplyr",
@@ -144,12 +132,12 @@
},
"sameAs": "https://CRAN.R-project.org/package=dplyr"
},
- "6": {
+ "5": {
"@type": "SoftwareApplication",
"identifier": "utils",
"name": "utils"
},
- "7": {
+ "6": {
"@type": "SoftwareApplication",
"identifier": "rlang",
"name": "rlang",
@@ -161,7 +149,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=rlang"
},
- "8": {
+ "7": {
"@type": "SoftwareApplication",
"identifier": "purrr",
"name": "purrr",
@@ -173,7 +161,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=purrr"
},
- "9": {
+ "8": {
"@type": "SoftwareApplication",
"identifier": "dataset",
"name": "dataset",
@@ -185,17 +173,17 @@
},
"sameAs": "https://CRAN.R-project.org/package=dataset"
},
- "10": {
+ "9": {
"@type": "SoftwareApplication",
"identifier": "stats",
"name": "stats"
},
- "11": {
+ "10": {
"@type": "SoftwareApplication",
"identifier": "tools",
"name": "tools"
},
- "12": {
+ "11": {
"@type": "SoftwareApplication",
"identifier": "glue",
"name": "glue",
@@ -207,7 +195,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=glue"
},
- "13": {
+ "12": {
"@type": "SoftwareApplication",
"identifier": "tibble",
"name": "tibble",
@@ -219,7 +207,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=tibble"
},
- "14": {
+ "13": {
"@type": "SoftwareApplication",
"identifier": "tidyr",
"name": "tidyr",
@@ -231,7 +219,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=tidyr"
},
- "15": {
+ "14": {
"@type": "SoftwareApplication",
"identifier": "magrittr",
"name": "magrittr",
@@ -243,7 +231,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=magrittr"
},
- "16": {
+ "15": {
"@type": "SoftwareApplication",
"identifier": "stringr",
"name": "stringr",
@@ -255,7 +243,7 @@
},
"sameAs": "https://CRAN.R-project.org/package=stringr"
},
- "17": {
+ "16": {
"@type": "SoftwareApplication",
"identifier": "labelled",
"name": "labelled",
@@ -267,6 +255,18 @@
},
"sameAs": "https://CRAN.R-project.org/package=labelled"
},
+ "17": {
+ "@type": "SoftwareApplication",
+ "identifier": "jsonlite",
+ "name": "jsonlite",
+ "provider": {
+ "@id": "https://cran.r-project.org",
+ "@type": "Organization",
+ "name": "Comprehensive R Archive Network (CRAN)",
+ "url": "https://cran.r-project.org"
+ },
+ "sameAs": "https://CRAN.R-project.org/package=jsonlite"
+ },
"18": {
"@type": "SoftwareApplication",
"identifier": "R",
@@ -275,7 +275,7 @@
},
"SystemRequirements": null
},
- "fileSize": "578.488KB",
+ "fileSize": "3450.424KB",
"citation": [
{
"@type": "CreativeWork",
diff --git a/cran-comments.md b/cran-comments.md
index 6a9c438..6fef07b 100644
--- a/cran-comments.md
+++ b/cran-comments.md
@@ -1,6 +1,8 @@
## Resubmission
-This is the first CRAN resubmission of `fscontext`.
+This is the second CRAN resubmission of `fscontext`.
+
+- The \dontrun{} example was made simpler and runs.
- Examples from internal, not exported function documentations were removed.
diff --git a/inst/WORDLIST b/inst/WORDLIST
index dba7087..28c7f2d 100644
--- a/inst/WORDLIST
+++ b/inst/WORDLIST
@@ -1,15 +1,15 @@
-HDTO
-ICA
-IIIF
+ARChive
+ISAD
Lebo
POSIXct
PROV
+Pomerantz
Prelabelled
RDF
RStudio
+RecordPart
RiC
Sahoo
-URIs
WACZ
WARC
WIP
@@ -19,19 +19,30 @@ codecov
contentReference
dataobservatory
deduplicate
+df
+dplyr
eviota
fonds
+ica
lifecycle
oaicite
operationalisation
operationalisations
+org
+pacakage
prelabelled
+purrr
recontextualisation
+recordset
regex
rel
rhub
+ric
+rico
roundtrip
testthat
tibble
tibbles
+tidyr
tidyverse
+www
diff --git a/inst/testdata/fscontext_020.wacz b/inst/testdata/fscontext_020.wacz
new file mode 100644
index 0000000..f6154b6
Binary files /dev/null and b/inst/testdata/fscontext_020.wacz differ
diff --git a/inst/testdata/minimal_R_folder.zip b/inst/testdata/minimal_R_folder.zip
new file mode 100644
index 0000000..49fca31
Binary files /dev/null and b/inst/testdata/minimal_R_folder.zip differ
diff --git a/man/add_structural_groups.Rd b/man/add_structural_groups.Rd
index 8f96869..6c3781d 100644
--- a/man/add_structural_groups.Rd
+++ b/man/add_structural_groups.Rd
@@ -59,7 +59,7 @@ provenance-aware Record Set construction
Future versions of the package may replace or extend this logic
with more explicit provenance-aware Record Set construction workflows
-(for example via \code{create_record_set()}).
+(for example via \code{record_set_projection()}).
}
\seealso{
\code{\link[=derive_structural_groups]{derive_structural_groups()}}
diff --git a/man/as_recordset_df.Rd b/man/as_recordset_df.Rd
deleted file mode 100644
index a72a5b4..0000000
--- a/man/as_recordset_df.Rd
+++ /dev/null
@@ -1,179 +0,0 @@
-% Generated by roxygen2: do not edit by hand
-% Please edit documentation in R/as_recordset_df.R
-\name{as_recordset_df}
-\alias{as_recordset_df}
-\title{Coerce a contextual Record Set projection into a semantically enriched
-\code{recordset_df}}
-\usage{
-as_recordset_df(
- x,
- title,
- creator,
- member_id = "resource_id",
- member_path = "locator_path",
- member_type = "resource_type",
- description = NULL,
- publisher = NULL,
- subject = NULL
-)
-}
-\arguments{
-\item{x}{A tibble or \code{data.frame}, typically created with
-\code{\link[=create_record_set]{create_record_set()}}.}
-
-\item{title}{Human-readable title of the contextual Record Set.}
-
-\item{creator}{Creator metadata passed to
-\code{\link[dataset:dublincore]{dataset::dublincore()}}.}
-
-\item{member_id}{Character scalar giving the column name in \code{x}
-that should be mapped to \code{member_id} in the resulting
-\code{recordset_df}.
-
-Defaults to \code{"resource_id"}.}
-
-\item{member_path}{Optional character scalar giving the column name
-in \code{x} that should be mapped to \code{member_path}.
-
-Defaults to \code{"locator_path"}.}
-
-\item{member_type}{Optional character scalar giving the column name
-in \code{x} that should be mapped to \code{member_type}.
-
-Defaults to \code{"resource_type"}.}
-
-\item{description}{Optional textual description documenting the
-contextual scope, construction logic, provenance assumptions,
-or analytical purpose of the Record Set.}
-
-\item{publisher}{Optional publisher metadata passed to
-\code{\link[dataset:dublincore]{dataset::dublincore()}}.}
-
-\item{subject}{Optional subject metadata for future semantic
-enrichment.}
-}
-\value{
-A semantically enriched \code{recordset_df} object inheriting from:
-\itemize{
-\item \code{recordset_df}
-\item \code{dataset_df}
-\item \code{tbl_df}
-}
-}
-\description{
-Converts a contextual Record Set projection created with
-\code{\link[=create_record_set]{create_record_set()}} into a semantically enriched \code{recordset_df}
-object.
-}
-\details{
-The function adds lightweight dataset-level semantics and publication
-metadata while preserving tidyverse compatibility.
-
-This staged design deliberately separates:
-\itemize{
-\item operational contextualisation (\code{create_record_set()})
-}
-
-from:
-\itemize{
-\item semantic stabilisation and publication (\code{as_recordset_df()})
-}
-
-The resulting object aligns with the package philosophy of:
-\itemize{
-\item observational acquisition,
-\item contextual enrichment,
-\item deferred semantic interpretation.
-}
-
-\code{as_recordset_df()} is conceptually aligned with:
-\itemize{
-\item RiC-O Record Set projections,
-\item contextual research workspaces,
-\item analytical Heritage Digital Twin layers,
-\item and semantically enriched reconstruction corpora.
-}
-
-The function builds on the \code{dataset_df} framework and therefore
-inherits:
-\itemize{
-\item tibble semantics,
-\item lightweight dataset metadata,
-\item publication-oriented enrichment,
-\item and future linked-data extensibility.
-}
-
-The function also acts as a lightweight semantic alignment layer
-between:
-\itemize{
-\item operational resource-oriented contextualisation
-}
-
-and:
-\itemize{
-\item semantically stabilised record set member representations.
-}
-
-Operational columns are mapped into the opinionated
-\code{recordset_df} vocabulary:
-\itemize{
-\item \code{resource_id} → \code{member_id}
-\item \code{locator_path} → \code{member_path}
-\item \code{resource_type} → \code{member_type}
-}
-
-by default, although alternative mappings may be supplied.
-
-The function creates lightweight semantic metadata but intentionally
-avoids:
-\itemize{
-\item authoritative archival description,
-\item full RiC-O graph construction,
-\item provenance reasoning,
-\item or ontology-complete archival modelling.
-}
-
-This lightweight semantic layer is intended for:
-\itemize{
-\item analytical reconstruction,
-\item contextual reporting,
-\item HDTO-like analytical workspaces,
-\item and iterative semantic enrichment workflows.
-}
-}
-\examples{
-toy_record_set <- tibble::tibble(
- structural_group = c(
- "_packages/eviota",
- "_packages/eviota",
- "_packages/iotables"
- ),
- path_id = c(
- "l480::R/import.R",
- "l480::data-raw/build.R",
- "l480::R/cube.R"
- ),
- rel_root_path = c(
- "R/import.R",
- "data-raw/build.R",
- "R/cube.R"
- )
-)
-
-toy_record_set <- toy_record_set |>
- create_record_set(
- record_set_id = "structural_group",
- resource_id = "path_id",
- locator_path = "rel_root_path",
- construction_rule =
- "filtered_project_roots|structural_group",
- resource_type = "file"
- ) |>
- as_recordset_df(
- title = "Toy reconstruction workspace",
- creator = person("Daniel", "Antal"),
- description =
- "Contextual reconstruction record set"
- )
-
-}
diff --git a/man/create_record_set.Rd b/man/create_record_set.Rd
deleted file mode 100644
index 9f0abce..0000000
--- a/man/create_record_set.Rd
+++ /dev/null
@@ -1,273 +0,0 @@
-% Generated by roxygen2: do not edit by hand
-% Please edit documentation in R/create_record_set.R
-\name{create_record_set}
-\alias{create_record_set}
-\title{Create a contextual Record Set projection from observational resources}
-\usage{
-create_record_set(
- x,
- record_set_id,
- resource_id,
- construction_rule,
- locator_path = NULL,
- resource_title = NULL,
- resource_type = NULL
-)
-}
-\arguments{
-\item{x}{A \code{data.frame} or tibble containing observational or derived
-resource rows.}
-
-\item{record_set_id}{Character scalar or existing column name defining
-the contextual Record Set membership of each resource.
-
-Typical examples include:
-\itemize{
-\item structural filesystem groupings;
-\item repository roots;
-\item WARC collection identifiers;
-\item digitisation batches;
-\item curatorial aggregation identifiers.
-}}
-
-\item{resource_id}{Character scalar or existing column name defining
-the operational identity of each resource within the Record Set.
-
-In many filesystem workflows, \code{resource_id} will often correspond
-to what users informally think of as a "file" or "file name".
-
-However, the identifier intentionally represents an operational or
-contextual resource approximation rather than an authoritative or
-permanent file identity.
-
-This distinction matters because digital resources frequently evolve
-over time:
-\itemize{
-\item filenames and paths may change;
-\item synchronized copies may diverge;
-\item local and cloud versions may coexist;
-\item files may be copied, renamed, or reorganised;
-\item multiple observations may refer to evolving versions of the
-same underlying resource.
-}
-
-For example:
-\itemize{
-\item the same digital resource ("file") may exist in multiple locations;
-\item a synchronized cloud copy may differ from a local working copy;
-\item a renamed file may still represent the continuation of the same
-evolving digital resource.
-}
-
-Typical examples include:
-\itemize{
-\item \code{storage_path_id}
-(storage-scoped filesystem resource approximation);
-\item URI identifiers;
-\item WARC record identifiers;
-\item repository-relative identifiers;
-\item IIIF resource identifiers.
-}}
-
-\item{construction_rule}{Character description documenting the
-deterministic operational rule used to construct the contextual
-Record Set projection.
-
-Examples:
-\itemize{
-\item \code{"filtered_project_roots|structural_group"}
-\item \code{"warc_collection|domain_partition"}
-\item \code{"iiif_manifest|folder_batch"}
-}
-
-The construction rule is stored as lightweight provenance metadata
-attached to the resulting tibble.}
-
-\item{locator_path}{Optional character scalar or existing column name
-providing a human-readable operational locator associated with the
-resource.
-
-Examples include:
-\itemize{
-\item filesystem paths;
-\item repository-relative paths;
-\item URIs;
-\item WARC locators;
-\item IIIF resource paths.
-}}
-
-\item{resource_title}{Optional character scalar or existing column name
-containing a human-readable resource title or label.}
-
-\item{resource_type}{Optional character scalar or existing column name
-describing the operational resource type.
-
-Examples:
-\itemize{
-\item \code{"file"}
-\item \code{"warc_record"}
-\item \code{"iiif_canvas"}
-\item \code{"rdf_resource"}
-\item \code{"digitised_page"}
-}}
-}
-\value{
-A tibble representing a contextual operational Record Set projection.
-
-The resulting tibble contains:
-\itemize{
-\item \code{record_set_id}
-\item \code{resource_id}
-\item optional contextual resource variables
-}
-
-together with lightweight provenance attributes:
-\itemize{
-\item \code{construction_rule}
-\item \code{created_by}
-\item \code{record_set_created_at}
-}
-}
-\description{
-Constructs a lightweight contextual Record Set projection from an
-observational resource table.
-
-\verb{Record Set} is a contextual aggregation concept defined by the
-International Council on Archives (ICA) Records in Contexts
-standard (RiC).
-
-In operational terms, a Record Set may represent:
-\itemize{
-\item a project workspace;
-\item a synchronized cloud folder;
-\item a repository inventory;
-\item a digitisation batch;
-\item a web archive collection;
-\item a reconstruction corpus;
-\item or another contextual grouping of related digital resources.
-}
-
-The function is designed as an operational bridge between:
-\itemize{
-\item filesystem observations;
-\item web archive inventories;
-\item digitised heritage collections;
-\item repository inventories;
-\item and later semantically enriched Record Set representations.
-}
-
-The returned object is intentionally a plain tibble rather than a
-semantically enriched \code{recordset_df}.
-
-This allows:
-\itemize{
-\item efficient tidyverse workflows;
-\item exploratory analytical pipelines;
-\item lightweight contextual reconstruction;
-\item deferred semantic stabilisation;
-\item provenance-aware iterative enrichment.
-}
-
-More information:
-\itemize{
-\item ICA Records in Contexts overview:
-\url{https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/}
-\item RiC-O ontology repository:
-\url{https://github.com/ica-egad/ric-o}
-}
-
-In RiC-aligned operational terminology:
-\itemize{
-\item rows typically represent observed or derived Record Resources,
-Instantiations, or other operational resource proxies;
-\item the resulting tibble represents a contextual Record Set projection
-constructed from deterministic operational rules;
-\item the function does not create authoritative archival arrangement,
-fonds hierarchy, or curatorial description;
-\item Record Set semantics remain analytical and operational unless
-later stabilised through curatorial or semantic workflows.
-}
-
-Typical use cases include:
-\itemize{
-\item grouping filesystem observations into project-level Record Sets;
-\item constructing analytical corpora from repository structures;
-\item creating candidate archival aggregations;
-\item preparing Heritage Digital Twin analytical spaces;
-\item contextualising WARC/WACZ collections;
-\item constructing enrichment workspaces for knowledge-graph workflows.
-}
-
-The function deliberately separates:
-\itemize{
-\item operational contextualisation (\code{create_record_set()})
-}
-
-from:
-\itemize{
-\item semantic publication and metadata enrichment
-(\code{as_recordset_df()}).
-}
-
-This mirrors the package philosophy used throughout the observational
-pipeline:
-\itemize{
-\item observe first;
-\item contextualise second;
-\item interpret later.
-}
-}
-\details{
-The function intentionally performs only lightweight contextual
-projection and validation.
-
-It does not:
-\itemize{
-\item infer authoritative documentary hierarchy;
-\item enforce archival arrangement;
-\item construct RiC-complete semantic graphs;
-\item perform provenance reasoning;
-\item stabilise resource identity across time.
-}
-
-Semantic enrichment and publication-oriented metadata are intended
-to be added later via \code{as_recordset_df()}.
-
-This staged architecture supports:
-\itemize{
-\item efficient analytical workflows;
-\item iterative reconstruction;
-\item provenance-aware contextualisation;
-\item future alignment with RiC-O and RiC-CM.
-}
-}
-\examples{
-toy_record_set <- tibble::tibble(
- structural_group = c(
- "heritage_collection",
- "heritage_collection",
- "digitisation_batch"
- ),
- storage_path_id = c(
- "laptop01::scans/photo_001.tif",
- "laptop01::ocr/photo_001.txt",
- "archive01::reports/summary.qmd"
- ),
- rel_root_path = c(
- "scans/photo_001.tif",
- "ocr/photo_001.txt",
- "reports/summary.qmd"
- )
-)
-
-toy_record_set <- create_record_set(
- toy_record_set,
- record_set_id = "structural_group",
- resource_id = "storage_path_id",
- locator_path = "rel_root_path",
- construction_rule =
- "filtered_project_roots|structural_group",
- resource_type = "file"
-)
-
-}
diff --git a/man/derive_group_path.Rd b/man/derive_group_path.Rd
index 3aef5d0..3706d05 100644
--- a/man/derive_group_path.Rd
+++ b/man/derive_group_path.Rd
@@ -30,7 +30,7 @@ layouts rather than authoritative documentary structure.
Future versions of the package are expected to replace or absorb
this functionality into higher-level Record Set construction logic
-(for example via \code{create_record_set()}), where grouping rules will
+(for example via \code{record_set_projection()}), where grouping rules will
be explicitly contextualised and provenance-aware.
}
\keyword{internal}
diff --git a/man/derive_structural_groups.Rd b/man/derive_structural_groups.Rd
index 6abe89d..0cb815c 100644
--- a/man/derive_structural_groups.Rd
+++ b/man/derive_structural_groups.Rd
@@ -2,93 +2,91 @@
% Please edit documentation in R/derive_structural_groups.R
\name{derive_structural_groups}
\alias{derive_structural_groups}
-\title{Derive structural grouping heuristics from relative paths}
+\title{Derive structural aggregation metadata from relative paths}
\usage{
-derive_structural_groups(rel_path)
+derive_structural_groups(rel_path, profile = "folder-depth-2")
}
\arguments{
\item{rel_path}{Character vector of relative filesystem paths.}
+
+\item{profile}{Character scalar specifying the structural
+aggregation strategy. Available profiles are:
+\describe{
+\item{"folder-depth-1"}{Group by the first directory level.}
+\item{"folder-depth-2"}{Group by the first two directory levels
+(default).}
+\item{"folder-depth-3"}{Group by the first three directory levels.}
+\item{"folder-depth-4"}{Group by the first four directory levels.}
+\item{"wacz"}{Use the first path component as the structural group
+and the second component as the structural subdivision, matching
+the standard organisation of WACZ archives.}
+}}
}
\value{
-A \code{data.frame} with columns:
+A \code{data.frame} with two columns:
\describe{
\item{structural_group}{
-Filesystem-based structural grouping heuristic derived from
-the first path component.
+Candidate structural aggregation derived from the selected path
+profile.
}
\item{component}{
-Immediate structural subdivision within the grouping,
-if present.
+Immediate structural subdivision within the aggregation, when
+present.
}
}
}
\description{
-Derives lightweight structural grouping heuristics from relative
-filesystem paths.
-}
-\details{
-The function extracts shallow structural patterns commonly found in
-software projects, research workflows, and digital working environments.
+Derives lightweight structural aggregation metadata from observed
+relative filesystem paths.
-It assigns:
-\itemize{
-\item \code{structural_group}:
-grouping heuristic derived from the first path component
-(e.g. \code{innolab25}, \verb{_packages}, \verb{_markdown})
-\item \code{component}:
-immediate structural subdivision within the grouping,
-if present
-(e.g. \code{eviota}, \code{filmledgerimport}, \code{iotables})
-}
+The function identifies recurring structural patterns in directory
+hierarchies and creates candidate aggregations that can support
+exploratory analysis, navigation, contextual reconstruction, and
+later semantic interpretation.
-These derived structures support:
-\itemize{
-\item exploratory grouping of filesystem observations
-\item navigation of large observational snapshots
-\item reconstruction of operational project environments
-\item identification of candidate documentary aggregations
+The resulting groupings are derived solely from path structure.
+They are analytical projections rather than authoritative Record
+Sets, provenance assertions, or documentary relationships.
}
+\details{
+Structural aggregation metadata provides a lightweight abstraction
+of observed directory organisation. It can increase the
+informativeness of filesystem observations by exposing recurring
+organisational patterns without asserting semantic meaning.
-The function performs deterministic structural projection only.
-It does not validate repository semantics, documentary structure,
-or authoritative Record Set boundaries.
-
-This function provides a lightweight structural interpretation layer
-on top of observational filesystem data.
-
-In RiC-aligned operational terms:
+Within the fscontext workflow:
\itemize{
-\item rows in observational snapshots represent filesystem
-Instantiations
-\item \code{rel_path} acts as an operational locator associated with
-observed filesystem occurrences
-\item the derived structural groupings provide analytical heuristics
-that may later support Record Set construction
+\item filesystem observations provide evidence about observed resources;
+\item relative paths provide structural organisation;
+\item structural aggregations expose candidate groups that may later
+support contextual reconstruction, Record Set construction,
+semantic stabilisation, or other downstream analyses.
}
-The derived groupings are operational analytical projections,
-not authoritative RiC Record Sets.
-
-The function is intended for analytical, navigational,
-and exploratory reconstruction workflows.
-
-Future versions of the package may replace or extend this logic with
-more explicit provenance-aware Record Set construction workflows
-(for example via \code{create_record_set()}).
+Future versions may introduce additional aggregation profiles based
+on repository structure, provenance, temporal patterns, or other
+observational evidence.
}
\examples{
-data("fscontextdemo_snapshot_02")
-
-example_paths <- c(
- "_packages/fscontextdemo/R/derive_fsdemo_country_data.R",
- "_packages/fscontextdemo/tests/testthat/test-country-data.R",
- "_packages/fscontextdemo/data-raw/create_fsdemo_country_data.R",
- "_packages/fscontextdemo/docs/index.html"
+rel_path <- c(
+ "_packages/demo/R/file.R",
+ "_packages/demo/tests/testthat/test-file.R",
+ "_packages/demo/data/input.csv"
)
-data.frame(
- rel_path = example_paths,
- derive_structural_groups(example_paths)
+derive_structural_groups(rel_path)
+
+derive_structural_groups(
+ rel_path,
+ profile = "folder-depth-1"
)
+derive_structural_groups(
+ c(
+ "archive/data.warc.gz",
+ "indexes/index.cdx",
+ "pages/pages.jsonl"
+ ),
+ profile = "wacz"
+)
}
diff --git a/man/new_recordset_df.Rd b/man/new_recordset_df.Rd
index 6c53062..d92c2d9 100644
--- a/man/new_recordset_df.Rd
+++ b/man/new_recordset_df.Rd
@@ -2,31 +2,36 @@
% Please edit documentation in R/recordset_df.R
\name{new_recordset_df}
\alias{new_recordset_df}
-\title{Internal constructor for \code{recordset_df}}
+\title{Internal constructor for recordset_df}
\usage{
-new_recordset_df(x)
+new_recordset_df(x, record_identifier = NULL, record_part_identifier = NULL)
}
\arguments{
-\item{x}{A data.frame or tibble containing at minimum:
-\itemize{
-\item \code{record_set_id}
-\item \code{member_id}
-}}
+\item{x}{A dataset_df object.}
+
+\item{record_identifier}{Optional name of the column containing
+Record identifiers.}
+
+\item{record_part_identifier}{Optional name of the column containing
+Record Part identifiers.}
}
\value{
-A \code{recordset_df} object.
+A recordset_df object inheriting from dataset_df.
}
\description{
-Low-level internal constructor for creating \code{recordset_df} objects.
+Low-level constructor for recordset_df objects.
-This function:
+This function extends an existing dataset_df with lightweight
+Record Set semantics by:
\itemize{
-\item validates required columns
-\item assigns the \code{recordset_df} class
-\item preserves existing classes
+\item assigning the recordset_df class;
+\item optionally assigning a Record Set identifier and provenance;
+\item optionally declaring Record and Record Part identifier columns as
+rico:Identifier using \code{\link[dataset:defined]{dataset::defined()}}.
}
-Unlike \code{\link{recordset_df}}, this constructor does not create
-semantic metadata structures or perform user-facing coercion.
+Unlike \code{\link[=recordset_df]{recordset_df()}}, this constructor assumes that dataset-level
+metadata have already been created and performs no coercion from
+ordinary data.frame objects.
}
\keyword{internal}
diff --git a/man/observe_wacz.Rd b/man/observe_wacz.Rd
new file mode 100644
index 0000000..842d623
--- /dev/null
+++ b/man/observe_wacz.Rd
@@ -0,0 +1,75 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/observe_wacz.R
+\name{observe_wacz}
+\alias{observe_wacz}
+\title{Observe a WACZ web archive}
+\usage{
+observe_wacz(wacz)
+}
+\arguments{
+\item{wacz}{Path to a \code{.wacz} archive.}
+}
+\value{
+A tibble containing observations extracted from the archive.
+
+The returned object carries two attributes:
+\itemize{
+\item \code{datapackage}, containing the parsed \code{datapackage.json}
+metadata supplied by the WACZ archive;
+\item \code{wacz}, containing the normalized path to the source archive.
+}
+
+Typical variables include:
+\itemize{
+\item page identifiers;
+\item resource locators (URLs);
+\item page titles;
+\item timestamps;
+\item extracted text;
+\item text signatures;
+\item MIME types;
+\item WARC digests;
+\item archive offsets;
+\item version counts.
+}
+}
+\description{
+Creates an observational data frame from a WACZ web archive.
+
+The function extracts structural metadata from the archive,
+combines page-level information with WARC index metadata, and returns
+one observational row for each archived web page.
+
+The resulting object represents observations only. It intentionally
+avoids making semantic assertions about Records, Record Parts,
+Instantiations, or other archival entities. Such interpretation can
+be added later with \code{\link[=wacz_to_recordset_df]{wacz_to_recordset_df()}} or downstream semantic
+enrichment workflows.
+}
+\details{
+The function performs the following steps:
+\itemize{
+\item extracts the WACZ archive into a temporary directory;
+\item reads the archive \code{datapackage.json};
+\item parses page metadata from \code{pages/pages.jsonl};
+\item parses WARC index metadata from \code{indexes/index.cdx};
+\item collapses multiple archived versions of the same resource;
+\item joins page observations with archive metadata.
+}
+
+The resulting observations preserve the evidence contained in the
+archive without interpreting its archival semantics.
+}
+\examples{
+wacz <- system.file("testdata", "fscontext_020.wacz", package = "fscontext")
+
+observe_wacz(wacz)
+
+}
+\references{
+The WACZ format specification:
+\url{https://specs.webrecorder.net/wacz/1.1.1/}
+}
+\seealso{
+\code{\link[=wacz_to_recordset_df]{wacz_to_recordset_df()}}
+}
diff --git a/man/quick_signature.Rd b/man/quick_signature.Rd
index 1125363..758bfad 100644
--- a/man/quick_signature.Rd
+++ b/man/quick_signature.Rd
@@ -2,65 +2,55 @@
% Please edit documentation in R/quick_signature.R
\name{quick_signature}
\alias{quick_signature}
-\title{Compute a fast content signature for a file}
+\title{Compute a fast operational signature for a file}
\usage{
quick_signature(path, n = 1024)
}
\arguments{
\item{path}{Character. Path to the file.}
-\item{n}{Integer. Number of bytes to read from selected regions
+\item{n}{Integer. Number of bytes sampled from selected regions
(default: 1024).}
}
\value{
-Character. A signature string representing sampled file content.
+Character. A lightweight operational signature.
}
\description{
-Generates a lightweight content signature based on hashing selected
-byte regions of a file. This provides a fast approximation for detecting
-identical or differing file instances without computing a full file hash.
+Generates a lightweight content signature by hashing sampled byte
+regions from a file. The signature provides a fast operational
+approximation for detecting identical or differing file instances
+without computing a full cryptographic hash.
}
\details{
-The function is designed for performance and is suitable for use in
-large-scale filesystem observations, where full hashing would be
-computationally expensive.
+The function is designed for large-scale observational workflows
+where complete file hashing would be unnecessarily expensive.
-The signature is constructed from hashed byte segments:
+The signature is constructed by hashing sampled byte regions:
\itemize{
-\item small files: hash of full content
-\item medium files: hash of first and last segments
-\item large files: hash of first, middle, and last segments
+\item small files: full file content
+\item medium files: beginning and end
+\item large files: beginning, middle and end
}
-The function provides a fast operational signal for probable
-content equivalence:
+The resulting signature is intended as a fast observational aid:
\itemize{
-\item identical signatures strongly suggest identical content
-\item different signatures indicate content differences
-\item collisions are possible but unlikely in practice
+\item identical signatures suggest identical file content;
+\item differing signatures indicate differing file content;
+\item collisions are possible but unlikely in operational use.
}
Missing or inaccessible files return \code{NA_character_}.
-In RiC-aligned operational terms, the signature supports later
-interpretation of observed filesystem Instantiations:
-\itemize{
-\item identifying likely identical Instantiations
-\item distinguishing likely versions or derivations
-\item detecting distributed or duplicated work
-\item supporting later Record Set construction and reconciliation
-}
-
-The function does not establish authoritative identity or provenance.
-It provides observational evidence that may later support analytical
-or curatorial interpretation.
+The signature does not establish authoritative identity or provenance.
+It provides lightweight observational evidence that may support later
+contextual reconstruction, duplicate detection, version analysis,
+or Record Set construction.
-This function is typically used in conjunction with:
-\itemize{
-\item \code{\link[=scan_storage]{scan_storage()}} for generating observational snapshots
-\item \code{\link[=summarise_duplicates]{summarise_duplicates()}} for detecting duplicate and versioned files
-}
+Unlike \code{\link[=quick_signature_text]{quick_signature_text()}}, this function operates on the
+binary representation of a file rather than its textual content.
}
\seealso{
+\code{\link[=quick_signature_text]{quick_signature_text()}},
+\code{\link[=scan_storage]{scan_storage()}},
\code{\link[=summarise_duplicates]{summarise_duplicates()}}
}
diff --git a/man/quick_signature_text.Rd b/man/quick_signature_text.Rd
new file mode 100644
index 0000000..49bcdcc
--- /dev/null
+++ b/man/quick_signature_text.Rd
@@ -0,0 +1,60 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/quick_signature.R
+\name{quick_signature_text}
+\alias{quick_signature_text}
+\title{Compute a fast operational signature for text}
+\usage{
+quick_signature_text(x, n = 1024)
+}
+\arguments{
+\item{x}{Character vector.}
+
+\item{n}{Integer. Number of characters sampled from selected regions
+(default: 1024).}
+}
+\value{
+Character vector of operational signatures that summarises the
+observed textual representation of a resource.
+}
+\description{
+Generates a lightweight content signature by hashing sampled character
+regions from one or more text strings. The signature provides a fast
+approximation for detecting identical or differing textual content
+without comparing complete strings.
+}
+\details{
+The function is intended for observational workflows where textual
+representations have already been extracted from digital resources,
+such as HTML pages, OCR output, PDFs, or office documents.
+
+The signature is constructed by hashing sampled character regions:
+\itemize{
+\item short texts: complete text;
+\item medium texts: beginning and end;
+\item long texts: beginning, middle and end.
+}
+
+The resulting signature is intended as a fast observational aid:
+\itemize{
+\item identical signatures suggest identical textual content;
+\item differing signatures indicate differing textual content;
+\item collisions are possible but unlikely in operational use.
+}
+
+Missing values return \code{NA_character_}.
+Empty strings return \code{"empty"}.
+
+Unlike \code{\link[=quick_signature]{quick_signature()}}, this function operates on extracted text
+rather than binary file content. Consequently, different file formats
+(for example DOCX, PDF and HTML) containing the same textual content
+may produce identical text signatures while retaining different file
+signatures.
+
+The function provides lightweight observational evidence that may
+support duplicate detection, content reconciliation, semantic
+stabilisation, or later contextual reconstruction.
+}
+\seealso{
+\code{\link[=quick_signature]{quick_signature()}},
+\code{\link[=observe_wacz]{observe_wacz()}}
+}
diff --git a/man/recordset_df.Rd b/man/recordset_df.Rd
index b872463..caaddad 100644
--- a/man/recordset_df.Rd
+++ b/man/recordset_df.Rd
@@ -2,193 +2,92 @@
% Please edit documentation in R/recordset_df.R
\name{recordset_df}
\alias{recordset_df}
-\title{Create a provenance-aware Record Set data frame}
+\title{Create a semantically annotated Record Set}
\usage{
recordset_df(
- ...,
- identifier = c(member = "http://example.com/recordset#member"),
- var_labels = NULL,
- units = NULL,
- concepts = NULL,
- dataset_bibentry = NULL,
- dataset_subject = NULL
+ x,
+ title = NULL,
+ creator = utils::person("Jane", "Doe"),
+ description = NULL,
+ record_set_identifier = NULL,
+ record_identifier = NULL,
+ record_part_identifier = NULL,
+ record_subject = "Record Set",
+ ...
)
}
\arguments{
-\item{...}{Vectors (columns) to include in the record set.}
+\item{x}{A \code{data.frame} or \code{dataset_df}.}
-\item{identifier}{A named vector of URI prefixes used to generate row identifiers.
+\item{title}{Character scalar giving the title of the Record Set.}
-Defaults to:
+\item{creator}{A \code{utils::person()} object describing the creator of
+the Record Set metadata.}
-\code{c(member = "http://example.com/recordset#member")}}
+\item{description}{Optional description of the Record Set.}
-\item{var_labels}{Optional named list of human-readable variable labels.}
+\item{record_set_identifier}{Optional identifier of the Record Set.}
-\item{units}{Optional named list of measurement units.}
+\item{record_identifier}{Name of the column containing Record
+identifiers. The selected column is annotated as
+\code{rico:Identifier} and labelled "Record Identifier".}
-\item{concepts}{Optional named list of semantic concept URIs.}
+\item{record_part_identifier}{Name of the column containing Record
+Part identifiers. The selected column is annotated as
+\code{rico:Identifier} and labelled "Record Part Identifier".}
-\item{dataset_bibentry}{Optional bibliographic metadata created with
-\code{dataset::dublincore()} or
-\code{dataset::datacite()}.}
+\item{record_subject}{Subject term describing the Record Set.
+Defaults to \code{"Record Set"}.}
-\item{dataset_subject}{Optional dataset subject metadata.}
+\item{...}{Reserved for future extensions.}
}
\value{
-A \code{recordset_df} object.
+A \code{recordset_df}, which inherits from \code{dataset_df}, \code{tbl_df},
+\code{tbl} and \code{data.frame}.
}
\description{
-Construct a \code{recordset_df}, a provenance-aware contextual dataset
-representing members of a Record Set.
+Create a \code{recordset_df}, a lightweight extension of
+\code{dataset::dataset_df} for representing archival Record Sets.
-\verb{Record Set} is a contextual aggregation concept defined by the
-International Council on Archives (ICA) Records in Contexts
-standard (RiC). In operational terms, a Record Set may represent:
-\itemize{
-\item a project workspace;
-\item a research corpus;
-\item a synchronized working environment;
-\item a digital collection;
-\item a reconstruction context;
-\item or another contextual grouping of related digital resources.
-}
-
-The Records in Contexts (RiC) standard provides a flexible and
-provenance-aware approach for describing evolving digital records,
-their relationships, and their contextual environments.
-
-Unlike rigid hierarchical archival models, RiC allows records and
-digital resources to participate in multiple overlapping contextual
-groupings while preserving provenance and contextual relationships.
-
-More information:
-\itemize{
-\item ICA Records in Contexts overview:
-\url{https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/}
-\item RiC-O ontology repository:
-\url{https://github.com/ica-egad/ric-o}
-}
-
-A \code{recordset_df} extends the
-\code{\link[dataset:dataset_df]{dataset_df}} class with lightweight
-contextual Record Set semantics suitable for:
-\itemize{
-\item filesystem observations;
-\item synchronized cloud folders;
-\item web archive members;
-\item digital surrogate collections;
-\item curation batches;
-\item Digital Twin workspaces;
-\item provenance-aware research collections;
-\item contextual digital preservation workflows.
-}
-
-The class is designed to work together with:
-\itemize{
-\item \code{\link[=read_snapshot]{read_snapshot()}}
-\item \code{\link[=snapshot_to_reconstruction_context]{snapshot_to_reconstruction_context()}}
-\item \code{\link[=snapshot_to_recordset_df]{snapshot_to_recordset_df()}}
-}
-
-while preserving the distinction between:
-\itemize{
-\item observed filesystem evidence;
-\item contextual grouping of related resources;
-\item later analytical interpretation;
-\item and archival or semantic enrichment workflows.
-}
-
-In operational terms:
-\itemize{
-\item \code{record_set_id}
-identifies a contextual grouping of related digital resources
-(similar to a project workspace, collection, or reconstruction
-environment);
-\item \code{member_id}
-identifies one observed or asserted member within that grouping.
-}
-
-The resulting object inherits from:
-\itemize{
-\item \code{recordset_df}
-\item \code{dataset_df}
-\item \code{tbl_df}
-\item \code{tbl}
-\item \code{data.frame}
-}
-}
-\details{
-The constructor requires at minimum the columns:
-\itemize{
-\item \code{record_set_id}
-\item \code{member_id}
-}
-
-Validation and class assignment are delegated to
-\code{\link{new_recordset_df}}.
-
-The constructor is intentionally lightweight and does not:
-\itemize{
-\item infer authoritative archival hierarchy;
-\item reconcile duplicate identities;
-\item infer canonical resources;
-\item construct ontology-complete provenance graphs;
-\item or replace curatorial or archival interpretation.
-}
+A \code{recordset_df} preserves ordinary tabular data while allowing
+selected columns to be declared as identifiers of RiC Records and
+Record Parts. It is intended for provenance-aware archival,
+curatorial and semantic enrichment workflows without requiring a
+complete implementation of the Records in Contexts (RiC) ontology.
-Instead, it provides a stable contextual preservation layer for
-provenance-aware reconstruction and human-in-the-loop workflows.
+See the \strong{recordset_df} vignette for a complete workflow starting
+from filesystem observations.
}
\examples{
-toy_recordset <- recordset_df(
- record_set_id = c(
- "heritage_digitisation",
- "heritage_digitisation",
- "heritage_digitisation"
- ),
- member_id = c(
- "inst_001",
- "inst_002",
- "inst_003"
- ),
- member_path = c(
- "scans/photo_001.tif",
- "ocr/photo_001.txt",
- "reports/collection_summary.qmd"
- ),
- member_type = c(
- "file",
- "file",
- "file"
+x <- data.frame(
+ resource_locator = c(
+ "https://example.org/1",
+ "https://example.org/2"
),
- source_type = c(
- "filesystem",
- "filesystem",
- "filesystem"
+ filename = c(
+ "a.html",
+ "b.html"
),
- identifier = c(
- member =
- "https://example.org/recordset/heritage#member"
- ),
- var_labels = list(
- record_set_id = "Record set identifier",
- member_id = "Member identifier",
- member_path = "Member path"
- ),
- concepts = list(
- record_set_id =
- "https://www.ica.org/standards/RiC/ontology#RecordSet",
- member_id =
- "https://www.ica.org/standards/RiC/ontology#Instantiation"
- ),
- dataset_bibentry = dataset::dublincore(
- title = "Toy Heritage Digitisation Record Set",
- creator = person("Jane", "Doe"),
- publisher = "fscontext"
- )
+ stringsAsFactors = FALSE
)
-toy_recordset
+rs <- recordset_df(
+ x,
+ title = "Demo Record Set",
+ creator = utils::person("Joe", "Doe", role = "aut"),
+ record_identifier = "resource_locator",
+ record_part_identifier = "filename"
+)
+rs
+
+}
+\references{
+International Council on Archives Expert Group on Archival
+Description (2023). Records in Contexts (RiC).
+https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/
+}
+\seealso{
+\code{\link[dataset:dataset_df]{dataset::dataset_df()}}, \code{\link[=observe_wacz]{observe_wacz()}},
+\code{\link[=wacz_to_recordset_df]{wacz_to_recordset_df()}}
}
diff --git a/man/save_scan.Rd b/man/save_scan.Rd
index 8971217..fce8eb7 100644
--- a/man/save_scan.Rd
+++ b/man/save_scan.Rd
@@ -41,24 +41,24 @@ This ensures:
}
}
\examples{
-\dontrun{
-root <- tempfile()
-dir.create(root)
+tmp_dir <- tempfile()
+dir.create(tmp_dir)
-dir.create(file.path(root, "R"))
-dir.create(file.path(root, "data"))
+dir.create(file.path(tmp_dir, "R"))
+dir.create(file.path(tmp_dir, "data"))
-file.create(file.path(root, "R", "a.R"))
-file.create(file.path(root, "R", "b.R"))
-file.create(file.path(root, "data", "c.csv"))
+file.create(file.path(tmp_dir, "R", "a.R"))
+file.create(file.path(tmp_dir, "R", "b.R"))
+file.create(file.path(tmp_dir, "data", "c.csv"))
-scan_storage(
- root = root,
- storage_id = "test-storage",
- path = tmp
+scan <- scan_storage(
+ root = tmp_dir,
+ storage_id = "test-storage"
)
-save_scan(scan, "test-storage")
-}
-
+save_scan(
+ df = scan,
+ storage_id = "test-storage",
+ path = tmp_dir
+)
}
diff --git a/man/scan_storage.Rd b/man/scan_storage.Rd
index 3795ad1..919eac4 100644
--- a/man/scan_storage.Rd
+++ b/man/scan_storage.Rd
@@ -6,8 +6,8 @@
\usage{
scan_storage(
root,
- storage_id = "l480-1-ssd",
- person_id = "antaldaniel",
+ storage_id = "local-storage",
+ person_id = "local-user",
scan_time = Sys.time(),
compute_signature = TRUE,
max_signature_size = 200 * 1024 * 1024
diff --git a/man/snapshot_storage.Rd b/man/snapshot_storage.Rd
index 19c94c0..f9e59d5 100644
--- a/man/snapshot_storage.Rd
+++ b/man/snapshot_storage.Rd
@@ -10,7 +10,7 @@ snapshot_storage(
person_id = "user",
scan_time = Sys.time(),
label = NULL,
- path = here::here("data-raw", "snapshots"),
+ path = tempdir(),
compute_signature = TRUE,
max_signature_size = 200 * 1024 * 1024
)
diff --git a/man/snapshot_to_reconstruction_context.Rd b/man/snapshot_to_reconstruction_context.Rd
index 1c21e7b..9461d2e 100644
--- a/man/snapshot_to_reconstruction_context.Rd
+++ b/man/snapshot_to_reconstruction_context.Rd
@@ -48,7 +48,7 @@ Contextual enrichment variables may include:
\item \code{observation_id};
\item \code{structural_group};
\item \code{component};
-\item \code{record_set_id};
+\item \code{record_set_identifier};
\item \code{resource_id};
\item \code{locator_path}.
}
@@ -121,6 +121,5 @@ snapshot_to_reconstruction_context(
\code{\link[=snapshot_to_recordset_df]{snapshot_to_recordset_df()}},
\code{\link[=subset_snapshot]{subset_snapshot()}},
\code{\link[=add_snapshot_context]{add_snapshot_context()}},
-\code{\link[=add_structural_groups]{add_structural_groups()}},
-\code{\link[=create_record_set]{create_record_set()}}
+\code{\link[=add_structural_groups]{add_structural_groups()}}.
}
diff --git a/man/snapshot_to_recordset_df.Rd b/man/snapshot_to_recordset_df.Rd
index 3ad0955..ce7151d 100644
--- a/man/snapshot_to_recordset_df.Rd
+++ b/man/snapshot_to_recordset_df.Rd
@@ -7,9 +7,9 @@
snapshot_to_recordset_df(
snapshot_files,
roots,
- record_set_id,
+ record_set_identifier,
record_set_title = NULL,
- person = utils::person("Jane", "Doe"),
+ creator = utils::person("Jane", "Doe", role = "aut"),
exclude_patterns = c("\\\\\\\\.Rcheck")
)
}
@@ -19,12 +19,12 @@ snapshot_to_recordset_df(
\item{roots}{Character vector of contextual root paths used
for observational selection.}
-\item{record_set_id}{Character scalar giving the asserted
+\item{record_set_identifier}{Character scalar giving the asserted
identifier of the resulting Record Set.}
\item{record_set_title}{Optional human-readable title.}
-\item{person}{A \code{\link[utils:person]{utils::person()}} object describing the creator
+\item{creator}{A \code{\link[utils:person]{utils::person()}} object describing the creator
of the semantic Record Set assertion.}
\item{exclude_patterns}{Character vector of exclusion patterns
@@ -35,7 +35,7 @@ A semantically enriched \code{recordset_df} object inheriting from
\code{dataset_df}.
}
\description{
-Creates a provenance-aware \code{recordset_df} from observational
+Creates a provenance-aware \code{\link[=recordset_df]{recordset_df()}} object from observational
filesystem snapshots and contextual reconstruction workflows.
The function preserves observed filesystem resources while adding:
diff --git a/man/wacz_to_recordset_df.Rd b/man/wacz_to_recordset_df.Rd
new file mode 100644
index 0000000..5cc63f9
--- /dev/null
+++ b/man/wacz_to_recordset_df.Rd
@@ -0,0 +1,82 @@
+% Generated by roxygen2: do not edit by hand
+% Please edit documentation in R/wacz_to_recordset_df.R
+\name{wacz_to_recordset_df}
+\alias{wacz_to_recordset_df}
+\title{Create a Record Set dataset from a WACZ observation}
+\usage{
+wacz_to_recordset_df(
+ wacz_observation,
+ record_set_id = NULL,
+ record_set_title = NULL,
+ record_identifier = "resource_locator",
+ record_part_identifier = NULL,
+ person = utils::person("Jane", "Doe")
+)
+}
+\arguments{
+\item{wacz_observation}{A \code{wacz_observation} object created with \code{\link[=observe_wacz]{observe_wacz()}}.}
+
+\item{record_set_id}{Optional identifier for the resulting Record Set. If \code{NULL}, the
+basename of the WACZ archive (without extension) is used.}
+
+\item{record_set_title}{Optional human-readable title for the Record Set. If omitted, a title
+is constructed automatically.}
+
+\item{record_identifier}{Name of the column whose values identify Records represented in the
+Record Set. The selected column is annotated as
+\code{rico:Identifier} using \code{\link[dataset:defined]{dataset::defined()}}. Set to \code{NULL} to skip
+annotation.}
+
+\item{record_part_identifier}{Optional name of a column whose values identify Record Parts. The
+selected column is annotated as \code{rico:Identifier}.}
+
+\item{person}{A \code{\link[utils:person]{utils::person()}} object describing the creator of the resulting
+dataset metadata.}
+}
+\value{
+A \code{dataset_df} object enriched with:
+\itemize{
+\item Dublin Core dataset metadata;
+\item a RiC Record Set subject;
+\item optional semantic annotations for Record and Record Part identifiers;
+\item the original \code{datapackage} and \code{wacz} attributes.
+}
+}
+\description{
+Converts a \code{wacz_observation} created with \code{\link[=observe_wacz]{observe_wacz()}} into a
+semantically enriched \code{dataset_df} representing a Record Set.
+
+The function preserves the original observations while attaching
+dataset-level metadata and lightweight Records in Contexts (RiC)
+semantics. Selected identifier columns may be declared as
+\code{rico:Identifier} values, allowing downstream workflows to distinguish
+identifiers intended to refer to Records or Record Parts without
+requiring a complete RiC-O implementation.
+
+The function intentionally performs only lightweight semantic
+enrichment. It does not infer Records, Record Parts, Instantiations,
+or other archival entities, nor does it reconcile identities or build
+provenance graphs. Such interpretation is expected to occur in later
+human-guided curation or semantic stabilisation workflows.
+}
+\details{
+This function occupies the boundary between observational data and
+semantic interpretation.
+
+\code{observe_wacz()} records observations extracted from a WACZ archive.
+\code{wacz_to_recordset_df()} adds curatorial assertions describing how
+particular observed identifiers should be interpreted within a Record
+Set, while deliberately avoiding stronger ontological commitments such
+as identity reconciliation or Record construction.
+
+The resulting object is intended for reproducible archival,
+curatorial, and semantic enrichment workflows.
+}
+\references{
+International Council on Archives Expert Group on Archival Description
+(2023). Records in Contexts (RiC).
+\url{https://www.ica.org/ica-network/expert-groups/egad/records-in-contexts-ric/}
+}
+\seealso{
+\code{\link[=observe_wacz]{observe_wacz()}}, \code{\link[dataset:dataset_df]{dataset::dataset_df()}}, \code{\link[dataset:defined]{dataset::defined()}}
+}
diff --git a/tests/testthat/test-as_recordset_df.R b/tests/testthat/test-as_recordset_df.R
deleted file mode 100644
index ad38c7b..0000000
--- a/tests/testthat/test-as_recordset_df.R
+++ /dev/null
@@ -1,247 +0,0 @@
-test_that(
- "as_recordset_df creates semantically enriched recordset_df",
- {
- toy_resources <- tibble::tibble(
- structural_group = c(
- "_packages/eviota",
- "_packages/eviota",
- "_packages/iotables"
- ),
- path_id = c(
- "l480::R/import.R",
- "l480::data-raw/build.R",
- "l480::R/cube.R"
- ),
- rel_root_path = c(
- "R/import.R",
- "data-raw/build.R",
- "R/cube.R"
- )
- )
-
- rs <- toy_resources |>
- create_record_set(
- record_set_id = "structural_group",
- resource_id = "path_id",
- locator_path = "rel_root_path",
- construction_rule =
- "filtered_project_roots|structural_group",
- resource_type = "file"
- ) |>
- as_recordset_df(
- title =
- "Toy reconstruction workspace",
- creator =
- utils::person(
- given = "Daniel",
- family = "Antal"
- ),
- description =
- "Contextual reconstruction record set"
- )
-
- expect_s3_class(
- rs,
- "recordset_df"
- )
-
- expect_s3_class(
- rs,
- "dataset_df"
- )
-
- expect_s3_class(
- rs,
- "tbl_df"
- )
-
- expect_true(
- all(
- c(
- "record_set_id",
- "member_id"
- ) %in% names(rs)
- )
- )
-
- expect_true(
- all(
- c(
- "member_path",
- "member_type"
- ) %in% names(rs)
- )
- )
-
- expect_equal(
- rs$member_id,
- toy_resources$path_id
- )
-
- expect_equal(
- rs$member_path,
- toy_resources$rel_root_path
- )
-
- expect_equal(
- unique(rs$member_type),
- "file"
- )
-
- expect_equal(
- dataset::dataset_title(rs),
- "Toy reconstruction workspace"
- )
- }
-)
-
-
-test_that(
- "as_recordset_df allows custom semantic mappings",
- {
- toy_resources <- tibble::tibble(
- record_identifier = c(
- "id_001",
- "id_002"
- ),
- record_locator = c(
- "a/file.txt",
- "b/file.txt"
- ),
- record_kind = c(
- "file",
- "file"
- ),
- grouping = c(
- "set_a",
- "set_a"
- )
- )
-
- rs <- toy_resources |>
- create_record_set(
- record_set_id = "grouping",
- resource_id = "record_identifier",
- locator_path = "record_locator",
- resource_type = "record_kind",
- construction_rule = "manual"
- ) |>
- as_recordset_df(
- title = "Custom mapped record set",
- creator =
- utils::person(
- given = "Daniel",
- family = "Antal"
- ),
- member_id = "resource_id",
- member_path = "locator_path",
- member_type = "resource_type"
- )
-
- expect_equal(
- rs$member_id,
- toy_resources$record_identifier
- )
-
- expect_equal(
- rs$member_path,
- toy_resources$record_locator
- )
-
- expect_equal(
- rs$member_type,
- toy_resources$record_kind
- )
- }
-)
-
-
-test_that(
- "as_recordset_df errors when semantic source column is missing",
- {
- toy_resources <- tibble::tibble(x = 1:3)
-
- expect_error(
- as_recordset_df(
- toy_resources,
- title = "Broken mapping",
- creator =
- utils::person(
- given = "Daniel",
- family = "Antal"
- ),
- member_id = "does_not_exist"
- ),
- regexp = "Column not found"
- )
- }
-)
-
-
-test_that(
- "as_recordset_df incorporates construction rule into description",
- {
- toy_resources <- tibble::tibble(
- structural_group = c(
- "_packages/eviota",
- "_packages/eviota"
- ),
- path_id = c(
- "l480::R/import.R",
- "l480::data-raw/build.R"
- ),
- rel_root_path = c(
- "R/import.R",
- "data-raw/build.R"
- )
- )
-
- rs <- toy_resources |>
- create_record_set(
- record_set_id = "structural_group",
- resource_id = "path_id",
- locator_path = "rel_root_path",
- construction_rule =
- "filtered_project_roots|structural_group",
- resource_type = "file"
- ) |>
- as_recordset_df(
- title =
- "Toy reconstruction workspace",
- creator =
- utils::person(
- given = "Daniel",
- family = "Antal"
- ),
- description =
- "Contextual reconstruction record set"
- )
-
- description_text <-
- attr(
- rs,
- "dataset_bibentry"
- )$description
-
- expect_true(
- grepl(
- "Contextual reconstruction record set",
- description_text
- )
- )
-
- expect_true(
- grepl(
- "Construction rule:",
- description_text
- )
- )
-
- expect_true(
- grepl(
- "filtered_project_roots\\|structural_group",
- description_text
- )
- )
- }
-)
diff --git a/tests/testthat/test-coverage_rules_path.R b/tests/testthat/test-coverage_rules_path.R
index 3d659d3..4bf49a1 100644
--- a/tests/testthat/test-coverage_rules_path.R
+++ b/tests/testthat/test-coverage_rules_path.R
@@ -1,65 +1,40 @@
-test_that(
- "coverage_rules_path matches recursive structural rules",
- {
- small_snapshot <- tibble::tibble(
- full_path = c(
- "D:/packages/fscontext/R/import/helpers.R",
- "D:/packages/fscontext/tests/testthat/test-import.R",
- "D:/packages/fscontext/data-raw/input.csv"
- ),
- rel_path = c(
- "R/import/helpers.R",
- "tests/testthat/test-import.R",
- "data-raw/input.csv"
- )
+test_that("coverage_rules_path matches recursive structural rules", {
+ small_snapshot <- tibble::tibble(
+ full_path = c(
+ "D:/proj/demo/R/a.R",
+ "D:/proj/demo/tests/testthat/b.R",
+ "D:/proj/demo/data-raw/c.csv"
+ ),
+ rel_path = c(
+ "R/a.R",
+ "tests/testthat/b.R",
+ "data-raw/c.csv"
)
+ )
- small_test_context <- list(
- contexts = list(
- fscontext = list(
- roots =
- "D:/packages/fscontext",
- rules = list(
- path = c(
- "R" =
- "software_development",
- "tests/testthat" =
- "unit_testing",
- "data-raw" =
- "etl"
- )
+ small_test_context <- list(
+ contexts = list(
+ demo = list(
+ roots = "D:/proj/demo",
+ rules = list(
+ path = c(
+ "R" = "software_development",
+ "tests/testthat" = "unit_testing",
+ "data-raw" = "etl"
)
)
)
)
+ )
- res <- coverage_rules_path(
- snapshot = small_snapshot,
- contexts = small_test_context
- )
-
- matched <- res |>
- dplyr::filter(matched)
-
- expect_true(
- any(
- matched$activity ==
- "software_development"
- )
- )
+ res <- coverage_rules_path(
+ snapshot = small_snapshot,
+ contexts = small_test_context
+ )
- expect_true(
- any(
- matched$activity ==
- "unit_testing"
- )
- )
+ matched <- dplyr::filter(res, matched)
- expect_true(
- any(
- matched$activity ==
- "etl"
- )
- )
- }
-)
+ expect_true(any(matched$activity == "software_development"))
+ expect_true(any(matched$activity == "unit_testing"))
+ expect_true(any(matched$activity == "etl"))
+})
diff --git a/tests/testthat/test-derive_group_path.R b/tests/testthat/test-derive_group_path.R
index 75f2e6e..c5aaa57 100644
--- a/tests/testthat/test-derive_group_path.R
+++ b/tests/testthat/test-derive_group_path.R
@@ -1,37 +1,37 @@
-# Group path derivation -----------------------------------------------
+# Group path derivation -----------------------------------------------
test_that("derive_group_path extracts project and module", {
- x <- "_packages/iotables/R/file.R"
+ x <- "_packages/mypackage/R/file.R"
res <- derive_group_path(x)
- expect_equal(res, "_packages/iotables/R")
+ expect_equal(res, "_packages/mypackage/R")
})
test_that("derive_group_path falls back to project if no module", {
- x <- "_packages/iotables/utils.R"
+ x <- "_packages/mypackage/utils.R"
res <- derive_group_path(x)
- expect_equal(res, "_packages/iotables")
+ expect_equal(res, "_packages/mypackage")
})
test_that("derive_group_path is vectorised", {
x <- c(
- "_packages/iotables/R/a.R",
- "_packages/iotables/tests/test-a.R"
+ "_packages/mypackage/R/a.R",
+ "_packages/mypackage/tests/test-a.R"
)
res <- derive_group_path(x)
expect_equal(
res,
- c("_packages/iotables/R", "_packages/iotables/tests")
+ c("_packages/mypackage/R", "_packages/mypackage/tests")
)
})
test_that("derive_group_path does not leak filenames into group_path", {
- x <- "_packages/iotables/utils.R"
+ x <- "_packages/mypackage/utils.R"
res <- derive_group_path(x)
@@ -40,23 +40,20 @@ test_that("derive_group_path does not leak filenames into group_path", {
test_that("derive_group_path keeps _packages and packages distinct", {
x <- c(
- "_packages/iotables/a.R",
- "packages/iotables/a.R"
+ "_packages/mypackage/a.R",
+ "packages/mypackage/a.R"
)
res <- derive_group_path(x)
- expect_equal(
- res,
- c("_packages/iotables", "packages/iotables")
- )
+ expect_equal(res, c("_packages/mypackage", "packages/mypackage"))
})
test_that("derive_group_path does not leak root-level filenames", {
x <- c(
- "_packages/iotables/a.R",
- "packages/iotables/a.R",
- "_eviota/iotables/DESCRIPTION"
+ "_packages/mypackage/a.R",
+ "packages/mypackage/a.R",
+ "_eviota/mypackage/DESCRIPTION"
)
res <- derive_group_path(x)
@@ -64,17 +61,17 @@ test_that("derive_group_path does not leak root-level filenames", {
expect_equal(
res,
c(
- "_packages/iotables",
- "packages/iotables",
- "_eviota/iotables"
+ "_packages/mypackage",
+ "packages/mypackage",
+ "_eviota/mypackage"
)
)
})
test_that("derive_group_path keeps module folders", {
x <- c(
- "_packages/iotables/R/a.R",
- "_packages/iotables/tests/testthat/test-a.R",
+ "_packages/mypackage/R/a.R",
+ "_packages/mypackage/tests/testthat/test-a.R",
"packages/filmledgerimport/data-raw/input.csv"
)
@@ -83,8 +80,8 @@ test_that("derive_group_path keeps module folders", {
expect_equal(
res,
c(
- "_packages/iotables/R",
- "_packages/iotables/tests",
+ "_packages/mypackage/R",
+ "_packages/mypackage/tests",
"packages/filmledgerimport/data-raw"
)
)
@@ -92,16 +89,13 @@ test_that("derive_group_path keeps module folders", {
test_that("derive_group_path normalises Windows separators", {
x <- c(
- "_packages\\iotables\\R\\a.R",
- "packages\\iotables\\a.R"
+ "_packages\\mypackage\\R\\a.R",
+ "packages\\mypackage\\a.R"
)
res <- derive_group_path(x)
- expect_equal(
- res,
- c("_packages/iotables/R", "packages/iotables")
- )
+ expect_equal(res, c("_packages/mypackage/R", "packages/mypackage"))
})
test_that("derive_group_path handles short and empty paths", {
@@ -115,9 +109,5 @@ test_that("derive_group_path handles short and empty paths", {
})
test_that("derive_group_path handles short paths", {
- expect_true(
- is.na(
- derive_group_path("file.R")
- )
- )
+ expect_true(is.na(derive_group_path("file.R")))
})
diff --git a/tests/testthat/test-derive_structural_groups.R b/tests/testthat/test-derive_structural_groups.R
index 64f7c95..cfa7df7 100644
--- a/tests/testthat/test-derive_structural_groups.R
+++ b/tests/testthat/test-derive_structural_groups.R
@@ -119,3 +119,335 @@ test_that("derive_group_path handles _packages and packages consistently", {
c("_packages/iotables/R", "packages/iotables/R")
)
})
+
+
+# Basic behaviour ------------------------------------------------------
+
+test_that("derive_structural_groups returns expected structure", {
+ rel_path <- c(
+ "a/b/c.txt",
+ "x/y",
+ "single"
+ )
+
+ res <- derive_structural_groups(rel_path)
+
+ expect_s3_class(res, "data.frame")
+
+ expect_setequal(names(res), c("structural_group", "component"))
+
+ expect_equal(nrow(res), length(rel_path))
+})
+
+
+# Standard paths -------------------------------------------------------
+
+test_that("derive_structural_groups parses standard paths correctly", {
+ rel_path <- c(
+ "_packages/eviota/R/file.R",
+ "_markdown/report/analysis.qmd"
+ )
+
+ res <- derive_structural_groups(rel_path)
+
+ expect_equal(res$structural_group[1], "_packages/eviota")
+ expect_equal(res$component[1], "R")
+
+ expect_equal(res$structural_group[2], "_markdown/report")
+ expect_equal(res$component[2], "analysis.qmd")
+})
+
+
+# Edge cases -----------------------------------------------------------
+
+test_that("derive_structural_groups handles short paths", {
+ rel_path <- c(
+ "file.R",
+ "folder/file.R"
+ )
+
+ res <- derive_structural_groups(rel_path)
+
+ # length 1 → no component
+ expect_equal(res$structural_group[1], "file.R")
+ expect_true(is.na(res$component[1]))
+
+ # length 2 → no component
+ expect_equal(res$structural_group[2], "folder/file.R")
+ expect_true(is.na(res$component[2]))
+})
+
+
+test_that("derive_structural_groups handles empty or malformed input", {
+ rel_path <- c("", NA)
+
+ res <- derive_structural_groups(rel_path)
+
+ expect_equal(nrow(res), 2)
+})
+
+
+# Determinism ----------------------------------------------------------
+
+test_that("derive_structural_groups is deterministic", {
+ rel_path <- c(
+ "a/b/c.txt",
+ "x/y/z.txt"
+ )
+
+ res1 <- derive_structural_groups(rel_path)
+ res2 <- derive_structural_groups(rel_path)
+
+ expect_equal(res1, res2)
+})
+
+
+# Integration with snapshot --------------------------------------------
+
+test_that("derive_structural_groups works on fscontextdemo_snapshot_02", {
+ data(fscontextdemo_snapshot_02, package = "fscontext")
+
+ res <- derive_structural_groups(fscontextdemo_snapshot_02$rel_path)
+
+ expect_equal(nrow(res), nrow(fscontextdemo_snapshot_02))
+
+ expect_true(all(!is.na(res$structural_group)))
+})
+
+
+# Conceptual consistency -----------------------------------------------
+
+test_that("structural_group is consistent with rel_path prefix", {
+ rel_path <- c(
+ "a/b/c/d.txt",
+ "x/y/z.txt"
+ )
+
+ res <- derive_structural_groups(rel_path)
+
+ expect_true(all(startsWith(rel_path, res$structural_group)))
+})
+
+
+test_that("derive_group_path handles _packages and packages consistently", {
+ x <- c(
+ "_packages/iotables/R/a.R",
+ "packages/iotables/R/a.R"
+ )
+
+ res <- derive_group_path(x)
+
+ expect_equal(
+ res,
+ c("_packages/iotables/R", "packages/iotables/R")
+ )
+})
+
+## WACZ ---------------------------------------------------------------------
+test_that("derive_structural_groups handles WACZ package members", {
+ rel_path <- c(
+ "archive/data.warc.gz",
+ "indexes/index.cdx",
+ "pages/pages.jsonl",
+ "datapackage.json"
+ )
+
+ res <- derive_structural_groups(rel_path)
+
+ expect_equal(
+ res$structural_group,
+ c(
+ "archive/data.warc.gz",
+ "indexes/index.cdx",
+ "pages/pages.jsonl",
+ "datapackage.json"
+ )
+ )
+
+ expect_true(all(is.na(res$component)))
+})
+
+test_that("derive_structural_groups works on WACZ observations", {
+ wacz <- scan_storage(
+ system.file(
+ "testdata/fscontext_020.wacz",
+ package = "fscontext"
+ )
+ )
+
+ res <- derive_structural_groups(
+ wacz$rel_path
+ )
+
+ expect_equal(
+ nrow(res),
+ nrow(wacz)
+ )
+
+ expect_true(
+ all(!is.na(res$structural_group))
+ )
+})
+
+# Profiles -------------------------------------------------------------
+
+test_that("folder-depth-1 groups by first folder level", {
+ rel_path <- c(
+ "_packages/eviota/R/file.R",
+ "_packages/iotables/R/file.R"
+ )
+
+ res <- derive_structural_groups(
+ rel_path,
+ profile = "folder-depth-1"
+ )
+
+ expect_equal(
+ res$structural_group,
+ c("_packages", "_packages")
+ )
+
+ expect_equal(
+ res$component,
+ c("eviota", "iotables")
+ )
+})
+
+test_that("folder-depth-3 groups by first three folder levels", {
+ rel_path <- "_packages/eviota/R/file.R"
+
+ res <- derive_structural_groups(
+ rel_path,
+ profile = "folder-depth-3"
+ )
+
+ expect_equal(
+ res$structural_group,
+ "_packages/eviota/R"
+ )
+
+ expect_equal(
+ res$component,
+ "file.R"
+ )
+})
+
+test_that("folder-depth-4 falls back gracefully on short paths", {
+ rel_path <- c(
+ "archive/data.warc.gz",
+ "datapackage.json"
+ )
+
+ res <- derive_structural_groups(
+ rel_path,
+ profile = "folder-depth-4"
+ )
+
+ expect_equal(
+ res$structural_group,
+ rel_path
+ )
+
+ expect_true(
+ all(is.na(res$component))
+ )
+})
+
+test_that("unknown profile throws an error", {
+ expect_error(
+ derive_structural_groups(
+ "a/b/c.txt",
+ profile = "banana"
+ )
+ )
+})
+
+## WACZ ---------------------------------------------------------------------
+## WACZ ----------------------------------------------------------------
+
+test_that("wacz profile derives top-level WACZ aggregations", {
+ rel_path <- c(
+ "archive/data.warc.gz",
+ "indexes/index.cdx",
+ "pages/pages.jsonl",
+ "datapackage.json"
+ )
+
+ res <- derive_structural_groups(
+ rel_path,
+ profile = "wacz"
+ )
+
+ expect_equal(
+ res$structural_group,
+ c(
+ "archive",
+ "indexes",
+ "pages",
+ "datapackage.json"
+ )
+ )
+
+ expect_equal(
+ res$component,
+ c(
+ "data.warc.gz",
+ "index.cdx",
+ "pages.jsonl",
+ NA_character_
+ )
+ )
+})
+
+test_that("wacz profile works on WACZ observations", {
+ wacz <- scan_storage(
+ system.file(
+ "testdata/fscontext_020.wacz",
+ package = "fscontext"
+ )
+ )
+
+ res <- derive_structural_groups(
+ wacz$rel_path,
+ profile = "wacz"
+ )
+
+ expect_equal(nrow(res), nrow(wacz))
+
+ expect_true(
+ all(!is.na(res$structural_group))
+ )
+
+ expect_setequal(
+ unique(res$structural_group),
+ c(
+ "archive", "indexes", "pages",
+ "datapackage-digest.json", "datapackage.json"
+ )
+ )
+})
+
+
+test_that("folder-depth-2 and wacz produce different aggregations", {
+ rel_path <- "archive/data.warc.gz"
+
+ depth2 <- derive_structural_groups(
+ rel_path,
+ profile = "folder-depth-2"
+ )
+
+ wacz <- derive_structural_groups(
+ rel_path,
+ profile = "wacz"
+ )
+
+ expect_equal(
+ depth2$structural_group,
+ "archive/data.warc.gz"
+ )
+
+ expect_equal(
+ wacz$structural_group,
+ "archive"
+ )
+})
diff --git a/tests/testthat/test-observe_wacz.R b/tests/testthat/test-observe_wacz.R
new file mode 100644
index 0000000..a980cb2
--- /dev/null
+++ b/tests/testthat/test-observe_wacz.R
@@ -0,0 +1,170 @@
+test_that("observe_wacz() reads a WACZ archive", {
+ wacz <- system.file(
+ "testdata",
+ "fscontext_020.wacz",
+ package = "fscontext"
+ )
+
+ obs <- observe_wacz(wacz)
+
+ expect_true(
+ all(!is.na(obs$quick_sig_text))
+ )
+
+ expect_true(
+ any(!is.na(obs$digest))
+ )
+
+ expect_s3_class(obs, "data.frame")
+
+ expect_gt(nrow(obs), 0)
+
+ expect_true("resource_locator" %in% names(obs))
+
+ expect_true("page_id" %in% names(obs))
+
+ expect_true("archive" %in% names(obs))
+
+ expect_equal(basename(attr(obs, "wacz")), "fscontext_020.wacz")
+
+ dp <- attr(obs, "datapackage")
+
+ expect_type(dp, "list")
+
+ expect_true(
+ "title" %in% names(dp)
+ )
+})
+
+
+test_that("read_pages_jsonl() reads page metadata", {
+ wacz <- system.file(
+ "testdata",
+ "fscontext_020.wacz",
+ package = "fscontext"
+ )
+
+ tmp <- tempfile("wacz")
+
+ extract_storage(
+ archive = wacz,
+ exdir = tmp
+ )
+
+ pages <- read_pages_jsonl(tmp)
+
+ expect_s3_class(pages, "tbl_df")
+
+ expect_gt(nrow(pages), 0)
+
+ expect_true(
+ all(
+ c(
+ "page_id",
+ "title",
+ "resource_locator",
+ "timestamp",
+ "favicon",
+ "text",
+ "text_length",
+ "quick_sig_text"
+ ) %in% names(pages)
+ )
+ )
+
+ expect_true(all(!is.na(pages$resource_locator)))
+
+ expect_true(all(pages$text_length >= 0))
+
+ expect_true(all(!is.na(pages$quick_sig_text)))
+})
+
+test_that("collapse_cdx_versions() collapses repeated resources", {
+ wacz <- system.file(
+ "testdata",
+ "fscontext_020.wacz",
+ package = "fscontext"
+ )
+
+ tmp <- tempfile("wacz")
+
+ extract_storage(
+ archive = wacz,
+ exdir = tmp
+ )
+
+ cdx <- read_cdx(tmp)
+
+ collapsed <- collapse_cdx_versions(cdx)
+
+ expect_s3_class(collapsed, "tbl_df")
+
+ expect_true(all(collapsed$mime == "text/html"))
+
+ expect_true(all(collapsed$n_versions >= 1))
+
+ expect_equal(anyDuplicated(collapsed$resource_locator), 0L)
+})
+
+test_that("match_pages_to_cdx() joins page observations and archive metadata", {
+ wacz <- system.file(
+ "testdata",
+ "fscontext_020.wacz",
+ package = "fscontext"
+ )
+
+ tmp <- tempfile("wacz")
+
+ extract_storage(
+ archive = wacz,
+ exdir = tmp
+ )
+
+ pages <- read_pages_jsonl(tmp)
+
+ cdx <- read_cdx(tmp) |>
+ collapse_cdx_versions()
+
+ matched <- match_pages_to_cdx(
+ pages,
+ cdx
+ )
+
+ expect_s3_class(matched, "tbl_df")
+
+ expect_equal(nrow(matched), nrow(pages))
+
+ expect_true(
+ any(!is.na(matched$digest))
+ )
+
+ expect_true(
+ any(!is.na(matched$offset))
+ )
+})
+
+
+test_that("read_datapackage() reads WACZ metadata", {
+ wacz <- system.file(
+ "testdata",
+ "fscontext_020.wacz",
+ package = "fscontext"
+ )
+
+ tmp <- tempfile("wacz")
+
+ extract_storage(
+ archive = wacz,
+ exdir = tmp
+ )
+
+ dp <- read_datapackage(tmp)
+
+ expect_true(
+ all(
+ c("profile", "title", "created", "resources") %in% names(dp)
+ )
+ )
+
+ expect_type(dp, "list")
+})
diff --git a/tests/testthat/test-create_record_set.R b/tests/testthat/test-record_set_projection.R
similarity index 50%
rename from tests/testthat/test-create_record_set.R
rename to tests/testthat/test-record_set_projection.R
index b3c1dc7..a8eea6c 100644
--- a/tests/testthat/test-create_record_set.R
+++ b/tests/testthat/test-record_set_projection.R
@@ -1,9 +1,9 @@
-test_that("create_record_set creates contextual record set projection", {
+test_that("record_set_projection creates contextual record set projection", {
toy_resources <- tibble::tibble(
structural_group = c(
- "_packages/eviota",
- "_packages/eviota",
- "_packages/iotables"
+ "_packages/pkg-a",
+ "_packages/pkg-a",
+ "_packages/pkg-b"
),
path_id = c(
"l480::R/import.R",
@@ -17,9 +17,9 @@ test_that("create_record_set creates contextual record set projection", {
)
)
- rs <- create_record_set(
+ rs <- record_set_projection(
toy_resources,
- record_set_id = "structural_group",
+ record_set_identifier = "structural_group",
resource_id = "path_id",
locator_path = "rel_root_path",
construction_rule =
@@ -32,7 +32,7 @@ test_that("create_record_set creates contextual record set projection", {
expect_true(
all(
c(
- "record_set_id",
+ "record_set_identifier",
"resource_id",
"locator_path",
"resource_type"
@@ -40,54 +40,30 @@ test_that("create_record_set creates contextual record set projection", {
)
)
- expect_equal(
- rs$record_set_id,
- toy_resources$structural_group
- )
+ expect_equal(rs$record_set_identifier, toy_resources$structural_group)
- expect_equal(
- rs$resource_id,
- toy_resources$path_id
- )
+ expect_equal(rs$resource_id, toy_resources$path_id)
- expect_equal(
- rs$locator_path,
- toy_resources$rel_root_path
- )
+ expect_equal(rs$locator_path, toy_resources$rel_root_path)
- expect_equal(
- unique(rs$resource_type),
- "file"
- )
+ expect_equal(unique(rs$resource_type), "file")
expect_equal(
attr(rs, "construction_rule"),
"filtered_project_roots|structural_group"
)
-
- expect_equal(
- attr(rs, "created_by"),
- "create_record_set"
- )
-
- expect_true(
- inherits(
- attr(rs, "record_set_created_at"),
- "POSIXct"
- )
- )
})
-test_that("create_record_set errors without required identifiers", {
+test_that("record_set_projection errors without required identifiers", {
toy_resources <- tibble::tibble(
rel_path = c("a.txt", "b.txt")
)
expect_error(
- create_record_set(
+ record_set_projection(
toy_resources,
- record_set_id = NULL,
+ record_set_identifier = NULL,
resource_id = NULL,
construction_rule = "test"
),
@@ -96,29 +72,20 @@ test_that("create_record_set errors without required identifiers", {
})
-test_that("create_record_set accepts scalar values", {
+test_that("record_set_projection accepts scalar values", {
toy_resources <- tibble::tibble(
- path_id = c(
- "id1",
- "id2"
- )
+ path_id = c("id1", "id2")
)
- rs <- create_record_set(
+ rs <- record_set_projection(
toy_resources,
- record_set_id = "toy_record_set",
+ record_set_identifier = "toy_record_set",
resource_id = "path_id",
construction_rule = "manual",
resource_type = "file"
)
- expect_equal(
- unique(rs$record_set_id),
- "toy_record_set"
- )
+ expect_equal(unique(rs$record_set_identifier), "toy_record_set")
- expect_equal(
- unique(rs$resource_type),
- "file"
- )
+ expect_equal(unique(rs$resource_type), "file")
})
diff --git a/tests/testthat/test-recordset_df.R b/tests/testthat/test-recordset_df.R
index f227dee..be960e0 100644
--- a/tests/testthat/test-recordset_df.R
+++ b/tests/testthat/test-recordset_df.R
@@ -1,90 +1,154 @@
-test_that("recordset_df creates a valid recordset_df object", {
- toy_recordset <- recordset_df(
- record_set_id = c(
- "eviota",
- "eviota",
- "eviota"
- ),
- member_id = c(
- "inst_001",
- "inst_002",
- "inst_003"
- ),
- member_path = c(
- "filmledgerimport/R/import.R",
- "eviota/data-raw/build.R",
- "eviota/reports/report.qmd"
- ),
- member_type = c(
- "file",
- "file",
- "file"
- ),
- source_type = c(
- "filesystem",
- "filesystem",
- "filesystem"
- ),
- identifier = c(
- member =
- "https://example.org/recordset/eviota#member"
- ),
- var_labels = list(
- record_set_id = "Record set identifier",
- member_id = "Member identifier",
- member_path = "Member path"
- ),
- concepts = list(
- record_set_id =
- "https://www.ica.org/standards/RiC/ontology#RecordSet",
- member_id =
- "https://www.ica.org/standards/RiC/ontology#Instantiation"
- ),
- dataset_bibentry = dataset::dublincore(
- title = "Toy Eviota Record Set",
- creator = person("Daniel", "Antal"),
- publisher = "fscontext"
- )
+test_that("recordset_df creates a dataset_df intherited record set df", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
)
- expect_s3_class(
- toy_recordset,
- "recordset_df"
+ rs <- recordset_df(x,
+ title = "Test"
)
- expect_s3_class(
- toy_recordset,
- "dataset_df"
+ expect_identical(
+ dataset::subject(rs)$term,
+ "Record Set"
)
- expect_true(
- is.data.frame(toy_recordset)
+ expect_s3_class(rs, "recordset_df")
+ expect_s3_class(rs, "dataset_df")
+ expect_true(is.data.frame(rs))
+ expect_equal(attr(rs, "dataset_bibentry")$title, "Test")
+
+ expect_identical(dataset::dataset_title(rs), "Test")
+})
+
+test_that("record_set_identifier is set properly", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
)
- expect_true(
- all(
- c("record_set_id", "member_id") %in%
- names(toy_recordset)
- )
+ rs <- recordset_df(
+ x,
+ record_set_identifier = "rs001"
+ )
+
+
+ expect_identical(dataset::identifier(rs), "rs001")
+})
+
+test_that("record_identifier is declared as a RiC identifier", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
)
+
+ rs <- recordset_df(
+ x,
+ record_identifier = "resource_locator"
+ )
+
+ expect_s3_class(rs$resource_locator, "haven_labelled_defined")
+ expect_identical(attr(rs$resource_locator, "concept"), "rico:Identifier")
})
-test_that("recordset_df requires record_set_id", {
+test_that("missing record identifier column throws an error", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
+ )
+
expect_error(
- recordset_df(
- member_id = c("inst_001"),
- member_path = c("file.R")
- ),
- "Missing required columns: record_set_id"
+ recordset_df(x, record_identifier = "missing"),
+ "Column not found"
)
})
-test_that("recordset_df requires member_id", {
+test_that("duplicate record identifiers produce a warning", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/1"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
+ )
+
+ expect_warning(
+ recordset_df(x, record_identifier = "resource_locator"),
+ "not unique"
+ )
+})
+
+test_that("record_part_identifier is declared as a RiC identifier", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
+ )
+
+ rs <- recordset_df(
+ x,
+ record_identifier = "resource_locator",
+ record_part_identifier = "filename"
+ )
+
+ expect_s3_class(rs$filename, "haven_labelled_defined")
+
+ expect_identical(attr(rs$filename, "concept"), "rico:Identifier")
+
+ expect_equal(as.character(unclass(rs$filename)), c("a.html", "b.html"))
+
expect_error(
recordset_df(
- record_set_id = c("eviota"),
- member_path = c("file.R")
+ x,
+ record_part_identifier = "missing"
),
- "Missing required columns: member_id"
+ "Column not found"
+ )
+
+ x <- data.frame(
+ id = c("r1", "r2"),
+ filename = c("a.html", "a.html")
+ )
+
+ expect_warning(
+ recordset_df(x, record_part_identifier = "filename"),
+ "not unique"
+ )
+})
+
+test_that("dataset-level metadata is preserved", {
+ x <- data.frame(
+ resource_locator = c("https://example.org/1", "https://example.org/2"),
+ filename = c("a.html", "b.html"),
+ stringsAsFactors = FALSE
+ )
+
+ rs <- recordset_df(
+ x,
+ record_identifier = "resource_locator",
+ record_part_identifier = "filename",
+ title = "Demo Record Set",
+ creator = person("Joe", "Doe", role = "aut")
+ )
+
+ # Test on core elements of the dataset_bibentry without class-specific
+ # implementation details.
+
+ expect_identical(
+ attr(rs, "dataset_bibentry")$title,
+ "Demo Record Set"
+ )
+
+ expect_identical(
+ attr(rs, "dataset_bibentry")$author,
+ person("Joe", "Doe", role = "aut")
+ )
+
+ rpt <- " "
+
+ expect_true(
+ any(grepl(pattern = rpt, x = attr(rs, "prov")))
)
})
diff --git a/tests/testthat/test-save_scan.R b/tests/testthat/test-save_scan.R
index 5b51464..1ef443d 100644
--- a/tests/testthat/test-save_scan.R
+++ b/tests/testthat/test-save_scan.R
@@ -38,7 +38,7 @@ test_that("save_scan writes file and returns path", {
df <- data.frame(x = 1)
attr(df, "created_at") <- as.POSIXct("2026-04-30 16:00:59", tz = "UTC")
- path <- save_scan(df, "test-storage", tmp)
+ path <- save_scan(df = df, storage_id = "test-storage", path = tmp)
expect_true(file.exists(path))
diff --git a/tests/testthat/test-scan_storage.R b/tests/testthat/test-scan_storage.R
index 8ccec8e..1b19e2c 100644
--- a/tests/testthat/test-scan_storage.R
+++ b/tests/testthat/test-scan_storage.R
@@ -1,10 +1,9 @@
-library(testthat)
-
# Structure ------------------------------------------------------------
test_that("scan_storage returns expected structure", {
- root <- system.file("testdata/minimal_R_folder", package = "fscontext")
- stopifnot(nzchar(root))
+ root <- system.file("testdata/minimal_R_folder",
+ package = "fscontext"
+ )
res <- scan_storage(root)
expect_s3_class(res, "data.frame")
@@ -303,3 +302,90 @@ test_that("full_path stores filesystem paths", {
expect_true(all(fs::file_exists(res$full_path)))
})
+
+## Working with ZIP files --------------------------------------------------
+test_that(
+ "zip storage reproduces directory observations",
+ {
+ folder_root <- system.file("testdata/minimal_R_folder",
+ package = "fscontext"
+ )
+
+ zip_root <- system.file(
+ "testdata/minimal_R_folder.zip",
+ package = "fscontext"
+ )
+
+ folder <- scan_storage(folder_root)
+
+ zip <- scan_storage(zip_root)
+
+ folder <- folder[!grepl("(^|/)\\.", folder$rel_path), ]
+ zip <- zip[!grepl("(^|/)\\.", zip$rel_path), ]
+
+ folder <- folder[order(folder$rel_path), ]
+
+ zip <- zip[order(zip$rel_path), ]
+
+ expect_equal(folder$rel_path, zip$rel_path)
+
+ expect_equal(folder$filename, zip$filename)
+
+ expect_equal(folder$extension, zip$extension)
+
+ expect_equal(folder$size, zip$size)
+
+ expect_equal(folder$quick_sig, zip$quick_sig)
+ }
+)
+
+test_that("zip files can be observed", {
+ zip_root <- system.file(
+ "testdata/minimal_R_folder.zip",
+ package = "fscontext"
+ )
+
+ res <- scan_storage(zip_root)
+
+ expect_s3_class(res, "data.frame")
+ expect_gt(nrow(res), 10)
+})
+
+
+test_that("zip observations can be contextualised", {
+ zip_root <- system.file(
+ "testdata/minimal_R_folder.zip",
+ package = "fscontext"
+ )
+
+ res <- scan_storage(zip_root)
+
+ ctx <- add_snapshot_context(res)
+
+ expect_true(
+ "observation_id" %in% names(ctx)
+ )
+})
+
+
+test_that(
+ "zip snapshots can be contextualised",
+ {
+ zip_root <- system.file(
+ "testdata/minimal_R_folder.zip",
+ package = "fscontext"
+ )
+
+ zip <- scan_storage(zip_root)
+
+ zip <- add_snapshot_context(zip)
+
+ expect_true(
+ "observation_id" %in% names(zip)
+ )
+
+ expect_true(
+ "storage_full_path" %in% names(zip)
+ )
+ }
+)
diff --git a/tests/testthat/test-snapshot_to_reconstruction_context.R b/tests/testthat/test-snapshot_to_reconstruction_context.R
index 4caee81..ee6d8d9 100644
--- a/tests/testthat/test-snapshot_to_reconstruction_context.R
+++ b/tests/testthat/test-snapshot_to_reconstruction_context.R
@@ -29,7 +29,7 @@ test_that("snapshot_to_reconstruction_context reconstructs expected columns", {
"observation_id",
"structural_group",
"component",
- "record_set_id",
+ "record_set_identifier",
"resource_id",
"locator_path"
) %in% names(out)))
@@ -37,7 +37,7 @@ test_that("snapshot_to_reconstruction_context reconstructs expected columns", {
expect_gt(nrow(out), 0)
})
-test_that("record_set_id is never missing", {
+test_that("record_set_identifier is never missing", {
data("fscontextdemo_snapshot_02")
f <- tempfile(fileext = ".rds")
@@ -50,8 +50,8 @@ test_that("record_set_id is never missing", {
roots = roots
)
- expect_false(any(is.na(out$record_set_id)))
- expect_false(any(out$record_set_id == ""))
+ expect_false(any(is.na(out$record_set_identifier)))
+ expect_false(any(out$record_set_identifier == ""))
})
diff --git a/tests/testthat/test-snapshot_to_recordset_df.R b/tests/testthat/test-snapshot_to_recordset_df.R
index 88ba376..e228c20 100644
--- a/tests/testthat/test-snapshot_to_recordset_df.R
+++ b/tests/testthat/test-snapshot_to_recordset_df.R
@@ -9,8 +9,8 @@ test_that("snapshot_to_recordset_df returns a semantic recordset_df", {
rs <- snapshot_to_recordset_df(
snapshot_files = snapshot_file,
roots = roots,
- record_set_id = "test-record-set",
- person = utils::person("Jane", "Doe")
+ record_set_identifier = "test-record-set",
+ creator = utils::person("Jane", "Doe")
)
expect_s3_class(rs, "recordset_df")
@@ -18,9 +18,6 @@ test_that("snapshot_to_recordset_df returns a semantic recordset_df", {
expect_s3_class(rs, "data.frame")
expect_gt(nrow(rs), 0)
-
- expect_true("record_set_id" %in% names(rs))
- expect_equal(unique(rs$record_set_id), "test-record-set")
})
test_that("snapshot_to_recordset_df applies asserted metadata", {
@@ -34,15 +31,11 @@ test_that("snapshot_to_recordset_df applies asserted metadata", {
rs <- snapshot_to_recordset_df(
snapshot_files = snapshot_file,
roots = roots,
- record_set_id = "test-record-set",
+ record_set_identifier = "test-record-set",
record_set_title = "The test-record-set filesystem record set",
- person = utils::person("Jane", "Doe")
+ creator = utils::person("Jane", "Doe")
)
- expect_equal(
- unique(rs$record_set_id),
- "test-record-set"
- )
expect_equal(
dataset::dataset_title(rs),
@@ -66,9 +59,9 @@ test_that("snapshot_to_recordset_df preserves reconstructed observations", {
rs <- snapshot_to_recordset_df(
snapshot_files = snapshot_file,
- person = utils::person("Jane", "Doe"),
+ creator = utils::person("Jane", "Doe"),
roots = roots,
- record_set_id = "test-record-set"
+ record_set_identifier = "test-record-set"
)
# ----------------------------------------------------------
@@ -83,7 +76,7 @@ test_that("snapshot_to_recordset_df preserves reconstructed observations", {
"filename",
"mtime",
"scan_time",
- "record_set_id"
+ "record_set_identifier"
)
expect_true(
@@ -114,8 +107,8 @@ test_that("snapshot_to_recordset_df preserves core observational columns", {
rs <- snapshot_to_recordset_df(
snapshot_files = snapshot_file,
roots = roots,
- record_set_id = "test-record-set",
- person = utils::person("Jane", "Doe")
+ record_set_identifier = "test-record-set",
+ creator = utils::person("Jane", "Doe")
)
required_cols <- c(
@@ -126,7 +119,7 @@ test_that("snapshot_to_recordset_df preserves core observational columns", {
"filename",
"mtime",
"scan_time",
- "record_set_id",
+ "record_set_identifier",
"resource_id",
"locator_path",
"observation_id"
@@ -157,9 +150,9 @@ test_that("snapshot_to_recordset_df attaches provenance metadata", {
rs <- snapshot_to_recordset_df(
snapshot_files = snapshot_files,
- person = utils::person("Jane", "Doe"),
+ creator = utils::person("Jane", "Doe"),
roots = roots,
- record_set_id = "test-record-set"
+ record_set_identifier = "test-record-set"
)
prov <- dataset::provenance(rs)
diff --git a/tests/testthat/test-summarise_observed_activity.R b/tests/testthat/test-summarise_observed_activity.R
index 789f3ac..a64bdfa 100644
--- a/tests/testthat/test-summarise_observed_activity.R
+++ b/tests/testthat/test-summarise_observed_activity.R
@@ -5,12 +5,12 @@ library(testthat)
make_test_df <- function() {
df <- data.frame(
rel_path = c(
- "_eviota/reporting/R/a.R",
- "_eviota/reporting/R/b.R",
- "_eviota/reporting/tests/test-a.R",
- "_packages/iotables/R/c.R",
- "_packages/iotables/data-raw/d.bak",
- "_packages/iotables/R/e.R"
+ "_devel/reporting/R/a.R",
+ "_devel/reporting/R/b.R",
+ "_devel/reporting/tests/test-a.R",
+ "_packages/mypackage/R/c.R",
+ "_packages/mypackage/data-raw/d.bak",
+ "_packages/mypackage/R/e.R"
),
extension = c("r", "r", "r", "r", "bak", "r"),
mtime = as.POSIXct(c(
diff --git a/vignettes/intro.Rmd b/vignettes/intro.Rmd
index 3ac6f4b..392e2ae 100644
--- a/vignettes/intro.Rmd
+++ b/vignettes/intro.Rmd
@@ -139,24 +139,24 @@ This creates lightweight contextual structures that support later reconstruction
One of the central goals of the package is to derive contextual Record Sets from filesystem observations.
-```{r}
+```{r snapshotrecordsetdf}
tmp <- tempfile(fileext = ".rds")
saveRDS(fscontextdemo_snapshot_02, tmp)
record_set <- snapshot_to_recordset_df(
- person = utils::person("Jane", "Doe"),
+ creator = utils::person("Jane", "Doe"),
snapshot_files = tmp,
roots = "D:/_packages/fscontextdemo",
- record_set_id = "fscontextdemo"
+ record_set_identifier = "fscontextdemo"
)
```
Record Sets provide contextual documentary groupings derived from filesystem evidence.
-```{r}
+```{r subsettingrecordset}
set.seed(12)
record_set |>
- dplyr::select(record_set_id, filename, quick_sig, size) |>
+ dplyr::select(record_set_identifier, filename, quick_sig, size) |>
sample_n(10)
```
@@ -166,7 +166,7 @@ A single snapshot provides a static view.
Multiple snapshots allow longitudinal analysis.
-```{r, eval=FALSE}
+```{r observeuniverse, eval=FALSE}
observe_universe(
snapshot_dir = snapshot_directory,
max_aggregation_depth = 2
diff --git a/vignettes/recordset_df.Rmd b/vignettes/recordset_df.Rmd
new file mode 100644
index 0000000..5963a47
--- /dev/null
+++ b/vignettes/recordset_df.Rmd
@@ -0,0 +1,111 @@
+---
+title: "Working with Record Sets"
+output: rmarkdown::html_vignette
+vignette: >
+ %\VignetteIndexEntry{Working with Record Sets}
+ %\VignetteEngine{knitr::rmarkdown}
+ %\VignetteEncoding{UTF-8}
+---
+
+```{r, include = FALSE}
+knitr::opts_chunk$set(
+ collapse = TRUE,
+ comment = "#>"
+)
+```
+
+This vignette demonstrates how to construct a semantically annotated `recordset_df` from ordinary filesystem observations. As a minimal example, we create five files representing two Records. Record `a` consists of a textual description and an image of a museum object. Record `b` consists of a textual description, a digital surrogate, and an OCR transcription of an archival document.
+
+```{r setup}
+library(fscontext)
+tmp_dir <- file.path(tempdir(), "recordset_df")
+if (dir.exists(tmp_dir)) unlink(tmp_dir, recursive = TRUE)
+dir.create(tmp_dir)
+
+writeLines(
+ c("", "", "Description of the A object.", "", ""),
+ file.path(tmp_dir, "a.html")
+)
+Sys.sleep(1)
+
+writeBin(charToRaw("JPEG"), file.path(tmp_dir, "a.jpg"))
+Sys.sleep(1)
+
+writeLines(
+ c("", "", "Description of the B document.", "", ""),
+ file.path(tmp_dir, "b.html")
+)
+Sys.sleep(1)
+
+writeLines(
+ c("%PDF-1.4", "Digital surrogate of B."),
+ file.path(tmp_dir, "b.pdf")
+)
+Sys.sleep(1)
+
+writeLines(
+ "Plain text transcription of B.",
+ file.path(tmp_dir, "b.txt")
+)
+```
+
+We observe the temporary directory using `snapshot_storage()`. The resulting snapshot records file-level observations such as names, timestamps, checksums and other filesystem metadata without making any assumptions about the semantic relationships between the files.
+
+```{r makesnapshot}
+rs001_snapshot_file <- snapshot_storage(
+ path = tmp_dir,
+ root = tmp_dir
+)
+```
+
+Next we add simple curatorial metadata. In this example we assign a human-readable description to each observed file and indicate which files belong to the same Record. You can add any further metadata or data about the records.
+
+```{r subsetsnapshot}
+rs001_snapshot <- readRDS(rs001_snapshot_file)
+rs001_snapshot$description <- c(
+ "Description of Object A", "Image of Object A",
+ "Description of Record B", "Surrogate of Record B", "OCR Text of Record B"
+)
+rs001_df <- rs001_snapshot[
+ ,
+ c("stem", "filename", "description", "quick_sig", "ctime")
+]
+```
+
+## recordset_df
+
+The recordset_df class extends `dataset::dataset_df` with lightweight semantics for describing Record Sets, Records and Record Parts. Rather than implementing the complete RiC ontology, it provides a small number of conventions that support reproducible workflows while remaining compatible with ordinary tidy data.
+
+The`recordset_df` uses `dataset_df` internally for metadata, provenance and serialisation, which is an extended `tibble::tibble()` `tbl_df` data frame. Users who require richer metadata or publication-oriented functionality can use the methods provided by the [dataset package](https://dataset.dataobservatory.eu/articles/dataset_df.html) directly.
+
+```{r createrecordset}
+rs001 <- recordset_df(
+ x = rs001_df,
+ creator = utils::person("Jane", "Doe", role = "aut"),
+ title = "Demonstrator Record Set",
+ record_set_identifier = "http://example.com/archive/sets/rs001",
+ description = "A demonstration of a record set",
+ record_identifier = "stem",
+ record_part_identifier = "filename"
+)
+```
+
+The constructor warns that the Record identifiers are not unique. This is expected because each Record is represented by multiple observed files. In this example, Record `a` has two Record Parts (individual files) and Record `b` has three Record Parts (files), so the Record identifier necessarily occurs more than once.
+
+```{r printrecordset}
+print(rs001)
+```
+
+The `stem` column is declared to contain identifiers of RiC Records. The values are annotated as `rico:Identifier` objects and labelled "Record Identifier", allowing downstream software to distinguish Record identifiers from other identifiers without requiring a complete RiC knowledge graph. (See: [rico:Record](https://www.ica.org/standards/RiC/RiC-O_1-0-2.html#Record))
+
+```{r recordlevel}
+rs001$stem
+```
+
+The `filename` column is declared to contain identifiers of RiC Record Parts. Record `a` consists of two Record Parts—a textual description and an image of the object—while Record `b` consists of three Record Parts: a textual description, a digital surrogate and an OCR transcription. Each file therefore identifies an individual Record Part within its parent Record. (See: [rico:RecordPart](https://www.ica.org/standards/RiC/RiC-O_1-0-2.html#RecordPart))
+
+```{r recorpart}
+rs001$filename
+```
+
+This example illustrates the intended role of `recordset_df`: observational evidence is acquired first, and lightweight semantic assertions are added afterwards. The resulting object remains an ordinary `data.frame` while carrying sufficient metadata to support reproducible archival, curatorial and semantic enrichment workflows.
diff --git a/vignettes/structural_aggregations.Rmd b/vignettes/structural_aggregations.Rmd
new file mode 100644
index 0000000..8a30b32
--- /dev/null
+++ b/vignettes/structural_aggregations.Rmd
@@ -0,0 +1,325 @@
+---
+title: "From Structural Aggregations to Record Sets"
+output: rmarkdown::html_vignette
+vignette: >
+ %\VignetteIndexEntry{From Structural Aggregations to Record Sets}
+ %\VignetteEngine{knitr::rmarkdown}
+ %\VignetteEncoding{UTF-8}
+---
+
+```{r, include = FALSE}
+knitr::opts_chunk$set(
+ collapse = TRUE,
+ comment = "#>"
+)
+```
+
+## Motivation
+
+A filesystem snapshot may contain hundreds, thousands, or millions of observed resources.
+
+Before creating Record Sets, users often need lightweight aggregation metadata that reveals potentially informative structures within the observations.
+
+`derive_structural_groups()` creates such aggregation metadata from observed locators.
+
+The resulting groupings are not Record Sets. They are candidate aggregations that may later support contextual reconstruction, semantic stabilisation, or human curation.
+
+## Example 1: An R package
+
+```{r setup, echo=FALSE, message=FALSE}
+library(fscontext)
+library(dplyr)
+```
+
+Observe the demo package included with the `fscontext` pacakage for documentation purposes:
+
+```{r show-folder}
+root <- system.file(
+ "testdata/minimal_R_folder",
+ package = "fscontext"
+)
+
+fs::dir_tree(root)
+```
+
+The package contains source code, documentation, data preparation scripts, and vignettes organised into a conventional project structure.
+
+We can observe the package and derive structural aggregations:
+
+```{r observedemo}
+snapshot <- scan_storage(
+ system.file(
+ "testdata/minimal_R_folder",
+ package = "fscontext"
+ )
+)
+
+groups <- derive_structural_groups(
+ snapshot$rel_path,
+ profile = "folder-depth-1"
+)
+```
+
+This yields candidate aggregations:
+
+```
+R
+man
+tests
+vignettes
+```
+
+```{r filtering}
+groups %>%
+ filter(structural_group %in% c("R", "man"))
+```
+
+The `R` structural group contains source code files, while the `man` structural group contains documentation files generated from the package source.
+
+These aggregations are not necessarily Record Sets. They are structural aggregations derived from observed filesystem organisation. In the terminology used throughout this vignette, they are examples of aggregation metadata that may help identify potentially informative objects.
+
+The usefulness of these aggregations comes from the fact that software projects often organise related resources into stable folders. The folder structure therefore provides evidence about how resources are grouped and used together.
+
+A curator, archivist, or researcher might later create Record Sets such as:
+
+- Source code
+
+- Documentation
+
+- Tests
+
+- Vignettes
+
+However, these Record Sets are not determined by the filesystem structure alone.
+
+For example, a curator may decide to create a Record Set containing all source code files from a single package such as **dplyr**. Alternatively, they may create a larger Record Set containing source files from multiple packages that form part of the **tidyverse** ecosystem. Such a Record Set could include functions from dplyr, tidyr, purrr, and related packages because these resources are frequently used together and share a common analytical context.
+
+The structural aggregations derived by `derive_structural_groups()` therefore provide evidence that may support Record Set construction, but they do not define Record Sets themselves. The final Record Set remains a curatorial, analytical, or archival assertion that depends on purpose, context, and intended use.
+
+## Example 2: A ZIP archive
+
+The ZIP archive contains the same observed resources as the original folder, but packaged into a different storage container.
+
+The structural aggregations remain stable because they are derived from the observed relative paths rather than from the physical storage mechanism.
+
+This illustrates an important principle in fscontext: aggregation metadata can often survive storage transformations.
+
+```{r zip}
+zip_snapshot <- scan_storage(
+ system.file(
+ "testdata/minimal_R_folder.zip",
+ package = "fscontext"
+ )
+)
+```
+
+The observations differ in storage form but preserve the same structural relationships.
+
+Structural aggregation metadata therefore remains stable across storage representations.
+
+This demonstrates that candidate aggregations may survive transformations between storage environments.
+
+```{r derivestructuralgroups}
+zip_groups <- derive_structural_groups(
+ zip_snapshot$rel_path,
+ profile = "folder-depth-1"
+)
+```
+
+```{r filterstructuralgroups}
+zip_groups %>%
+ filter(structural_group %in% c("R", "man"))
+```
+
+## Example 3: A WACZ package
+
+WACZ (Web Archive Collection Zipped) is an open archival packaging format designed for the preservation, exchange, and analysis of web archives. A WACZ package combines archived web content, indexes, metadata, and access information into a single portable file.
+
+The format builds upon established web archiving standards, including the WARC (Web ARChive) format standard published by the World Wide Web Consortium (W3C):
+
+The WACZ specification:
+
+A WACZ package may therefore contain both archived web content and the metadata required to discover, index, and replay that content.
+
+Observe a WACZ package:
+
+```{r observewacz}
+wacz <- scan_storage(
+ system.file(
+ "testdata/fscontext_020.wacz",
+ package = "fscontext"
+ )
+)
+
+derive_structural_groups(
+ wacz$rel_path,
+ profile = "wacz"
+)
+```
+
+These groupings correspond to the structural organisation of a WACZ package.
+
+Again, they are not Record Sets.
+
+They are aggregation metadata that help users identify potentially informative objects within the archive.
+
+## Structural Aggregations and Informative Objects
+
+Following Pomerantz (2015), observed resources are not necessarily informative in isolation. Whether an object becomes informative depends on the context in which it is used and interpreted. Pomerantz defines metadata as statements about an object, and such statements may increase the capacity of an object to become informative.
+
+Structural aggregation metadata can increase the potential informativeness of observations by exposing recurring organisational patterns. One common pattern is the use of folder structures to organise related resources. In projects that follow relatively disciplined document or record organisation practices, folders often reflect recurring workflows or functional groupings.
+
+For example, a standard CRAN package organises source code into an `R` folder, documentation into `man`, and long-form tutorials into `vignettes`. These folder structures provide useful contextual information about the resources they contain. As a result, the contents of `vignettes` may be more informative when considered together than when viewed as isolated files.
+
+In fscontext:
+
+```
+Filesystem observation
+ ↓
+Aggregation metadata
+ ↓
+Potentially informative object
+ ↓
+Record Set candidate
+```
+
+Structural aggregations are therefore a lightweight analytical layer between observation and interpretation.
+
+## Structural Aggregations and RiC
+
+In Records in Contexts (RiC), Record Sets are curated aggregations of Records.
+
+`derive_structural_groups()` does not create Record Sets. Instead, it creates aggregation metadata derived from observed locators. These aggregations may provide evidence that supports Record Set construction.
+
+For example, in an R software development context, the presence of an `R` folder strongly suggests a grouping of source code files, while a `man` folder suggests a grouping of documentation resources.
+
+In this example,
+
+```
+R/
+```
+
+may support the creation of a source-code Record Set.
+
+In a web archiving context,
+
+```
+pages/
+```
+
+may support the creation of a Record Set describing archived web pages, while
+
+```
+archive/
+```
+
+may support the creation of a Record Set containing archived web resources stored in WARC files.
+
+To give a less engineering context, a wedding photographer may organise photographs on a NAS drive as\
+
+```
+Photos/
+ Personal/
+ family/
+ vacation/
+ Clients/
+ 2026/
+ Smith_wedding/
+ raw/
+ processed/
+ Doe_wedding/
+ raw/
+ processed/
+ 2025/
+ Miller_wedding
+ raw/
+ processed/
+```
+
+This organisation illustrates how structural aggregations emerge from folder hierarchies. The photographer may wish to maintain a clear separation between personal and professional photographs, suggesting different aggregation boundaries for different purposes.
+
+Using a profile such as `folder-depth-3`, the structural aggregation metadata might identify groupings such as:
+
+```
+
+Clients/2026/Smith_wedding
+Clients/2026/Doe_wedding
+Clients/2025/Miller_wedding
+
+```
+
+These aggregations are not yet Record Sets. They are candidate groupings derived from observed organisational structure.
+
+Depending on the intended purpose, a curator or archivist could create Record Sets at several different levels. For example:
+
+- all photographs relating to a particular wedding;
+
+- all weddings photographed during a given year;
+
+- all professional client work;
+
+- all processed photographs;
+
+- all photographs, both personal and professional, relating to a particular family.
+
+The folder structure therefore provides evidence about how resources were organised and used, but it does not determine the final Record Sets. The resulting Record Sets remain contextual assertions that depend on the needs of creators, users, curators, or archivists.
+
+## From Folder Hierarchies to Record Sets
+
+Historically, archives were organised around physical storage constraints. Documents were placed into folders, folders into boxes, and boxes onto shelves. Archival description standards such as ISAD(G) emerged in a world where the physical arrangement of records was often inseparable from their intellectual organisation.
+
+Many contemporary backup and archiving solutions for personal computers continue this tradition. File synchronisation systems, backup software, and operating-system tools such as Time Machine on macOS or File History on Windows preserve folder hierarchies because these structures allow users to restore files, projects, and working environments quickly and efficiently.
+
+These approaches remain extremely useful. They are fast, practical, and well suited to recovering a lost laptop, restoring a working directory, or retrieving an accidentally deleted document.
+
+However, physical and filesystem structures also impose limitations. The organisation of a laptop, network drive, or backup archive reflects the needs of a particular moment in time, a particular user, and a particular technical environment. As projects evolve, resources become distributed across multiple devices, repositories, cloud services, and institutions.
+
+Records in Contexts (RiC) offers a different perspective. Rather than treating a folder hierarchy as the primary organising principle, RiC allows Record Sets to be created according to contextual relationships that may transcend physical storage locations.
+
+For example, a researcher might wish to create a Record Set containing all materials relating to a project, regardless of whether those materials originated on an old laptop, a current workstation, a cloud storage service, or a web archive. Similarly, an organisation might wish to create annual project summaries that combine reports, correspondence, datasets, presentations, and archived web resources stored across many different systems.
+
+Other examples include:
+
+- all documents associated with a particular grant application;
+
+- all correspondence relating to a specific research collaboration;
+
+- all versions of a manuscript created across multiple devices;
+
+- all photographs associated with a particular client;
+
+- all source code and documentation contributing to a software release;
+
+- all web resources archived during a specific investigation or event.
+
+In these situations, folder structures remain valuable because they provide evidence about how resources were originally organised and used. Structural aggregations derived from observed locators can therefore serve as useful aggregation metadata and provide candidate groupings for further analysis.
+
+The purpose of `derive_structural_groups()` is not to replace archival description or to automatically create Record Sets. Instead, it provides a lightweight analytical layer that helps users move from observed storage structures toward more meaningful contextual aggregations. In this sense, structural aggregations act as a bridge between filesystem observations and the richer contextual relationships supported by RiC.
+
+Structural aggregation is only one possible way to derive aggregation metadata from observations. Future versions of fscontext may introduce additional analytical grouping strategies based on other observable characteristics of digital resources.
+
+For example, files may be grouped according to temporal characteristics, such as creation time (`birth_time`) or modification time (`mtime`), allowing users to identify activity periods, project phases, or clusters of work. Similarly, resources may be grouped according to authorship, ownership, repository affiliation, storage context, or other observable provenance indicators.
+
+Conceptually, these approaches follow the same pattern:
+
+```
+Filesystem observation
+ ↓
+Aggregation metadata
+ ↓
+Potentially informative object
+ ↓
+Record Set candidate
+ ↓
+ Human curation
+ ↓
+ Record Set
+```
+
+The difference lies in the evidence used to derive the aggregation. `derive_structural_groups()` uses filesystem organisation as evidence. Other analytical grouping methods may use temporal, authorship, provenance, repository, or content-related observations.
+
+Together, these analytical layers can help users identify potentially informative objects and candidate Record Sets before undertaking more formal contextual reconstruction, semantic stabilisation, or archival description.
+
+## References
+
+Pomerantz, J. (2015). *Metadata*. MIT Press.