Skip to contents

Create reference libraries from source files and OpenSpecy objects. When output_dir is supplied, build_lib() runs the official end-to-end workflow and returns raw, processed, medoid, model, and assessment artifacts in one object. Supporting functions remain available for advanced composition.

Usage

build_lib(
  x,
  recipes = .default_lib_recipes(),
  range = "full",
  res = 6,
  id_col = "sample_name",
  exclude_ids = NULL,
  dedupe = TRUE,
  metadata_lookups = NULL,
  material_hierarchy = NULL,
  metadata_name_lookup = lib_metadata_name_lookup(),
  clean_metadata_values = NULL,
  convert_intensity = TRUE,
  restrict_range_args = NULL,
  signal_noise = TRUE,
  assess = FALSE,
  prune = NULL,
  progress = TRUE,
  workflow_data = NULL,
  output_dir = NULL,
  previous_library_dir = "system",
  reuse = TRUE,
  remove_other = TRUE,
  seed = 123,
  holdout = 0.1,
  ...
)

rebuild_lib_artifacts(
  x,
  output_dir,
  previous_library_dir = "system",
  reuse = TRUE,
  seed = 123,
  holdout = 0.1,
  progress = TRUE
)

make_lib_lookup_template(x, columns, add = NULL, path = NULL)

join_lib_metadata(
  x,
  lookup,
  by,
  require_complete = FALSE,
  return = c("object", "table", "report"),
  suffixes = c(".x", ".y")
)

join_material_hierarchy(
  x,
  hierarchy,
  key_col = "material",
  levels = c("material", "material_class", "material_type"),
  output_names = levels,
  require_complete = FALSE,
  return = c("object", "table", "report")
)

dedupe_spec(
  x,
  id_col = "sample_name",
  exclude_ids = NULL,
  duplicate = c("first", "remove_all", "none"),
  scale = 100,
  algo = "md5"
)

prune_lib(
  x,
  class_col = "material_class",
  type_col = "spectrum_type",
  material_type_col = "material_type",
  id_col = "sample_name",
  min_n = 10,
  exclude = c(2200, 2420),
  return = c("object", "ids", "report"),
  progress = TRUE
)

reduce_lib(
  x,
  group_cols = "material_class",
  id_col = "sample_name",
  k = 50,
  min_n = k,
  return = c("object", "ids"),
  progress = FALSE,
  ...
)

build_model_lib(
  x,
  class_col = "material_class",
  type_col = "spectrum_type",
  min_n = 10,
  alpha = 0.1,
  seed = 123,
  grouped = TRUE,
  weights = TRUE,
  make_relative = TRUE,
  method = c("logistic_regression", "random_forest"),
  ...
)

train_spec_model(
  x,
  class_col = "material_class",
  type_col = "spectrum_type",
  min_n = 10,
  alpha = 0.1,
  seed = 123,
  grouped = TRUE,
  weights = TRUE,
  make_relative = TRUE,
  method = c("logistic_regression", "random_forest"),
  ...
)

assess_lib(
  x,
  class_col = NULL,
  id_col = "sample_name",
  nearest = !is.null(class_col)
)

Arguments

x

an OpenSpecy or Specs object for metadata helpers. For build_lib(), one OpenSpecy, a nonempty list containing only OpenSpecy objects, or a nonempty character vector of file paths. The official workflow requires x; it does not guess source-library locations. Each RDS path may store either one OpenSpecy or a list of them; other paths are read with read_any(). Large same-axis source lists are prepared in bulk to avoid repeated legacy object coercion. For rebuild_lib_artifacts(), a completed four-part build object, its RDS path, or a build/output directory containing completed libraries checkpoints.

recipes

named list of process_spec() argument lists or functions. Names become names of the returned libraries.

range, res

wavenumber range and resolution passed to c_spec() when build_lib() combines multiple sources.

id_col

metadata column used as the spectrum identifier.

exclude_ids

identifiers to remove before returning a library.

dedupe

logical; whether to generate stable IDs and remove duplicated spectra in build_lib().

metadata_lookups

a lookup table, csv path, or list of lookup tables and paths. A lookup may instead be supplied as list(lookup = x, by = key) to use an explicit key (including a named metadata-to-lookup key mapping). fill_only = TRUE preserves existing nonblank metadata values while filling gaps from the lookup. The older fallback_by lookup field is deprecated; canonical metadata keys are now filled from reviewed internal aliases before any external join. If non-NULL, each is joined with join_lib_metadata(). Automatic ordinary lookups use the single shared column that has overlapping values and unique lookup keys. Lookups with no usable shared key are skipped with a message; lookups with multiple usable shared keys are considered ambiguous and stop. Lookup values that share non-key metadata column names are coalesced back into those columns, with non-missing lookup values taking precedence.

material_hierarchy

hierarchy table or csv path used when non-NULL. It is joined with join_material_hierarchy() using the default "material" metadata key.

metadata_name_lookup

a data.frame or data.table with canonical_name, source_name, and optional regex columns. The default is returned by lib_metadata_name_lookup(); use NULL to clean names without coalescing aliases.

clean_metadata_values

logical or NULL; whether build_lib() should lowercase, trim, ASCII-normalize, and normalize blank/unknown character metadata values before joining. NULL enables cleaning for the official output workflow and disables it for composable in-memory builds. Invalid byte sequences are always converted to UTF-8.

convert_intensity

logical; whether to infer reflectance, transmittance, or absorbance units from each source and convert known non-absorbance spectra with adj_intens() before merging. Object attribute intensity_unit is authoritative when supplied; otherwise metadata column intensity_units is evaluated per spectrum.

restrict_range_args

optional named list of arguments passed to restrict_range() after unit conversion and source merging. Supplying the list triggers restriction; make_rel = FALSE is used unless explicitly overridden.

signal_noise

logical; whether to append the default sig_noise() result as metadata column sn.

assess

logical; whether to run assess_spec() on each output library and append assessment summaries to its metadata.

prune

NULL, or a named list mapping recipe names to argument lists for prune_lib(). Selected recipes are pruned independently after processing and assessment. NULL preserves unpruned outputs.

progress

logical; whether build_lib() reports named processing stages and elapsed time, or reduce_lib() reports group sizes and correlation/PAM timings.

workflow_data

optional directory containing the curated reference CSV tables. If NULL, the directory is discovered beside the calling script or under the current working directory.

output_dir

NULL for the composable in-memory return, or a directory that triggers the complete checkpointed workflow. The official workflow requires an explicit output directory.

previous_library_dir

directory containing the seven legacy artifacts used for complete old/new assessment, "system", or NULL to skip external comparison. The end-to-end workflow defaults to "system" and retrieves missing artifacts with get_lib().

reuse

logical; whether manifest-compatible completed checkpoints and versioned release files may be reused.

remove_other

logical; in the official end-to-end workflow, whether spectra with blank spectrum_identity or the unresolved literal "other" class are removed before quality control, medoid selection, and model fitting. Reviewed broad "other plastic" and "other material" categories remain in the reference libraries and are eligible for prune_lib()'s constrained nearest-class reassignment. Removed and reviewed rows remain visible in assessments$cleanup$summary. Source-only composable builds do not apply the official filter.

seed

random seed used before model training.

holdout

fraction of stable spectrum groups reserved for assessment.

columns

metadata columns to deduplicate into a template.

add

blank columns to add to a template.

path

optional csv path. If NULL, template helpers return a data.table.

lookup

a data.frame, data.table, or csv file path used as a metadata lookup table.

by

named character vector mapping metadata columns to lookup columns, or an unnamed character vector when the names are the same in both tables.

require_complete

logical; if TRUE, incomplete joins fail.

return

whether to return an updated OpenSpecy object, joined table, report list, or selected ids depending on the helper.

suffixes

suffixes used when joined metadata and lookup tables share non-key column names.

hierarchy

a data.frame, data.table, or csv file path with hierarchical material metadata.

key_col

metadata column containing material labels to match.

levels

hierarchy columns ordered from most-specific to most-general.

output_names

names to use for hierarchy columns added to metadata.

duplicate

how duplicated generated identifiers should be handled.

scale

numeric multiplier used before hashing intensity values.

algo

hash algorithm passed to digest().

class_col, type_col

metadata columns used for model labels.

material_type_col

metadata column used to require plastic candidates for "other plastic" and to update the type after generic-class reassignment. "other material" candidates are restricted by class_col to "organic matter" or "mineral"; "other" may match any established class in the spectral pool.

min_n

For prune_lib(), the minimum spectra required for a resolved class within one spectrum type: smaller groups are reassigned as a whole to their most-correlated eligible class, or removed only when no valid destination exists; groups exactly at the threshold are retained whole, and larger groups are never reduced below it. For reduce_lib(), groups with min_n or fewer spectra are kept whole. Model trainers fit only classes meeting the threshold.

exclude

numeric length-two wavenumber interval excluded from pruning correlations.

group_cols

metadata columns defining groups for reduction.

k

maximum representatives to keep for groups larger than min_n.

alpha

alpha value passed to glmnet().

grouped

logical; whether multinomial coefficients use grouped penalties.

weights

logical; whether to use inverse class-frequency weights for logistic regression or inverse-frequency case sampling for random forest.

make_relative

logical; whether to normalize model inputs with make_rel().

method

classifier to train: "logistic_regression" (the backward-compatible default) or "random_forest". The latter requires the suggested ranger package.

nearest

logical; if TRUE, assess_lib() compares each spectrum with its highest-correlation neighbor and reports the fraction where that neighbor has the same class_col value.

...

further arguments passed to the underlying operation.

Value

Each library returned by build_lib() includes a spectrum_identity_cleanup_report attribute listing changed original and normalized identities with their counts. In composable mode, build_lib() returns a named list of OpenSpecy libraries. Its end-to-end mode returns one list containing libraries, medoids, models, and assessments. Official libraries and medoids are nested by recipe and then ftir, raman, or nir. Models are nested by algorithm, recipe, and spectrum type. FTIR and Raman medoids/models use 800–3200 while the NIR interval is derived from finite coverage within 4000–12000. Assessments use five ordered process lists: cleanup, ref_lib, medoid, model, and functionality, with no more than ten nonempty review tables in total. Every spectrum receives a derived library_name: populated organization first, otherwise user_name. The cleanup summary includes source-library counts at each major stage and identifies the first stage and reason whenever an entire source library is dropped. Accuracy tables contain overall aggregate metrics only in long form, confusion tables retain misidentifications only, and model error-mode tables compare accuracy percentages with and without each automated-test flag. Review tables reject columns with more than 10 percent missing values. Row-level tests, split manifests, model-training diagnostics, and release manifests remain hash-addressed evidence attributes rather than additional review leaves. Each in-memory training model contains one tests data.table and a one-spectrum fill object. Versioned release directories instead store global build and model diagnostics only in assessments.rds; library, medoid, and model files retain only runtime data, scientific attributes, and prediction state. The companion reference_library_build.rds is a lightweight release index. join_lib_metadata(), join_material_hierarchy(), dedupe_spec(), prune_lib(), and reduce_lib() return an updated spectral object unless return requests a table, report, or ids. make_lib_lookup_template() returns a data.table unless path is supplied, in which case it writes the csv and invisibly returns the table. train_spec_model() and build_model_lib() return a list suitable for AI classification with match_spec() and one tidy tests table instead of separate accuracy/confusion summaries. It also contains typed lambda_metrics and support tables. Random-forest results also contain out-of-bag metrics and feature importance. assess_lib() returns a data.table summary.

Details

build_lib() combines sources over their full wavenumber range, optionally adds ordinary and hierarchical metadata, removes requested identifiers, optionally generates stable source-stage duplicate IDs, and applies named processing recipes. Source-stage IDs follow the reference library's legacy hash recipe: each source spectrum is trimmed with manage_na(type = "remove"), conformed at resolution 8, smoothed, and hashed from the resulting wavenumber/intensity vectors before later merging and range restriction. The older 100–4000 cm-1 hash is kept in sample_name_old when id_col = "sample_name" so exclude_ids can remove both current and legacy curated bad IDs. Metadata column names are first converted to lowercase underscore names and known aliases are coalesced using metadata_name_lookup; see lib_clean_metadata() for automatic and regular-expression matching. Metadata values can optionally be normalized to lowercase trimmed character values before lookup joins. spectrum_identity is also reduced to a basename when it is a recognizable path, then trailing extensions supported by read_any() are removed. The same normalization is applied to exact lookup keys. Regex class rules belong in a separate table and can be applied afterward with predict_class_reference(). This keeps filenames usable as identities without treating file containers as part of a material name. By default, each source is also converted to absorbance before merging when its intensity units are known. A nonempty intensity_unit object attribute takes precedence over the per-spectrum intensity_units metadata column. Each recipe is either a named list of arguments passed to process_spec() or a function accepting one OpenSpecy object. An empty recipe returns an unprocessed copy. Signal-to-noise is added by default, and optional assess_spec() results are summarized into one metadata row per spectrum. Progress messages report named stages and elapsed time by default so long-running builds remain observable.

The official workflow requires explicit source paths and an output directory. When workflow_data is omitted, build_lib() looks for the curated helper tables under data/ beside the calling script, then under data/ or workflows/data/ in the current working directory. It writes each completed stage under output_dir/checkpoints, and promotes validated legacy-compatible files into a versioned release directory. With reuse = TRUE, a checkpoint is reused only when its manifest signature matches the source files, curated tables, relevant arguments, package version, and builder implementation. Full assessments use the complete candidate and legacy artifacts. Seeded ten-percent holdouts are allocated independently within each source across class/type strata after physical identifiers and exact spectral-content duplicates have been joined into stable groups. Candidate artifacts are assessed on candidate data and legacy artifacts on legacy data, so taxonomy changes do not require fuzzy cross-version class matching. Query identifiers are removed from full and references by group before matching to prevent transformed duplicate or physical-replicate self-matches. Each medoid artifact separately identifies its complete corresponding processed library, measuring the deployed medoid search directly without another split. Model assessments do not retrain models: each existing candidate or legacy model identifies its complete corresponding source dataset once. This measures the deployed artifact directly and keeps assessment generation bounded by prediction rather than model fitting. After derivative and baseline-removal processing, FTIR spectra whose 2200–2420 CO2-region maximum exceeds twice the 2420–2550 silent-region maximum are flattened and reassessed; failed postconditions are removed. High-tail checks use each spectrum's finite support; tails are trimmed and reassessed, failed corrections are removed, and spectra with running signal-to-noise below two are removed before pruning. Full artifacts are then partitioned into Raman (200–4000), FTIR (400–4000), and NIR (4000–12000) OpenSpecy objects. After full and medoid libraries are complete, metadata columns containing only missing values are removed and the remainder are stably ordered from the fewest to the most missing values. Spectra, metadata rows, identifiers, axes, and object attributes are unchanged. Official class completion temporarily assigns unresolved identities to "other". By default, spectra with a blank identity or that unresolved literal class are removed before quality control and retained in the other_review assessment table. Reviewed "other plastic" and "other material" rows stay in the reference library and enter prune_lib()'s nearest-class semisupervised pathway. Before pruning, prune_lib() reassigns generic classes by nearest same-technique correlation: "other" may use any established class, "other plastic" requires a plastic candidate, and "other material" requires "organic matter" or "mineral". The matched material type and a correlation audit are retained. Once those labels are resolved, each class/spectrum-type group with fewer than min_n spectra is reassigned as a whole to its most-correlated established class in the same technique pool and material type. A group is removed only when no eligible correlated destination exists. The report identifies its support, destination, class-level correlation, action, reason, and affected spectrum identifiers.

make_lib_lookup_template() creates a deduplicated table of metadata values from an OpenSpecy or Specs object. Users can fill the added columns in R or write the template to CSV and curate it elsewhere.

join_lib_metadata() left-joins lookup columns onto object metadata and reports unmatched metadata keys, duplicate lookup keys, and missing joined values. Joins are exact; clean or harmonize values before calling this helper.

join_material_hierarchy() joins user-defined hierarchical material metadata. The supplied levels are tried from most-specific to most-general so a material label can match any level in the hierarchy.

dedupe_spec() hashes the current spectra and wavenumber axis to create stable IDs and remove duplicated spectra. Process or conform spectra before this step when that should affect duplicate detection.

reduce_lib() uses PAM medoids to keep representative spectra within each metadata group. It uses OpenSpecy's optimized correlation routine on relative spectra whose missing values are temporarily replaced by each spectrum's finite mean. Groups of at most 3,000 spectra use exact PAM; oversized groups use five deterministic 1,000-spectrum PAM samples and keep the candidate set with the best full-group correlation-distance objective. Official medoids are then selected from the original object so their genuine missing values are preserved.

train_spec_model() trains either OpenSpecy's multinomial logistic regression model (method = "logistic_regression") or an experimental probability random forest (method = "random_forest"). Logistic regression uses inverse class weights and stratified cross-validation to select the lambda with the highest out-of-fold macro class accuracy. Random forest uses inverse-frequency balanced case sampling and out-of-bag predictions. This improves minority-class representation in each bootstrap sample without applying a second class-vote correction. Treat the stored out-of-bag metrics as fit diagnostics: balanced resampling can make them optimistic. The library workflow separately reports accuracy from applying each deployed model to its complete corresponding dataset. Missing training values are replaced with the finite mean at each wavenumber. The returned model carries the same training-mean filler so match_spec() can identify partially covered spectra. build_model_lib() is the backward-compatible wrapper used by older scripts; build_lib() calls the dedicated trainer for official models.

assess_lib() returns a compact summary of object validity, library size, class balance, and optionally nearest-neighbor class consistency.

rebuild_lib_artifacts() starts from completed type-keyed libraries and rebuilds only medoids, models, and assessments. Its input and output locations are explicit, and every downstream component is checkpointed so a compatible interrupted run can resume without repeating completed work. Checkpoint and release payloads carry SHA-256 hashes, and an existing versioned release path is accepted only when its payload is unchanged.

See also

match_spec() for deploying trained models and plotly_spec() for logistic coefficient overlays.

Author

Win Cowger

Examples

wavenumber <- seq(100, 6100, by = 100)
base_a <- dnorm(seq(-3, 3, length.out = length(wavenumber)))
base_b <- rev(cumsum(seq_along(wavenumber)))
spectra <- cbind(base_a, base_a + 0.1, base_a + 0.2,
                 base_b, base_b + 0.1, base_b + 0.2)
colnames(spectra) <- paste0("s", seq_len(ncol(spectra)))
mini <- as_OpenSpecy(
  wavenumber,
  spectra = spectra,
  metadata = data.table::data.table(
    sample_name = colnames(spectra),
    source = rep(c("A", "B"), each = 3),
    label = c("nylon 6", "polyamides", "nylon 6",
              "pet", "polyesters", "pet"),
    material_class = rep(c("polyamides", "polyesters"), each = 3),
    spectrum_type = rep("ftir", 6),
    intensity_units = rep("absorbance", 6)
  ),
  attributes = list(intensity_unit = "absorbance")
)

name_lookup <- lib_metadata_name_lookup()
name_lookup[name_lookup$canonical_name == "material_color", ]
#>    canonical_name    source_name  regex
#>            <char>         <char> <char>
#> 1: material_color material_color   <NA>
#> 2: material_color          color   <NA>
#> 3: material_color         colour   <NA>

make_lib_lookup_template(mini, columns = "source", add = "library_type")
#>    source library_type
#>    <char>       <char>
#> 1:      A         <NA>
#> 2:      B         <NA>

source_lookup <- data.frame(
  source = c("A", "B"),
  library_type = c("lab", "field"),
  material = c("nylon 6", "pet")
)
joined <- join_lib_metadata(mini, source_lookup, by = "source",
                            require_complete = TRUE)

hierarchy <- data.frame(
  material = c("nylon 6", "pet"),
  material_class = c("polyamides", "polyesters"),
  material_type = c("plastic", "plastic")
)
joined <- join_material_hierarchy(joined, hierarchy, key_col = "label",
                                  require_complete = TRUE)

deduped <- dedupe_spec(joined)
reduced <- reduce_lib(deduped, group_cols = "material_class",
                      k = 1, min_n = 1)
libs <- build_lib(
  mini,
  recipes = list(
    raw = list(),
    derivative = list(
      conform_spec = FALSE,
      smooth_intens = TRUE,
      smooth_intens_args = list(window = 15, derivative = 1),
      make_rel = TRUE
    )
  ),
  metadata_lookups = source_lookup,
  material_hierarchy = hierarchy,
  restrict_range_args = list(min = 100, max = 6000),
  assess = TRUE,
  dedupe = FALSE
)
#> build_lib [0.0s]: starting
#> build_lib [0.0s]: using one in-memory OpenSpecy source
#> build_lib [0.0s]: preparing 1 source object(s)
#> build_lib [0.0s]: preparing source object 1/1
#> build_lib [0.0s]: restricting the wavenumber range
#> build_lib [0.0s]: joining metadata lookup 1/1
#> build_lib [0.0s]: metadata lookup 1/1 complete (matched=6; unmatched=0)
#> build_lib [0.0s]: joining the material hierarchy
#> build_lib [0.0s]: material hierarchy complete (unmatched=0)
#> build_lib [0.0s]: processing recipe 1/2 (raw)
#> build_lib [0.0s]: calculating signal-to-noise (raw)
#> build_lib [0.0s]: assessing spectra (raw)
#> build_lib [0.0s]: processing recipe 2/2 (derivative)
#> build_lib [0.0s]: calculating signal-to-noise (derivative)
#> build_lib [0.0s]: assessing spectra (derivative)
#> build_lib [0.0s]: complete

model <- suppressWarnings(train_spec_model(
  joined, class_col = "material_class", type_col = NULL, min_n = 2,
  nlambda = 3
))
assess_lib(libs$raw, class_col = "material_class", nearest = FALSE)
#>             metric  value
#>             <char> <char>
#> 1: valid_OpenSpecy   TRUE
#> 2:         spectra      6
#> 3:     wavenumbers     60
#> 4:         classes      2
#> 5:  smallest_class      3