Create reference libraries from source files and OpenSpecy objects.
When output_dir is supplied, build_lib() runs the official
end-to-end workflow and returns raw,
processed, medoid, model, and assessment artifacts in one object. Supporting
functions remain available for advanced composition.
Usage
build_lib(
x,
recipes = .default_lib_recipes(),
range = "full",
res = 6,
id_col = "sample_name",
exclude_ids = NULL,
dedupe = TRUE,
metadata_lookups = NULL,
material_hierarchy = NULL,
metadata_name_lookup = lib_metadata_name_lookup(),
clean_metadata_values = NULL,
convert_intensity = TRUE,
restrict_range_args = NULL,
signal_noise = TRUE,
assess = FALSE,
prune = NULL,
progress = TRUE,
workflow_data = NULL,
output_dir = NULL,
previous_library_dir = "system",
reuse = TRUE,
remove_other = TRUE,
seed = 123,
holdout = 0.1,
...
)
rebuild_lib_artifacts(
x,
output_dir,
previous_library_dir = "system",
reuse = TRUE,
seed = 123,
holdout = 0.1,
progress = TRUE
)
make_lib_lookup_template(x, columns, add = NULL, path = NULL)
join_lib_metadata(
x,
lookup,
by,
require_complete = FALSE,
return = c("object", "table", "report"),
suffixes = c(".x", ".y")
)
join_material_hierarchy(
x,
hierarchy,
key_col = "material",
levels = c("material", "material_class", "material_type"),
output_names = levels,
require_complete = FALSE,
return = c("object", "table", "report")
)
dedupe_spec(
x,
id_col = "sample_name",
exclude_ids = NULL,
duplicate = c("first", "remove_all", "none"),
scale = 100,
algo = "md5"
)
prune_lib(
x,
class_col = "material_class",
type_col = "spectrum_type",
material_type_col = "material_type",
id_col = "sample_name",
min_n = 10,
exclude = c(2200, 2420),
return = c("object", "ids", "report"),
progress = TRUE
)
reduce_lib(
x,
group_cols = "material_class",
id_col = "sample_name",
k = 50,
min_n = k,
return = c("object", "ids"),
progress = FALSE,
...
)
build_model_lib(
x,
class_col = "material_class",
type_col = "spectrum_type",
min_n = 10,
alpha = 0.1,
seed = 123,
grouped = TRUE,
weights = TRUE,
make_relative = TRUE,
method = c("logistic_regression", "random_forest"),
...
)
train_spec_model(
x,
class_col = "material_class",
type_col = "spectrum_type",
min_n = 10,
alpha = 0.1,
seed = 123,
grouped = TRUE,
weights = TRUE,
make_relative = TRUE,
method = c("logistic_regression", "random_forest"),
...
)
assess_lib(
x,
class_col = NULL,
id_col = "sample_name",
nearest = !is.null(class_col)
)Arguments
- x
an
OpenSpecyorSpecsobject for metadata helpers. Forbuild_lib(), oneOpenSpecy, a nonempty list containing onlyOpenSpecyobjects, or a nonempty character vector of file paths. The official workflow requiresx; it does not guess source-library locations. Each RDS path may store either oneOpenSpecyor a list of them; other paths are read withread_any(). Large same-axis source lists are prepared in bulk to avoid repeated legacy object coercion. Forrebuild_lib_artifacts(), a completed four-part build object, its RDS path, or a build/output directory containing completedlibrariescheckpoints.- recipes
named list of
process_spec()argument lists or functions. Names become names of the returned libraries.- range, res
wavenumber range and resolution passed to
c_spec()whenbuild_lib()combines multiple sources.- id_col
metadata column used as the spectrum identifier.
- exclude_ids
identifiers to remove before returning a library.
- dedupe
logical; whether to generate stable IDs and remove duplicated spectra in
build_lib().- metadata_lookups
a lookup table, csv path, or list of lookup tables and paths. A lookup may instead be supplied as
list(lookup = x, by = key)to use an explicit key (including a named metadata-to-lookup key mapping).fill_only = TRUEpreserves existing nonblank metadata values while filling gaps from the lookup. The olderfallback_bylookup field is deprecated; canonical metadata keys are now filled from reviewed internal aliases before any external join. If non-NULL, each is joined withjoin_lib_metadata(). Automatic ordinary lookups use the single shared column that has overlapping values and unique lookup keys. Lookups with no usable shared key are skipped with a message; lookups with multiple usable shared keys are considered ambiguous and stop. Lookup values that share non-key metadata column names are coalesced back into those columns, with non-missing lookup values taking precedence.- material_hierarchy
hierarchy table or csv path used when non-
NULL. It is joined withjoin_material_hierarchy()using the default"material"metadata key.- metadata_name_lookup
a data.frame or data.table with
canonical_name,source_name, and optionalregexcolumns. The default is returned bylib_metadata_name_lookup(); useNULLto clean names without coalescing aliases.- clean_metadata_values
logical or
NULL; whetherbuild_lib()should lowercase, trim, ASCII-normalize, and normalize blank/unknown character metadata values before joining.NULLenables cleaning for the official output workflow and disables it for composable in-memory builds. Invalid byte sequences are always converted to UTF-8.- convert_intensity
logical; whether to infer reflectance, transmittance, or absorbance units from each source and convert known non-absorbance spectra with
adj_intens()before merging. Object attributeintensity_unitis authoritative when supplied; otherwise metadata columnintensity_unitsis evaluated per spectrum.- restrict_range_args
optional named list of arguments passed to
restrict_range()after unit conversion and source merging. Supplying the list triggers restriction;make_rel = FALSEis used unless explicitly overridden.- signal_noise
logical; whether to append the default
sig_noise()result as metadata columnsn.- assess
logical; whether to run
assess_spec()on each output library and append assessment summaries to its metadata.- prune
NULL, or a named list mapping recipe names to argument lists forprune_lib(). Selected recipes are pruned independently after processing and assessment.NULLpreserves unpruned outputs.- progress
logical; whether
build_lib()reports named processing stages and elapsed time, orreduce_lib()reports group sizes and correlation/PAM timings.- workflow_data
optional directory containing the curated reference CSV tables. If
NULL, the directory is discovered beside the calling script or under the current working directory.- output_dir
NULLfor the composable in-memory return, or a directory that triggers the complete checkpointed workflow. The official workflow requires an explicit output directory.- previous_library_dir
directory containing the seven legacy artifacts used for complete old/new assessment,
"system", orNULLto skip external comparison. The end-to-end workflow defaults to"system"and retrieves missing artifacts withget_lib().- reuse
logical; whether manifest-compatible completed checkpoints and versioned release files may be reused.
- remove_other
logical; in the official end-to-end workflow, whether spectra with blank
spectrum_identityor the unresolved literal"other"class are removed before quality control, medoid selection, and model fitting. Reviewed broad"other plastic"and"other material"categories remain in the reference libraries and are eligible forprune_lib()'s constrained nearest-class reassignment. Removed and reviewed rows remain visible inassessments$cleanup$summary. Source-only composable builds do not apply the official filter.- seed
random seed used before model training.
- holdout
fraction of stable spectrum groups reserved for assessment.
- columns
metadata columns to deduplicate into a template.
- add
blank columns to add to a template.
- path
optional csv path. If
NULL, template helpers return a data.table.- lookup
a data.frame, data.table, or csv file path used as a metadata lookup table.
- by
named character vector mapping metadata columns to lookup columns, or an unnamed character vector when the names are the same in both tables.
- require_complete
logical; if
TRUE, incomplete joins fail.- return
whether to return an updated
OpenSpecyobject, joined table, report list, or selected ids depending on the helper.- suffixes
suffixes used when joined metadata and lookup tables share non-key column names.
- hierarchy
a data.frame, data.table, or csv file path with hierarchical material metadata.
- key_col
metadata column containing material labels to match.
- levels
hierarchy columns ordered from most-specific to most-general.
- output_names
names to use for hierarchy columns added to metadata.
- duplicate
how duplicated generated identifiers should be handled.
- scale
numeric multiplier used before hashing intensity values.
- algo
hash algorithm passed to
digest().- class_col, type_col
metadata columns used for model labels.
- material_type_col
metadata column used to require plastic candidates for
"other plastic"and to update the type after generic-class reassignment."other material"candidates are restricted byclass_colto"organic matter"or"mineral";"other"may match any established class in the spectral pool.- min_n
For
prune_lib(), the minimum spectra required for a resolved class within one spectrum type: smaller groups are reassigned as a whole to their most-correlated eligible class, or removed only when no valid destination exists; groups exactly at the threshold are retained whole, and larger groups are never reduced below it. Forreduce_lib(), groups withmin_nor fewer spectra are kept whole. Model trainers fit only classes meeting the threshold.- exclude
numeric length-two wavenumber interval excluded from pruning correlations.
- group_cols
metadata columns defining groups for reduction.
- k
maximum representatives to keep for groups larger than
min_n.- alpha
alpha value passed to
glmnet().- grouped
logical; whether multinomial coefficients use grouped penalties.
- weights
logical; whether to use inverse class-frequency weights for logistic regression or inverse-frequency case sampling for random forest.
- make_relative
logical; whether to normalize model inputs with
make_rel().- method
classifier to train:
"logistic_regression"(the backward-compatible default) or"random_forest". The latter requires the suggested ranger package.- nearest
logical; if
TRUE,assess_lib()compares each spectrum with its highest-correlation neighbor and reports the fraction where that neighbor has the sameclass_colvalue.- ...
further arguments passed to the underlying operation.
Value
Each library returned by build_lib() includes a
spectrum_identity_cleanup_report attribute listing changed original
and normalized identities with their counts.
In composable mode, build_lib() returns a named list of
OpenSpecy libraries. Its end-to-end mode returns one list containing
libraries, medoids, models, and assessments.
Official libraries and medoids are nested by recipe and then
ftir, raman, or nir. Models are nested by algorithm,
recipe, and spectrum type. FTIR and Raman medoids/models use
800–3200 while the NIR interval is derived from finite coverage within
4000–12000. Assessments use five ordered process lists:
cleanup, ref_lib, medoid, model, and
functionality, with no more than ten nonempty review tables in total.
Every spectrum receives a derived library_name: populated
organization first, otherwise user_name. The cleanup summary
includes source-library counts at each major stage and identifies the first
stage and reason whenever an entire source library is dropped. Accuracy
tables contain overall aggregate metrics only in long form, confusion tables
retain misidentifications only, and model error-mode tables compare accuracy
percentages with and without each automated-test flag. Review tables reject
columns with more than 10 percent missing values. Row-level tests, split manifests, model-training
diagnostics, and release manifests remain hash-addressed evidence attributes
rather than additional review leaves. Each in-memory training model contains
one tests data.table and a one-spectrum fill object. Versioned
release directories instead store global build and model diagnostics only in
assessments.rds; library, medoid, and model files retain only runtime
data, scientific attributes, and prediction state. The companion
reference_library_build.rds is a lightweight release index.
join_lib_metadata(), join_material_hierarchy(),
dedupe_spec(), prune_lib(), and reduce_lib() return an updated spectral
object unless return requests a table, report, or ids.
make_lib_lookup_template() returns a data.table unless path is
supplied, in which case it writes the csv and invisibly returns the table.
train_spec_model() and build_model_lib() return a list suitable
for AI classification with
match_spec() and one tidy tests table instead of
separate accuracy/confusion summaries. It also contains typed
lambda_metrics and support tables. Random-forest results also
contain out-of-bag metrics and feature importance. assess_lib()
returns a data.table summary.
Details
build_lib() combines sources over their full wavenumber range,
optionally adds ordinary and hierarchical metadata, removes requested
identifiers, optionally generates stable source-stage duplicate IDs, and
applies named processing recipes. Source-stage IDs follow the reference
library's legacy hash recipe: each source spectrum is trimmed with
manage_na(type = "remove"), conformed at resolution 8,
smoothed, and hashed from the resulting wavenumber/intensity vectors before
later merging and range restriction. The older 100–4000 cm-1 hash is kept in
sample_name_old when id_col = "sample_name" so
exclude_ids can remove both current and legacy curated bad IDs.
Metadata column names are first converted to lowercase
underscore names and known aliases are coalesced using
metadata_name_lookup; see lib_clean_metadata() for
automatic and regular-expression matching. Metadata values can optionally be
normalized to lowercase trimmed character values before lookup joins.
spectrum_identity is also reduced to a basename when it is a
recognizable path, then trailing extensions supported by
read_any() are removed. The same normalization is applied to
exact lookup keys. Regex class rules belong in a separate table and can be
applied afterward with predict_class_reference().
This keeps filenames usable as identities without treating file containers
as part of a material name. By default, each source is also
converted to absorbance before merging when its intensity units are known.
A nonempty intensity_unit object attribute takes precedence over the
per-spectrum intensity_units metadata column. Each recipe is either a
named list of arguments passed to process_spec() or a function
accepting one OpenSpecy object. An empty recipe returns an unprocessed
copy. Signal-to-noise is added by default, and optional
assess_spec() results are summarized into one metadata row per
spectrum.
Progress messages report named stages and elapsed time by default so
long-running builds remain observable.
The official workflow requires explicit source paths and an output
directory. When workflow_data is omitted, build_lib() looks
for the curated helper tables under data/ beside the calling script,
then under data/ or workflows/data/ in the current working
directory. It writes each completed stage under
output_dir/checkpoints, and promotes validated legacy-compatible files
into a versioned release directory. With reuse = TRUE, a checkpoint
is reused only when its manifest signature matches the source files, curated
tables, relevant arguments, package version, and builder implementation.
Full assessments use the complete candidate and legacy artifacts. Seeded
ten-percent holdouts are allocated independently within each source across
class/type strata after physical identifiers and exact spectral-content
duplicates have been joined into stable groups. Candidate artifacts are assessed on candidate data and
legacy artifacts on legacy data, so taxonomy changes do not require fuzzy
cross-version class matching. Query identifiers are removed from full and
references by group before matching to prevent transformed duplicate or
physical-replicate self-matches. Each medoid artifact separately identifies
its complete corresponding processed library, measuring the deployed medoid
search directly without another split. Model assessments do not retrain models:
each existing candidate or legacy model identifies its complete corresponding
source dataset once. This measures the deployed artifact directly and keeps
assessment generation bounded by prediction rather than model fitting.
After derivative and baseline-removal processing, FTIR spectra whose
2200–2420 CO2-region maximum exceeds twice the 2420–2550 silent-region
maximum are flattened and reassessed; failed postconditions are removed.
High-tail checks use each spectrum's finite support;
tails are trimmed and reassessed, failed corrections are removed, and
spectra with running signal-to-noise below two are removed before pruning.
Full artifacts are then partitioned into Raman (200–4000), FTIR
(400–4000), and NIR (4000–12000) OpenSpecy objects.
After full and medoid libraries are complete, metadata columns containing
only missing values are removed and the remainder are stably ordered from
the fewest to the most missing values. Spectra, metadata rows, identifiers,
axes, and object attributes are unchanged.
Official class completion temporarily assigns unresolved identities to
"other". By default, spectra with a blank identity or that unresolved
literal class are removed before quality control and retained in the
other_review assessment table. Reviewed "other plastic" and
"other material" rows stay in the reference library and enter
prune_lib()'s nearest-class semisupervised pathway.
Before pruning, prune_lib() reassigns generic classes by nearest
same-technique correlation: "other" may use any established class,
"other plastic" requires a plastic candidate, and
"other material" requires "organic matter" or
"mineral". The matched material type and a correlation audit are
retained. Once those labels are resolved, each class/spectrum-type group with
fewer than min_n spectra is reassigned as a whole to its
most-correlated established class in the same technique pool and material
type. A group is removed only when no eligible correlated destination
exists. The report identifies its support, destination, class-level
correlation, action, reason, and affected spectrum identifiers.
make_lib_lookup_template() creates a deduplicated table of metadata
values from an OpenSpecy or Specs object. Users can fill the
added columns in R or write the template to CSV and curate it elsewhere.
join_lib_metadata() left-joins lookup columns onto object metadata and
reports unmatched metadata keys, duplicate lookup keys, and missing joined
values. Joins are exact; clean or harmonize values before calling this helper.
join_material_hierarchy() joins user-defined hierarchical material
metadata. The supplied levels are tried from most-specific to
most-general so a material label can match any level in the hierarchy.
dedupe_spec() hashes the current spectra and wavenumber axis to create
stable IDs and remove duplicated spectra. Process or conform spectra before
this step when that should affect duplicate detection.
reduce_lib() uses PAM medoids to keep representative spectra within
each metadata group. It uses OpenSpecy's optimized correlation routine on
relative spectra whose missing values are temporarily replaced by each
spectrum's finite mean. Groups of at most 3,000 spectra use exact PAM;
oversized groups use five deterministic 1,000-spectrum PAM samples and keep
the candidate set with the best full-group correlation-distance objective.
Official medoids are then selected from the original object so their genuine
missing values are preserved.
train_spec_model() trains either OpenSpecy's multinomial logistic
regression model (method = "logistic_regression") or an experimental
probability random forest (method = "random_forest"). Logistic
regression uses inverse class weights and stratified cross-validation to
select the lambda with the highest out-of-fold macro class accuracy. Random
forest uses inverse-frequency balanced case sampling and out-of-bag
predictions. This improves minority-class representation in each bootstrap
sample without applying a second class-vote correction.
Treat the stored out-of-bag metrics as fit diagnostics: balanced resampling
can make them optimistic. The library workflow separately reports accuracy
from applying each deployed model to its complete corresponding dataset.
Missing training values are replaced with the finite mean at each
wavenumber. The returned model carries the same training-mean filler so
match_spec() can identify partially covered spectra.
build_model_lib() is the backward-compatible wrapper used by older
scripts; build_lib() calls the dedicated trainer for official models.
assess_lib() returns a compact summary of object validity, library
size, class balance, and optionally nearest-neighbor class consistency.
rebuild_lib_artifacts() starts from completed type-keyed libraries
and rebuilds only medoids, models, and assessments. Its input and output
locations are explicit, and every downstream component is checkpointed so a
compatible interrupted run can resume without repeating completed work.
Checkpoint and release payloads carry SHA-256 hashes, and an existing
versioned release path is accepted only when its payload is unchanged.
See also
match_spec() for deploying trained models and
plotly_spec() for logistic coefficient overlays.
Examples
wavenumber <- seq(100, 6100, by = 100)
base_a <- dnorm(seq(-3, 3, length.out = length(wavenumber)))
base_b <- rev(cumsum(seq_along(wavenumber)))
spectra <- cbind(base_a, base_a + 0.1, base_a + 0.2,
base_b, base_b + 0.1, base_b + 0.2)
colnames(spectra) <- paste0("s", seq_len(ncol(spectra)))
mini <- as_OpenSpecy(
wavenumber,
spectra = spectra,
metadata = data.table::data.table(
sample_name = colnames(spectra),
source = rep(c("A", "B"), each = 3),
label = c("nylon 6", "polyamides", "nylon 6",
"pet", "polyesters", "pet"),
material_class = rep(c("polyamides", "polyesters"), each = 3),
spectrum_type = rep("ftir", 6),
intensity_units = rep("absorbance", 6)
),
attributes = list(intensity_unit = "absorbance")
)
name_lookup <- lib_metadata_name_lookup()
name_lookup[name_lookup$canonical_name == "material_color", ]
#> canonical_name source_name regex
#> <char> <char> <char>
#> 1: material_color material_color <NA>
#> 2: material_color color <NA>
#> 3: material_color colour <NA>
make_lib_lookup_template(mini, columns = "source", add = "library_type")
#> source library_type
#> <char> <char>
#> 1: A <NA>
#> 2: B <NA>
source_lookup <- data.frame(
source = c("A", "B"),
library_type = c("lab", "field"),
material = c("nylon 6", "pet")
)
joined <- join_lib_metadata(mini, source_lookup, by = "source",
require_complete = TRUE)
hierarchy <- data.frame(
material = c("nylon 6", "pet"),
material_class = c("polyamides", "polyesters"),
material_type = c("plastic", "plastic")
)
joined <- join_material_hierarchy(joined, hierarchy, key_col = "label",
require_complete = TRUE)
deduped <- dedupe_spec(joined)
reduced <- reduce_lib(deduped, group_cols = "material_class",
k = 1, min_n = 1)
libs <- build_lib(
mini,
recipes = list(
raw = list(),
derivative = list(
conform_spec = FALSE,
smooth_intens = TRUE,
smooth_intens_args = list(window = 15, derivative = 1),
make_rel = TRUE
)
),
metadata_lookups = source_lookup,
material_hierarchy = hierarchy,
restrict_range_args = list(min = 100, max = 6000),
assess = TRUE,
dedupe = FALSE
)
#> build_lib [0.0s]: starting
#> build_lib [0.0s]: using one in-memory OpenSpecy source
#> build_lib [0.0s]: preparing 1 source object(s)
#> build_lib [0.0s]: preparing source object 1/1
#> build_lib [0.0s]: restricting the wavenumber range
#> build_lib [0.0s]: joining metadata lookup 1/1
#> build_lib [0.0s]: metadata lookup 1/1 complete (matched=6; unmatched=0)
#> build_lib [0.0s]: joining the material hierarchy
#> build_lib [0.0s]: material hierarchy complete (unmatched=0)
#> build_lib [0.0s]: processing recipe 1/2 (raw)
#> build_lib [0.0s]: calculating signal-to-noise (raw)
#> build_lib [0.0s]: assessing spectra (raw)
#> build_lib [0.0s]: processing recipe 2/2 (derivative)
#> build_lib [0.0s]: calculating signal-to-noise (derivative)
#> build_lib [0.0s]: assessing spectra (derivative)
#> build_lib [0.0s]: complete
model <- suppressWarnings(train_spec_model(
joined, class_col = "material_class", type_col = NULL, min_n = 2,
nlambda = 3
))
assess_lib(libs$raw, class_col = "material_class", nearest = FALSE)
#> metric value
#> <char> <char>
#> 1: valid_OpenSpecy TRUE
#> 2: spectra 6
#> 3: wavenumbers 60
#> 4: classes 2
#> 5: smallest_class 3