| build_lib | R Documentation |
Create reference libraries from source files and OpenSpecy objects.
When output_dir is supplied, build_lib() runs the official
end-to-end workflow and returns raw,
processed, medoid, model, and assessment artifacts in one object. Supporting
functions remain available for advanced composition.
build_lib(
x,
recipes = .default_lib_recipes(),
range = "full",
res = 6,
id_col = "sample_name",
exclude_ids = NULL,
dedupe = TRUE,
metadata_lookups = NULL,
material_hierarchy = NULL,
metadata_name_lookup = lib_metadata_name_lookup(),
clean_metadata_values = NULL,
convert_intensity = TRUE,
restrict_range_args = NULL,
signal_noise = TRUE,
assess = FALSE,
prune = NULL,
progress = TRUE,
workflow_data = NULL,
output_dir = NULL,
previous_library_dir = "system",
reuse = TRUE,
remove_other = TRUE,
seed = 123,
holdout = 0.1,
...
)
rebuild_lib_artifacts(
x,
output_dir,
previous_library_dir = "system",
reuse = TRUE,
seed = 123,
holdout = 0.1,
progress = TRUE
)
make_lib_lookup_template(x, columns, add = NULL, path = NULL)
join_lib_metadata(
x,
lookup,
by,
require_complete = FALSE,
return = c("object", "table", "report"),
suffixes = c(".x", ".y")
)
join_material_hierarchy(
x,
hierarchy,
key_col = "material",
levels = c("material", "material_class", "material_type"),
output_names = levels,
require_complete = FALSE,
return = c("object", "table", "report")
)
dedupe_spec(
x,
id_col = "sample_name",
exclude_ids = NULL,
duplicate = c("first", "remove_all", "none"),
scale = 100,
algo = "md5"
)
prune_lib(
x,
class_col = "material_class",
type_col = "spectrum_type",
material_type_col = "material_type",
id_col = "sample_name",
min_n = 10,
cross_class = FALSE,
cross_class_threshold = 0.9,
exclude = c(2200, 2420),
return = c("object", "ids", "report"),
progress = TRUE
)
reduce_lib(
x,
group_cols = "material_class",
id_col = "sample_name",
k = 50,
min_n = k,
return = c("object", "ids"),
progress = FALSE,
...
)
build_model_lib(
x,
class_col = "material_class",
type_col = "spectrum_type",
min_n = 10,
alpha = 0.1,
seed = 123,
grouped = TRUE,
weights = TRUE,
make_relative = TRUE,
method = c("logistic_regression", "random_forest"),
...
)
train_spec_model(
x,
class_col = "material_class",
type_col = "spectrum_type",
min_n = 10,
alpha = 0.1,
seed = 123,
grouped = TRUE,
weights = TRUE,
make_relative = TRUE,
method = c("logistic_regression", "random_forest"),
...
)
assess_lib(
x,
class_col = NULL,
id_col = "sample_name",
nearest = !is.null(class_col)
)
x |
an |
recipes |
named list of |
range, res |
wavenumber range and resolution passed to |
id_col |
metadata column used as the spectrum identifier. |
exclude_ids |
identifiers to remove before returning a library. |
dedupe |
logical; whether to generate stable IDs and remove duplicated
spectra in |
metadata_lookups |
a lookup table, csv path, or list of lookup tables and
paths. A lookup may instead be supplied as |
material_hierarchy |
hierarchy table or csv path used when
non- |
metadata_name_lookup |
a data.frame or data.table with
|
clean_metadata_values |
logical or |
convert_intensity |
logical; whether to infer reflectance,
transmittance, or absorbance units from each source and convert known
non-absorbance spectra with |
restrict_range_args |
optional named list of arguments passed to
|
signal_noise |
logical; whether to append the default
|
assess |
logical; whether to run |
prune |
|
progress |
logical; whether |
workflow_data |
optional directory containing the curated reference CSV
tables. If |
output_dir |
|
previous_library_dir |
directory containing the seven legacy artifacts
used for complete old/new assessment, |
reuse |
logical; whether manifest-compatible completed checkpoints and versioned release files may be reused. |
remove_other |
logical; in the official end-to-end workflow, whether
spectra with blank |
seed |
random seed used before model training. |
holdout |
fraction of stable spectrum groups reserved for assessment. |
columns |
metadata columns to deduplicate into a template. |
add |
blank columns to add to a template. |
path |
optional csv path. If |
lookup |
a data.frame, data.table, or csv file path used as a metadata lookup table. |
by |
named character vector mapping metadata columns to lookup columns, or an unnamed character vector when the names are the same in both tables. |
require_complete |
logical; if |
return |
whether to return an updated |
suffixes |
suffixes used when joined metadata and lookup tables share non-key column names. |
hierarchy |
a data.frame, data.table, or csv file path with hierarchical material metadata. |
key_col |
metadata column containing material labels to match. |
levels |
hierarchy columns ordered from most-specific to most-general. |
output_names |
names to use for hierarchy columns added to metadata. |
duplicate |
how duplicated generated identifiers should be handled. |
scale |
numeric multiplier used before hashing intensity values. |
algo |
hash algorithm passed to |
class_col, type_col |
metadata columns used for model labels. |
material_type_col |
metadata column used to require plastic candidates
for |
min_n |
For |
cross_class |
logical; whether |
cross_class_threshold |
numeric Pearson-correlation threshold in
|
exclude |
numeric length-two wavenumber interval excluded from pruning correlations. |
group_cols |
metadata columns defining groups for reduction. |
k |
maximum representatives to keep for groups larger than
|
alpha |
alpha value passed to |
grouped |
logical; whether multinomial coefficients use grouped penalties. |
weights |
logical; whether to use inverse class-frequency weights for logistic regression or inverse-frequency case sampling for random forest. |
make_relative |
logical; whether to normalize model inputs with
|
method |
classifier to train: |
nearest |
logical; if |
... |
further arguments passed to the underlying operation. |
build_lib() combines sources over their full wavenumber range,
optionally adds ordinary and hierarchical metadata, removes requested
identifiers, optionally generates stable source-stage duplicate IDs, and
applies named processing recipes. Source-stage IDs follow the reference
library's legacy hash recipe: each source spectrum is trimmed with
manage_na(type = "remove"), conformed at resolution 8,
smoothed, and hashed from the resulting wavenumber/intensity vectors before
later merging and range restriction. The older 100–4000 cm-1 hash is kept in
sample_name_old when id_col = "sample_name" so
exclude_ids can remove both current and legacy curated bad IDs.
Metadata column names are first converted to lowercase
underscore names and known aliases are coalesced using
metadata_name_lookup; see lib_clean_metadata() for
automatic and regular-expression matching. Metadata values can optionally be
normalized to lowercase trimmed character values before lookup joins.
spectrum_identity is also reduced to a basename when it is a
recognizable path, then trailing extensions supported by
read_any() are removed. The same normalization is applied to
exact lookup keys. Regex class rules belong in a separate table and can be
applied afterward with predict_class_reference().
This keeps filenames usable as identities without treating file containers
as part of a material name. By default, each source is also
converted to absorbance before merging when its intensity units are known.
A nonempty intensity_unit object attribute takes precedence over the
per-spectrum intensity_units metadata column. Each recipe is either a
named list of arguments passed to process_spec() or a function
accepting one OpenSpecy object. An empty recipe returns an unprocessed
copy. Signal-to-noise is added by default, and optional
assess_spec() results are summarized into one metadata row per
spectrum.
Progress messages report named stages and elapsed time by default so
long-running builds remain observable.
The official workflow requires explicit source paths and an output
directory. When workflow_data is omitted, build_lib() looks
for the curated helper tables under data/ beside the calling script,
then under data/ or workflows/data/ in the current working
directory. It writes each completed stage under
output_dir/checkpoints, and promotes validated legacy-compatible files
into a versioned release directory. With reuse = TRUE, a checkpoint
is reused only when its manifest signature matches the source files, curated
tables, relevant arguments, package version, and builder implementation.
Full assessments use the complete candidate and legacy artifacts. Seeded
ten-percent holdouts are allocated independently within each source across
class/type strata after physical identifiers and exact spectral-content
duplicates have been joined into stable groups. Candidate artifacts are assessed on candidate data and
legacy artifacts on legacy data, so taxonomy changes do not require fuzzy
cross-version class matching. Query identifiers are removed from full and
references by group before matching to prevent transformed duplicate or
physical-replicate self-matches. Each medoid artifact separately identifies
its complete corresponding processed library, measuring the deployed medoid
search directly without another split. Model assessments do not retrain models:
each existing candidate or legacy model identifies its complete corresponding
source dataset once. This measures the deployed artifact directly and keeps
assessment generation bounded by prediction rather than model fitting.
After derivative and baseline-removal processing, FTIR spectra whose
2200–2420 CO2-region maximum exceeds twice the 2420–2550 silent-region
maximum are flattened and reassessed; failed postconditions are removed.
High-tail checks use each spectrum's finite support;
tails are trimmed and reassessed, failed corrections are removed, and
spectra with running signal-to-noise below two are removed before pruning.
Full artifacts are then partitioned into Raman (200–4000), FTIR
(400–4000), and NIR (4000–12000) OpenSpecy objects.
After pruning and class reassignment, the official workflow derives
material_form from the curated form-regex CSV and all atomic metadata
values, then joins evidence-backed common_use by final material class.
Existing form values take precedence when uniquely standardized; conflicting
form categories remain missing and are audited. Common use may be
"consumer", "industrial", "mixed", or missing;
quantitative mass shares are retained when available, while sourced
qualitative proposals remain explicit in the review table.
After full and medoid libraries are complete, metadata columns containing
only missing values are removed, except that these two standardized fields
are retained, and the remainder are stably ordered from the fewest to the
most missing values. Spectra, metadata rows, identifiers, axes, and object
attributes are unchanged.
Official class completion temporarily assigns unresolved identities to
"other". By default, spectra with a blank identity or that unresolved
literal class are removed before quality control and retained in the
other_review assessment table. Reviewed "other plastic" and
"other material" rows stay in the reference library and enter
prune_lib()'s nearest-class semisupervised pathway.
When requested, prune_lib() first resolves high-correlation conflicts
between different reviewed classes within each source library. It repeatedly
removes the spectrum with the most active wrong-class neighbors, recalculates
after each round, and removes both endpoints of an adjacent maximum-score
tie. It then resolves between-library conflicts by the number of distinct
independent libraries supporting the opposing class. Generic and
unclassified labels do not participate. Official builds repeat closure on
the rounded typed and model-range views before deriving medoids, and retain
excluded spectra plus conflict provenance in
quarantined_spectra.rds. Pruning then reassigns generic classes by nearest same-technique
correlation: "other" may use any established class,
"other plastic" requires a plastic candidate, and
"other material" requires "organic matter" or
"mineral". The matched material type and a correlation audit are
retained. Once those labels are resolved, each class/spectrum-type group with
fewer than min_n spectra is reassigned as a whole to its
most-correlated established class in the same technique pool and material
type. A group is removed only when no eligible correlated destination
exists. The report identifies its support, destination, class-level
correlation, action, reason, and affected spectrum identifiers.
make_lib_lookup_template() creates a deduplicated table of metadata
values from an OpenSpecy or Specs object. Users can fill the
added columns in R or write the template to CSV and curate it elsewhere.
join_lib_metadata() left-joins lookup columns onto object metadata and
reports unmatched metadata keys, duplicate lookup keys, and missing joined
values. Joins are exact; clean or harmonize values before calling this helper.
join_material_hierarchy() joins user-defined hierarchical material
metadata. The supplied levels are tried from most-specific to
most-general so a material label can match any level in the hierarchy.
dedupe_spec() hashes the current spectra and wavenumber axis to create
stable IDs and remove duplicated spectra. Process or conform spectra before
this step when that should affect duplicate detection.
reduce_lib() uses PAM medoids to keep representative spectra within
each metadata group. It uses OpenSpecy's optimized correlation routine on
relative spectra whose missing values are temporarily replaced by each
spectrum's finite mean. Groups of at most 3,000 spectra use exact PAM;
oversized groups use five deterministic 1,000-spectrum PAM samples and keep
the candidate set with the best full-group correlation-distance objective.
Official medoids are then selected from the original object so their genuine
missing values are preserved.
train_spec_model() trains either OpenSpecy's multinomial logistic
regression model (method = "logistic_regression") or an experimental
probability random forest (method = "random_forest"). Logistic
regression uses inverse class weights and stratified cross-validation to
select the lambda with the highest out-of-fold macro class accuracy. Random
forest uses inverse-frequency balanced case sampling and out-of-bag
predictions. This improves minority-class representation in each bootstrap
sample without applying a second class-vote correction.
Treat the stored out-of-bag metrics as fit diagnostics: balanced resampling
can make them optimistic. The library workflow separately reports accuracy
from applying each deployed model to its complete corresponding dataset.
Missing training values are replaced with the finite mean at each
wavenumber. The returned model carries the same training-mean filler so
match_spec() can identify partially covered spectra.
build_model_lib() is the backward-compatible wrapper used by older
scripts; build_lib() calls the dedicated trainer for official models.
assess_lib() returns a compact summary of object validity, library
size, class balance, and optionally nearest-neighbor class consistency.
rebuild_lib_artifacts() starts from completed type-keyed libraries
and rebuilds only medoids, models, and assessments. Its input and output
locations are explicit, and every downstream component is checkpointed so a
compatible interrupted run can resume without repeating completed work.
Checkpoint and release payloads carry SHA-256 hashes, and an existing
versioned release path is accepted only when its payload is unchanged.
Each library returned by build_lib() includes a
spectrum_identity_cleanup_report attribute listing changed original
and normalized identities with their counts.
In composable mode, build_lib() returns a named list of
OpenSpecy libraries. Its end-to-end mode returns one list containing
libraries, medoids, models, and assessments.
Official libraries and medoids are nested by recipe and then
ftir, raman, or nir. Models are nested by algorithm,
recipe, and spectrum type. FTIR and Raman medoids/models use
800–3200 while the NIR interval is derived from finite coverage within
4000–12000. Assessments use five ordered process lists:
cleanup, ref_lib, medoid, model, and
functionality, with no more than ten nonempty review tables in total.
Every spectrum receives a derived library_name: populated
organization first, otherwise user_name. The cleanup summary
includes source-library counts at each major stage and identifies the first
stage and reason whenever an entire source library is dropped. Accuracy
tables contain overall aggregate metrics only in long form, confusion tables
retain misidentifications only, and model error-mode tables compare accuracy
percentages with and without each automated-test flag. Review tables reject
columns with more than 10 percent missing values. Row-level tests, split manifests, model-training
diagnostics, and release manifests remain hash-addressed evidence attributes
rather than additional review leaves. Each in-memory training model contains
one tests data.table and a one-spectrum fill object. Versioned
release directories instead store global build and model diagnostics only in
assessments.rds; library, medoid, and model files retain only runtime
data, scientific attributes, and prediction state. The companion
reference_library_build.rds is a lightweight release index. The
companion quarantined_spectra.rds stores valid OpenSpecy
objects by recipe/type, a long conflict table, and build provenance for
spectra excluded by cross-class closure.
join_lib_metadata(), join_material_hierarchy(),
dedupe_spec(), prune_lib(), and reduce_lib() return an updated spectral
object unless return requests a table, report, or ids.
make_lib_lookup_template() returns a data.table unless path is
supplied, in which case it writes the csv and invisibly returns the table.
train_spec_model() and build_model_lib() return a list suitable
for AI classification with
match_spec() and one tidy tests table instead of
separate accuracy/confusion summaries. It also contains typed
lambda_metrics and support tables. Random-forest results also
contain out-of-bag metrics and feature importance. assess_lib()
returns a data.table summary.
Win Cowger
match_spec() for deploying trained models and
plotly_spec() for logistic coefficient overlays.
wavenumber <- seq(100, 6100, by = 100)
base_a <- dnorm(seq(-3, 3, length.out = length(wavenumber)))
base_b <- rev(cumsum(seq_along(wavenumber)))
spectra <- cbind(base_a, base_a + 0.1, base_a + 0.2,
base_b, base_b + 0.1, base_b + 0.2)
colnames(spectra) <- paste0("s", seq_len(ncol(spectra)))
mini <- as_OpenSpecy(
wavenumber,
spectra = spectra,
metadata = data.table::data.table(
sample_name = colnames(spectra),
source = rep(c("A", "B"), each = 3),
label = c("nylon 6", "polyamides", "nylon 6",
"pet", "polyesters", "pet"),
material_class = rep(c("polyamides", "polyesters"), each = 3),
spectrum_type = rep("ftir", 6),
intensity_units = rep("absorbance", 6)
),
attributes = list(intensity_unit = "absorbance")
)
name_lookup <- lib_metadata_name_lookup()
name_lookup[name_lookup$canonical_name == "material_color", ]
make_lib_lookup_template(mini, columns = "source", add = "library_type")
source_lookup <- data.frame(
source = c("A", "B"),
library_type = c("lab", "field"),
material = c("nylon 6", "pet")
)
joined <- join_lib_metadata(mini, source_lookup, by = "source",
require_complete = TRUE)
hierarchy <- data.frame(
material = c("nylon 6", "pet"),
material_class = c("polyamides", "polyesters"),
material_type = c("plastic", "plastic")
)
joined <- join_material_hierarchy(joined, hierarchy, key_col = "label",
require_complete = TRUE)
deduped <- dedupe_spec(joined)
reduced <- reduce_lib(deduped, group_cols = "material_class",
k = 1, min_n = 1)
libs <- build_lib(
mini,
recipes = list(
raw = list(),
derivative = list(
conform_spec = FALSE,
smooth_intens = TRUE,
smooth_intens_args = list(window = 15, derivative = 1),
make_rel = TRUE
)
),
metadata_lookups = source_lookup,
material_hierarchy = hierarchy,
restrict_range_args = list(min = 100, max = 6000),
assess = TRUE,
dedupe = FALSE
)
model <- suppressWarnings(train_spec_model(
joined, class_col = "material_class", type_col = NULL, min_n = 2,
nlambda = 3
))
assess_lib(libs$raw, class_col = "material_class", nearest = FALSE)
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.