Skip to content

Python API

The public surface: what from gatle_ignite import ... gives you. Everything else is internal and may move.

Most tasks need only BaseTrainer, base_config and to_device.

gatle_ignite.BaseTrainer

The training loop, so your task does not have to contain one.

Subclass it and implement prep_batch; engines, metrics, checkpointing, resume, logging, AMP and DDP are handled here and driven by the config. Point cfg.main_runner at the module holding it: the launcher constructs Trainer(local_rank, cfg) and calls fit().

Source code in src/gatle_ignite/trainer/base.py
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
class BaseTrainer:
    """The training loop, so your task does not have to contain one.

    Subclass it and implement `prep_batch`; engines, metrics, checkpointing, resume, logging,
    AMP and DDP are handled here and driven by the config. Point `cfg.main_runner` at the
    module holding it: the launcher constructs `Trainer(local_rank, cfg)` and calls `fit()`.
    """

    def __init__(self, local_rank, cfg):
        self.local_rank = local_rank
        self.cfg = cfg
        validate_config(cfg, specs=self.eval_specs())
        setup_seed(cfg.seed, cudnn_benchmark=cfg.cudnn_benchmark)
        self.dtype, self.autocast_enabled = resolve_amp_dtype(cfg.amp_dtype)
        self.device_type = autocast_device_type()
        # This rank's device, for a task that creates tensors (torch.randn in a sampler).
        self.device = idist.device()

        self.model = None
        # What build_model resolved cfg.compile to. None until it has run.
        self._compiling = None
        self.criterion = None
        self.optimizer = None
        self.scheduler = None
        self.scaler = None
        self.logger = None
        self.grad_clip_norm = None
        self.grad_clip_value = None
        self.accum_steps = 1
        self.transform = None
        self.dls = {}
        self.infos = {}
        self.engines = {}
        # engine key -> handlers run after that engine's run() returns, outside eval_context.
        self._post_eval = {}
        # The "latest" Checkpoint attach_checkpoints builds and fit() registers, or None.
        self._latest = None
        # Set when a score goes non-finite, so nothing after it in that epoch is saved.
        self._diverged = False
        # checkpoint key -> EarlyStop. setup() fills it before anything is checkpointed.
        self._early_stoppers = {}
        self._built = False

    # ---- Hooks: what a task actually writes ----

    def prep_batch(self, batch, split="train", **kwargs):
        """Map a raw batch to ``{"model_input": {...}, "targets": {...}}``.

        ``model_input`` is splatted into the model's forward; ``targets`` is what the loss and
        metrics select from by name. ``split`` is the engine's ``engine_type``: "train",
        "valid", "test", or an added engine's name. Read ``split``, not ``kwargs``, where a
        mistyped key is silently None.
        """
        raise NotImplementedError(
            f'{type(self).__name__} must implement prep_batch(batch, split="train") '
            'returning {"model_input": {...}, "targets": {...}}'
        )

    def eval_specs(self):
        """The eval engines, as data. Override to add one.

        def eval_specs(self):
            return super().eval_specs() + (EngineSpec.for_split("dictionary"),)
        """
        return EVAL_SPECS

    def train_spec(self):
        """The training engine, as data. Override to change what it reports.

        `total_loss=False` keeps only the per-component averages, for components that do not
        sum to anything (a GAN's loss_d + loss_g). `with_losses=False` drops all of them.
        """
        return TRAIN_SPEC

    @contextlib.contextmanager
    def eval_context(self, spec):
        """Model state for one engine's run. Released before anything is written.

        Swap in an EMA shadow or an SWA average: anything `spec`'s engine must see and the best
        checkpoint must NOT store. Use try/finally, or an exception mid-eval leaks the swap into
        the next epoch's training. Only engines run by `self.run_eval_engine(spec)` pass here.
        """
        yield

    def run_eval_engine(self, spec):
        """Run one eval engine inside its eval_context, then write its checkpoint outside.

        The post-eval handlers run after run() returns, because ignite fires COMPLETED from
        inside it, still inside the context.
        """
        engine = self.engines.get(spec.key)
        if engine is None or self.dls.get(spec.split) is None:
            return None
        with self.eval_context(spec):
            engine.run(self.dls[spec.split], max_epochs=1, epoch_length=spec.length(self.cfg))
        # Reached only on success: a failed eval must not select a checkpoint.
        for handler in self._post_eval.get(spec.key, ()):
            handler(engine)
        return engine.state.metrics

    def forward(self, model_input):
        return self.model(**model_input)

    def build_model(self):
        """Build the model. -> nn.Module, which the framework assigns to self.model.

        What it returns is THE model: the only one DDP-wrapped, handed to the optimizer, and
        checkpointed. A second model (a distillation teacher, a GAN discriminator) is built
        here too and held by you, not returned. Hold it in a list if `self` might ever become an
        nn.Module, or __setattr__ registers it into the optimizer and every checkpoint.
        """
        Model = import_entrypoint(self.cfg.model_name, "Model", field="model_name")
        model = Model(**(self.cfg.model_params or {}))
        # ENHANCEMENT (compile): on the unwrapped model, before the DDP wrap below.
        self._compiling = compile_built_model(model, self.cfg)
        if idist.get_world_size() == 1:
            # Not auto_model: with several GPUs visible it wraps in DataParallel instead.
            return model.to(self.device)
        return idist.auto_model(model, **dict(self.cfg.auto_model_params))

    def dict_metric_from_list(self, engine_type, spec, dict_metrics, info=None):
        """Build the metric dict for one engine.

        Default: dotted-path dispatch from the config, each module exposing
        get_metric(engine_type, info, **params). Override to build metrics imperatively.
        Keys are namespaced `{engine_type}/{key}`, so cfg.score_name reads "valid/acc".
        """
        if not spec:
            return dict_metrics
        # Any mapping: ml_collections turns an assigned dict into a ConfigDict, not a dict.
        if not hasattr(spec, "items"):
            raise ConfigError(
                f"{engine_type} metrics must be a mapping of name -> {{'cls_name', 'params'}}, "
                f"got {type(spec).__name__}.\n"
                f"  To build metrics imperatively instead, override dict_metric_from_list()."
            )

        for key, entry in spec.items():
            if "cls_name" not in entry:
                raise ConfigError(f"metric entry {key!r} is missing 'cls_name'")
            get_metric = import_entrypoint(
                entry["cls_name"], "get_metric", field=f"{engine_type}_metrics.{key}.cls_name"
            )
            dict_metrics[f"{engine_type}/{key}"] = get_metric(
                engine_type, info or {}, **entry.get("params", {})
            )
        return dict_metrics

    # ---- Steps: overridable, but the defaults suit most supervised tasks ----

    def loss_fn(self, y_pred, target, **kwargs):
        return self.criterion(y_pred, target, **kwargs)

    def backward(self, loss, step=True):
        """Backward for one batch, and the optimizer step when this batch ends a window.

        `loss` arrives ALREADY divided by the accumulation window. `step` is False for every
        batch but the last of a window: an override must honour it, or accumulation becomes a
        step on every micro-batch, which trains and is wrong.
        """
        model, optimizer = self.model, self.optimizer
        if self.scaler is not None:
            self.scaler.scale(loss).backward()
            if not step:
                return
            if self.grad_clip_value is not None or self.grad_clip_norm is not None:
                self.scaler.unscale_(optimizer)
                self._clip_grads(model)
            self.scaler.step(optimizer)
            self.scaler.update()
        else:
            loss.backward()
            if not step:
                return
            self._clip_grads(model)
            optimizer.step()

    def _clip_grads(self, model):
        if self.grad_clip_value is not None:
            torch.nn.utils.clip_grad_value_(model.parameters(), self.grad_clip_value)
        if self.grad_clip_norm is not None:
            torch.nn.utils.clip_grad_norm_(model.parameters(), self.grad_clip_norm)

    def train_step(self, engine, batch, split="train"):
        engine.state.batch = None
        engine.state.output = None
        self.model.train()

        window, is_first, is_step = self._accum_window(engine)
        if is_first:
            self.optimizer.zero_grad(set_to_none=True)

        # no_sync must wrap the forward too: DDP decides during forward whether to all-reduce.
        with self._maybe_no_sync(is_step):
            x = self.prep_batch(batch, split=split)
            with torch.autocast(
                device_type=self.device_type, dtype=self.dtype, enabled=self.autocast_enabled
            ):
                y_pred = self.forward(x["model_input"])
                loss, dict_losses = self.loss_fn(y_pred, x, iteration=engine.state.iteration)

            # Outside autocast. By the window, not accum_steps: an epoch's last one can be shorter.
            self.backward(loss / window if window > 1 else loss, step=is_step)

        # The unscaled loss is reported, so train/loss_avg is comparable across accum_steps.
        return {"y_pred": y_pred, "target": x, "losses": {"loss": loss, **dict_losses}}

    def _accum_window(self, engine):
        """(window size, is this batch first in the window, does it end the window).

        Windows never straddle an epoch, and a short final window still steps.
        """
        accum = self.accum_steps
        if accum == 1:
            return 1, True, True

        length = engine.state.epoch_length or 1
        index = (engine.state.iteration - 1) % length  # 0-based, within this epoch
        start = (index // accum) * accum
        window = min(accum, length - start)
        return window, index == start, index == start + window - 1

    def _maybe_no_sync(self, is_step):
        if is_step or not hasattr(self.model, "no_sync"):
            return contextlib.nullcontext()
        return self.model.no_sync()

    def eval_step(self, engine, batch, split="valid"):
        engine.state.batch = None
        engine.state.output = None
        self.model.eval()
        x = self.prep_batch(batch, split=split)
        with torch.no_grad():
            with torch.autocast(
                device_type=self.device_type,
                dtype=self.dtype,
                enabled=self.autocast_enabled,
            ):
                y_pred = self.forward(x["model_input"])
        return {"y_pred": y_pred, "target": x}

    # ---- Framework below. A task never needs to touch any of it. ----

    def setup(self):
        """Build everything. Idempotent, so fit() and eval paths can both call it."""
        if self._built:
            return self
        cfg = self.cfg

        if "run" not in cfg:
            from gatle_ignite.callbacks.logging import LoggingCallback

            self.logger = LoggingCallback(cfg)
            self.logger.start()

        self.build_dataloaders()
        self.model = self.build_model()
        # ENHANCEMENT (compile): an overridden build_model owns it, so check it did.
        check_build_model_override(self.model, getattr(self, "_compiling", None), cfg)
        self.build_criterion()
        self.build_optimizer()
        self.build_engines()
        self.build_scheduler()
        self.attach_metrics()
        # ENHANCEMENT (early stopping): built before attach_checkpoints saves their counters.
        # Not in inference mode, which never trains and may build a different set of engines.
        if "run" not in cfg:
            self._early_stoppers = build_early_stoppers(
                self.eval_specs(), self.engines, cfg, score_function
            )
        attach_checkpoints(self)
        self._built = True
        return self

    def fit(self):
        cfg = self.cfg
        self.setup()
        load_checkpoints(self)
        self.attach_runner()
        if self._latest is not None:
            # After the eval runners, so "latest" holds what that epoch's evaluation changed.
            trainer = self.engines["trainer"]
            trainer.add_event_handler(Events.EPOCH_COMPLETED(every=1), _clear_state)

            def save_latest(engine):
                if not self._diverged:
                    self._latest(engine)

            trainer.add_event_handler(Events.EPOCH_COMPLETED(every=1), save_latest)
        self.attach_progress()
        try:
            self.engines["trainer"].run(
                self.dls["train"], max_epochs=cfg.max_epochs, epoch_length=cfg.train_length
            )
        except BaseException:
            # Mark the run failed, or the loggers record a crash as a clean finish.
            self.teardown(failed=True)
            raise
        self.teardown()
        return self

    def evaluate(self, load_checkpoint=True):
        """Load the configured checkpoint, run the eval engines once, return the metrics.

        Needs inference mode, so it never writes a "best" checkpoint of what it evaluates.
        load_checkpoint=False evaluates the weights in memory.
        """
        cfg = self.cfg
        if "run" not in cfg:
            raise ConfigError(
                "evaluate() needs inference mode, or checkpoint handlers could overwrite what "
                "it evaluates.\n"
                "  Use `gatle-ignite eval --config=...`, or set cfg.run = True before building "
                "the trainer."
            )
        self.setup()
        if load_checkpoint:
            load_eval_checkpoint(self)
        results = {}
        for spec in self.eval_specs():
            metrics = self.run_eval_engine(spec)
            if metrics:
                results.update(metrics)

        for name, value in results.items():
            print(f"{name}: {value}")
        return results

    def teardown(self, failed=False):
        if self.logger is not None:
            self.logger.finish(failed=failed)

    def build_dataloaders(self):
        """One dataloader per declared engine, from each spec's `*_ds_name`.

        Override to add a loader belonging to no engine (a dictionary pass, a prototype
        source). Call `super()` first, or the declared splits are never built.
        """
        cfg = self.cfg
        self.transform = get_aug(cfg.aug_name, cfg.aug_params)

        self.dls["train"], self.infos["train"] = get_dataset(
            cfg.train_ds_name, cfg.train_ds_params, self.transform, field="train_ds_name"
        )
        for spec in self.eval_specs():
            split = prefix = spec.split
            # Two specs may share a split (`ds_prefix`); build its loader once.
            if split in self.dls:
                continue
            name = cfg.get(f"{prefix}_ds_name", None)
            if not name:
                self.dls[split], self.infos[split] = None, None
                continue

            # Under DDP, DistributedSampler pads an uneven eval set and scores the duplicates.
            # A get_ds that builds its own loader ignores this key; _warn_if_eval_padded checks.
            ds_params = dict(cfg.get(f"{prefix}_ds_params", {}))
            ds_params.setdefault("exact_sharding", True)

            self.dls[split], self.infos[split] = get_dataset(
                name,
                ds_params,
                self.transform,
                field=f"{prefix}_ds_name",
            )
            self._warn_if_eval_padded(split, self.dls[split])
        return self.dls, self.infos

    @staticmethod
    def _warn_if_eval_padded(split, dataloader):
        """Warn if a loader the framework could not shard exactly scores duplicates under DDP."""
        if idist.get_world_size() <= 1 or dataloader is None:
            return
        sampler = getattr(dataloader, "sampler", None)
        if isinstance(sampler, ExactDistributedSampler):
            return
        dataset = getattr(dataloader, "dataset", None)
        try:
            remainder = len(dataset) % idist.get_world_size()
        except TypeError:
            return  # an iterable dataset has no length to check
        if remainder:
            warnings.warn(
                f"{split} split: {len(dataset)} samples do not divide across "
                f"{idist.get_world_size()} ranks, so {idist.get_world_size() - remainder} "
                f"repeated sample(s) are scored.\n"
                f"  Build this split with gatle_ignite.build_dataloader to shard it exactly.",
                stacklevel=2,
            )

    def build_criterion(self):
        cfg = self.cfg
        Loss = import_entrypoint(cfg.criterion_name, "Loss", field="criterion_name")
        self.criterion = Loss(**(cfg.criterion_params or {})).to(idist.device())
        return self.criterion

    def build_optimizer(self):
        cfg = self.cfg
        get_optimizer = import_entrypoint(
            cfg.optimizer_name, "get_optimizer", field="optimizer_name"
        )
        self.optimizer = get_optimizer(self.model, **(cfg.optimizer_params or {}))

        trainable = sum(p.numel() for p in self.model.parameters() if p.requires_grad)
        print(f"MODEL: {trainable:,} trainable parameters")

        # Only fp16 needs a GradScaler: bf16 has fp32's exponent range.
        if self.dtype == torch.float16 and torch.cuda.is_available():
            self.scaler = torch.amp.GradScaler("cuda")

        self.grad_clip_norm = cfg.grad_clip_norm
        self.grad_clip_value = cfg.grad_clip_value
        self.accum_steps = self._resolve_accum_steps()
        return self.optimizer

    def _resolve_accum_steps(self):
        accum = self.cfg.get("accum_steps", 1)
        if not isinstance(accum, int) or isinstance(accum, bool) or accum < 1:
            raise ConfigError(f"accum_steps must be an integer >= 1, got {accum!r}")
        if accum > 1 and type(self).train_step is not BaseTrainer.train_step:
            # Accumulation lives inside train_step, so an override skips it unless it copies it.
            warnings.warn(
                f"{type(self).__name__} overrides train_step, so accum_steps={accum} has no "
                f"effect unless the override accumulates too.\n"
                f"  Use _accum_window() there, and pass its `step` through to backward().",
                stacklevel=3,
            )
        return accum

    def build_engines(self):
        specs = self.eval_specs()
        self._check_specs_are_distinct(specs)

        train_spec = self.train_spec()
        self.engines[train_spec.key] = self._engine_for(train_spec)
        for spec in specs:
            # Present even when unbuilt: callers ask `engines["tester"] is None`.
            self.engines[spec.key] = self._engine_for(spec) if self._wants(spec) else None
        return self.engines

    def _engine_for(self, spec):
        return Engine(partial(getattr(self, spec.step), split=spec.engine_type))

    def _wants(self, spec):
        if self.dls.get(spec.split) is None:
            return False
        # Built only if it will run. Inference mode runs every engine, whatever its cadence.
        return spec.cadence(self.cfg) > 0 or "run" in self.cfg

    @staticmethod
    def _check_specs_are_distinct(specs):
        """Two engines sharing a key or an engine_type would silently overwrite each other."""
        for attr in ("key", "engine_type"):
            seen = [getattr(s, attr) for s in specs]
            dupes = sorted({v for v in seen if seen.count(v) > 1})
            if dupes:
                raise ConfigError(
                    f"eval_specs() has more than one engine with {attr} {dupes!r}. "
                    f"Each engine needs its own {attr}."
                )

    def build_scheduler(self):
        # The whole engines dict, so a schedule can be driven by an engine the task declared.
        scheduler, engine_name, event_name = prep_scheduler(
            self.cfg.lr_scheduler,
            self.cfg,
            self.dls["train"],
            self.optimizer,
            self.engines,
        )
        self.scheduler = attach_scheduler(scheduler, engine_name, event_name, self.engines)
        return self.scheduler

    def attach_metrics(self):
        for spec in (self.train_spec(), *self.eval_specs()):
            engine = self.engines.get(spec.key)
            if engine is None:
                continue
            self.init_metrics(
                engine,
                spec.engine_type,
                self.cfg.get(spec.metrics_field, {}),
                info=self.infos.get(spec.split),
                with_losses=spec.with_losses,
                total_loss=spec.total_loss,
            )

    def init_metrics(
        self, engine, engine_type, metrics_cfg, info=None, with_losses=True, total_loss=True
    ):
        dict_metrics = self.dict_metric_from_list(engine_type, metrics_cfg, {}, info=info)

        if with_losses:

            def loss_display(key):
                return lambda output: output["losses"][key]

            keys = ["loss"] if total_loss else []
            keys += [f"loss_{c}" for c in getattr(self.criterion, "crit_keys", [])]
            for key in keys:
                dict_metrics[f"{engine_type}/{key}_avg"] = Average(
                    output_transform=loss_display(key)
                )

        for name, metric in dict_metrics.items():
            metric.attach(engine, name)
        return dict_metrics

    def extra_to_save(self):
        """Your own state to put in the checkpoint. Values need state_dict/load_state_dict.

        For an EMA shadow, a prototype bank or a running normaliser. One override covers save,
        resume and `gatle-ignite eval`, so all three agree on what a run holds.
        """
        return {}

    def attach_runner(self):
        trainer = self.engines["trainer"]
        # Stop at the first non-finite loss: the weights cannot recover from it.
        trainer.add_event_handler(
            Events.ITERATION_COMPLETED, TerminateOnNan(output_transform=_losses_on_every_rank)
        )

        if self.logger is not None:
            self.logger.on_train_epoch_end(trainer, self.optimizer)
            self.logger.on_train_iteration(trainer, self.model)

        for spec in self.eval_specs():
            self._attach_eval_runner(trainer, spec)

        # ENHANCEMENT (early stopping). Not in attach_checkpoints, which returns early on inference.
        for stopper in self._early_stoppers.values():
            self._post_eval.setdefault(stopper.spec.key, []).append(stopper)

        if self.logger is not None:
            self.logger.on_completion(trainer)

    def _attach_eval_runner(self, trainer, spec):
        engine = self.engines.get(spec.key)
        if engine is None:
            return

        # Its own method: inlined in the caller's loop, every handler would late-bind the last spec.
        @trainer.on(Events.EPOCH_COMPLETED(every=spec.cadence(self.cfg)))
        def _run(_trainer):
            _clear_state(_trainer)
            self.run_eval_engine(spec)

        # Beside the handler that runs the engine, so no engine can run without being logged.
        if self.logger is not None:
            self.logger.on_valid_epoch_end(trainer, engine)

    def attach_progress(self):
        """Progress reporting, memory cleanup, and DDP epoch wiring."""
        cfg = self.cfg
        trainer = self.engines["trainer"]
        engines = [e for e in self.engines.values() if e is not None]

        if "pbar" in cfg.logger_name and idist.get_rank() == 0:
            for engine in engines:
                ProgressBar(persist=False).attach(engine)
        else:
            every = max((cfg.train_length or len(self.dls["train"])) // 10, 1)
            trainer.add_event_handler(
                Events.ITERATION_COMPLETED(every=every), _progress_printer("TRAIN")
            )
            for spec in self.eval_specs():
                if self.engines.get(spec.key) is not None:
                    self.engines[spec.key].add_event_handler(
                        Events.ITERATION_COMPLETED(every=spec.progress_every),
                        _progress_printer(spec.progress_label),
                    )

        for engine in engines:
            engine.add_event_handler(Events.EPOCH_STARTED, _clear_state)
            engine.add_event_handler(Events.EPOCH_COMPLETED, empty_cuda_cache)

        sampler = getattr(self.dls["train"], "sampler", None)
        if hasattr(sampler, "set_epoch"):
            # DistributedSampler needs the epoch, or it replays the same shuffle every epoch.
            trainer.add_event_handler(
                Events.EPOCH_STARTED, lambda engine: sampler.set_epoch(engine.state.epoch)
            )

prep_batch

prep_batch(batch, split='train', **kwargs)

Map a raw batch to {"model_input": {...}, "targets": {...}}.

model_input is splatted into the model's forward; targets is what the loss and metrics select from by name. split is the engine's engine_type: "train", "valid", "test", or an added engine's name. Read split, not kwargs, where a mistyped key is silently None.

Source code in src/gatle_ignite/trainer/base.py
def prep_batch(self, batch, split="train", **kwargs):
    """Map a raw batch to ``{"model_input": {...}, "targets": {...}}``.

    ``model_input`` is splatted into the model's forward; ``targets`` is what the loss and
    metrics select from by name. ``split`` is the engine's ``engine_type``: "train",
    "valid", "test", or an added engine's name. Read ``split``, not ``kwargs``, where a
    mistyped key is silently None.
    """
    raise NotImplementedError(
        f'{type(self).__name__} must implement prep_batch(batch, split="train") '
        'returning {"model_input": {...}, "targets": {...}}'
    )

forward

forward(model_input)
Source code in src/gatle_ignite/trainer/base.py
def forward(self, model_input):
    return self.model(**model_input)

build_model

build_model()

Build the model. -> nn.Module, which the framework assigns to self.model.

What it returns is THE model: the only one DDP-wrapped, handed to the optimizer, and checkpointed. A second model (a distillation teacher, a GAN discriminator) is built here too and held by you, not returned. Hold it in a list if self might ever become an nn.Module, or setattr registers it into the optimizer and every checkpoint.

Source code in src/gatle_ignite/trainer/base.py
def build_model(self):
    """Build the model. -> nn.Module, which the framework assigns to self.model.

    What it returns is THE model: the only one DDP-wrapped, handed to the optimizer, and
    checkpointed. A second model (a distillation teacher, a GAN discriminator) is built
    here too and held by you, not returned. Hold it in a list if `self` might ever become an
    nn.Module, or __setattr__ registers it into the optimizer and every checkpoint.
    """
    Model = import_entrypoint(self.cfg.model_name, "Model", field="model_name")
    model = Model(**(self.cfg.model_params or {}))
    # ENHANCEMENT (compile): on the unwrapped model, before the DDP wrap below.
    self._compiling = compile_built_model(model, self.cfg)
    if idist.get_world_size() == 1:
        # Not auto_model: with several GPUs visible it wraps in DataParallel instead.
        return model.to(self.device)
    return idist.auto_model(model, **dict(self.cfg.auto_model_params))

dict_metric_from_list

dict_metric_from_list(engine_type, spec, dict_metrics, info=None)

Build the metric dict for one engine.

Default: dotted-path dispatch from the config, each module exposing get_metric(engine_type, info, **params). Override to build metrics imperatively. Keys are namespaced {engine_type}/{key}, so cfg.score_name reads "valid/acc".

Source code in src/gatle_ignite/trainer/base.py
def dict_metric_from_list(self, engine_type, spec, dict_metrics, info=None):
    """Build the metric dict for one engine.

    Default: dotted-path dispatch from the config, each module exposing
    get_metric(engine_type, info, **params). Override to build metrics imperatively.
    Keys are namespaced `{engine_type}/{key}`, so cfg.score_name reads "valid/acc".
    """
    if not spec:
        return dict_metrics
    # Any mapping: ml_collections turns an assigned dict into a ConfigDict, not a dict.
    if not hasattr(spec, "items"):
        raise ConfigError(
            f"{engine_type} metrics must be a mapping of name -> {{'cls_name', 'params'}}, "
            f"got {type(spec).__name__}.\n"
            f"  To build metrics imperatively instead, override dict_metric_from_list()."
        )

    for key, entry in spec.items():
        if "cls_name" not in entry:
            raise ConfigError(f"metric entry {key!r} is missing 'cls_name'")
        get_metric = import_entrypoint(
            entry["cls_name"], "get_metric", field=f"{engine_type}_metrics.{key}.cls_name"
        )
        dict_metrics[f"{engine_type}/{key}"] = get_metric(
            engine_type, info or {}, **entry.get("params", {})
        )
    return dict_metrics

setup

setup()

Build everything. Idempotent, so fit() and eval paths can both call it.

Source code in src/gatle_ignite/trainer/base.py
def setup(self):
    """Build everything. Idempotent, so fit() and eval paths can both call it."""
    if self._built:
        return self
    cfg = self.cfg

    if "run" not in cfg:
        from gatle_ignite.callbacks.logging import LoggingCallback

        self.logger = LoggingCallback(cfg)
        self.logger.start()

    self.build_dataloaders()
    self.model = self.build_model()
    # ENHANCEMENT (compile): an overridden build_model owns it, so check it did.
    check_build_model_override(self.model, getattr(self, "_compiling", None), cfg)
    self.build_criterion()
    self.build_optimizer()
    self.build_engines()
    self.build_scheduler()
    self.attach_metrics()
    # ENHANCEMENT (early stopping): built before attach_checkpoints saves their counters.
    # Not in inference mode, which never trains and may build a different set of engines.
    if "run" not in cfg:
        self._early_stoppers = build_early_stoppers(
            self.eval_specs(), self.engines, cfg, score_function
        )
    attach_checkpoints(self)
    self._built = True
    return self

fit

fit()
Source code in src/gatle_ignite/trainer/base.py
def fit(self):
    cfg = self.cfg
    self.setup()
    load_checkpoints(self)
    self.attach_runner()
    if self._latest is not None:
        # After the eval runners, so "latest" holds what that epoch's evaluation changed.
        trainer = self.engines["trainer"]
        trainer.add_event_handler(Events.EPOCH_COMPLETED(every=1), _clear_state)

        def save_latest(engine):
            if not self._diverged:
                self._latest(engine)

        trainer.add_event_handler(Events.EPOCH_COMPLETED(every=1), save_latest)
    self.attach_progress()
    try:
        self.engines["trainer"].run(
            self.dls["train"], max_epochs=cfg.max_epochs, epoch_length=cfg.train_length
        )
    except BaseException:
        # Mark the run failed, or the loggers record a crash as a clean finish.
        self.teardown(failed=True)
        raise
    self.teardown()
    return self

evaluate

evaluate(load_checkpoint=True)

Load the configured checkpoint, run the eval engines once, return the metrics.

Needs inference mode, so it never writes a "best" checkpoint of what it evaluates. load_checkpoint=False evaluates the weights in memory.

Source code in src/gatle_ignite/trainer/base.py
def evaluate(self, load_checkpoint=True):
    """Load the configured checkpoint, run the eval engines once, return the metrics.

    Needs inference mode, so it never writes a "best" checkpoint of what it evaluates.
    load_checkpoint=False evaluates the weights in memory.
    """
    cfg = self.cfg
    if "run" not in cfg:
        raise ConfigError(
            "evaluate() needs inference mode, or checkpoint handlers could overwrite what "
            "it evaluates.\n"
            "  Use `gatle-ignite eval --config=...`, or set cfg.run = True before building "
            "the trainer."
        )
    self.setup()
    if load_checkpoint:
        load_eval_checkpoint(self)
    results = {}
    for spec in self.eval_specs():
        metrics = self.run_eval_engine(spec)
        if metrics:
            results.update(metrics)

    for name, value in results.items():
        print(f"{name}: {value}")
    return results

eval_specs

eval_specs()

The eval engines, as data. Override to add one.

def eval_specs(self): return super().eval_specs() + (EngineSpec.for_split("dictionary"),)

Source code in src/gatle_ignite/trainer/base.py
def eval_specs(self):
    """The eval engines, as data. Override to add one.

    def eval_specs(self):
        return super().eval_specs() + (EngineSpec.for_split("dictionary"),)
    """
    return EVAL_SPECS

train_spec

train_spec()

The training engine, as data. Override to change what it reports.

total_loss=False keeps only the per-component averages, for components that do not sum to anything (a GAN's loss_d + loss_g). with_losses=False drops all of them.

Source code in src/gatle_ignite/trainer/base.py
def train_spec(self):
    """The training engine, as data. Override to change what it reports.

    `total_loss=False` keeps only the per-component averages, for components that do not
    sum to anything (a GAN's loss_d + loss_g). `with_losses=False` drops all of them.
    """
    return TRAIN_SPEC

eval_context

eval_context(spec)

Model state for one engine's run. Released before anything is written.

Swap in an EMA shadow or an SWA average: anything spec's engine must see and the best checkpoint must NOT store. Use try/finally, or an exception mid-eval leaks the swap into the next epoch's training. Only engines run by self.run_eval_engine(spec) pass here.

Source code in src/gatle_ignite/trainer/base.py
@contextlib.contextmanager
def eval_context(self, spec):
    """Model state for one engine's run. Released before anything is written.

    Swap in an EMA shadow or an SWA average: anything `spec`'s engine must see and the best
    checkpoint must NOT store. Use try/finally, or an exception mid-eval leaks the swap into
    the next epoch's training. Only engines run by `self.run_eval_engine(spec)` pass here.
    """
    yield

run_eval_engine

run_eval_engine(spec)

Run one eval engine inside its eval_context, then write its checkpoint outside.

The post-eval handlers run after run() returns, because ignite fires COMPLETED from inside it, still inside the context.

Source code in src/gatle_ignite/trainer/base.py
def run_eval_engine(self, spec):
    """Run one eval engine inside its eval_context, then write its checkpoint outside.

    The post-eval handlers run after run() returns, because ignite fires COMPLETED from
    inside it, still inside the context.
    """
    engine = self.engines.get(spec.key)
    if engine is None or self.dls.get(spec.split) is None:
        return None
    with self.eval_context(spec):
        engine.run(self.dls[spec.split], max_epochs=1, epoch_length=spec.length(self.cfg))
    # Reached only on success: a failed eval must not select a checkpoint.
    for handler in self._post_eval.get(spec.key, ()):
        handler(engine)
    return engine.state.metrics

backward

backward(loss, step=True)

Backward for one batch, and the optimizer step when this batch ends a window.

loss arrives ALREADY divided by the accumulation window. step is False for every batch but the last of a window: an override must honour it, or accumulation becomes a step on every micro-batch, which trains and is wrong.

Source code in src/gatle_ignite/trainer/base.py
def backward(self, loss, step=True):
    """Backward for one batch, and the optimizer step when this batch ends a window.

    `loss` arrives ALREADY divided by the accumulation window. `step` is False for every
    batch but the last of a window: an override must honour it, or accumulation becomes a
    step on every micro-batch, which trains and is wrong.
    """
    model, optimizer = self.model, self.optimizer
    if self.scaler is not None:
        self.scaler.scale(loss).backward()
        if not step:
            return
        if self.grad_clip_value is not None or self.grad_clip_norm is not None:
            self.scaler.unscale_(optimizer)
            self._clip_grads(model)
        self.scaler.step(optimizer)
        self.scaler.update()
    else:
        loss.backward()
        if not step:
            return
        self._clip_grads(model)
        optimizer.step()

extra_to_save

extra_to_save()

Your own state to put in the checkpoint. Values need state_dict/load_state_dict.

For an EMA shadow, a prototype bank or a running normaliser. One override covers save, resume and gatle-ignite eval, so all three agree on what a run holds.

Source code in src/gatle_ignite/trainer/base.py
def extra_to_save(self):
    """Your own state to put in the checkpoint. Values need state_dict/load_state_dict.

    For an EMA shadow, a prototype bank or a running normaliser. One override covers save,
    resume and `gatle-ignite eval`, so all three agree on what a run holds.
    """
    return {}

build_dataloaders

build_dataloaders()

One dataloader per declared engine, from each spec's *_ds_name.

Override to add a loader belonging to no engine (a dictionary pass, a prototype source). Call super() first, or the declared splits are never built.

Source code in src/gatle_ignite/trainer/base.py
def build_dataloaders(self):
    """One dataloader per declared engine, from each spec's `*_ds_name`.

    Override to add a loader belonging to no engine (a dictionary pass, a prototype
    source). Call `super()` first, or the declared splits are never built.
    """
    cfg = self.cfg
    self.transform = get_aug(cfg.aug_name, cfg.aug_params)

    self.dls["train"], self.infos["train"] = get_dataset(
        cfg.train_ds_name, cfg.train_ds_params, self.transform, field="train_ds_name"
    )
    for spec in self.eval_specs():
        split = prefix = spec.split
        # Two specs may share a split (`ds_prefix`); build its loader once.
        if split in self.dls:
            continue
        name = cfg.get(f"{prefix}_ds_name", None)
        if not name:
            self.dls[split], self.infos[split] = None, None
            continue

        # Under DDP, DistributedSampler pads an uneven eval set and scores the duplicates.
        # A get_ds that builds its own loader ignores this key; _warn_if_eval_padded checks.
        ds_params = dict(cfg.get(f"{prefix}_ds_params", {}))
        ds_params.setdefault("exact_sharding", True)

        self.dls[split], self.infos[split] = get_dataset(
            name,
            ds_params,
            self.transform,
            field=f"{prefix}_ds_name",
        )
        self._warn_if_eval_padded(split, self.dls[split])
    return self.dls, self.infos

gatle_ignite.base_config

base_config()

Return a ConfigDict with every optional field already set.

A task's get_config() starts here and overrides only what it changes. Fields the framework knows but does not pre-set live in OPTIONAL_KNOWN above.

Source code in src/gatle_ignite/config/base.py
def base_config():
    """Return a ConfigDict with every optional field already set.

    A task's `get_config()` starts here and overrides only what it changes. Fields the
    framework knows but does not pre-set live in `OPTIONAL_KNOWN` above.
    """
    cfg = config_dict.ConfigDict()

    # --- identity / wiring ---
    cfg.name = config_dict.placeholder(
        str
    )  # Identifies the run: the W&B id, and the default save_dir stem.
    cfg.save_dir = config_dict.placeholder(str)  # Where checkpoints go. Resume reads from here too.
    cfg.project_name = "gatle"  # Groups runs in W&B. Not used by the text logger.
    cfg.main_runner = config_dict.placeholder(
        str
    )  # Dotted path to the module exposing your `Trainer`.
    cfg.seed = 42  # Seeds python, numpy and torch before anything is built.

    # --- model ---
    cfg.model_name = config_dict.placeholder(
        str
    )  # Dotted path to a module exposing `Model(**model_params)`.
    cfg.model_params = {}  # Splatted into `Model(...)`.
    cfg.auto_model_params = {
        "find_unused_parameters": False,
        "sync_bn": True,
    }  # kwargs for idist.auto_model (DDP wrapping). sync_bn needs a GPU backend.

    # --- data ---
    cfg.train_ds_name = config_dict.placeholder(
        str
    )  # Dotted path to a module exposing `get_ds(ds_params, transform)`.
    cfg.train_ds_params = {}  # Passed to your `get_ds`. Batch size lives here (`bs`), not at top level.
    cfg.aug_name = config_dict.placeholder(
        str
    )  # Optional. Built once and passed to EVERY split; the dataset decides who gets it.
    cfg.aug_params = {}  # Splatted into `Transformation(...)`.

    # --- loss ---
    cfg.criterion_name = config_dict.placeholder(
        str
    )  # Usually `gatle_ignite.losses.composite`, even for a single term.
    cfg.criterion_params = {}  # For the composite: `{'dict_of_loss_params': {...}}`.

    # --- optimizer / scheduler ---
    cfg.optimizer_name = config_dict.placeholder(
        str
    )  # Dotted path. Builtins are named in full: `gatle_ignite.optimizers.adamw`.
    cfg.optimizer_params = {}  # Splatted into `get_optimizer(model, ...)`. The LR schedule reads `lr` from here.
    cfg.lr_scheduler = config_dict.placeholder(str)  # Dotted path. None = no scheduling.
    cfg.lr_scheduler_params = {}  # Splatted into `get_scheduler(...)`.
    cfg.grad_clip_norm = config_dict.placeholder(float)  # Clip gradients by total norm. None = off.
    cfg.grad_clip_value = config_dict.placeholder(float)  # Clip gradients elementwise. None = off.
    cfg.accum_steps = 1  # Batches per optimizer step. N emulates N*bs at the memory of bs. Clipping applies to the accumulated gradient.

    # --- loop ---
    cfg.max_epochs = (
        1  # Also sets the LR schedule's geometry, so changing it on resume replays the old curve.
    )
    cfg.every_val = (
        1  # Run the evaluator every N epochs. 0 disables validation, mirroring every_test.
    )
    cfg.every_test = 0  # 0 disables the tester engine entirely
    cfg.train_length = config_dict.placeholder(
        int
    )  # Cap an epoch to N iterations. None = the full dataloader. Handy for smoke runs.
    cfg.early_stop_patience = (
        0  # Stop after N evaluations with no improvement in score_name. 0 = never stop early.
    )
    cfg.early_stop_after = (
        0  # Epochs to train before the patience counter arms. 0 = arm immediately.
    )
    cfg.val_length = config_dict.placeholder(int)  # As train_length, for the evaluator.
    cfg.test_length = config_dict.placeholder(int)  # As train_length, for the tester.

    # --- precision ---
    cfg.amp_dtype = "auto"  # auto (bf16 where supported, else fp32) | bf16 | fp16 | fp32
    cfg.cudnn_benchmark = (
        False  # Autotunes per input shape. A pessimisation when shapes vary; off by default.
    )
    cfg.compile = config_dict.placeholder(
        object
    )  # torch.compile the model: True | False | "auto" (on where it is usable). Off by default; the first step pays a one-off warm-up.
    cfg.compile_params = {}  # Splatted into nn.Module.compile(): mode, dynamic, backend, fullgraph.

    # --- metrics / scoring ---
    cfg.train_metrics = {}  # {"acc": {"cls_name": "...", "params": {...}}}. Loss averages are added automatically.
    cfg.val_metrics = {}  # As train_metrics, on the evaluator. Keys are namespaced "valid/<name>".
    cfg.tester_metrics = {}  # As train_metrics, on the tester. Keys are namespaced "test/<name>".
    cfg.score_name = config_dict.placeholder(
        str
    )  # Metric that selects the best checkpoint, e.g. "valid/acc". None = latest only.
    cfg.score_factor = 1  # Checkpointing keeps the MAXIMUM, so use -1 for loss/CER/WER.
    cfg.tester_score_name = config_dict.placeholder(
        str
    )  # As score_name, for the test engine -> `test_best_result*`.
    cfg.tester_score_factor = 1  # As score_factor, for the test engine.

    # --- checkpointing ---
    cfg.save_ckpt = True  # Write checkpoints at all.
    cfg.n_saved = (
        1  # How many of each kind to keep, >= 1 or None for all. Above 1, `best` is by score.
    )
    cfg.resume = False  # Continue from the newest `latest_epoch*` in save_dir. Safe to leave on; warns if there is none.
    cfg.model_checkpoint_dir = (
        ""  # Load WEIGHTS ONLY from this file (fine-tuning, or a clean warm start).
    )
    cfg.strict = True  # strict= for the model_checkpoint_dir load.

    # --- logging ---
    cfg.logger_name = [
        "text"
    ]  # Sinks to enable. The one field taking short names: text, pbar, wandb, discord.
    cfg.log_every = 100  # Iterations between LR / grad-norm points.
    cfg.watch_grad = False  # Log gradient norms. Costs a pass over the parameters.
    cfg.tags = []  # Passed to W&B.

    # --- distributed ---
    cfg.dist_backend = (
        "nccl"  # Only applies with >1 process; below that a run is single-process with no backend.
    )
    cfg.nnodes = 1  # Number of machines. >1 needs the same command, and node_rank, on every node.
    cfg.node_rank = 0  # This machine's index, 0..nnodes-1. Node 0 is where master_addr must point.
    cfg.nproc_per_node = config_dict.placeholder(
        int
    )  # Processes per machine. None = one per visible GPU (or 1 on CPU).
    cfg.master_addr = config_dict.placeholder(
        str
    )  # Rendezvous host. None = MASTER_ADDR, else 127.0.0.1. Required for nnodes > 1.
    cfg.master_port = config_dict.placeholder(
        int
    )  # Rendezvous port. None = MASTER_PORT, else random on one node. Required for nnodes > 1.

    return cfg

gatle_ignite.build_dataloader

build_dataloader(dataset, ds_params, sampler=None)

Build a DDP-aware dataloader from a dataset and the config's ds_params.

Recognised keys: bs (the total across GPUs on every split), num_workers, shuffle, drop_last, pin_memory, collate_fn, sampler_params, exact_sharding; anything else is the dataset's own business. sampler_params and collate_fn are dotted: get_sampler(dataset, params) and get_collate_fn(params). BaseTrainer sets exact_sharding on valid and test itself.

Source code in src/gatle_ignite/data/helpers.py
def build_dataloader(dataset, ds_params, sampler=None):
    """Build a DDP-aware dataloader from a dataset and the config's ds_params.

    Recognised keys: bs (the total across GPUs on every split), num_workers, shuffle, drop_last,
    pin_memory, collate_fn, sampler_params, exact_sharding; anything else is the dataset's own
    business. sampler_params and collate_fn are dotted: get_sampler(dataset, **params) and
    get_collate_fn(**params). BaseTrainer sets exact_sharding on valid and test itself.
    """
    if sampler is None:
        sampler = _build_sampler(dataset, ds_params.get("sampler_params", None))

    if ds_params.get("exact_sharding", False) and idist.get_world_size() > 1:
        if sampler is not None:
            # We cannot tell whether a custom sampler selects or orders, so keep it and warn.
            warnings.warn(
                "exact_sharding is set but this split has a custom sampler, so under DDP padded "
                "duplicates will be scored. Remove the sampler to shard this split exactly.",
                stacklevel=2,
            )
        else:
            return _build_exact_sharded_loader(dataset, ds_params)

    batch_size = ds_params.get("bs", 1)
    drop_last = ds_params.get("drop_last", False)

    kwargs = {
        "batch_size": batch_size,
        "num_workers": ds_params.get("num_workers", 0),
        "drop_last": drop_last,
        # Pinning only helps for host->device copies; on CPU it is a no-op that warns.
        "pin_memory": ds_params.get("pin_memory", torch.cuda.is_available()),
    }

    if sampler is not None:
        # torch rejects sampler+shuffle, and shuffle dropping out silently deserves a warning.
        if ds_params.get("shuffle", False):
            warnings.warn(
                "ds_params sets both 'shuffle': True and a sampler, so shuffle is ignored and "
                "the sampler decides the order. Remove 'shuffle' to make that explicit.",
                stacklevel=2,
            )
        kwargs["sampler"] = sampler
    else:
        kwargs["shuffle"] = ds_params.get("shuffle", False)

    collate_fn = resolve_collate_fn(ds_params)
    if collate_fn is not None:
        kwargs["collate_fn"] = collate_fn

    if drop_last and len(dataset) < batch_size:
        # An empty loader trains on nothing and still reports success.
        warnings.warn(
            f"drop_last=True with {len(dataset)} sample(s) and bs={batch_size} yields no batches, "
            f"so this split is silently skipped. Set drop_last=False or lower bs.",
            stacklevel=2,
        )

    return idist.auto_dataloader(dataset, **kwargs)

gatle_ignite.to_device

to_device(batch, device=None, non_blocking=True)

Move a (possibly nested) batch structure to the active device.

ignite's convert_tensor, so a prep_batch need not import idist to name the device.

Source code in src/gatle_ignite/utils/batch.py
def to_device(batch, device=None, non_blocking=True):
    """Move a (possibly nested) batch structure to the active device.

    ignite's convert_tensor, so a prep_batch need not import idist to name the device.
    """
    return convert_tensor(
        batch,
        device=device if device is not None else idist.device(),
        non_blocking=non_blocking,
    )

gatle_ignite.get_value

get_value(container, name)

get_value(y, "logits") or get_value(x, ("targets", "labels")).

A string is a top-level key and a tuple or list is a path, which is what lets a config wire a loss or metric by naming its src_name and tgt_name.

Source code in src/gatle_ignite/utils/batch.py
def get_value(container, name):
    """get_value(y, "logits") or get_value(x, ("targets", "labels")).

    A string is a top-level key and a tuple or list is a path, which is what lets a config
    wire a loss or metric by naming its `src_name` and `tgt_name`.
    """
    if isinstance(name, (tuple, list)):
        value = container
        for i, key in enumerate(name):
            try:
                value = value[key]
            except (KeyError, TypeError) as e:
                path = " -> ".join(map(str, name[: i + 1]))
                raise ConfigError(f"could not resolve path {tuple(name)!r} at {path!r}: {e}") from e
        return value
    try:
        return container[name]
    except (KeyError, TypeError) as e:
        available = list(container) if hasattr(container, "keys") else type(container)
        raise ConfigError(f"could not resolve {name!r}; available: {available}") from e

gatle_ignite.ConfigError

Bases: Exception

A config names something that cannot be resolved, or is missing/invalid.

Source code in src/gatle_ignite/dispatch.py
class ConfigError(Exception):
    """A config names something that cannot be resolved, or is missing/invalid."""