Compare commits
32 Commits
6a21c3b908
...
master
| Author | SHA1 | Date | |
|---|---|---|---|
| a1ecf0df1d | |||
| d858226294 | |||
| 4f092c4528 | |||
| cc37a55183 | |||
| c4b12b5e7a | |||
| 1a3c907571 | |||
| c71210f006 | |||
| 4692cee699 | |||
| b63edcb8f9 | |||
| 593c5f4d34 | |||
| c00ee91a74 | |||
| f301fd98d2 | |||
| 9752ddf79c | |||
| f8722e347e | |||
| c8f52259d6 | |||
| 0f95e0eaae | |||
| dc4cad7d11 | |||
| f3f7645bf7 | |||
| c83e72b689 | |||
| f505fe7f22 | |||
| 32aa5a5f92 | |||
| 899ca3a7d5 | |||
| da717971b6 | |||
| c3fc768b40 | |||
| a4b5a6c3bf | |||
| 30a448927c | |||
| 81eb14d75c | |||
| 72f5a891bf | |||
| a4f4cba58b | |||
| e6261cea03 | |||
| 733c13c31c | |||
| 818c380fd0 |
@@ -22,7 +22,7 @@ giant analyze render <run_dir> --gallery # render PDFs + HTML
|
||||
dwarf --help # dataset/tooling CLI: convert, migrate, bump-gen,
|
||||
# bump-schema, status, update-manifest, create-manifest,
|
||||
# make-root, build-geometry-oracle, warm-cache, hparam-scan
|
||||
# (see scripts/dwarf.py)
|
||||
# (see giant/tools/dwarf.py)
|
||||
```
|
||||
|
||||
`cpu` and `cuda` are mutually exclusive — pick one to select the torch build (pinned to 2.3.x; newer torch requires newer NVIDIA drivers). Plain `uv sync` with no extra will not install torch at all; uv has no concept of a "default extra", so `--extra cpu` should always be included unless you need GPU support.
|
||||
@@ -91,6 +91,6 @@ GIANT is a conditional generative surrogate for the Geant4 step function. It rep
|
||||
|
||||
A sampling-calorimeter (multi-material) dataset is still a planned future direction, not yet built. See the knowledge base (`/home/lars/knowledge-base/meta/roadmap.md`).
|
||||
|
||||
**v0.3.0 — Stage-2 autoregressive redesign (designed, not implemented; branch `v0.3.0-stage2-autoregressive`):** the 2026-08-03 WGAN rollout benchmark failed specifically at the secondary-species level (zero photon secondaries, ~4M hallucinated `-14` muon antineutrinos). The agreed response pivots Stage 2 to **autoregressive generation** in descending-energy order with teacher forcing, and switches the particle-type representation back to **categorical** (top N−1 by training-set count + an "other" bucket), reversing the 2026-07-17 continuous `(log-mass, charge)` target. This requires a config break: `[conditioning]` / `[stage1_model]` / `[stage2_model]` / `[train]` blocks replace the single global `train.mode` + `[model]`, so per-stage generators (`stage1 = flow` + `stage2 = wgan`), stage-2-only training, and one-shot-vs-autoregressive comparison are all expressible. `network.py` is refactored from ten permutation classes into composable parts (encoder × trunk × objective), which also makes routed WGAN work for the first time.
|
||||
**v0.3.0 — Stage-2 autoregressive redesign (designed, not implemented; branch `v0.3.0-stage2-autoregressive`):** the 2026-08-03 WGAN rollout benchmark failed specifically at the secondary-species level (zero photon secondaries, ~4M hallucinated `-14` muon antineutrinos). The agreed response pivots Stage 2 to **autoregressive generation** in descending-energy order with teacher forcing, and switches the particle-type representation back to **categorical** (top N−1 by training-set count + an "other" bucket), reversing the 2026-07-17 continuous `(log-mass, charge)` target. This requires a config break: `[conditioning]` / `[stage1_model]` / `[stage2_model]` / `[train]` blocks replace the single global `train.mode` + `[model]`, so per-stage generators (`stage1 = flow` + `stage2 = wgan`), stage-2-only training, and one-shot-vs-autoregressive comparison are all expressible. `network.py` is refactored from ten permutation classes into composable parts (encoder × trunk × objective), which also makes routed WGAN work for the first time. This config break is why v0.2-shaped configs/checkpoints need migrating at all (`config.migrate_config`, `model.network._migrate_legacy_model_config`, both drawing on shared facts in `giant/_migration.py`) — v0.2 checkpoint-loading support has **no expiry decided yet**: `/ceph` still holds pre-v0.3.0 checkpoints and analysis runs referencing them, so don't delete or substantially alter either migration function or `tests/legacy/network_v02_snapshot.py` (the frozen v0.2 snapshot they're tested against) without an explicit decision to do so first.
|
||||
|
||||
**Condor-submitted GPU training/rollout (in progress, `condor-gpu-train-rollout` branch, not yet merged):** moves `giant train`/`giant rollout` off the shared portal GPU dev machines (see Compute environment) onto remote-GPU HTCondor submission on TOpAS/NEMO2 (`giant/condor.py`). Partway between "needs major features" and feature-complete — not ready to merge yet.
|
||||
|
||||
@@ -93,7 +93,7 @@ giant/
|
||||
│ │ ├── condor.py # prep / compute-one / submit-description plumbing
|
||||
│ │ └── render.py # PDFs + HTML gallery (only module importing plotstyle/LaTeX)
|
||||
│ └── cli.py # `giant train` / `new-run` / `predict` / `rollout` / `analyze` Typer app
|
||||
├── scripts/ # dataset/tooling logic, unified under the `dwarf` CLI (`dwarf --help`)
|
||||
├── giant/tools/ # dataset/tooling logic, unified under the `dwarf` CLI (`dwarf --help`)
|
||||
│ ├── dwarf.py # Typer app: convert, migrate, bump-gen, bump-schema, status,
|
||||
│ │ # update-manifest, create-manifest, make-root,
|
||||
│ │ # build-geometry-oracle, warm-cache, hparam-scan
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
# GIANT reference baseline (v0.3 schema).
|
||||
#
|
||||
# The fixed comparison point every future architecture variant is measured
|
||||
# against. Chosen so that each experimental axis the roadmap cares about
|
||||
# (routed trunk, WGAN generators, attention history, shared conditioning,
|
||||
# embedding/onehot conditioning) is a *single* edit away from this file.
|
||||
#
|
||||
# Rationale for the choices below, from the runs already on record
|
||||
# (analysis_runs/ + the `giant` W&B project):
|
||||
#
|
||||
# * flow, not wgan, for both stages. Ranking the five existing rollouts by
|
||||
# mean Jensen-Shannon divergence against the Geant4 reference, the plain
|
||||
# non-routed flow model wins (0.172) over the routed flow runs
|
||||
# (0.197/0.200) and both WGAN runs (0.218/0.234) — and it beats them by
|
||||
# ~7x on per-event total deposited energy and by 3-10x on every
|
||||
# per-PDG marginal. WGAN stays a variant, not the reference.
|
||||
#
|
||||
# * no router. The routed runs are not better, and soft-mixing 10 small
|
||||
# experts costs ~10x per-pass throughput at train time (29k samples/s vs
|
||||
# the WGAN runs' 52-116k), which is what made those runs take ~110 h for
|
||||
# 30 epochs.
|
||||
#
|
||||
# * hidden_dim 512 / 6 blocks per stage. The best-scoring rollout so far
|
||||
# was hidden_dim 1024, but at 4x the trunk FLOPs of 512. 512/6 sits in
|
||||
# the same weight class as the variants it will be compared against and
|
||||
# leaves headroom to train it properly rather than cheaply.
|
||||
#
|
||||
# * dropout 0.0. Training set is ~5e8 steps against <1e7 parameters;
|
||||
# capacity overfitting is not the binding constraint, and every recent
|
||||
# run used 0.0.
|
||||
#
|
||||
# Known weak spots this baseline is expected to *exhibit* (they are the
|
||||
# reason for the comparisons, not a reason to retune this file): every model
|
||||
# on record under-produces steps per event by ~2x (rollout ~7e4 vs Geant4
|
||||
# ~1.4e5) and secondaries per event by 2-3.5x (~2-3e4 vs 7.2e4), and n_sec
|
||||
# head accuracy sits at 0.863-0.867 regardless of size or objective.
|
||||
|
||||
[meta]
|
||||
# REQUIRED. Without it config.migrate_config reads this file as v0.2 and
|
||||
# rewrites it from V02_FIXED_FACTS — silently forcing decoder = "one_shot",
|
||||
# particle_type.target = "physical" and the v0.2 default sizes, while still
|
||||
# passing validate_config.
|
||||
config_version = 3
|
||||
|
||||
[conditioning]
|
||||
# Physical-property MLPs rather than learned vocab embeddings: computable for
|
||||
# any PDG code / material, which is what the held-out-species and
|
||||
# held-out-material generalization comparisons need.
|
||||
out_dim = 128
|
||||
share_stages = false
|
||||
|
||||
# n_layers = 2 rather than the v0.3 default of 1: v0.2's conditioning MLP was
|
||||
# always 2 deep (see _migration.V02_FIXED_FACTS), so this keeps the encoder
|
||||
# identical to the architecture that produced the results cited above.
|
||||
[conditioning.particle]
|
||||
type = "physical"
|
||||
emb_dim = 16
|
||||
n_layers = 2
|
||||
|
||||
[conditioning.material]
|
||||
type = "physical"
|
||||
emb_dim = 16
|
||||
n_layers = 2
|
||||
|
||||
[stage1_model]
|
||||
generator = "flow"
|
||||
hidden_dim = 512
|
||||
n_res_blocks = 6
|
||||
dropout = 0.0
|
||||
|
||||
[stage2_model]
|
||||
# The v0.3 pivot: autoregressive in descending-energy order with a
|
||||
# categorical species target, which is the agreed response to the 2026-08-03
|
||||
# secondary-species failure. Flow (not the schema default wgan) so the
|
||||
# baseline varies only the decoder relative to the best v0.2 result.
|
||||
#
|
||||
# COST, measured (RTX 4070, bs 4096, 10 ODE steps), not estimated:
|
||||
# sample.sample_secondaries_ar loops `for k in range(k_max)` unconditionally
|
||||
# — all 15 slots regardless of predicted n_sec — so a flow AR token costs
|
||||
# k_max * steps = 150 stage-2 calls per physics step. That makes this block
|
||||
# the dominant cost on both sides:
|
||||
# training flow AR 29.5k samp/s vs flow one-shot 190.7k samp/s (6.5x)
|
||||
# inference flow AR 8.5k step/s vs flow one-shot 68.7k step/s (8.1x)
|
||||
# Accepted deliberately: one-shot is the configuration whose secondary
|
||||
# species distribution failed, and that failure is what v0.3 exists to fix.
|
||||
decoder = "autoregressive"
|
||||
generator = "flow"
|
||||
hidden_dim = 512
|
||||
n_res_blocks = 6
|
||||
dropout = 0.0
|
||||
k_max = 15
|
||||
|
||||
[stage2_model.autoregressive]
|
||||
history = "markov"
|
||||
teacher_forcing = "always"
|
||||
|
||||
[stage2_model.particle_type]
|
||||
target = "onehot"
|
||||
# Decoupled from conditioning.particle.emb_dim (gitea #29). 32 classes + the
|
||||
# "other" bucket keeps essentially all real secondary species out of "other"
|
||||
# without making the head expensive.
|
||||
n_classes = 32
|
||||
other_policy = "sample"
|
||||
|
||||
[train]
|
||||
epochs = 50
|
||||
# Sized for ONE NVIDIA L40S on deepthought2 (46068 MiB; the box has two, and
|
||||
# CLAUDE.md's shared-machine rule allows a single GPU). From a measured
|
||||
# linear fit of this exact config's training step on the local RTX 4070:
|
||||
# peak reserved MiB = 0.9736 * batch_size + 115
|
||||
# so 36864 reserves ~36.0 GiB, i.e. 78% of the card, leaving ~10 GiB of
|
||||
# headroom for fragmentation and the CUDA context. Throughput is already
|
||||
# flat above bs~4096 on the 4070, so this is chosen for occupancy on the
|
||||
# larger card, not for step efficiency — and it sits next to the 43008/32768
|
||||
# of the runs lr = 3e-4 was proven at.
|
||||
batch_size = 36864
|
||||
lr = 3e-4
|
||||
warmup_epochs = 3
|
||||
weight_decay = 0.01
|
||||
ema_decay = 0.9999
|
||||
val_fraction = 0.1
|
||||
num_workers = 4
|
||||
seed = 0
|
||||
# The marginal/KL pass is expensive (~5000 s on top of an epoch), so keep it
|
||||
# to every 10th epoch; the cheap per-epoch val loss still runs every epoch.
|
||||
validate_every = 10
|
||||
validate_steps = 10
|
||||
wandb = true
|
||||
wandb_project = "giant"
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Shared v0.2 -> v0.3 migration knowledge.
|
||||
|
||||
v0.3.0 broke the config format (single `[train]` + `[model]` -> `[conditioning]`/
|
||||
`[stage1_model]`/`[stage2_model]`/`[train]`), and that break has to be absorbed by two
|
||||
independent migration surfaces: `giant.config.migrate_config` (a v0.2 `config.toml`) and
|
||||
`giant.model.network._migrate_legacy_model_config` (a v0.2 checkpoint's flat
|
||||
`model_config` dict). Both translate the same v0.2 facts into the same v0.3 shape, so
|
||||
the facts live here once rather than as two hand-maintained copies — see issues.md
|
||||
Issue 6.
|
||||
|
||||
A dependency-free leaf module so neither `config.py` nor `network.py` has to import the
|
||||
other to share this.
|
||||
"""
|
||||
|
||||
# v0.2 model-shaped keys (config.toml's [model] table, or a checkpoint's flat
|
||||
# model_config dict — same key names in both) applied identically to both v0.3 stage
|
||||
# blocks, because v0.2 had only one trunk shape shared by both stages.
|
||||
V02_MODEL_KEY_TO_STAGES: tuple[tuple[str, str], ...] = (
|
||||
("hidden_dim", "hidden_dim"),
|
||||
("n_blocks", "n_res_blocks"),
|
||||
("dropout", "dropout"),
|
||||
)
|
||||
|
||||
# v0.2 architectural facts that had no corresponding config key at all — always true of
|
||||
# a v0.2 model, so both migration surfaces inject them unconditionally. Keyed by dotted
|
||||
# path relative to the migrated dict's root. NOTE: conditioning.*.n_layers (2) differs
|
||||
# from the v0.3 *default* (1) — not a typo, v0.2's conditioning MLP was always 2 layers
|
||||
# deep.
|
||||
V02_FIXED_FACTS: dict[str, object] = {
|
||||
"conditioning.out_dim": 128,
|
||||
"conditioning.particle.n_layers": 2,
|
||||
"conditioning.material.n_layers": 2,
|
||||
"stage1_model.active": True,
|
||||
"stage1_model.flow.time_dim": 64,
|
||||
"stage1_model.ddpm.time_dim": 64,
|
||||
"stage2_model.active": True,
|
||||
"stage2_model.flow.time_dim": 64,
|
||||
"stage2_model.ddpm.time_dim": 64,
|
||||
"stage2_model.context_dim": 64,
|
||||
"stage2_model.decoder": "one_shot",
|
||||
"stage2_model.particle_type.target": "physical",
|
||||
}
|
||||
|
||||
|
||||
def reject_legacy_router_expert_sizing(router_cfg: dict, *, source: str) -> None:
|
||||
"""Pop and validate v0.2's per-expert width/depth override, in place.
|
||||
|
||||
v0.3.0 removed per-expert sizing — experts always inherit the stage's
|
||||
hidden_dim/n_res_blocks — so a v0.2 router config/checkpoint that set a non-default
|
||||
`expert_hidden_dim`/`expert_n_blocks` describes experts with a different width/depth
|
||||
than the monolith, and can only be reproduced by v0.2 code. Silently dropping these
|
||||
keys (a router builder's kwarg filtering would do this for free) would resize the
|
||||
experts instead of refusing, so this raises loudly.
|
||||
|
||||
Always pops both keys, whether or not they were non-default, so callers can go on
|
||||
to use the (now-cleaned) `router_cfg` unconditionally. `source` names what's being
|
||||
migrated (e.g. "v0.2 config's model.router" or "this checkpoint's
|
||||
model_config.router") for the error message.
|
||||
"""
|
||||
expert_hidden_dim = router_cfg.pop("expert_hidden_dim", 0)
|
||||
expert_n_blocks = router_cfg.pop("expert_n_blocks", 0)
|
||||
if not (expert_hidden_dim or expert_n_blocks):
|
||||
return
|
||||
raise ValueError(
|
||||
f"{source} sets expert_hidden_dim/expert_n_blocks to a non-default value "
|
||||
f"({expert_hidden_dim!r}, {expert_n_blocks!r}); v0.3.0 removed per-expert "
|
||||
"sizing (experts always inherit the stage's hidden_dim/n_res_blocks), so "
|
||||
"this router's experts have a different width/depth than the monolith. "
|
||||
"This checkpoint/config can only be loaded by v0.2 code."
|
||||
)
|
||||
@@ -66,27 +66,11 @@ class _RouterHandle:
|
||||
router_type: str
|
||||
|
||||
|
||||
def _conditioning_axes(model_cfg: dict, default: str = "embedding") -> tuple[str, str]:
|
||||
"""(particle_conditioning, material_conditioning) for
|
||||
`giant.data.transforms.build_cond_features` — from either a v0.2
|
||||
checkpoint's flat `model_config["conditioning"]` (one shared string, same
|
||||
for both axes) or a new-format one (independent
|
||||
`model_config["conditioning"]["particle"/"material"]["type"]` — the two
|
||||
axes may differ). Mirrors
|
||||
`giant.cli._conditioning_axes`."""
|
||||
raw = model_cfg.get("conditioning", default)
|
||||
if isinstance(raw, dict):
|
||||
return (
|
||||
raw.get("particle", {}).get("type", default),
|
||||
raw.get("material", {}).get("type", default),
|
||||
)
|
||||
return raw, raw
|
||||
|
||||
|
||||
def load_router(checkpoint: str | Path) -> _RouterHandle | None:
|
||||
"""Load a checkpoint's Stage-1 router, or None if it isn't a MoE checkpoint."""
|
||||
import torch
|
||||
|
||||
from giant.checkpoint_io import conditioning_axes
|
||||
from giant.data.transforms import Normalizer
|
||||
from giant.model.network import build_models
|
||||
|
||||
@@ -110,7 +94,7 @@ def load_router(checkpoint: str | Path) -> _RouterHandle | None:
|
||||
if router is None:
|
||||
return None
|
||||
|
||||
particle_conditioning, material_conditioning = _conditioning_axes(model_cfg)
|
||||
particle_conditioning, material_conditioning = conditioning_axes(model_cfg)
|
||||
return _RouterHandle(
|
||||
router=router,
|
||||
pdg_map={int(k): v for k, v in ckpt["pdg_map"].items()},
|
||||
|
||||
@@ -27,7 +27,7 @@ would then wrongly scale up with a bigger dataset. `RUNTIME_SAFETY_MARGIN` is
|
||||
deliberately generous (4x total) specifically to absorb that kind of
|
||||
contention spike instead. Rerun this calibration (pull fresh
|
||||
`condor_history`/`run_meta.json`, refit) if the catalog changes or timings
|
||||
drift — a synthetic local rebaseline via `scripts/profile_analysis_costs.py`
|
||||
drift — a synthetic local rebaseline via `giant/tools/profile_analysis_costs.py`
|
||||
is a reasonable fallback when no real cluster data is available yet, but
|
||||
undershoots real wall time badly (it can't see docker pull / `/ceph` I/O
|
||||
latency), which is exactly why this file moved off it.
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
"""Load a trained checkpoint into ready-to-run models (giant.cli's `predict`/`rollout`).
|
||||
|
||||
Both commands need the same ~15 steps to go from a checkpoint path to two
|
||||
`eval()`-mode models plus their normalizers/vocab maps: load the pickle,
|
||||
validate it carries what current code expects, resolve which conditioning
|
||||
mode each axis was trained with, restore the top-N vocab maps (if the
|
||||
checkpoint used one-hot conditioning), rebuild the normalizers, construct the
|
||||
model from `model_config`, and load the requested (raw or EMA) weights. This
|
||||
used to be duplicated near-verbatim in both commands (issues.md Issue 5) —
|
||||
`load_for_inference` is the single implementation.
|
||||
|
||||
This module intentionally has no Typer dependency, so it can be unit-tested
|
||||
directly and imported from non-CLI code (`giant.analysis.router_gating`,
|
||||
lazily — see that module's docstring for why). Failures raise
|
||||
`CheckpointCompatibilityError` with the same wording the CLI has always
|
||||
shown; the CLI layer catches it and does the `typer.echo`/`Exit(1)`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
|
||||
from giant import config as gconfig
|
||||
from giant.constants import K_MAX
|
||||
from giant.data.loader import TopNMap
|
||||
from giant.data.setup_cache import topnmap_from_json
|
||||
from giant.data.transforms import Normalizer
|
||||
from giant.model.network import build_models
|
||||
|
||||
|
||||
class CheckpointCompatibilityError(Exception):
|
||||
"""Checkpoint is missing something `load_for_inference` needs."""
|
||||
|
||||
|
||||
def conditioning_axes(model_cfg: dict, default: str = "embedding") -> tuple[str, str]:
|
||||
"""(particle_conditioning, material_conditioning) for
|
||||
`giant.data.transforms.build_cond_features`/`build_features` — from
|
||||
either a v0.2 checkpoint's flat `model_config["conditioning"]` (one
|
||||
shared string, same for both axes) or a new-format one (independent
|
||||
`model_config["conditioning"]["particle"/"material"]["type"]` — the two
|
||||
axes are configured independently and may differ)."""
|
||||
raw = model_cfg.get("conditioning", default)
|
||||
if isinstance(raw, dict):
|
||||
return (
|
||||
raw.get("particle", {}).get("type", default),
|
||||
raw.get("material", {}).get("type", default),
|
||||
)
|
||||
return raw, raw
|
||||
|
||||
|
||||
def stage_cfg(model_cfg: dict, stage: str) -> dict:
|
||||
"""`model_cfg[f"{stage}_model"]` for a new-format model_config, `{}` for
|
||||
a v0.2 flat one (whose ddpm schedule always used `CosineSchedule`'s own
|
||||
default `T=1000` — never a config key — and which never had
|
||||
`particle_type` at all, so `{}` is the correct fallback for both
|
||||
`ddpm_steps`/`particle_type_other_policy` below)."""
|
||||
val = model_cfg.get(f"{stage}_model")
|
||||
return val if isinstance(val, dict) else {}
|
||||
|
||||
|
||||
def ddpm_steps(model_cfg: dict, stage: str) -> int:
|
||||
return stage_cfg(model_cfg, stage).get("ddpm", {}).get("n_steps", 1000)
|
||||
|
||||
|
||||
def particle_type_other_policy(model_cfg: dict) -> str:
|
||||
return stage_cfg(model_cfg, "stage2").get("particle_type", {}).get("other_policy", "sample")
|
||||
|
||||
|
||||
def load_pdg_topn_map(ckpt: dict) -> TopNMap | None:
|
||||
"""`ckpt["pdg_topn_map"]` as a `giant.data.loader.TopNMap`, or `None` if
|
||||
this checkpoint's conditioning/particle_type never needed one (see
|
||||
`giant.pipeline.run_setup_stage`, which only populates it when
|
||||
`conditioning.particle.type` or `stage2_model.particle_type.target` is
|
||||
`"onehot"`)."""
|
||||
raw = ckpt.get("pdg_topn_map")
|
||||
return topnmap_from_json(raw, axis="pdg") if raw is not None else None
|
||||
|
||||
|
||||
def load_mat_topn_map(ckpt: dict) -> TopNMap | None:
|
||||
"""`ckpt["mat_topn_map"]` as a `giant.data.loader.TopNMap`, or `None` if
|
||||
this checkpoint's `conditioning.material.type` was never `"onehot"` (see
|
||||
`giant.pipeline.run_setup_stage`)."""
|
||||
raw = ckpt.get("mat_topn_map")
|
||||
return topnmap_from_json(raw, axis="material") if raw is not None else None
|
||||
|
||||
|
||||
def load_sec_type_topn_map(ckpt: dict) -> TopNMap | None:
|
||||
"""`ckpt["sec_type_topn_map"]` as a `giant.data.loader.TopNMap`, or
|
||||
`None` if this checkpoint's `stage2_model.particle_type.target` was never
|
||||
`"onehot"` (see `giant.pipeline.run_setup_stage`).
|
||||
|
||||
Pre-gitea-#29 checkpoints have no `sec_type_topn_map` key at all — before
|
||||
#29, the secondary-species decode map and the conditioning PDG onehot map
|
||||
were always numerically the same map, saved once under `pdg_topn_map`.
|
||||
For those, fall back to `load_pdg_topn_map` to reproduce that exact
|
||||
behavior; a current checkpoint always has the key (possibly `null`, if
|
||||
`particle_type.target != "onehot"`), so this fallback never fires for one."""
|
||||
if "sec_type_topn_map" in ckpt:
|
||||
raw = ckpt["sec_type_topn_map"]
|
||||
return topnmap_from_json(raw, axis="pdg") if raw is not None else None
|
||||
return load_pdg_topn_map(ckpt)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InferenceContext:
|
||||
"""Everything needed to run a trained checkpoint forward, resolved once."""
|
||||
|
||||
stage1: nn.Module | None
|
||||
stage2: nn.Module | None
|
||||
cond_norm: Normalizer
|
||||
tgt_norm: Normalizer
|
||||
sec_phys_norm: Normalizer
|
||||
pdg_map: dict[int, int]
|
||||
mat_map: dict[str, int]
|
||||
pdg_topn_map: TopNMap | None
|
||||
mat_topn_map: TopNMap | None
|
||||
sec_type_topn_map: TopNMap | None
|
||||
particle_conditioning: str
|
||||
material_conditioning: str
|
||||
k_max: int
|
||||
stage1_ddpm_steps: int
|
||||
stage2_ddpm_steps: int
|
||||
other_policy: str
|
||||
model_config: dict
|
||||
epoch: int | None
|
||||
best_val_loss: float | None
|
||||
|
||||
|
||||
def load_for_inference(
|
||||
checkpoint: Path,
|
||||
device: torch.device,
|
||||
command_name: str,
|
||||
weights: str = "raw",
|
||||
require_stage2: bool = True,
|
||||
) -> InferenceContext:
|
||||
"""Load *checkpoint* and reconstruct everything `predict`/`rollout` need
|
||||
to run it forward, on *device*, in `eval()` mode.
|
||||
|
||||
*command_name* (e.g. `"predict"`/`"rollout"`) only feeds the "needs both"
|
||||
error message below. *weights* is `"raw"` (the live training weights) or
|
||||
`"ema"` (the EMA shadow copy, see `--ema-decay`). *require_stage2*
|
||||
controls whether a checkpoint with an inactive stage 2
|
||||
(`stage2_model.active = false`) is an error (both current callers need
|
||||
both stages) or an acceptable `stage2 = None` result — kept as a real
|
||||
parameter since `stage{1,2}_model.active` is a real, if currently
|
||||
stage1+stage2-only-in-practice, config option.
|
||||
"""
|
||||
ckpt = torch.load(checkpoint, map_location="cpu", weights_only=False)
|
||||
for key in ("model_config", "sec_decoder"):
|
||||
if key not in ckpt:
|
||||
raise CheckpointCompatibilityError(f"checkpoint has no {key} — retrain with the current code")
|
||||
|
||||
if "sec_phys" not in ckpt.get("normalizer", {}):
|
||||
raise CheckpointCompatibilityError("checkpoint has no normalizer.sec_phys — retrain with the current code")
|
||||
|
||||
gconfig.warn_if_checkpoint_config_mismatch(checkpoint)
|
||||
|
||||
model_cfg = ckpt["model_config"]
|
||||
particle_conditioning, material_conditioning = conditioning_axes(model_cfg)
|
||||
pdg_topn_map = load_pdg_topn_map(ckpt)
|
||||
mat_topn_map = load_mat_topn_map(ckpt)
|
||||
if particle_conditioning == "onehot" and pdg_topn_map is None:
|
||||
raise CheckpointCompatibilityError(
|
||||
"checkpoint's conditioning.particle.type='onehot' but has no pdg_topn_map — retrain with the current code"
|
||||
)
|
||||
if material_conditioning == "onehot" and mat_topn_map is None:
|
||||
raise CheckpointCompatibilityError(
|
||||
"checkpoint's conditioning.material.type='onehot' but has no mat_topn_map — retrain with the current code"
|
||||
)
|
||||
sec_type_topn_map = load_sec_type_topn_map(ckpt)
|
||||
particle_type_target = stage_cfg(model_cfg, "stage2").get("particle_type", {}).get("target", "onehot")
|
||||
if particle_type_target == "onehot" and sec_type_topn_map is None:
|
||||
raise CheckpointCompatibilityError(
|
||||
"checkpoint's stage2_model.particle_type.target='onehot' but has no "
|
||||
"sec_type_topn_map — retrain with the current code"
|
||||
)
|
||||
other_policy = particle_type_other_policy(model_cfg)
|
||||
stage1_ddpm_steps = ddpm_steps(model_cfg, "stage1")
|
||||
stage2_ddpm_steps = ddpm_steps(model_cfg, "stage2")
|
||||
k_max = stage_cfg(model_cfg, "stage2").get("k_max", K_MAX)
|
||||
pdg_map = {int(k): v for k, v in ckpt["pdg_map"].items()}
|
||||
mat_map = {str(k): v for k, v in ckpt["mat_map"].items()}
|
||||
cond_norm = Normalizer.from_dict(ckpt["normalizer"]["cond"])
|
||||
tgt_norm = Normalizer.from_dict(ckpt["normalizer"]["target"])
|
||||
sec_phys_norm = Normalizer.from_dict(ckpt["normalizer"]["sec_phys"])
|
||||
|
||||
built = build_models(model_cfg)
|
||||
stage1, stage2 = built["stage1"], built["stage2"]
|
||||
if require_stage2 and (stage1 is None or stage2 is None):
|
||||
raise CheckpointCompatibilityError(
|
||||
f"checkpoint has an inactive stage1 or stage2 — {command_name} needs both (see stage{{1,2}}_model.active)"
|
||||
)
|
||||
|
||||
if weights == "raw":
|
||||
model_key, sec_key = "model", "sec_decoder"
|
||||
else:
|
||||
model_key, sec_key = "model_ema", "sec_decoder_ema"
|
||||
if model_key not in ckpt or sec_key not in ckpt:
|
||||
raise CheckpointCompatibilityError(
|
||||
f"{checkpoint} has no EMA weights (trained before --ema-decay, "
|
||||
"or with --ema-decay 0) — use --weights raw"
|
||||
)
|
||||
if stage1 is not None:
|
||||
stage1.load_state_dict(ckpt[model_key])
|
||||
stage1.to(device).eval()
|
||||
if stage2 is not None:
|
||||
stage2.load_state_dict(ckpt[sec_key])
|
||||
stage2.to(device).eval()
|
||||
|
||||
return InferenceContext(
|
||||
stage1=stage1,
|
||||
stage2=stage2,
|
||||
cond_norm=cond_norm,
|
||||
tgt_norm=tgt_norm,
|
||||
sec_phys_norm=sec_phys_norm,
|
||||
pdg_map=pdg_map,
|
||||
mat_map=mat_map,
|
||||
pdg_topn_map=pdg_topn_map,
|
||||
mat_topn_map=mat_topn_map,
|
||||
sec_type_topn_map=sec_type_topn_map,
|
||||
particle_conditioning=particle_conditioning,
|
||||
material_conditioning=material_conditioning,
|
||||
k_max=k_max,
|
||||
stage1_ddpm_steps=stage1_ddpm_steps,
|
||||
stage2_ddpm_steps=stage2_ddpm_steps,
|
||||
other_policy=other_policy,
|
||||
model_config=model_cfg,
|
||||
epoch=ckpt.get("epoch"),
|
||||
best_val_loss=ckpt.get("best_val_loss"),
|
||||
)
|
||||
+79
-210
@@ -19,7 +19,6 @@ from tqdm import tqdm
|
||||
|
||||
from giant import config as gconfig
|
||||
from giant.constants import (
|
||||
K_MAX,
|
||||
LOCAL_TARGET_NAMES,
|
||||
PREDICT_COORD_METADATA_KEY,
|
||||
PREDICT_SCHEMA_VERSION,
|
||||
@@ -39,11 +38,9 @@ from giant.data.transforms import (
|
||||
inv_local_frame_rotation,
|
||||
inv_log_transform,
|
||||
reconstruct_post_pos,
|
||||
Normalizer,
|
||||
)
|
||||
from giant.data.setup_cache import topnmap_from_json
|
||||
from giant.checkpoint_io import CheckpointCompatibilityError, load_for_inference
|
||||
from giant.geometry import GeometryOracle
|
||||
from giant.model.network import build_models
|
||||
from giant.pipeline import run_train_job
|
||||
from giant.rollout import (
|
||||
L1DistCollector,
|
||||
@@ -71,58 +68,6 @@ def _router_total_experts(router_cfg: dict) -> int:
|
||||
return int(router_cfg.get("n_experts", 1))
|
||||
|
||||
|
||||
def _conditioning_axes(model_cfg: dict, default: str = "embedding") -> tuple[str, str]:
|
||||
"""(particle_conditioning, material_conditioning) for
|
||||
`giant.data.transforms.build_cond_features`/`build_features` — from
|
||||
either a v0.2 checkpoint's flat `model_config["conditioning"]` (one
|
||||
shared string, same for both axes) or a new-format one (independent
|
||||
`model_config["conditioning"]["particle"/"material"]["type"]` — the two
|
||||
axes are configured independently and may differ)."""
|
||||
raw = model_cfg.get("conditioning", default)
|
||||
if isinstance(raw, dict):
|
||||
return (
|
||||
raw.get("particle", {}).get("type", default),
|
||||
raw.get("material", {}).get("type", default),
|
||||
)
|
||||
return raw, raw
|
||||
|
||||
|
||||
def _stage_cfg(model_cfg: dict, stage: str) -> dict:
|
||||
"""`model_cfg[f"{stage}_model"]` for a new-format model_config, `{}` for
|
||||
a v0.2 flat one (whose ddpm schedule always used `CosineSchedule`'s own
|
||||
default `T=1000` — never a config key — and which never had
|
||||
`particle_type` at all, so `{}` is the correct fallback for both
|
||||
`_ddpm_steps`/`_particle_type_other_policy` below)."""
|
||||
val = model_cfg.get(f"{stage}_model")
|
||||
return val if isinstance(val, dict) else {}
|
||||
|
||||
|
||||
def _ddpm_steps(model_cfg: dict, stage: str) -> int:
|
||||
return _stage_cfg(model_cfg, stage).get("ddpm", {}).get("n_steps", 1000)
|
||||
|
||||
|
||||
def _particle_type_other_policy(model_cfg: dict) -> str:
|
||||
return _stage_cfg(model_cfg, "stage2").get("particle_type", {}).get("other_policy", "sample")
|
||||
|
||||
|
||||
def _load_pdg_topn_map(ckpt: dict):
|
||||
"""`ckpt["pdg_topn_map"]` as a `giant.data.loader.TopNMap`, or `None` if
|
||||
this checkpoint's conditioning/particle_type never needed one (see
|
||||
`giant.pipeline.run_setup_stage`, which only populates it when
|
||||
`conditioning.particle.type` or `stage2_model.particle_type.target` is
|
||||
`"onehot"`)."""
|
||||
raw = ckpt.get("pdg_topn_map")
|
||||
return topnmap_from_json(raw, axis="pdg") if raw is not None else None
|
||||
|
||||
|
||||
def _load_mat_topn_map(ckpt: dict):
|
||||
"""`ckpt["mat_topn_map"]` as a `giant.data.loader.TopNMap`, or `None` if
|
||||
this checkpoint's `conditioning.material.type` was never `"onehot"` (see
|
||||
`giant.pipeline.run_setup_stage`)."""
|
||||
raw = ckpt.get("mat_topn_map")
|
||||
return topnmap_from_json(raw, axis="material") if raw is not None else None
|
||||
|
||||
|
||||
def _batch_size_estimate_dims(model_cfg: dict, training: bool, stage: str = "stage1") -> tuple[int, int]:
|
||||
"""Pick the (hidden_dim, n_blocks) that dominate per-call activation memory.
|
||||
|
||||
@@ -283,7 +228,7 @@ class Stage1Context(str, Enum):
|
||||
|
||||
|
||||
# Conditioning itself lives in giant.config (imported below as gconfig) —
|
||||
# shared with scripts/dwarf.py's Typer commands so the two CLIs can't
|
||||
# shared with giant/tools/dwarf.py's Typer commands so the two CLIs can't
|
||||
# silently drift apart on the option's valid values.
|
||||
Conditioning = gconfig.Conditioning
|
||||
|
||||
@@ -298,34 +243,6 @@ class Weights(str, Enum):
|
||||
ema = "ema"
|
||||
|
||||
|
||||
def _load_model_weights(
|
||||
model: torch.nn.Module,
|
||||
sec_decoder: torch.nn.Module,
|
||||
ckpt: dict,
|
||||
weights: "Weights",
|
||||
checkpoint_path: Path,
|
||||
) -> None:
|
||||
"""Load either the raw or EMA state dicts from a training checkpoint.
|
||||
|
||||
EMA weights (giant.training's shadow copy, see --ema-decay) only exist in
|
||||
checkpoints written after that feature landed, so `ema` fails loudly
|
||||
rather than silently falling back to raw weights a caller didn't ask for.
|
||||
"""
|
||||
if weights == Weights.raw:
|
||||
model_key, sec_key = "model", "sec_decoder"
|
||||
else:
|
||||
model_key, sec_key = "model_ema", "sec_decoder_ema"
|
||||
if model_key not in ckpt or sec_key not in ckpt:
|
||||
typer.echo(
|
||||
f"error: {checkpoint_path} has no EMA weights (trained before "
|
||||
"--ema-decay, or with --ema-decay 0) — use --weights raw",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
model.load_state_dict(ckpt[model_key])
|
||||
sec_decoder.load_state_dict(ckpt[sec_key])
|
||||
|
||||
|
||||
@app.command()
|
||||
def train(
|
||||
data: Annotated[Path, typer.Argument(help="Parquet file or directory of parquet files")],
|
||||
@@ -542,6 +459,34 @@ def train(
|
||||
Optional[float],
|
||||
typer.Option("--stage2-critic-lr", help="Overrides --critic-lr for stage 2 only"),
|
||||
] = None,
|
||||
stage1_critic_hidden_dim: Annotated[
|
||||
Optional[int],
|
||||
typer.Option(
|
||||
"--stage1-critic-hidden-dim",
|
||||
help="WGAN-GP (--mode wgan only): critic width for stage 1 (default: same as generator's hidden_dim)",
|
||||
),
|
||||
] = None,
|
||||
stage1_critic_n_res_blocks: Annotated[
|
||||
Optional[int],
|
||||
typer.Option(
|
||||
"--stage1-critic-n-res-blocks",
|
||||
help="WGAN-GP (--mode wgan only): critic depth for stage 1 (default: same as generator's n_res_blocks)",
|
||||
),
|
||||
] = None,
|
||||
stage2_critic_hidden_dim: Annotated[
|
||||
Optional[int],
|
||||
typer.Option(
|
||||
"--stage2-critic-hidden-dim",
|
||||
help="WGAN-GP (--mode wgan only): critic width for stage 2 (default: same as generator's hidden_dim)",
|
||||
),
|
||||
] = None,
|
||||
stage2_critic_n_res_blocks: Annotated[
|
||||
Optional[int],
|
||||
typer.Option(
|
||||
"--stage2-critic-n-res-blocks",
|
||||
help="WGAN-GP (--mode wgan only): critic depth for stage 2 (default: same as generator's n_res_blocks)",
|
||||
),
|
||||
] = None,
|
||||
val_fraction: Annotated[Optional[float], typer.Option("--val-fraction", "-f")] = None,
|
||||
seed: Annotated[
|
||||
Optional[int],
|
||||
@@ -702,6 +647,10 @@ def train(
|
||||
"stage2_gp_weight": stage2_gp_weight,
|
||||
"stage2_noise_dim": stage2_noise_dim,
|
||||
"stage2_critic_lr": stage2_critic_lr,
|
||||
"stage1_critic_hidden_dim": stage1_critic_hidden_dim,
|
||||
"stage1_critic_n_res_blocks": stage1_critic_n_res_blocks,
|
||||
"stage2_critic_hidden_dim": stage2_critic_hidden_dim,
|
||||
"stage2_critic_n_res_blocks": stage2_critic_n_res_blocks,
|
||||
}
|
||||
overrides = gconfig.overrides_from_flags(flag_values)
|
||||
|
||||
@@ -992,37 +941,26 @@ def predict(
|
||||
typer.echo(f"device: {_device}")
|
||||
|
||||
# --- Load checkpoint ---
|
||||
ckpt = torch.load(checkpoint, map_location="cpu", weights_only=False)
|
||||
if "model_config" not in ckpt:
|
||||
typer.echo(
|
||||
"error: checkpoint has no model_config — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
try:
|
||||
ctx = load_for_inference(checkpoint, _device, "predict", weights=weights.value)
|
||||
except CheckpointCompatibilityError as exc:
|
||||
typer.echo(f"error: {exc}", err=True)
|
||||
raise typer.Exit(1)
|
||||
typer.echo(f"loaded checkpoint: {checkpoint} (weights: {weights.value})")
|
||||
|
||||
if "sec_decoder" not in ckpt:
|
||||
typer.echo(
|
||||
"error: checkpoint has no sec_decoder — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
|
||||
if "sec_phys" not in ckpt.get("normalizer", {}):
|
||||
typer.echo(
|
||||
"error: checkpoint has no normalizer.sec_phys — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
|
||||
model_cfg = ckpt["model_config"]
|
||||
pdg_topn_map = _load_pdg_topn_map(ckpt)
|
||||
mat_topn_map = _load_mat_topn_map(ckpt)
|
||||
other_policy = _particle_type_other_policy(model_cfg)
|
||||
stage1_ddpm_steps = _ddpm_steps(model_cfg, "stage1")
|
||||
stage2_k_max = _stage_cfg(model_cfg, "stage2").get("k_max", K_MAX)
|
||||
assert ctx.stage1 is not None and ctx.stage2 is not None # require_stage2=True (default) guarantees this
|
||||
model, sec_decoder = ctx.stage1, ctx.stage2
|
||||
cond_norm, tgt_norm, sec_phys_norm = ctx.cond_norm, ctx.tgt_norm, ctx.sec_phys_norm
|
||||
pdg_map, mat_map = ctx.pdg_map, ctx.mat_map
|
||||
pdg_topn_map, mat_topn_map = ctx.pdg_topn_map, ctx.mat_topn_map
|
||||
sec_type_topn_map = ctx.sec_type_topn_map
|
||||
particle_conditioning, material_conditioning = ctx.particle_conditioning, ctx.material_conditioning
|
||||
other_policy = ctx.other_policy
|
||||
stage1_ddpm_steps = ctx.stage1_ddpm_steps
|
||||
stage2_k_max = ctx.k_max
|
||||
|
||||
if batch_size_auto:
|
||||
est_hidden_dim, est_n_blocks = _batch_size_estimate_dims(model_cfg, training=False)
|
||||
est_hidden_dim, est_n_blocks = _batch_size_estimate_dims(ctx.model_config, training=False)
|
||||
try:
|
||||
batch_size_value = gconfig.estimate_batch_size(
|
||||
est_hidden_dim,
|
||||
@@ -1037,42 +975,6 @@ def predict(
|
||||
|
||||
assert batch_size_value is not None
|
||||
bs = batch_size_value
|
||||
particle_conditioning, material_conditioning = _conditioning_axes(model_cfg)
|
||||
if particle_conditioning == "onehot" and pdg_topn_map is None:
|
||||
typer.echo(
|
||||
"error: checkpoint's conditioning.particle.type='onehot' but has "
|
||||
"no pdg_topn_map — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
if material_conditioning == "onehot" and mat_topn_map is None:
|
||||
typer.echo(
|
||||
"error: checkpoint's conditioning.material.type='onehot' but has "
|
||||
"no mat_topn_map — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
pdg_map = {int(k): v for k, v in ckpt["pdg_map"].items()}
|
||||
mat_map = {str(k): v for k, v in ckpt["mat_map"].items()}
|
||||
cond_norm = Normalizer.from_dict(ckpt["normalizer"]["cond"])
|
||||
tgt_norm = Normalizer.from_dict(ckpt["normalizer"]["target"])
|
||||
sec_phys_norm = Normalizer.from_dict(ckpt["normalizer"]["sec_phys"])
|
||||
|
||||
built = build_models(model_cfg)
|
||||
model, sec_decoder = built["stage1"], built["stage2"]
|
||||
if model is None or sec_decoder is None:
|
||||
typer.echo(
|
||||
"error: checkpoint has an inactive stage1 or stage2 — giant "
|
||||
"predict needs both (see stage{1,2}_model.active)",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
_load_model_weights(model, sec_decoder, ckpt, weights, checkpoint)
|
||||
model.to(_device).eval()
|
||||
sec_decoder.to(_device).eval()
|
||||
|
||||
typer.echo(f"loaded checkpoint: {checkpoint} (weights: {weights.value})")
|
||||
gconfig.warn_if_checkpoint_config_mismatch(checkpoint)
|
||||
|
||||
# --- Output path ---
|
||||
out, dataset_path, pred_uuid = _resolve_prediction_output(data, out)
|
||||
@@ -1094,8 +996,12 @@ def predict(
|
||||
return iter_file_chunks(path, offset=offset, k_max=stage2_k_max)
|
||||
return iter_cond_chunks(path, offset=offset)
|
||||
|
||||
cond_pdg_topn = pdg_topn_map.class_map if particle_conditioning == "onehot" else None
|
||||
cond_mat_topn = mat_topn_map.class_map if material_conditioning == "onehot" else None
|
||||
# load_for_inference already guarantees pdg_topn_map/mat_topn_map are not
|
||||
# None whenever the matching conditioning axis is "onehot" — the extra
|
||||
# `is not None` conjuncts below are redundant at runtime, just narrowing
|
||||
# for the type checker.
|
||||
cond_pdg_topn = pdg_topn_map.class_map if pdg_topn_map is not None and particle_conditioning == "onehot" else None
|
||||
cond_mat_topn = mat_topn_map.class_map if mat_topn_map is not None and material_conditioning == "onehot" else None
|
||||
|
||||
def _concat(a: dict[str, np.ndarray], b: dict[str, np.ndarray]) -> dict[str, np.ndarray]:
|
||||
return {k: np.concatenate([a[k], b[k]], axis=0) for k in a}
|
||||
@@ -1104,7 +1010,7 @@ def predict(
|
||||
nonlocal writer, total
|
||||
|
||||
if coord == Coord.local:
|
||||
cond_cont, cond_cat, target_raw, _, _, _, _, _, _ = build_features(
|
||||
feats = build_features(
|
||||
piece,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
@@ -1114,7 +1020,9 @@ def predict(
|
||||
mat_topn_map=cond_mat_topn,
|
||||
k_max=stage2_k_max,
|
||||
)
|
||||
cond_cont = cond_norm.transform(cond_cont)
|
||||
cond_cat = feats.cond_cat
|
||||
target_raw = feats.target_s1
|
||||
cond_cont = cond_norm.transform(feats.cond_cont)
|
||||
else:
|
||||
cond_cont, cond_cat = build_cond_features(
|
||||
piece,
|
||||
@@ -1198,7 +1106,7 @@ def predict(
|
||||
piece["pre_dir"],
|
||||
sec_phys_norm,
|
||||
pdg_map,
|
||||
pdg_topn_map,
|
||||
sec_type_topn_map,
|
||||
other_policy,
|
||||
None,
|
||||
)
|
||||
@@ -1405,65 +1313,25 @@ def rollout(
|
||||
_device = torch.device(device) if device else gconfig.auto_device()
|
||||
typer.echo(f"device: {_device}")
|
||||
|
||||
ckpt = torch.load(checkpoint, map_location="cpu", weights_only=False)
|
||||
for key in ("model_config", "sec_decoder"):
|
||||
if key not in ckpt:
|
||||
typer.echo(
|
||||
f"error: checkpoint has no {key} — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
|
||||
if "sec_phys" not in ckpt.get("normalizer", {}):
|
||||
typer.echo(
|
||||
"error: checkpoint has no normalizer.sec_phys — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
try:
|
||||
ctx = load_for_inference(checkpoint, _device, "rollout", weights=weights.value)
|
||||
except CheckpointCompatibilityError as exc:
|
||||
typer.echo(f"error: {exc}", err=True)
|
||||
raise typer.Exit(1)
|
||||
typer.echo(f"loaded checkpoint: {checkpoint} (weights: {weights.value})")
|
||||
|
||||
gconfig.warn_if_checkpoint_config_mismatch(checkpoint)
|
||||
training_cfg = gconfig.load_checkpoint_config(checkpoint)
|
||||
|
||||
model_cfg = ckpt["model_config"]
|
||||
particle_conditioning, material_conditioning = _conditioning_axes(model_cfg)
|
||||
pdg_topn_map = _load_pdg_topn_map(ckpt)
|
||||
mat_topn_map = _load_mat_topn_map(ckpt)
|
||||
if particle_conditioning == "onehot" and pdg_topn_map is None:
|
||||
typer.echo(
|
||||
"error: checkpoint's conditioning.particle.type='onehot' but has "
|
||||
"no pdg_topn_map — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
if material_conditioning == "onehot" and mat_topn_map is None:
|
||||
typer.echo(
|
||||
"error: checkpoint's conditioning.material.type='onehot' but has "
|
||||
"no mat_topn_map — retrain with the current code",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
other_policy = _particle_type_other_policy(model_cfg)
|
||||
stage1_ddpm_steps = _ddpm_steps(model_cfg, "stage1")
|
||||
stage2_ddpm_steps = _ddpm_steps(model_cfg, "stage2")
|
||||
pdg_map = {int(k): v for k, v in ckpt["pdg_map"].items()}
|
||||
mat_map = {str(k): v for k, v in ckpt["mat_map"].items()}
|
||||
cond_norm = Normalizer.from_dict(ckpt["normalizer"]["cond"])
|
||||
tgt_norm = Normalizer.from_dict(ckpt["normalizer"]["target"])
|
||||
sec_phys_norm = Normalizer.from_dict(ckpt["normalizer"]["sec_phys"])
|
||||
|
||||
built = build_models(model_cfg)
|
||||
model, sec_decoder = built["stage1"], built["stage2"]
|
||||
if model is None or sec_decoder is None:
|
||||
typer.echo(
|
||||
"error: checkpoint has an inactive stage1 or stage2 — giant "
|
||||
"rollout needs both (see stage{1,2}_model.active)",
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
_load_model_weights(model, sec_decoder, ckpt, weights, checkpoint)
|
||||
model.to(_device).eval()
|
||||
sec_decoder.to(_device).eval()
|
||||
typer.echo(f"loaded checkpoint: {checkpoint} (weights: {weights.value})")
|
||||
assert ctx.stage1 is not None and ctx.stage2 is not None # require_stage2=True (default) guarantees this
|
||||
model, sec_decoder = ctx.stage1, ctx.stage2
|
||||
cond_norm, tgt_norm, sec_phys_norm = ctx.cond_norm, ctx.tgt_norm, ctx.sec_phys_norm
|
||||
pdg_map, mat_map = ctx.pdg_map, ctx.mat_map
|
||||
pdg_topn_map, mat_topn_map = ctx.pdg_topn_map, ctx.mat_topn_map
|
||||
sec_type_topn_map = ctx.sec_type_topn_map
|
||||
particle_conditioning, material_conditioning = ctx.particle_conditioning, ctx.material_conditioning
|
||||
other_policy = ctx.other_policy
|
||||
stage1_ddpm_steps, stage2_ddpm_steps = ctx.stage1_ddpm_steps, ctx.stage2_ddpm_steps
|
||||
model_cfg = ctx.model_config
|
||||
|
||||
oracle = GeometryOracle.load(geometry)
|
||||
typer.echo(f"loaded geometry oracle: {geometry} (escape_threshold={oracle.escape_threshold:.3f})")
|
||||
@@ -1520,6 +1388,7 @@ def rollout(
|
||||
material_conditioning=material_conditioning,
|
||||
pdg_topn_map=pdg_topn_map,
|
||||
mat_topn_map=mat_topn_map,
|
||||
sec_type_topn_map=sec_type_topn_map,
|
||||
other_policy=other_policy,
|
||||
seed=seed,
|
||||
stage1_ddpm_steps=stage1_ddpm_steps,
|
||||
@@ -1560,8 +1429,8 @@ def rollout(
|
||||
# model knob (router type/n_experts, noise_dim, vocab sizes, ...)
|
||||
# is available downstream without touching this command again.
|
||||
"model_config": dict(model_cfg),
|
||||
"training_epoch": ckpt.get("epoch"),
|
||||
"best_val_loss": ckpt.get("best_val_loss"),
|
||||
"training_epoch": ctx.epoch,
|
||||
"best_val_loss": ctx.best_val_loss,
|
||||
# [train]/[meta] from the sibling config.toml (giant.config.save_config)
|
||||
# — empty dicts if the checkpoint has no config.toml next to it.
|
||||
"training_config": dict(training_cfg.get("train", {})),
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
"""Single source of truth for the conditioning arrays' column layout (gitea #37).
|
||||
|
||||
`cond_cont` and `cond_cat` are built in `giant.data.transforms` and consumed in
|
||||
`giant.model.encoders` / `giant.model.routers`. Their column order used to be
|
||||
written down independently on each side, kept in sync only by parallel comments
|
||||
— so getting it wrong produced silently mis-indexed columns rather than an
|
||||
exception, and adding a conditioning axis meant a coordinated multi-file edit.
|
||||
|
||||
`CondLayout` owns that order. Both sides construct one from the same
|
||||
`conditioning.particle.type` / `conditioning.material.type` pair and read named
|
||||
slices off it, so the layout is stated exactly once. This module depends only on
|
||||
`giant.constants`, so both the data and model packages can import it.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import ClassVar
|
||||
|
||||
from giant.constants import COND_DIM, COND_DIM_BASE, MATERIAL_PHYS_DIM, PARTICLE_PHYS_DIM
|
||||
|
||||
# The three per-axis conditioning modes. Mirrors giant.config.Conditioning,
|
||||
# which this module deliberately does not import (giant.config pulls in the
|
||||
# whole model package).
|
||||
AXIS_TYPES = ("physical", "embedding", "onehot")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CondLayout:
|
||||
"""Column layout of `cond_cont`/`cond_cat` for one (particle, material) mode pair.
|
||||
|
||||
`cond_cont` is unconditionally `COND_DIM` wide regardless of mode: the base
|
||||
block, then the particle physical block, then the material physical block.
|
||||
An axis that isn't `"physical"` gets its block zero-filled and never reads
|
||||
it (see `giant.data.transforms._physical_cond_columns`), so the widths are
|
||||
mode-independent and only the *meaning* of a block changes.
|
||||
|
||||
`cond_cat` is 2 to 4 wide. Columns `PDG_COL`/`MAT_COL` are always the dense
|
||||
training-vocab index; an axis in `"onehot"` mode appends one more column
|
||||
holding its top-N-plus-other class index, particle before material.
|
||||
"""
|
||||
|
||||
particle_type: str
|
||||
material_type: str
|
||||
|
||||
# cond_cat's dense-vocab columns, present in every mode. Under
|
||||
# "physical"/"onehot" they are a reporting/router convenience the
|
||||
# ConditionEncoder never reads; under "embedding" they are the signal.
|
||||
PDG_COL: ClassVar[int] = 0
|
||||
MAT_COL: ClassVar[int] = 1
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.particle_type not in AXIS_TYPES:
|
||||
raise ValueError(f"unknown conditioning.particle.type {self.particle_type!r}")
|
||||
if self.material_type not in AXIS_TYPES:
|
||||
raise ValueError(f"unknown conditioning.material.type {self.material_type!r}")
|
||||
|
||||
@classmethod
|
||||
def from_types(cls, particle_type: str, material_type: str) -> "CondLayout":
|
||||
"""Named constructor — the entry point both sides use."""
|
||||
return cls(particle_type=particle_type, material_type=material_type)
|
||||
|
||||
# --- cond_cont ---------------------------------------------------------
|
||||
|
||||
@property
|
||||
def base(self) -> slice:
|
||||
"""pre_pos(3), log(pre_E)(1), pre_dir(3), layer_id(1)."""
|
||||
return slice(0, COND_DIM_BASE)
|
||||
|
||||
@property
|
||||
def particle_phys(self) -> slice:
|
||||
"""log(mass), charge — see `giant.particles`."""
|
||||
return slice(COND_DIM_BASE, COND_DIM_BASE + PARTICLE_PHYS_DIM)
|
||||
|
||||
@property
|
||||
def material_phys(self) -> slice:
|
||||
"""Z_eff, A_eff, log(density), log(X0), log(lambda_int) — see `giant.materials`."""
|
||||
start = COND_DIM_BASE + PARTICLE_PHYS_DIM
|
||||
return slice(start, start + MATERIAL_PHYS_DIM)
|
||||
|
||||
@property
|
||||
def cont_dim(self) -> int:
|
||||
return COND_DIM
|
||||
|
||||
# --- cond_cat ----------------------------------------------------------
|
||||
|
||||
@property
|
||||
def particle_topn_col(self) -> int | None:
|
||||
"""Column of the particle top-N class index, or `None` if not `"onehot"`."""
|
||||
return self.MAT_COL + 1 if self.particle_type == "onehot" else None
|
||||
|
||||
@property
|
||||
def material_topn_col(self) -> int | None:
|
||||
"""Column of the material top-N class index, or `None` if not `"onehot"`.
|
||||
|
||||
Comes after the particle top-N column when both axes are `"onehot"`.
|
||||
"""
|
||||
if self.material_type != "onehot":
|
||||
return None
|
||||
return self.MAT_COL + (2 if self.particle_type == "onehot" else 1)
|
||||
|
||||
@property
|
||||
def cat_dim(self) -> int:
|
||||
"""Total `cond_cat` width: 2, plus one column per `"onehot"` axis."""
|
||||
return self.MAT_COL + 1 + (self.particle_type == "onehot") + (self.material_type == "onehot")
|
||||
+163
-57
@@ -14,10 +14,13 @@ from pathlib import Path
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from giant._migration import V02_FIXED_FACTS, V02_MODEL_KEY_TO_STAGES, reject_legacy_router_expert_sizing
|
||||
from giant.model.history import HISTORY_REGISTRY
|
||||
|
||||
|
||||
class Conditioning(str, Enum):
|
||||
"""`conditioning.particle.type` / `conditioning.material.type` choices —
|
||||
shared by `giant.cli` and `scripts.dwarf`'s Typer commands so the two
|
||||
shared by `giant.cli` and `giant.tools.dwarf`'s Typer commands so the two
|
||||
CLIs can't silently drift apart on the option's valid values (see
|
||||
DEFAULT_CONFIG["conditioning"] for what each value means)."""
|
||||
|
||||
@@ -47,13 +50,11 @@ CONFIG_VERSION = 3
|
||||
# `lambda` is a Python keyword, so dict key "lambda" is always exposed as the
|
||||
# field `lambda_weight`.
|
||||
#
|
||||
# Two sub-blocks — router and n_sec — carry genuinely dynamic keys that don't
|
||||
# fit a fixed schema: composed-router `axis{i}_{field}` flags (see
|
||||
# giant.model.network._parse_composed_axes) and pipeline.py's runtime-seeded
|
||||
# `centers_init`, plus n_sec's `legacy_owner` (injected only by
|
||||
# _migrate_legacy_model_config for v0.2 checkpoints). Both dataclasses carry
|
||||
# an `extra: dict` catch-all so these keys round-trip losslessly without
|
||||
# becoming named fields that would leak into every new run's config.toml.
|
||||
# The router sub-block carries genuinely dynamic keys that don't fit a fixed schema:
|
||||
# composed-router `axis{i}_{field}` flags (see giant.model.network._parse_composed_axes)
|
||||
# and pipeline.py's runtime-seeded `centers_init`. It carries an `extra: dict` catch-all
|
||||
# so these keys round-trip losslessly without becoming named fields that would leak into
|
||||
# every new run's config.toml.
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -336,6 +337,35 @@ class RouterConfig:
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrunkConfig:
|
||||
"""`stage1_model.trunk`/`stage2_model.trunk`: selects the trunk's expert
|
||||
*body* architecture from `giant.model.trunks.TRUNK_REGISTRY` (default
|
||||
`"resmlp"` — today's only body, `input_proj -> ResBlock stack ->
|
||||
out_proj`). Orthogonal to whether that body is mixed: mixing is still
|
||||
controlled entirely by `router.enabled`/`router.n_experts` on the same
|
||||
stage, unaffected by this block. A future body's own hyperparameters
|
||||
(e.g. a transformer's `n_heads`/`n_layers`) would get their own sibling
|
||||
field here, matching how `flow`/`ddpm`/`wgan` already coexist selected by
|
||||
`generator`.
|
||||
|
||||
`block_conditioning` selects each body's conditioning-injection mechanism
|
||||
from `giant.model.layers.BLOCK_REGISTRY` — `"add"` (default, today's
|
||||
conditional-bias `ResBlock`, bit-identical to pre-gitea-#34 behaviour),
|
||||
`"film"`, or `"adaln"`."""
|
||||
|
||||
type: str = "resmlp"
|
||||
block_conditioning: str = "add"
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "TrunkConfig":
|
||||
d = d or {}
|
||||
return cls(type=d.get("type", "resmlp"), block_conditioning=d.get("block_conditioning", "add"))
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"type": self.type, "block_conditioning": self.block_conditioning}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Stage2RouterConfig(RouterConfig):
|
||||
# true: stage 2 shares stage 1's Router module instance, so expert i in
|
||||
@@ -384,26 +414,24 @@ class NSecConfig:
|
||||
# only, never for rollout.
|
||||
mode: str = "head"
|
||||
lambda_weight: float = 0.1 # dict key "lambda" — cross-entropy weight for the head
|
||||
# Holds "legacy_owner" when injected by _migrate_legacy_model_config
|
||||
# (v0.2 checkpoints only) — not a user-facing config.toml key.
|
||||
extra: dict = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def legacy_owner(self) -> str | None:
|
||||
return self.extra.get("legacy_owner")
|
||||
# Which stage's module physically owns the n_sec_head weights: "stage2" (default,
|
||||
# fresh v0.3.0 runs — Stage2OneShot/Stage2Autoregressive builds it) or "stage1"
|
||||
# (a migrated v0.2 checkpoint — see network._migrate_legacy_model_config, whose
|
||||
# n_sec head was trained against Stage 1's own ConditionEncoder output and so has
|
||||
# to stay attached there, not just be labeled as such).
|
||||
owner: str = "stage2"
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "NSecConfig":
|
||||
d = d or {}
|
||||
known = {"mode", "lambda"}
|
||||
return cls(
|
||||
mode=d.get("mode", "head"),
|
||||
lambda_weight=d.get("lambda", 0.1),
|
||||
extra={k: v for k, v in d.items() if k not in known},
|
||||
owner=d.get("owner", "stage2"),
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"mode": self.mode, "lambda": self.lambda_weight, **self.extra}
|
||||
return {"mode": self.mode, "lambda": self.lambda_weight, "owner": self.owner}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -420,6 +448,11 @@ class ParticleTypeConfig:
|
||||
# at map-build time. "modal": always the most common member. "drop":
|
||||
# discard the secondary. Read only under target = "onehot".
|
||||
other_policy: str = "sample"
|
||||
# Secondary-species class count under target = "onehot" — independent of
|
||||
# conditioning.particle.emb_dim (see gitea #29: the two used to be
|
||||
# silently the same number). 0 = inherit conditioning.particle.emb_dim,
|
||||
# preserving pre-#29 behavior.
|
||||
n_classes: int = 0
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "ParticleTypeConfig":
|
||||
@@ -428,10 +461,16 @@ class ParticleTypeConfig:
|
||||
target=d.get("target", "onehot"),
|
||||
lambda_weight=d.get("lambda", 1.0),
|
||||
other_policy=d.get("other_policy", "sample"),
|
||||
n_classes=d.get("n_classes", 0),
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"target": self.target, "lambda": self.lambda_weight, "other_policy": self.other_policy}
|
||||
return {
|
||||
"target": self.target,
|
||||
"lambda": self.lambda_weight,
|
||||
"other_policy": self.other_policy,
|
||||
"n_classes": self.n_classes,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -477,6 +516,64 @@ class AutoregressiveConfig:
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HeadConfig:
|
||||
"""A single classifier head's shape — `n_sec_head`/`type_head` (gitea
|
||||
#36 deduplicated their five identical hand-rolled
|
||||
`Linear -> SiLU -> Linear` definitions into
|
||||
`giant.model.layers.build_mlp_head`, which this config drives).
|
||||
`hidden_ratio=0.5`/`depth=2` are the exact pre-#36 hardcoded values
|
||||
(hidden width = `hidden_dim // 2`, one hidden layer), so omitting a
|
||||
`heads` block — including every migrated v0.2 config — reproduces the
|
||||
old architecture bit-for-bit."""
|
||||
|
||||
hidden_ratio: float = 0.5 # hidden width = round(hidden_dim * hidden_ratio)
|
||||
depth: int = 2 # matches build_mlp_head's depth
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "HeadConfig":
|
||||
d = d or {}
|
||||
return cls(hidden_ratio=d.get("hidden_ratio", 0.5), depth=d.get("depth", 2))
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"hidden_ratio": self.hidden_ratio, "depth": self.depth}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Stage1HeadsConfig:
|
||||
"""Stage 1 only ever owns `n_sec_head`, and only for a migrated v0.2
|
||||
checkpoint (`stage2_model.n_sec.owner = "stage1"`) — see
|
||||
`Stage1Model`'s docstring."""
|
||||
|
||||
n_sec: HeadConfig = field(default_factory=HeadConfig)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "Stage1HeadsConfig":
|
||||
d = d or {}
|
||||
return cls(n_sec=HeadConfig.from_dict(d.get("n_sec")))
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"n_sec": self.n_sec.to_dict()}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Stage2HeadsConfig:
|
||||
"""`n_sec` and `type` are independently configurable — n_sec accuracy
|
||||
and secondary-species accuracy are separately known weak spots (gitea
|
||||
#36)."""
|
||||
|
||||
n_sec: HeadConfig = field(default_factory=HeadConfig)
|
||||
type: HeadConfig = field(default_factory=HeadConfig)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "Stage2HeadsConfig":
|
||||
d = d or {}
|
||||
return cls(n_sec=HeadConfig.from_dict(d.get("n_sec")), type=HeadConfig.from_dict(d.get("type")))
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"n_sec": self.n_sec.to_dict(), "type": self.type.to_dict()}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Stage1ModelConfig:
|
||||
# false skips building/training stage 1 entirely. The resulting
|
||||
@@ -500,6 +597,8 @@ class Stage1ModelConfig:
|
||||
ddpm: DdpmConfig = field(default_factory=DdpmConfig)
|
||||
wgan: Stage1WganConfig = field(default_factory=Stage1WganConfig)
|
||||
router: RouterConfig = field(default_factory=RouterConfig)
|
||||
trunk: TrunkConfig = field(default_factory=TrunkConfig)
|
||||
heads: Stage1HeadsConfig = field(default_factory=Stage1HeadsConfig)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "Stage1ModelConfig":
|
||||
@@ -515,6 +614,8 @@ class Stage1ModelConfig:
|
||||
ddpm=DdpmConfig.from_dict(d.get("ddpm")),
|
||||
wgan=Stage1WganConfig.from_dict(d.get("wgan")),
|
||||
router=RouterConfig.from_dict(d.get("router")),
|
||||
trunk=TrunkConfig.from_dict(d.get("trunk")),
|
||||
heads=Stage1HeadsConfig.from_dict(d.get("heads")),
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
@@ -529,6 +630,8 @@ class Stage1ModelConfig:
|
||||
"ddpm": self.ddpm.to_dict(),
|
||||
"wgan": self.wgan.to_dict(),
|
||||
"router": self.router.to_dict(),
|
||||
"trunk": self.trunk.to_dict(),
|
||||
"heads": self.heads.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
@@ -566,6 +669,8 @@ class Stage2ModelConfig:
|
||||
ddpm: DdpmConfig = field(default_factory=DdpmConfig)
|
||||
wgan: Stage2WganConfig = field(default_factory=Stage2WganConfig)
|
||||
router: Stage2RouterConfig = field(default_factory=Stage2RouterConfig)
|
||||
trunk: TrunkConfig = field(default_factory=TrunkConfig)
|
||||
heads: Stage2HeadsConfig = field(default_factory=Stage2HeadsConfig)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict | None) -> "Stage2ModelConfig":
|
||||
@@ -588,6 +693,8 @@ class Stage2ModelConfig:
|
||||
ddpm=DdpmConfig.from_dict(d.get("ddpm")),
|
||||
wgan=Stage2WganConfig.from_dict(d.get("wgan")),
|
||||
router=Stage2RouterConfig.from_dict(d.get("router")),
|
||||
trunk=TrunkConfig.from_dict(d.get("trunk")),
|
||||
heads=Stage2HeadsConfig.from_dict(d.get("heads")),
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
@@ -609,6 +716,8 @@ class Stage2ModelConfig:
|
||||
"ddpm": self.ddpm.to_dict(),
|
||||
"wgan": self.wgan.to_dict(),
|
||||
"router": self.router.to_dict(),
|
||||
"trunk": self.trunk.to_dict(),
|
||||
"heads": self.heads.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
@@ -981,6 +1090,13 @@ FLAG_SPECS: tuple[FlagSpec, ...] = (
|
||||
FlagSpec("critic_lr", ("stage1_model.wgan.critic_lr", "stage2_model.wgan.critic_lr"), precedence=0),
|
||||
FlagSpec("stage1_critic_lr", ("stage1_model.wgan.critic_lr",), precedence=1),
|
||||
FlagSpec("stage2_critic_lr", ("stage2_model.wgan.critic_lr",), precedence=1),
|
||||
# Critic sizing: stage-scoped only, no shared alias — this is an
|
||||
# architectural per-stage knob like hidden_dim/n_res_blocks above, not a
|
||||
# shared training hyperparameter like the wgan knobs above it.
|
||||
FlagSpec("stage1_critic_hidden_dim", ("stage1_model.wgan.critic_hidden_dim",)),
|
||||
FlagSpec("stage1_critic_n_res_blocks", ("stage1_model.wgan.critic_n_res_blocks",)),
|
||||
FlagSpec("stage2_critic_hidden_dim", ("stage2_model.wgan.critic_hidden_dim",)),
|
||||
FlagSpec("stage2_critic_n_res_blocks", ("stage2_model.wgan.critic_n_res_blocks",)),
|
||||
)
|
||||
|
||||
|
||||
@@ -1024,14 +1140,6 @@ _V02_TRAIN_PASSTHROUGH = (
|
||||
"wandb_log_every",
|
||||
)
|
||||
|
||||
# v0.2 model.hidden_dim/n_blocks/dropout applied identically to both stages
|
||||
# (there was only ever one trunk shape) -> copied to both stage{1,2}_model.
|
||||
_V02_MODEL_TO_BOTH_STAGES = (
|
||||
("hidden_dim", "hidden_dim"),
|
||||
("n_blocks", "n_res_blocks"),
|
||||
("dropout", "dropout"),
|
||||
)
|
||||
|
||||
# v0.2 train.{n_critic,gp_weight,critic_lr} applied identically to both
|
||||
# stages' wgan sub-table (there was only ever one wgan objective, shared).
|
||||
_V02_TRAIN_TO_BOTH_STAGES_WGAN = (
|
||||
@@ -1091,7 +1199,7 @@ def migrate_config(cfg: dict) -> dict:
|
||||
_set_path(new, f"stage1_model.wgan.{new_key}", old_train[old_key])
|
||||
_set_path(new, f"stage2_model.wgan.{new_key}", old_train[old_key])
|
||||
|
||||
for old_key, new_key in _V02_MODEL_TO_BOTH_STAGES:
|
||||
for old_key, new_key in V02_MODEL_KEY_TO_STAGES:
|
||||
if old_key in old_model:
|
||||
_set_path(new, f"stage1_model.{new_key}", old_model[old_key])
|
||||
_set_path(new, f"stage2_model.{new_key}", old_model[old_key])
|
||||
@@ -1108,18 +1216,7 @@ def migrate_config(cfg: dict) -> dict:
|
||||
_set_path(new, "stage2_model.k_max", old_model["k_max"])
|
||||
|
||||
if old_router:
|
||||
expert_hidden_dim = old_router.pop("expert_hidden_dim", 0)
|
||||
expert_n_blocks = old_router.pop("expert_n_blocks", 0)
|
||||
if expert_hidden_dim or expert_n_blocks:
|
||||
raise ValueError(
|
||||
"v0.2 config sets model.router.expert_hidden_dim/"
|
||||
f"expert_n_blocks to a non-default value "
|
||||
f"({expert_hidden_dim!r}, {expert_n_blocks!r}); v0.3.0 removed "
|
||||
"per-expert sizing (experts always inherit the stage's "
|
||||
"hidden_dim/n_res_blocks), so this config's routed experts "
|
||||
"have a different width/depth than the monolith and its "
|
||||
"checkpoint can only be loaded by v0.2 code."
|
||||
)
|
||||
reject_legacy_router_expert_sizing(old_router, source="v0.2 config's model.router")
|
||||
_set_path(new, "stage1_model.router", dict(old_router))
|
||||
stage2_router = dict(old_router)
|
||||
stage2_router["tie_to_stage1"] = False
|
||||
@@ -1127,21 +1224,9 @@ def migrate_config(cfg: dict) -> dict:
|
||||
|
||||
# v0.2 architectural facts with no corresponding config key at all —
|
||||
# always set once we've determined we're migrating a v0.2 dict,
|
||||
# independent of what the file did/didn't specify. NOTE: n_layers here
|
||||
# (2) differs from the v0.3 *default* (1) — this is not a typo, see the
|
||||
# docstring above.
|
||||
_set_path(new, "conditioning.out_dim", 128)
|
||||
_set_path(new, "conditioning.particle.n_layers", 2)
|
||||
_set_path(new, "conditioning.material.n_layers", 2)
|
||||
_set_path(new, "stage1_model.active", True)
|
||||
_set_path(new, "stage1_model.flow.time_dim", 64)
|
||||
_set_path(new, "stage1_model.ddpm.time_dim", 64)
|
||||
_set_path(new, "stage2_model.active", True)
|
||||
_set_path(new, "stage2_model.flow.time_dim", 64)
|
||||
_set_path(new, "stage2_model.ddpm.time_dim", 64)
|
||||
_set_path(new, "stage2_model.context_dim", 64)
|
||||
_set_path(new, "stage2_model.decoder", "one_shot")
|
||||
_set_path(new, "stage2_model.particle_type.target", "physical")
|
||||
# independent of what the file did/didn't specify (see giant._migration).
|
||||
for path, value in V02_FIXED_FACTS.items():
|
||||
_set_path(new, path, value)
|
||||
|
||||
new_meta = dict(cfg.pop("meta", {}))
|
||||
new_meta["config_version"] = CONFIG_VERSION
|
||||
@@ -1289,6 +1374,14 @@ def validate_config(cfg: dict) -> None:
|
||||
"(standalone stage-2 evaluation only, never for rollout)"
|
||||
)
|
||||
|
||||
if _get_path(cfg, "stage2_model.stage1_context") == "sampled":
|
||||
raise ValueError(
|
||||
"stage2_model.stage1_context = 'sampled' is accepted by the schema "
|
||||
"but not implemented — trainers.py always trains stage 2 against "
|
||||
"the ground-truth stage-1 output; use 'truth' (default) instead "
|
||||
"(see issues.md Issue 16 for the planned implementation)"
|
||||
)
|
||||
|
||||
if (
|
||||
_get_path(cfg, "stage2_model.n_sec.mode") == "truth"
|
||||
and _get_path(cfg, "stage1_model.active")
|
||||
@@ -1305,9 +1398,18 @@ def validate_config(cfg: dict) -> None:
|
||||
)
|
||||
|
||||
if _get_path(cfg, "stage2_model.decoder") == "autoregressive":
|
||||
order = _get_path(cfg, "stage2_model.autoregressive.order")
|
||||
if order != "energy_desc":
|
||||
raise ValueError(
|
||||
f"stage2_model.autoregressive.order = {order!r} — must be "
|
||||
"'energy_desc' (the only implemented ordering; see "
|
||||
"AutoregressiveConfig.order's docstring)"
|
||||
)
|
||||
history = _get_path(cfg, "stage2_model.autoregressive.history")
|
||||
if history not in ("markov", "attention"):
|
||||
raise ValueError(f"stage2_model.autoregressive.history = {history!r} — must be 'markov' or 'attention'")
|
||||
if history not in HISTORY_REGISTRY:
|
||||
raise ValueError(
|
||||
f"stage2_model.autoregressive.history = {history!r} — must be one of {sorted(HISTORY_REGISTRY)}"
|
||||
)
|
||||
teacher_forcing = _get_path(cfg, "stage2_model.autoregressive.teacher_forcing")
|
||||
if teacher_forcing not in ("always", "scheduled", "never"):
|
||||
raise ValueError(
|
||||
@@ -1400,6 +1502,10 @@ _OUT_DIR_NAME_CANDIDATES = [
|
||||
"particle_type_target",
|
||||
_path_candidate("stage2_model.particle_type.target", "pt-"),
|
||||
),
|
||||
("stage1_trunk_type", _path_candidate("stage1_model.trunk.type", "s1t-")),
|
||||
("stage2_trunk_type", _path_candidate("stage2_model.trunk.type", "s2t-")),
|
||||
("stage1_block_cond", _path_candidate("stage1_model.trunk.block_conditioning", "s1bc-")),
|
||||
("stage2_block_cond", _path_candidate("stage2_model.trunk.block_conditioning", "s2bc-")),
|
||||
("stage1_router", _router_candidate("stage1_model", "s1")),
|
||||
("stage2_router", _router_candidate("stage2_model", "s2")),
|
||||
(
|
||||
|
||||
+50
-45
@@ -1,6 +1,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import NamedTuple
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
@@ -11,6 +12,37 @@ from giant.data.loader import event_id_offset, iter_file_chunks
|
||||
from giant.data.transforms import Normalizer, build_features, sorted_membership
|
||||
|
||||
|
||||
class StepBatch(NamedTuple):
|
||||
"""One training batch, as yielded by `StreamingStepsDataset`. Field order
|
||||
is load-bearing for existing positional unpacking elsewhere (`trainers.py`,
|
||||
`validate.py`, test fixtures) — append only, never insert or reorder.
|
||||
|
||||
cond_cont: (B, COND_DIM) float32
|
||||
cond_cat: (B, 2/3/4) int64 — width 2 unless conditioning="onehot"
|
||||
target_s1: (B, 9) float32 — normalised Stage-1 primary target
|
||||
n_sec: (B,) int64 — true secondary count per step
|
||||
sec_cont: (B, k_max, SEC_SLOT_DIM) float32 — [stick_logit,
|
||||
local_dir, log_mass, charge] per slot (mass/charge
|
||||
normalised iff `sec_phys_normalizer` was given); always
|
||||
computed the same way regardless of
|
||||
stage2_model.particle_type.target, only actually used
|
||||
downstream under target="physical"
|
||||
proc_idx: (B,) int64 — process-class label (ProcessRouter supervision
|
||||
only; zeros when `proc_map` is None)
|
||||
sec_type_idx: (B, k_max) int64 — per-slot class index into
|
||||
`sec_type_class_map`, for particle_type.target in
|
||||
("onehot", "embedding"); zeros (unused) otherwise
|
||||
"""
|
||||
|
||||
cond_cont: torch.Tensor
|
||||
cond_cat: torch.Tensor
|
||||
target_s1: torch.Tensor
|
||||
n_sec: torch.Tensor
|
||||
sec_cont: torch.Tensor
|
||||
proc_idx: torch.Tensor
|
||||
sec_type_idx: torch.Tensor
|
||||
|
||||
|
||||
def make_event_split(
|
||||
all_event_ids: np.ndarray,
|
||||
val_fraction: float = 0.1,
|
||||
@@ -39,24 +71,7 @@ class StreamingStepsDataset(IterableDataset):
|
||||
rather than single rows, so the batch is assembled with vectorized
|
||||
numpy slicing instead of a per-row Python loop in the default collate.
|
||||
|
||||
Each batch is a tuple:
|
||||
(cond_cont, cond_cat, target_s1, n_sec, sec_cont, proc_idx, sec_type_idx)
|
||||
where:
|
||||
cond_cont: (B, COND_DIM) float32
|
||||
cond_cat: (B, 2/3/4) int64 — width 2 unless conditioning="onehot"
|
||||
target_s1: (B, 9) float32 — normalised Stage-1 primary target
|
||||
n_sec: (B,) int64 — true secondary count per step
|
||||
sec_cont: (B, k_max, SEC_SLOT_DIM) float32 — [stick_logit,
|
||||
local_dir, log_mass, charge] per slot (mass/charge
|
||||
normalised iff `sec_phys_normalizer` was given); always
|
||||
computed the same way regardless of
|
||||
stage2_model.particle_type.target, only actually used
|
||||
downstream under target="physical"
|
||||
proc_idx: (B,) int64 — process-class label (ProcessRouter supervision
|
||||
only; zeros when `proc_map` is None)
|
||||
sec_type_idx: (B, k_max) int64 — per-slot class index into
|
||||
`sec_type_class_map`, for particle_type.target in
|
||||
("onehot", "embedding"); zeros (unused) otherwise
|
||||
Each batch is a `StepBatch` — see its docstring for field meanings.
|
||||
|
||||
`k_max` (constructor arg, default the module constant) should match
|
||||
`stage2_model.k_max` — it sets the padded
|
||||
@@ -129,17 +144,7 @@ class StreamingStepsDataset(IterableDataset):
|
||||
continue
|
||||
chunk = {k: v[mask] for k, v in chunk.items()}
|
||||
|
||||
(
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
target_s1,
|
||||
n_sec,
|
||||
sec_cont,
|
||||
proc_idx,
|
||||
sec_type_idx,
|
||||
_,
|
||||
_,
|
||||
) = build_features(
|
||||
feats = build_features(
|
||||
chunk,
|
||||
self.pdg_map,
|
||||
self.mat_map,
|
||||
@@ -155,14 +160,14 @@ class StreamingStepsDataset(IterableDataset):
|
||||
sec_type_class_map=self.sec_type_class_map,
|
||||
k_max=self.k_max,
|
||||
)
|
||||
buf_cont.append(cond_cont)
|
||||
buf_cat.append(cond_cat)
|
||||
buf_tgt.append(target_s1)
|
||||
buf_nsec.append(n_sec)
|
||||
buf_sec.append(sec_cont)
|
||||
buf_proc.append(proc_idx)
|
||||
buf_type.append(sec_type_idx)
|
||||
buf_n += len(cond_cont)
|
||||
buf_cont.append(feats.cond_cont)
|
||||
buf_cat.append(feats.cond_cat)
|
||||
buf_tgt.append(feats.target_s1)
|
||||
buf_nsec.append(feats.n_sec)
|
||||
buf_sec.append(feats.sec_cont)
|
||||
buf_proc.append(feats.proc_idx)
|
||||
buf_type.append(feats.sec_type_idx)
|
||||
buf_n += len(feats.cond_cont)
|
||||
|
||||
if buf_n >= self.shuffle_buffer:
|
||||
(
|
||||
@@ -226,14 +231,14 @@ class StreamingStepsDataset(IterableDataset):
|
||||
n_full = n // bs if not final else (n + bs - 1) // bs
|
||||
for start in range(0, n_full * bs, bs):
|
||||
end = min(start + bs, n)
|
||||
yield (
|
||||
torch.from_numpy(cont[start:end]).float(),
|
||||
torch.from_numpy(cat[start:end]).long(),
|
||||
torch.from_numpy(tgt[start:end]).float(),
|
||||
torch.from_numpy(nsec[start:end]).long(),
|
||||
torch.from_numpy(sec[start:end]).float(),
|
||||
torch.from_numpy(proc[start:end]).long(),
|
||||
torch.from_numpy(styp[start:end]).long(),
|
||||
yield StepBatch(
|
||||
cond_cont=torch.from_numpy(cont[start:end]).float(),
|
||||
cond_cat=torch.from_numpy(cat[start:end]).long(),
|
||||
target_s1=torch.from_numpy(tgt[start:end]).float(),
|
||||
n_sec=torch.from_numpy(nsec[start:end]).long(),
|
||||
sec_cont=torch.from_numpy(sec[start:end]).float(),
|
||||
proc_idx=torch.from_numpy(proc[start:end]).long(),
|
||||
sec_type_idx=torch.from_numpy(styp[start:end]).long(),
|
||||
)
|
||||
|
||||
if final:
|
||||
|
||||
@@ -16,7 +16,7 @@ from giant.constants import K_MAX
|
||||
MANIFEST_SUFFIX = ".manifest"
|
||||
|
||||
# Each input parquet file is a separate Geant4 job converted 1:1 from its own
|
||||
# ROOT file (scripts/steps_to_parquet.py), and a job's event_id numbering
|
||||
# ROOT file (giant/tools/steps_to_parquet.py), and a job's event_id numbering
|
||||
# always restarts from 0 — so when multiple files are loaded together (a
|
||||
# directory or .manifest), raw event_id values collide across files even
|
||||
# though they refer to unrelated events. Every per-file event_id column gets
|
||||
|
||||
+129
-129
@@ -1,7 +1,9 @@
|
||||
import warnings
|
||||
from typing import NamedTuple
|
||||
|
||||
import numpy as np
|
||||
|
||||
from giant.cond_layout import CondLayout
|
||||
from giant.constants import K_MAX
|
||||
|
||||
_EPS = 1e-8
|
||||
@@ -692,18 +694,13 @@ def decode_secondaries(
|
||||
return sec_E, sec_dir_world, sec_mass, sec_charge, sec_valid
|
||||
|
||||
|
||||
def _physical_cond_columns(
|
||||
data: dict[str, np.ndarray],
|
||||
particle_conditioning: str,
|
||||
material_conditioning: str,
|
||||
) -> np.ndarray:
|
||||
def _physical_cond_columns(data: dict[str, np.ndarray], layout: CondLayout) -> np.ndarray:
|
||||
"""(N, PARTICLE_PHYS_DIM + MATERIAL_PHYS_DIM) physical conditioning columns.
|
||||
|
||||
The particle and material blocks are gated independently and may mix
|
||||
freely — e.g. material `physical` with particle `embedding` — so e.g.
|
||||
`particle_conditioning="embedding"` + `material_conditioning="physical"`
|
||||
zero-fills only the particle columns and computes the material ones for
|
||||
real.
|
||||
`particle_type="embedding"` + `material_type="physical"` zero-fills only
|
||||
the particle columns and computes the material ones for real.
|
||||
|
||||
"embedding"/"onehot" zero-fill their block (cheap, and ConditionEncoder
|
||||
never reads these columns in either mode — so an unfilled
|
||||
@@ -720,7 +717,7 @@ def _physical_cond_columns(
|
||||
|
||||
n = len(next(iter(data.values())))
|
||||
|
||||
if particle_conditioning == "physical":
|
||||
if layout.particle_type == "physical":
|
||||
from giant.particles import particle_phys_array
|
||||
|
||||
if "mass" in data and "charge" in data:
|
||||
@@ -729,12 +726,10 @@ def _physical_cond_columns(
|
||||
else:
|
||||
mass, charge = particle_phys_array(data["pdg"]).T
|
||||
particle_cols = np.column_stack([log_transform(mass), charge])
|
||||
elif particle_conditioning in ("embedding", "onehot"):
|
||||
particle_cols = np.zeros((n, PARTICLE_PHYS_DIM), dtype=np.float32)
|
||||
else:
|
||||
raise ValueError(f"unknown conditioning.particle.type {particle_conditioning!r}")
|
||||
particle_cols = np.zeros((n, PARTICLE_PHYS_DIM), dtype=np.float32)
|
||||
|
||||
if material_conditioning == "physical":
|
||||
if layout.material_type == "physical":
|
||||
from giant.materials import material_properties_array
|
||||
|
||||
z_eff, a_eff, density, x0, lambda_int = material_properties_array(data["material"]).T
|
||||
@@ -747,14 +742,66 @@ def _physical_cond_columns(
|
||||
log_transform(lambda_int),
|
||||
]
|
||||
)
|
||||
elif material_conditioning in ("embedding", "onehot"):
|
||||
material_cols = np.zeros((n, MATERIAL_PHYS_DIM), dtype=np.float32)
|
||||
else:
|
||||
raise ValueError(f"unknown conditioning.material.type {material_conditioning!r}")
|
||||
material_cols = np.zeros((n, MATERIAL_PHYS_DIM), dtype=np.float32)
|
||||
|
||||
return np.column_stack([particle_cols, material_cols]).astype(np.float32)
|
||||
|
||||
|
||||
def _build_cond_arrays(
|
||||
data: dict[str, np.ndarray],
|
||||
pdg_map: dict[int, int],
|
||||
mat_map: dict[str, int],
|
||||
layout: CondLayout,
|
||||
pdg_topn_map: dict[int, int] | None,
|
||||
mat_topn_map: dict[str, int] | None,
|
||||
) -> tuple[np.ndarray, np.ndarray]:
|
||||
"""The un-normalized `(cond_cont, cond_cat)` pair, in `layout`'s column order.
|
||||
|
||||
Both `build_cond_features` and `build_features` go through here, so the
|
||||
column order — and everything that depends on it — is stated once. See
|
||||
`giant.cond_layout.CondLayout` for the layout itself.
|
||||
"""
|
||||
cond_cont = np.column_stack(
|
||||
[
|
||||
data["pre_pos"],
|
||||
log_transform(data["pre_E"]),
|
||||
data["pre_dir"],
|
||||
data["layer_id"].astype(np.float32),
|
||||
]
|
||||
).astype(np.float32) # (N, COND_DIM_BASE=8)
|
||||
cond_cont = np.column_stack([cond_cont, _physical_cond_columns(data, layout)]).astype(
|
||||
np.float32
|
||||
) # (N, COND_DIM=15)
|
||||
|
||||
# In "physical" mode cond_cat's first two columns are only a
|
||||
# reporting/router convenience — ConditionEncoder never reads them
|
||||
# (giant/model/encoders.py) — so a species/material outside the training
|
||||
# vocab (the whole point of physical-property conditioning) gets a dummy
|
||||
# index instead of raising. In "embedding" mode those columns ARE the
|
||||
# conditioning signal, so an unmapped value must still raise loudly
|
||||
# rather than silently misassign. In "onehot" mode they again go unread
|
||||
# (the topN columns below are the real signal), so they're as permissive
|
||||
# as "physical". Each axis's strictness is independent.
|
||||
pdg_idx = _vectorized_map_lookup(data["pdg"], pdg_map, strict=layout.particle_type == "embedding")
|
||||
mat_idx = _vectorized_map_lookup(data["material"], mat_map, strict=layout.material_type == "embedding")
|
||||
# Which extra columns exist is the layout's call, not "did the caller
|
||||
# happen to pass a map" — that's what used to let the producer and
|
||||
# ConditionEncoder disagree. A map for a non-"onehot" axis is unused.
|
||||
cat_cols = [pdg_idx, mat_idx]
|
||||
if layout.particle_topn_col is not None:
|
||||
if pdg_topn_map is None:
|
||||
raise ValueError("conditioning.particle.type='onehot' needs pdg_topn_map")
|
||||
cat_cols.append(_vectorized_map_lookup(data["pdg"], pdg_topn_map))
|
||||
if layout.material_topn_col is not None:
|
||||
if mat_topn_map is None:
|
||||
raise ValueError("conditioning.material.type='onehot' needs mat_topn_map")
|
||||
cat_cols.append(_vectorized_map_lookup(data["material"], mat_topn_map))
|
||||
cond_cat = np.column_stack(cat_cols) # (N, layout.cat_dim)
|
||||
|
||||
return cond_cont, cond_cat
|
||||
|
||||
|
||||
def build_cond_features(
|
||||
data: dict[str, np.ndarray],
|
||||
pdg_map: dict[int, int],
|
||||
@@ -772,49 +819,16 @@ def build_cond_features(
|
||||
`material_conditioning="physical"` is a valid mix.
|
||||
|
||||
`pdg_topn_map`/`mat_topn_map` (a top-N-plus-other `class_map`, see
|
||||
`giant.data.loader.build_topn_map_from_files`) append extra `cond_cat`
|
||||
columns read by `ConditionEncoder`'s `"onehot"` mode: pdg topN index at
|
||||
column 2 (iff `pdg_topn_map` given), material topN index at column 3
|
||||
(iff `mat_topn_map` given, after column 2 if both are). Only ever given when
|
||||
the corresponding axis is `"onehot"`; `cond_cat` stays `(N, 2)` otherwise.
|
||||
`giant.data.loader.build_topn_map_from_files`) supply the extra `cond_cat`
|
||||
columns read by `ConditionEncoder`'s `"onehot"` mode, and are required
|
||||
whenever the corresponding axis is `"onehot"`. See
|
||||
`giant.cond_layout.CondLayout` for which columns exist where.
|
||||
"""
|
||||
cond_cont = np.column_stack(
|
||||
[
|
||||
data["pre_pos"],
|
||||
log_transform(data["pre_E"]),
|
||||
data["pre_dir"],
|
||||
data["layer_id"].astype(np.float32),
|
||||
]
|
||||
).astype(np.float32)
|
||||
cond_cont = np.column_stack(
|
||||
[
|
||||
cond_cont,
|
||||
_physical_cond_columns(data, particle_conditioning, material_conditioning),
|
||||
]
|
||||
).astype(np.float32)
|
||||
|
||||
# In "physical" mode cond_cat's first two columns are only a
|
||||
# reporting/router convenience — ConditionEncoder never reads them
|
||||
# (giant/model/network.py) — so a species/material outside the training
|
||||
# vocab (the whole point of physical-property conditioning) gets a dummy
|
||||
# index instead of raising. In "embedding" mode those columns ARE the
|
||||
# conditioning signal, so an unmapped value must still raise loudly
|
||||
# rather than silently misassign. In "onehot" mode they again go unread
|
||||
# (the topN columns below are the real signal), so they're as permissive
|
||||
# as "physical". Each axis's strictness is independent.
|
||||
pdg_strict = particle_conditioning == "embedding"
|
||||
mat_strict = material_conditioning == "embedding"
|
||||
pdg_idx = _vectorized_map_lookup(data["pdg"], pdg_map, strict=pdg_strict)
|
||||
mat_idx = _vectorized_map_lookup(data["material"], mat_map, strict=mat_strict)
|
||||
cat_cols = [pdg_idx, mat_idx]
|
||||
if pdg_topn_map is not None:
|
||||
cat_cols.append(_vectorized_map_lookup(data["pdg"], pdg_topn_map))
|
||||
if mat_topn_map is not None:
|
||||
cat_cols.append(_vectorized_map_lookup(data["material"], mat_topn_map))
|
||||
cond_cat = np.column_stack(cat_cols)
|
||||
layout = CondLayout.from_types(particle_conditioning, material_conditioning)
|
||||
cond_cont, cond_cat = _build_cond_arrays(data, pdg_map, mat_map, layout, pdg_topn_map, mat_topn_map)
|
||||
|
||||
if cond_normalizer is not None:
|
||||
cond_cont = _cond_normalizer_transform(cond_cont, cond_normalizer, particle_conditioning, material_conditioning)
|
||||
cond_cont = _cond_normalizer_transform(cond_cont, cond_normalizer, layout)
|
||||
|
||||
return cond_cont, cond_cat
|
||||
|
||||
@@ -822,8 +836,7 @@ def build_cond_features(
|
||||
def _cond_normalizer_transform(
|
||||
cond_cont: np.ndarray,
|
||||
cond_normalizer: "Normalizer",
|
||||
particle_conditioning: str,
|
||||
material_conditioning: str,
|
||||
layout: CondLayout,
|
||||
) -> np.ndarray:
|
||||
"""Apply ``cond_normalizer``, padding a legacy narrower normalizer if needed.
|
||||
|
||||
@@ -831,7 +844,7 @@ def _cond_normalizer_transform(
|
||||
8->15, ``giant/constants.py``) saved a ``COND_DIM_BASE``-wide (8) cond
|
||||
normalizer, fit before ``build_cond_features`` grew the extra physical
|
||||
columns. When NEITHER axis is "physical" those columns are never read by
|
||||
``ConditionEncoder`` (``giant/model/network.py``), so padding the missing
|
||||
``ConditionEncoder`` (``giant/model/encoders.py``), so padding the missing
|
||||
entries with mean=0/std=1 is a safe no-op that keeps such checkpoints
|
||||
usable under the current, always-``COND_DIM``-wide contract. If EITHER
|
||||
axis is "physical" its columns are load-bearing, so a mismatch there is a
|
||||
@@ -842,14 +855,14 @@ def _cond_normalizer_transform(
|
||||
width = cond_cont.shape[-1]
|
||||
if mean.shape[-1] < width:
|
||||
physical_load_bearing = "physical" in (
|
||||
particle_conditioning,
|
||||
material_conditioning,
|
||||
layout.particle_type,
|
||||
layout.material_type,
|
||||
)
|
||||
if physical_load_bearing:
|
||||
raise ValueError(
|
||||
f"cond normalizer has {mean.shape[-1]} columns, expected "
|
||||
f"{width}, and particle_conditioning={particle_conditioning!r}/"
|
||||
f"material_conditioning={material_conditioning!r} reads the "
|
||||
f"{width}, and particle_conditioning={layout.particle_type!r}/"
|
||||
f"material_conditioning={layout.material_type!r} reads the "
|
||||
"physical columns directly — this checkpoint predates "
|
||||
"physical-property conditioning and can't be safely padded; "
|
||||
"retrain it under the current code."
|
||||
@@ -860,6 +873,41 @@ def _cond_normalizer_transform(
|
||||
return ((cond_cont - mean) / std).astype(np.float32)
|
||||
|
||||
|
||||
class StepFeatures(NamedTuple):
|
||||
"""Output of `build_features`. Field order is load-bearing for existing
|
||||
positional unpacking (tests, `StreamingStepsDataset`) — append only,
|
||||
never insert or reorder.
|
||||
|
||||
target_s1: (N, 9) Stage-1 primary post-step target (unchanged from Phase 1)
|
||||
n_sec: (N,) integer secondary counts (target for n_sec head)
|
||||
sec_cont: (N, K_MAX, SEC_SLOT_DIM=6) continuous secondary targets
|
||||
[stick_logit, dir_local, log_mass, charge] — mass/charge are
|
||||
the secondary's real physical identity (from its ground-truth
|
||||
PDG code), a fixed regression target, not a learned/snapped one.
|
||||
Always computed the same way regardless of
|
||||
`stage2_model.particle_type.target` — only actually used
|
||||
downstream under `target = "physical"`.
|
||||
proc_idx: (N,) integer process-class label (ProcessRouter supervision only —
|
||||
never conditioning). Zeros when `proc_map` is None or the loaded
|
||||
data has no "process" column (e.g. pre-conversion parquet files).
|
||||
sec_type_idx: (N, K_MAX) integer secondary class index into
|
||||
`sec_type_class_map`, for `stage2_model.particle_type.target`
|
||||
in `("onehot", "embedding")` — see `encode_secondary_type_idx`.
|
||||
Zero-filled (and unused) when `sec_type_class_map` is None
|
||||
(i.e. `target = "physical"`).
|
||||
"""
|
||||
|
||||
cond_cont: np.ndarray
|
||||
cond_cat: np.ndarray
|
||||
target_s1: np.ndarray
|
||||
n_sec: np.ndarray
|
||||
sec_cont: np.ndarray
|
||||
proc_idx: np.ndarray
|
||||
sec_type_idx: np.ndarray
|
||||
cond_normalizer: Normalizer | None
|
||||
target_normalizer: Normalizer | None
|
||||
|
||||
|
||||
def build_features(
|
||||
data: dict[str, np.ndarray],
|
||||
pdg_map: dict[int, int],
|
||||
@@ -877,37 +925,10 @@ def build_features(
|
||||
mat_topn_map: dict[str, int] | None = None,
|
||||
sec_type_class_map: dict | None = None,
|
||||
k_max: int = K_MAX,
|
||||
) -> tuple[
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
np.ndarray,
|
||||
Normalizer | None,
|
||||
Normalizer | None,
|
||||
]:
|
||||
"""Assemble (cond_cont, cond_cat, target_s1, n_sec, sec_cont, proc_idx,
|
||||
sec_type_idx) arrays.
|
||||
|
||||
target_s1: (N, 9) Stage-1 primary post-step target (unchanged from Phase 1)
|
||||
n_sec: (N,) integer secondary counts (target for n_sec head)
|
||||
sec_cont: (N, K_MAX, SEC_SLOT_DIM=6) continuous secondary targets
|
||||
[stick_logit, dir_local, log_mass, charge] — mass/charge are
|
||||
the secondary's real physical identity (from its ground-truth
|
||||
PDG code), a fixed regression target, not a learned/snapped one.
|
||||
Always computed the same way regardless of
|
||||
`stage2_model.particle_type.target` — only actually used
|
||||
downstream under `target = "physical"`.
|
||||
proc_idx: (N,) integer process-class label (ProcessRouter supervision only —
|
||||
never conditioning). Zeros when `proc_map` is None or the loaded
|
||||
data has no "process" column (e.g. pre-conversion parquet files).
|
||||
sec_type_idx: (N, K_MAX) integer secondary class index into
|
||||
`sec_type_class_map`, for `stage2_model.particle_type.target`
|
||||
in `("onehot", "embedding")` — see `encode_secondary_type_idx`.
|
||||
Zero-filled (and unused) when `sec_type_class_map` is None
|
||||
(i.e. `target = "physical"`).
|
||||
) -> StepFeatures:
|
||||
"""Assemble a `StepFeatures` of (cond_cont, cond_cat, target_s1, n_sec,
|
||||
sec_cont, proc_idx, sec_type_idx, cond_normalizer, target_normalizer) —
|
||||
see `StepFeatures` for field meanings.
|
||||
|
||||
require_secondaries: when True, raise if any step has n_sec > 0 but the
|
||||
per-secondary list columns are absent (a mis-converted file that would
|
||||
@@ -919,7 +940,7 @@ def build_features(
|
||||
instead) for callers (normalizer fitting) that only read
|
||||
`sec_cont[:, :, 4:6]` and would otherwise discard that work.
|
||||
|
||||
pdg_topn_map/mat_topn_map: appended `cond_cat` columns for
|
||||
pdg_topn_map/mat_topn_map: source of the extra `cond_cat` columns for
|
||||
`ConditionEncoder`'s `"onehot"` mode — see `build_cond_features`.
|
||||
|
||||
sec_type_class_map: the map `sec_type_idx` is looked up against — a
|
||||
@@ -950,29 +971,8 @@ def build_features(
|
||||
).astype(np.float32) # (N, 9)
|
||||
|
||||
# Phase 2: conditioning drops n_sec and log(e_sec)
|
||||
cond_cont = np.column_stack(
|
||||
[
|
||||
data["pre_pos"],
|
||||
log_transform(data["pre_E"]),
|
||||
data["pre_dir"],
|
||||
data["layer_id"].astype(np.float32),
|
||||
]
|
||||
).astype(np.float32) # (N, COND_DIM_BASE=8)
|
||||
cond_cont = np.column_stack(
|
||||
[
|
||||
cond_cont,
|
||||
_physical_cond_columns(data, particle_conditioning, material_conditioning),
|
||||
]
|
||||
).astype(np.float32) # (N, COND_DIM=15)
|
||||
|
||||
pdg_idx = _vectorized_map_lookup(data["pdg"], pdg_map)
|
||||
mat_idx = _vectorized_map_lookup(data["material"], mat_map)
|
||||
cat_cols = [pdg_idx, mat_idx]
|
||||
if pdg_topn_map is not None:
|
||||
cat_cols.append(_vectorized_map_lookup(data["pdg"], pdg_topn_map))
|
||||
if mat_topn_map is not None:
|
||||
cat_cols.append(_vectorized_map_lookup(data["material"], mat_topn_map))
|
||||
cond_cat = np.column_stack(cat_cols) # (N, 2/3/4)
|
||||
layout = CondLayout.from_types(particle_conditioning, material_conditioning)
|
||||
cond_cont, cond_cat = _build_cond_arrays(data, pdg_map, mat_map, layout, pdg_topn_map, mat_topn_map)
|
||||
|
||||
n_sec_raw = data["n_sec"].astype(np.int64) # (N,) unclamped, for the valid-slot mask
|
||||
|
||||
@@ -1039,7 +1039,7 @@ def build_features(
|
||||
target_normalizer = Normalizer().fit(target_s1)
|
||||
|
||||
if cond_normalizer is not None:
|
||||
cond_cont = cond_normalizer.transform(cond_cont)
|
||||
cond_cont = _cond_normalizer_transform(cond_cont, cond_normalizer, layout)
|
||||
if target_normalizer is not None:
|
||||
target_s1 = target_normalizer.transform(target_s1)
|
||||
if sec_phys_normalizer is not None:
|
||||
@@ -1054,14 +1054,14 @@ def build_features(
|
||||
else:
|
||||
proc_idx = np.zeros(len(cond_cat), dtype=np.int64)
|
||||
|
||||
return (
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
target_s1,
|
||||
n_sec,
|
||||
sec_cont,
|
||||
proc_idx,
|
||||
sec_type_idx,
|
||||
cond_normalizer,
|
||||
target_normalizer,
|
||||
return StepFeatures(
|
||||
cond_cont=cond_cont,
|
||||
cond_cat=cond_cat,
|
||||
target_s1=target_s1,
|
||||
n_sec=n_sec,
|
||||
sec_cont=sec_cont,
|
||||
proc_idx=proc_idx,
|
||||
sec_type_idx=sec_type_idx,
|
||||
cond_normalizer=cond_normalizer,
|
||||
target_normalizer=target_normalizer,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
"""v0.2 -> v0.3 checkpoint migration: translates a v0.2 checkpoint's flat
|
||||
`model_config`/state dicts into the current nested shape (issues.md Issue 8;
|
||||
see also `giant._migration` and `giant.config.migrate_config`, the sibling
|
||||
config.toml migration surface — issues.md Issue 6)."""
|
||||
|
||||
from giant._migration import V02_FIXED_FACTS, reject_legacy_router_expert_sizing
|
||||
from giant.constants import EMB_DIM, K_MAX
|
||||
|
||||
|
||||
def _migrate_legacy_model_config(model_config: dict) -> dict:
|
||||
"""Translate a v0.2 checkpoint's flat `model_config` (giant/pipeline.py's
|
||||
old shape: `hidden_dim`/`n_blocks`/`emb_dim`/`dropout`/`conditioning`/
|
||||
`router`/`mode`/... all at one level) into the nested
|
||||
`{"pdg_vocab", "mat_vocab", "conditioning", "stage1_model",
|
||||
"stage2_model"}` shape `build_models` expects.
|
||||
|
||||
Sets `stage2_model.n_sec.owner = "stage1"` so the n_sec_head weights a v0.2
|
||||
checkpoint carries on its Stage-1 module keep loading there instead of the new
|
||||
default location (`Stage2OneShot`) — the n_sec head was trained against Stage 1's
|
||||
own `ConditionEncoder` output, so it has to stay attached to Stage 1's module, not
|
||||
just be labeled as such.
|
||||
|
||||
Only the monolithic (non-routed) trunk shape is exercised by the step-2
|
||||
migration test; a routed v0.2 checkpoint still builds correctly here
|
||||
(the router config passes through), but its
|
||||
state dict isn't covered by `migrate_legacy_state_dict` below.
|
||||
"""
|
||||
m = model_config
|
||||
conditioning_mode = m.get("conditioning", "embedding")
|
||||
generator = m.get("mode", "flow")
|
||||
hidden_dim = m.get("hidden_dim", 256)
|
||||
n_blocks = m.get("n_blocks", 6)
|
||||
emb_dim = m.get("emb_dim", EMB_DIM)
|
||||
dropout = m.get("dropout", 0.1)
|
||||
k_max = m.get("k_max", K_MAX)
|
||||
noise_dim = m.get("noise_dim", 64)
|
||||
router_cfg = dict(m.get("router") or {})
|
||||
reject_legacy_router_expert_sizing(router_cfg, source="this checkpoint's model_config.router")
|
||||
router_cfg.setdefault("enabled", False)
|
||||
|
||||
F = V02_FIXED_FACTS
|
||||
cond_n_layers = F["conditioning.particle.n_layers"] # same fact for both axes
|
||||
return {
|
||||
"pdg_vocab": m["pdg_vocab"],
|
||||
"mat_vocab": m["mat_vocab"],
|
||||
"conditioning": {
|
||||
"out_dim": F["conditioning.out_dim"],
|
||||
"share_stages": False,
|
||||
"particle": {"type": conditioning_mode, "emb_dim": emb_dim, "n_layers": cond_n_layers},
|
||||
"material": {"type": conditioning_mode, "emb_dim": emb_dim, "n_layers": cond_n_layers},
|
||||
},
|
||||
"stage1_model": {
|
||||
"active": F["stage1_model.active"],
|
||||
"generator": generator,
|
||||
"hidden_dim": hidden_dim,
|
||||
"n_res_blocks": n_blocks,
|
||||
"dropout": dropout,
|
||||
"flow": {"time_dim": F["stage1_model.flow.time_dim"]},
|
||||
"ddpm": {"time_dim": F["stage1_model.ddpm.time_dim"]},
|
||||
"wgan": {"noise_dim": noise_dim},
|
||||
"router": dict(router_cfg),
|
||||
},
|
||||
"stage2_model": {
|
||||
"active": F["stage2_model.active"],
|
||||
"decoder": F["stage2_model.decoder"],
|
||||
"generator": generator,
|
||||
"hidden_dim": hidden_dim,
|
||||
"n_res_blocks": n_blocks,
|
||||
"dropout": dropout,
|
||||
"k_max": k_max,
|
||||
"context_dim": F["stage2_model.context_dim"],
|
||||
"n_sec": {"mode": "head", "owner": "stage1"},
|
||||
"particle_type": {"target": F["stage2_model.particle_type.target"]},
|
||||
"flow": {"time_dim": F["stage2_model.flow.time_dim"]},
|
||||
"ddpm": {"time_dim": F["stage2_model.ddpm.time_dim"]},
|
||||
"wgan": {"noise_dim": noise_dim},
|
||||
"router": {**router_cfg, "tie_to_stage1": False},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def migrate_legacy_state_dict(old_stage1_sd: dict, old_stage2_sd: dict) -> tuple[dict, dict]:
|
||||
"""Remap a v0.2 checkpoint's (`DenoisingMLP`-or-`WGANGenerator`,
|
||||
`SecondaryDecoder`-or-`WGANSecondaryGenerator`) state dicts onto the new
|
||||
`(Stage1Model, Stage2OneShot)` module structure produced by
|
||||
`build_models(_migrate_legacy_model_config(model_config))`.
|
||||
|
||||
Only the monolithic (non-routed) trunk shape is handled.
|
||||
"""
|
||||
|
||||
def _trunk_prefix(k: str) -> str:
|
||||
if k.startswith(("input_proj.", "blocks.", "out_proj.")):
|
||||
return f"trunk.{k}"
|
||||
return k
|
||||
|
||||
new_stage1 = {}
|
||||
for k, v in old_stage1_sd.items():
|
||||
if k.startswith("n_sec_head."):
|
||||
new_stage1[k] = v # stays top-level (n_sec.owner="stage1")
|
||||
else:
|
||||
new_stage1[_trunk_prefix(k)] = v
|
||||
|
||||
new_stage2 = {}
|
||||
for k, v in old_stage2_sd.items():
|
||||
if k.startswith("cond_enc.base."):
|
||||
new_stage2["cond_enc." + k[len("cond_enc.base.") :]] = v
|
||||
elif k.startswith("cond_enc.stage1_proj."):
|
||||
new_stage2["context_adapter.proj." + k[len("cond_enc.stage1_proj.") :]] = v
|
||||
elif k.startswith("cond_enc.fuse."):
|
||||
new_stage2["fuse." + k[len("cond_enc.fuse.") :]] = v
|
||||
else:
|
||||
new_stage2[_trunk_prefix(k)] = v
|
||||
|
||||
return new_stage1, new_stage2
|
||||
@@ -0,0 +1,226 @@
|
||||
"""Factories: `build_models`/`build_critics` assemble the top-level stage
|
||||
models from a config dict (issues.md Issue 8)."""
|
||||
|
||||
import torch.nn as nn
|
||||
|
||||
from giant.config import ConditioningConfig, Stage1ModelConfig, Stage2ModelConfig
|
||||
from giant.constants import X_DIM
|
||||
from giant.model._legacy import _migrate_legacy_model_config
|
||||
from giant.model.encoders import ConditionEncoder
|
||||
from giant.model.models import (
|
||||
CriticModel,
|
||||
Stage1Model,
|
||||
Stage2Autoregressive,
|
||||
Stage2OneShot,
|
||||
resolve_type_n_classes,
|
||||
stage2_trunk_sec_dim,
|
||||
)
|
||||
from giant.model.objectives import build_objective
|
||||
from giant.model.routers import Router, _build_router_from_cfg
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Factories
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def build_models(model_config: dict) -> dict[str, nn.Module | None]:
|
||||
"""Construct `{"stage1": ..., "stage2": ...}` from a config dict — either
|
||||
the new nested shape (has a `"stage1_model"` key, plus `"pdg_vocab"`/
|
||||
`"mat_vocab"`/`"conditioning"` at the top level) or a v0.2 checkpoint's
|
||||
flat `model_config`, auto-migrated via `_migrate_legacy_model_config`.
|
||||
|
||||
A stage is `None` in the result when that stage's `active = False`.
|
||||
`stage2_model.router.tie_to_stage1` shares stage 1's literal `Router`
|
||||
instance rather than building a second, independently-parameterized one
|
||||
(v0.2's actual — probably accidental — behaviour: two routers built from
|
||||
one config with no semantic relationship between them).
|
||||
|
||||
`conditioning.share_stages = true` builds one `ConditionEncoder`
|
||||
instance here and passes it to both stages (`Stage1Model`/`Stage2OneShot`/
|
||||
`Stage2Autoregressive`'s `cond_enc` param), instead of each stage
|
||||
building its own — halving the conditioning parameter count and forcing a
|
||||
common representation. `false` (default) keeps v0.2 behaviour:
|
||||
independent instances with identical config but independent weights.
|
||||
"""
|
||||
cfg = model_config if "stage1_model" in model_config else _migrate_legacy_model_config(model_config)
|
||||
pdg_vocab = cfg["pdg_vocab"]
|
||||
mat_vocab = cfg["mat_vocab"]
|
||||
conditioning_cfg = ConditioningConfig.from_dict(cfg["conditioning"])
|
||||
particle_cfg = conditioning_cfg.particle
|
||||
material_cfg = conditioning_cfg.material
|
||||
particle_conditioning = particle_cfg.type
|
||||
s1_spec = Stage1ModelConfig.from_dict(cfg["stage1_model"])
|
||||
s2_spec = Stage2ModelConfig.from_dict(cfg["stage2_model"])
|
||||
cond_out_dim = conditioning_cfg.out_dim
|
||||
shared_cond_enc: ConditionEncoder | None = None
|
||||
if conditioning_cfg.share_stages:
|
||||
shared_cond_enc = ConditionEncoder(pdg_vocab, mat_vocab, particle_cfg, material_cfg, out_dim=cond_out_dim)
|
||||
|
||||
result: dict[str, nn.Module | None] = {"stage1": None, "stage2": None}
|
||||
|
||||
stage1_router: Router | None = None
|
||||
if s1_spec.active:
|
||||
router_cfg = cfg["stage1_model"].get("router") or {}
|
||||
if s1_spec.router.enabled:
|
||||
stage1_router = _build_router_from_cfg(router_cfg, pdg_vocab, mat_vocab, particle_conditioning)
|
||||
generator = s1_spec.generator
|
||||
objective = build_objective(generator)
|
||||
# wgan has no time_dim concept (no diffusion/flow time variable) —
|
||||
# matches the pre-dataclass .get("time_dim", 64) fallback, which
|
||||
# always hit its default for a wgan sub-block too.
|
||||
time_dim = getattr(s1_spec, generator).time_dim if objective.needs_time else 64
|
||||
n_sec_owner = s2_spec.n_sec.owner
|
||||
n_sec_head_k_max = s2_spec.k_max if n_sec_owner == "stage1" else None
|
||||
result["stage1"] = Stage1Model(
|
||||
pdg_vocab=pdg_vocab,
|
||||
mat_vocab=mat_vocab,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=s1_spec.hidden_dim,
|
||||
n_res_blocks=s1_spec.n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
dropout=s1_spec.dropout,
|
||||
generator=generator,
|
||||
time_dim=time_dim,
|
||||
noise_dim=s1_spec.wgan.noise_dim,
|
||||
router=stage1_router,
|
||||
trunk_type=s1_spec.trunk.type,
|
||||
block_conditioning=s1_spec.trunk.block_conditioning,
|
||||
n_sec_head_k_max=n_sec_head_k_max,
|
||||
cond_enc=shared_cond_enc,
|
||||
n_sec_head_cfg=s1_spec.heads.n_sec.to_dict(),
|
||||
)
|
||||
|
||||
if s2_spec.active:
|
||||
decoder = s2_spec.decoder
|
||||
router_cfg = cfg["stage2_model"].get("router") or {}
|
||||
stage2_router: Router | None = None
|
||||
if s2_spec.router.enabled:
|
||||
if s2_spec.router.tie_to_stage1 and stage1_router is not None:
|
||||
stage2_router = stage1_router
|
||||
else:
|
||||
stage2_router = _build_router_from_cfg(router_cfg, pdg_vocab, mat_vocab, particle_conditioning)
|
||||
generator = s2_spec.generator
|
||||
objective = build_objective(generator)
|
||||
# wgan has no time_dim concept — see the matching comment in stage 1
|
||||
# above.
|
||||
time_dim = getattr(s2_spec, generator).time_dim if objective.needs_time else 64
|
||||
n_sec_owner = s2_spec.n_sec.owner
|
||||
k_max = s2_spec.k_max
|
||||
particle_type_cfg = s2_spec.particle_type
|
||||
|
||||
if decoder == "autoregressive":
|
||||
ar_cfg = s2_spec.autoregressive
|
||||
result["stage2"] = Stage2Autoregressive(
|
||||
pdg_vocab=pdg_vocab,
|
||||
mat_vocab=mat_vocab,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=s2_spec.hidden_dim,
|
||||
n_res_blocks=s2_spec.n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
context_dim=s2_spec.context_dim,
|
||||
dropout=s2_spec.dropout,
|
||||
generator=generator,
|
||||
time_dim=time_dim,
|
||||
noise_dim=s2_spec.wgan.noise_dim,
|
||||
k_max=k_max,
|
||||
router=stage2_router,
|
||||
trunk_type=s2_spec.trunk.type,
|
||||
block_conditioning=s2_spec.trunk.block_conditioning,
|
||||
build_n_sec_head=n_sec_owner != "stage1",
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
history=ar_cfg.history,
|
||||
attn_n_heads=ar_cfg.attn_n_heads,
|
||||
attn_n_layers=ar_cfg.attn_n_layers,
|
||||
cond_enc=shared_cond_enc,
|
||||
n_sec_head_cfg=s2_spec.heads.n_sec.to_dict(),
|
||||
type_head_cfg=s2_spec.heads.type.to_dict(),
|
||||
)
|
||||
else:
|
||||
sec_dim = stage2_trunk_sec_dim(
|
||||
particle_type_cfg, generator, k_max, resolve_type_n_classes(particle_type_cfg, particle_cfg.emb_dim)
|
||||
)
|
||||
result["stage2"] = Stage2OneShot(
|
||||
pdg_vocab=pdg_vocab,
|
||||
mat_vocab=mat_vocab,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=s2_spec.hidden_dim,
|
||||
n_res_blocks=s2_spec.n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
context_dim=s2_spec.context_dim,
|
||||
sec_dim=sec_dim,
|
||||
dropout=s2_spec.dropout,
|
||||
generator=generator,
|
||||
time_dim=time_dim,
|
||||
noise_dim=s2_spec.wgan.noise_dim,
|
||||
k_max=k_max,
|
||||
router=stage2_router,
|
||||
trunk_type=s2_spec.trunk.type,
|
||||
block_conditioning=s2_spec.trunk.block_conditioning,
|
||||
build_n_sec_head=n_sec_owner != "stage1",
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
cond_enc=shared_cond_enc,
|
||||
n_sec_head_cfg=s2_spec.heads.n_sec.to_dict(),
|
||||
type_head_cfg=s2_spec.heads.type.to_dict(),
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def build_critics(model_config: dict) -> dict[str, nn.Module | None]:
|
||||
"""Construct `{"stage1": ..., "stage2": ...}` critics for `generator =
|
||||
"wgan"` training. Training-only — never persisted for inference the way
|
||||
`build_models`'s pair is. `None` for a stage that's inactive or not
|
||||
WGAN."""
|
||||
cfg = model_config if "stage1_model" in model_config else _migrate_legacy_model_config(model_config)
|
||||
pdg_vocab = cfg["pdg_vocab"]
|
||||
mat_vocab = cfg["mat_vocab"]
|
||||
conditioning_cfg = ConditioningConfig.from_dict(cfg["conditioning"])
|
||||
particle_cfg = conditioning_cfg.particle
|
||||
material_cfg = conditioning_cfg.material
|
||||
cond_out_dim = conditioning_cfg.out_dim
|
||||
s1_spec = Stage1ModelConfig.from_dict(cfg["stage1_model"])
|
||||
s2_spec = Stage2ModelConfig.from_dict(cfg["stage2_model"])
|
||||
|
||||
result: dict[str, nn.Module | None] = {"stage1": None, "stage2": None}
|
||||
|
||||
if s1_spec.active and build_objective(s1_spec.generator).is_adversarial:
|
||||
result["stage1"] = CriticModel(
|
||||
pdg_vocab=pdg_vocab,
|
||||
mat_vocab=mat_vocab,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
in_dim=X_DIM,
|
||||
hidden_dim=s1_spec.wgan.critic_hidden_dim or s1_spec.hidden_dim,
|
||||
n_res_blocks=s1_spec.wgan.critic_n_res_blocks or s1_spec.n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
dropout=s1_spec.dropout,
|
||||
stage="stage1",
|
||||
)
|
||||
|
||||
if s2_spec.active and build_objective(s2_spec.generator).is_adversarial:
|
||||
k_max = s2_spec.k_max
|
||||
particle_type_cfg = s2_spec.particle_type
|
||||
in_dim = stage2_trunk_sec_dim(
|
||||
particle_type_cfg,
|
||||
s2_spec.generator,
|
||||
k_max,
|
||||
resolve_type_n_classes(particle_type_cfg, particle_cfg.emb_dim),
|
||||
)
|
||||
result["stage2"] = CriticModel(
|
||||
pdg_vocab=pdg_vocab,
|
||||
mat_vocab=mat_vocab,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
in_dim=in_dim,
|
||||
hidden_dim=s2_spec.wgan.critic_hidden_dim or s2_spec.hidden_dim,
|
||||
n_res_blocks=s2_spec.wgan.critic_n_res_blocks or s2_spec.n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
dropout=s2_spec.dropout,
|
||||
stage="stage2",
|
||||
context_dim=s2_spec.context_dim,
|
||||
)
|
||||
|
||||
return result
|
||||
@@ -0,0 +1,101 @@
|
||||
"""Conditioning encoder — fuses continuous conditioning with particle/material
|
||||
identity (issues.md Issue 8)."""
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
|
||||
from giant.cond_layout import CondLayout
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import COND_DIM, COND_DIM_BASE, MATERIAL_PHYS_DIM, PARTICLE_PHYS_DIM
|
||||
from giant.model.layers import _make_axis_mlp
|
||||
|
||||
|
||||
class ConditionEncoder(nn.Module):
|
||||
"""Fuses continuous conditioning with particle/material identity.
|
||||
|
||||
The particle and material axes are configured independently
|
||||
(`particle_cfg`/`material_cfg`, each a `ConditioningAxisConfig`) and may
|
||||
mix freely, e.g. material "physical" with particle "embedding". Three
|
||||
modes per axis:
|
||||
- "embedding": a learned `nn.Embedding` lookup, indexed by `cond_cat`'s
|
||||
dense training-vocab index. Memorizes the training menu.
|
||||
- "physical": an `n_layers`-deep MLP over the axis's raw physical
|
||||
properties (already present in `cond_cont`'s physical block — see
|
||||
giant.data.transforms.build_features), computable for any PDG code /
|
||||
material name rather than only ones seen in training.
|
||||
- "onehot": a fixed, unlearned one-hot vector over a top-N-plus-other
|
||||
class map (`giant.data.loader.build_topn_map_from_files`/
|
||||
`build_pdg_topn_map_from_files`), read from `cond_cat`'s extra
|
||||
top-N-index column(s).
|
||||
|
||||
Every column index/slice comes from `self.layout`
|
||||
(`giant.cond_layout.CondLayout`), the same object the feature builders
|
||||
lay the arrays out with, so the two sides cannot drift apart.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
cont_dim: int = COND_DIM,
|
||||
out_dim: int = 128,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.particle_cfg = particle_cfg
|
||||
self.material_cfg = material_cfg
|
||||
# Also validates both axis types — an unknown one raises here.
|
||||
self.layout = CondLayout.from_types(particle_cfg.type, material_cfg.type)
|
||||
|
||||
p_type = particle_cfg.type
|
||||
p_emb_dim = particle_cfg.emb_dim
|
||||
if p_type == "embedding":
|
||||
self.pdg_emb = nn.Embedding(pdg_vocab, p_emb_dim)
|
||||
elif p_type == "physical":
|
||||
self.particle_mlp = _make_axis_mlp(PARTICLE_PHYS_DIM, p_emb_dim, particle_cfg.n_layers)
|
||||
|
||||
m_type = material_cfg.type
|
||||
m_emb_dim = material_cfg.emb_dim
|
||||
if m_type == "embedding":
|
||||
self.mat_emb = nn.Embedding(mat_vocab, m_emb_dim)
|
||||
elif m_type == "physical":
|
||||
self.material_mlp = _make_axis_mlp(MATERIAL_PHYS_DIM, m_emb_dim, material_cfg.n_layers)
|
||||
|
||||
in_dim = COND_DIM_BASE + p_emb_dim + m_emb_dim
|
||||
self.mlp = nn.Sequential(
|
||||
nn.Linear(in_dim, out_dim),
|
||||
nn.SiLU(),
|
||||
nn.Linear(out_dim, out_dim),
|
||||
)
|
||||
|
||||
def _particle_embed(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor):
|
||||
p_type = self.particle_cfg.type
|
||||
if p_type == "embedding":
|
||||
return self.pdg_emb(cond_cat[:, self.layout.PDG_COL])
|
||||
if p_type == "physical":
|
||||
return self.particle_mlp(cond_cont[:, self.layout.particle_phys])
|
||||
assert self.layout.particle_topn_col is not None
|
||||
return F.one_hot(
|
||||
cond_cat[:, self.layout.particle_topn_col],
|
||||
num_classes=self.particle_cfg.emb_dim,
|
||||
).float()
|
||||
|
||||
def _material_embed(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor):
|
||||
m_type = self.material_cfg.type
|
||||
if m_type == "embedding":
|
||||
return self.mat_emb(cond_cat[:, self.layout.MAT_COL])
|
||||
if m_type == "physical":
|
||||
return self.material_mlp(cond_cont[:, self.layout.material_phys])
|
||||
assert self.layout.material_topn_col is not None
|
||||
return F.one_hot(
|
||||
cond_cat[:, self.layout.material_topn_col],
|
||||
num_classes=self.material_cfg.emb_dim,
|
||||
).float()
|
||||
|
||||
def forward(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
pdg_e = self._particle_embed(cond_cont, cond_cat)
|
||||
mat_e = self._material_embed(cond_cont, cond_cat)
|
||||
x = torch.cat([cond_cont[:, self.layout.base], pdg_e, mat_e], dim=-1)
|
||||
return self.mlp(x)
|
||||
@@ -0,0 +1,200 @@
|
||||
"""History encoders — stage-2 autoregressive only. Self-contained, no
|
||||
dependency on any other `giant.model` submodule (issues.md Issue 8), except
|
||||
for the `HISTORY_REGISTRY`/`build_history` factory, which mirrors
|
||||
`giant.model.routers`'s `Router`/`ROUTER_REGISTRY` pattern (gitea #35)."""
|
||||
|
||||
import inspect
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
class HistoryEncoder(nn.Module):
|
||||
"""Interface for stage-2 autoregressive per-token history summaries:
|
||||
`forward(feat, has_prev) -> (B, K, out_dim)`, a single parallel pass over
|
||||
a full (teacher-forced) token sequence — used by training. `MarkovHistory`
|
||||
and `AttentionHistory` are the two registered implementations (see
|
||||
`HISTORY_REGISTRY`/`build_history`). Inference (`giant/sample.py`)
|
||||
generates one token at a time and cannot afford `forward`'s per-step cost
|
||||
to be O(K) (attention would then be O(K^2) over a rollout's k_max loop),
|
||||
so this interface also declares `init_cache`/`step` for that incremental
|
||||
path, with working O(1) defaults here (`init_cache` -> `None`, `step` ->
|
||||
one `forward` call ignoring `cache`) — correct for any encoder whose
|
||||
per-step cost is already O(1) (i.e. it only ever looks at the previous
|
||||
token, not the full prefix), which is what `MarkovHistory` relies on.
|
||||
`AttentionHistory` overrides both with real incremental-cache versions,
|
||||
since its `forward` genuinely needs the full prefix."""
|
||||
|
||||
def forward(self, feat: torch.Tensor, has_prev: torch.Tensor) -> torch.Tensor:
|
||||
raise NotImplementedError
|
||||
|
||||
def init_cache(self) -> object:
|
||||
return None
|
||||
|
||||
def step(self, feat: torch.Tensor, has_prev: torch.Tensor, cache: object) -> tuple[torch.Tensor, object]:
|
||||
return self.forward(feat, has_prev), cache
|
||||
|
||||
|
||||
HISTORY_REGISTRY: dict[str, type[HistoryEncoder]] = {}
|
||||
|
||||
|
||||
def register_history(name: str):
|
||||
def decorator(cls: type[HistoryEncoder]) -> type[HistoryEncoder]:
|
||||
HISTORY_REGISTRY[name] = cls
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def build_history(name: str, in_dim: int, out_dim: int, **kwargs) -> HistoryEncoder:
|
||||
"""Factory: look up a `HistoryEncoder` subclass by name from the registry.
|
||||
|
||||
Every registered history type is fed the same `stage2_model.autoregressive`
|
||||
kwargs; kwargs not declared by that type's constructor are silently
|
||||
dropped, so per-type hyperparameters (e.g. `AttentionHistory`'s
|
||||
`n_heads`/`n_layers`) can coexist in one config without special-casing —
|
||||
mirrors `giant.model.routers.build_router`.
|
||||
"""
|
||||
if name not in HISTORY_REGISTRY:
|
||||
raise ValueError(f"unknown history type {name!r}; available: {sorted(HISTORY_REGISTRY)}")
|
||||
cls = HISTORY_REGISTRY[name]
|
||||
accepted = set(inspect.signature(cls.__init__).parameters) - {"self", "in_dim", "out_dim"}
|
||||
filtered = {k: v for k, v in kwargs.items() if k in accepted}
|
||||
return cls(in_dim, out_dim, **filtered)
|
||||
|
||||
|
||||
@register_history("markov")
|
||||
class MarkovHistory(HistoryEncoder):
|
||||
"""Summarizes the previous secondary's own `(energy_fraction, direction,
|
||||
type_representation)` through one small MLP — the "markov" history:
|
||||
token i+1 only ever sees token i plus the running scalars
|
||||
(`remaining_frac`/`slot_idx`, fused in separately by
|
||||
`Stage2Autoregressive._token_cond`), not the full prefix.
|
||||
|
||||
At slot 0 (`has_prev` False) substitutes a learned start vector rather
|
||||
than zeros — a reasonable default.
|
||||
"""
|
||||
|
||||
def __init__(self, in_dim: int, out_dim: int) -> None:
|
||||
super().__init__()
|
||||
self.start = nn.Parameter(torch.zeros(in_dim))
|
||||
self.mlp = nn.Sequential(nn.Linear(in_dim, out_dim), nn.SiLU())
|
||||
|
||||
def forward(self, feat: torch.Tensor, has_prev: torch.Tensor) -> torch.Tensor:
|
||||
start = self.start.view(1, 1, -1).expand_as(feat)
|
||||
x = torch.where(has_prev.unsqueeze(-1), feat, start)
|
||||
return self.mlp(x)
|
||||
|
||||
|
||||
class _CausalAttnBlock(nn.Module):
|
||||
"""One pre-norm causal self-attention block for `AttentionHistory`.
|
||||
|
||||
Exposes two forward paths that must agree (see
|
||||
`test_attention_history_step_matches_forward` in `tests/test_network.py`):
|
||||
`forward` — the full-sequence, causally-masked pass used for training;
|
||||
`step` — an incremental pass for inference, given the *pre-attention*
|
||||
normalized hidden states of every earlier position (`kv_cache`, i.e.
|
||||
`norm1(x)` for positions `< t`, not `x` itself). Caching `norm1(x)` rather
|
||||
than raw `x` is what makes `step` correct: this block's attention needs
|
||||
exactly that quantity as keys/values, and `LayerNorm` has no cross-position
|
||||
interaction, so recomputing it per position instead of caching it would
|
||||
still be correct but pointlessly repeat work. The *next* block's cache is
|
||||
built from a different sequence (this block's output), so each block owns
|
||||
an independent cache entry.
|
||||
"""
|
||||
|
||||
def __init__(self, dim: int, n_heads: int, dropout: float = 0.0) -> None:
|
||||
super().__init__()
|
||||
self.norm1 = nn.LayerNorm(dim)
|
||||
self.attn = nn.MultiheadAttention(dim, n_heads, dropout=dropout, batch_first=True)
|
||||
self.norm2 = nn.LayerNorm(dim)
|
||||
self.mlp = nn.Sequential(nn.Linear(dim, 4 * dim), nn.GELU(), nn.Linear(4 * dim, dim))
|
||||
|
||||
def forward(self, x: torch.Tensor, causal_mask: torch.Tensor) -> torch.Tensor:
|
||||
h = self.norm1(x)
|
||||
attn_out, _ = self.attn(h, h, h, attn_mask=causal_mask, need_weights=False)
|
||||
x = x + attn_out
|
||||
x = x + self.mlp(self.norm2(x))
|
||||
return x
|
||||
|
||||
def step(self, x_new: torch.Tensor, kv_cache: torch.Tensor | None) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
"""`x_new`: `(B, 1, dim)`, this position's input. `kv_cache`: `None`
|
||||
(first position) or `(B, T, dim)` — `norm1(x)` of every earlier
|
||||
position at this same block. Returns `(out, new_kv_cache)`, `out`
|
||||
being this position's block output (`(B, 1, dim)`, to feed the next
|
||||
block's `step`), `new_kv_cache` the same cache extended by this
|
||||
position (to reuse at this block's *next* `step` call)."""
|
||||
h_new = self.norm1(x_new)
|
||||
kv = h_new if kv_cache is None else torch.cat([kv_cache, h_new], dim=1)
|
||||
attn_out, _ = self.attn(h_new, kv, kv, need_weights=False)
|
||||
x = x_new + attn_out
|
||||
x = x + self.mlp(self.norm2(x))
|
||||
return x, kv
|
||||
|
||||
|
||||
@register_history("attention")
|
||||
class AttentionHistory(HistoryEncoder):
|
||||
"""Causal self-attention over the emitted-token prefix — the more
|
||||
expressive alternative to `MarkovHistory`'s fixed previous-token-only
|
||||
summary. `feat`/`has_prev`
|
||||
follow the same shifted-by-one convention `MarkovHistory` and
|
||||
`Stage2Autoregressive._token_cond` use: `feat[:, i]` is token `i - 1`'s
|
||||
own `(energy_fraction, direction, type_representation)`, with a learned
|
||||
start vector substituted at `has_prev == False` positions (only slot 0 in
|
||||
practice — see `giant.training.stage2_inputs._ar_has_prev`). Causal masking then makes
|
||||
position `i`'s output a function of `feat[:, 1:i+1]` — i.e. tokens
|
||||
`0..i-1` — exactly the prefix available when predicting token `i`.
|
||||
|
||||
`forward` is the parallel training path (one pass over the whole
|
||||
teacher-forced sequence); `init_cache`/`step` are the incremental
|
||||
inference path `giant/sample.py` uses, one new token per call, to avoid
|
||||
re-encoding the whole prefix from scratch every slot — `step` must be
|
||||
called exactly once per slot (its cache-extension is not idempotent),
|
||||
so a slot's output must be reused for
|
||||
every model call within that slot (`forward`'s ODE substeps, or a separate
|
||||
`predict_type` call) rather than re-derived — see
|
||||
`Stage2Autoregressive.history_step`.
|
||||
"""
|
||||
|
||||
def __init__(self, in_dim: int, out_dim: int, n_heads: int = 4, n_layers: int = 2) -> None:
|
||||
super().__init__()
|
||||
self.start = nn.Parameter(torch.zeros(in_dim))
|
||||
self.in_proj = nn.Linear(in_dim, out_dim)
|
||||
self.blocks = nn.ModuleList([_CausalAttnBlock(out_dim, n_heads) for _ in range(n_layers)])
|
||||
|
||||
def _embed(self, feat: torch.Tensor, has_prev: torch.Tensor) -> torch.Tensor:
|
||||
start = self.start.view(1, 1, -1).expand_as(feat)
|
||||
x = torch.where(has_prev.unsqueeze(-1), feat, start)
|
||||
return self.in_proj(x)
|
||||
|
||||
def forward(self, feat: torch.Tensor, has_prev: torch.Tensor) -> torch.Tensor:
|
||||
B, K, _ = feat.shape
|
||||
x = self._embed(feat, has_prev)
|
||||
mask = nn.Transformer.generate_square_subsequent_mask(K, device=feat.device)
|
||||
for block in self.blocks:
|
||||
x = block(x, mask)
|
||||
return x
|
||||
|
||||
def init_cache(self) -> list[torch.Tensor | None]:
|
||||
return [None for _ in self.blocks]
|
||||
|
||||
def step(
|
||||
self,
|
||||
feat: torch.Tensor,
|
||||
has_prev: torch.Tensor,
|
||||
cache: object,
|
||||
) -> tuple[torch.Tensor, object]:
|
||||
"""`feat`/`has_prev`: `(B, 1, in_dim)`/`(B, 1)` — the newest token's
|
||||
own features (what would be `feat[:, k]` in `forward`). `cache`: the
|
||||
`list[Tensor | None]` from `init_cache`/a previous `step` call (typed
|
||||
`object` here to match `HistoryEncoder.step`'s base signature).
|
||||
Advances every block's cache by this position and returns this
|
||||
position's output (`(B, 1, out_dim)`, the correct history summary for
|
||||
the NEXT slot) plus the updated cache."""
|
||||
assert isinstance(cache, list)
|
||||
x = self._embed(feat, has_prev)
|
||||
new_cache: list[torch.Tensor | None] = []
|
||||
for block, kv in zip(self.blocks, cache):
|
||||
x, kv_new = block.step(x, kv)
|
||||
new_cache.append(kv_new)
|
||||
return x, new_cache
|
||||
@@ -0,0 +1,184 @@
|
||||
"""Small stateless-ish building blocks shared across encoders/trunks/models —
|
||||
no dependency on any other `giant.model` submodule (issues.md Issue 8)."""
|
||||
|
||||
import math
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
class SinusoidalEmbedding(nn.Module):
|
||||
def __init__(self, dim: int) -> None:
|
||||
super().__init__()
|
||||
assert dim % 2 == 0, "dim must be even"
|
||||
half = dim // 2
|
||||
freqs = torch.exp(-math.log(10000) * torch.arange(half, dtype=torch.float32) / max(half - 1, 1))
|
||||
self.register_buffer("freqs", freqs)
|
||||
|
||||
def forward(self, t: torch.Tensor) -> torch.Tensor:
|
||||
t = t.reshape(-1, 1).float()
|
||||
args = t * self.freqs.unsqueeze(0) # (B, half)
|
||||
return torch.cat([args.sin(), args.cos()], dim=-1) # (B, dim)
|
||||
|
||||
|
||||
def _make_axis_mlp(in_dim: int, emb_dim: int, n_layers: int) -> nn.Sequential:
|
||||
"""`n_layers`-deep MLP producing an `emb_dim`-wide vector from `in_dim`
|
||||
physical properties (`conditioning.{particle,material}.n_layers`).
|
||||
|
||||
`n_layers=1` (the v0.3.0 default): a single `Linear`, no hidden
|
||||
activation. `n_layers=2` reproduces v0.2's hardcoded depth exactly —
|
||||
`Linear -> SiLU -> Linear` — which is why `migrate_config` back-fills
|
||||
`n_layers=2` for migrated configs rather than the v0.3 default of 1 (see
|
||||
its docstring).
|
||||
"""
|
||||
if n_layers < 1:
|
||||
raise ValueError(f"n_layers must be >= 1, got {n_layers}")
|
||||
if n_layers == 1:
|
||||
return nn.Sequential(nn.Linear(in_dim, emb_dim))
|
||||
layers: list[nn.Module] = [nn.Linear(in_dim, emb_dim), nn.SiLU()]
|
||||
for _ in range(n_layers - 2):
|
||||
layers += [nn.Linear(emb_dim, emb_dim), nn.SiLU()]
|
||||
layers.append(nn.Linear(emb_dim, emb_dim))
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
|
||||
def build_mlp_head(
|
||||
in_dim: int, out_dim: int, hidden: int, depth: int = 2, act: type[nn.Module] = nn.SiLU
|
||||
) -> nn.Sequential:
|
||||
"""`depth`-layer MLP head (gitea #36) — factors out the n_sec_head/
|
||||
type_head pattern duplicated five times across `giant.model.models`.
|
||||
|
||||
`depth=1` is a bare `Linear(in_dim, out_dim)` (no hidden layer/
|
||||
activation); `depth>=2` is `Linear(in_dim, hidden) -> act -> [Linear
|
||||
(hidden, hidden) -> act] * (depth-2) -> Linear(hidden, out_dim)` —
|
||||
`depth=2` reproduces every pre-#36 n_sec_head/type_head exactly when
|
||||
`hidden == hidden_dim // 2`. Mirrors `_make_axis_mlp`'s depth
|
||||
convention above, but takes `hidden` and `out_dim` as independent
|
||||
widths (n_sec_head/type_head's hidden width is not their output width,
|
||||
unlike the particle/material axis MLPs)."""
|
||||
if depth < 1:
|
||||
raise ValueError(f"depth must be >= 1, got {depth}")
|
||||
if depth == 1:
|
||||
return nn.Sequential(nn.Linear(in_dim, out_dim))
|
||||
layers: list[nn.Module] = [nn.Linear(in_dim, hidden), act()]
|
||||
for _ in range(depth - 2):
|
||||
layers += [nn.Linear(hidden, hidden), act()]
|
||||
layers.append(nn.Linear(hidden, out_dim))
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
|
||||
class ContextAdapter(nn.Module):
|
||||
"""Projects a stage's outcome (e.g. Stage 1's 9D target) down to a
|
||||
fixed-width context vector for a downstream stage's conditioning —
|
||||
`stage2_model.context_dim`. Was `SecondaryConditionEncoder.stage1_proj`
|
||||
(+ its `tanh`) in v0.2; pulled out as its own module in v0.3.0 since
|
||||
`SecondaryConditionEncoder` as a wrapper class disappears."""
|
||||
|
||||
def __init__(self, in_dim: int, context_dim: int) -> None:
|
||||
super().__init__()
|
||||
self.proj = nn.Linear(in_dim, context_dim)
|
||||
|
||||
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
||||
return torch.tanh(self.proj(x))
|
||||
|
||||
|
||||
BLOCK_REGISTRY: dict[str, type[nn.Module]] = {}
|
||||
|
||||
|
||||
def register_block(name: str):
|
||||
def decorator(cls: type[nn.Module]) -> type[nn.Module]:
|
||||
BLOCK_REGISTRY[name] = cls
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def build_block(name: str, dim: int, cond_dim: int, dropout: float = 0.0) -> nn.Module:
|
||||
"""Factory: look up a registered conditioning-injection block by name and
|
||||
construct one instance — `trunk.block_conditioning` (gitea #34)."""
|
||||
if name not in BLOCK_REGISTRY:
|
||||
raise ValueError(f"unknown block conditioning type {name!r}; available: {sorted(BLOCK_REGISTRY)}")
|
||||
return BLOCK_REGISTRY[name](dim, cond_dim, dropout)
|
||||
|
||||
|
||||
@register_block("add")
|
||||
class ResBlock(nn.Module):
|
||||
def __init__(self, dim: int, cond_dim: int, dropout: float = 0.0) -> None:
|
||||
super().__init__()
|
||||
self.norm = nn.LayerNorm(dim)
|
||||
self.linear1 = nn.Linear(dim, dim)
|
||||
self.cond_proj = nn.Linear(cond_dim, dim, bias=False)
|
||||
self.act = nn.SiLU()
|
||||
self.dropout = nn.Dropout(dropout)
|
||||
self.linear2 = nn.Linear(dim, dim)
|
||||
|
||||
def forward(self, x: torch.Tensor, cond: torch.Tensor) -> torch.Tensor:
|
||||
h = self.norm(x)
|
||||
h = self.linear1(h) + self.cond_proj(cond)
|
||||
h = self.act(h)
|
||||
h = self.dropout(h)
|
||||
h = self.linear2(h)
|
||||
return x + h
|
||||
|
||||
|
||||
@register_block("film")
|
||||
class FilmResBlock(nn.Module):
|
||||
"""FiLM conditioning (Perez et al. 2018): a per-channel scale+shift
|
||||
modulates the normalized features, on top of the norm's own affine —
|
||||
an *additional* modulation, unlike `AdaLNResBlock` below, which replaces
|
||||
the norm's affine outright. `film_proj` is zero-initialized so
|
||||
`gamma=beta=0` at construction — conditioning has no effect on the
|
||||
output until training moves it, a stable starting point (though not a
|
||||
literal identity block, since `linear1`/`linear2` aren't zero-init)."""
|
||||
|
||||
def __init__(self, dim: int, cond_dim: int, dropout: float = 0.0) -> None:
|
||||
super().__init__()
|
||||
self.norm = nn.LayerNorm(dim)
|
||||
self.linear1 = nn.Linear(dim, dim)
|
||||
self.film_proj = nn.Linear(cond_dim, 2 * dim)
|
||||
nn.init.zeros_(self.film_proj.weight)
|
||||
nn.init.zeros_(self.film_proj.bias)
|
||||
self.act = nn.SiLU()
|
||||
self.dropout = nn.Dropout(dropout)
|
||||
self.linear2 = nn.Linear(dim, dim)
|
||||
|
||||
def forward(self, x: torch.Tensor, cond: torch.Tensor) -> torch.Tensor:
|
||||
h = self.norm(x)
|
||||
gamma, beta = self.film_proj(cond).chunk(2, dim=-1)
|
||||
h = h * (1 + gamma) + beta
|
||||
h = self.linear1(h)
|
||||
h = self.act(h)
|
||||
h = self.dropout(h)
|
||||
h = self.linear2(h)
|
||||
return x + h
|
||||
|
||||
|
||||
@register_block("adaln")
|
||||
class AdaLNResBlock(nn.Module):
|
||||
"""AdaLN-Zero conditioning (DiT, Peebles & Xie 2022): the norm's own
|
||||
affine is replaced by a conditioning-derived scale/shift, and the
|
||||
residual branch is scaled by a conditioning-derived gate. `adaln_proj`
|
||||
is zero-initialized, so `scale=shift=gate=0` at construction — the block
|
||||
is the exact identity function at init (`x + 0 * h' == x`), regardless
|
||||
of `x`/`cond`."""
|
||||
|
||||
def __init__(self, dim: int, cond_dim: int, dropout: float = 0.0) -> None:
|
||||
super().__init__()
|
||||
self.norm = nn.LayerNorm(dim, elementwise_affine=False)
|
||||
self.linear1 = nn.Linear(dim, dim)
|
||||
self.adaln_proj = nn.Linear(cond_dim, 3 * dim)
|
||||
nn.init.zeros_(self.adaln_proj.weight)
|
||||
nn.init.zeros_(self.adaln_proj.bias)
|
||||
self.act = nn.SiLU()
|
||||
self.dropout = nn.Dropout(dropout)
|
||||
self.linear2 = nn.Linear(dim, dim)
|
||||
|
||||
def forward(self, x: torch.Tensor, cond: torch.Tensor) -> torch.Tensor:
|
||||
h = self.norm(x)
|
||||
scale, shift, gate = self.adaln_proj(cond).chunk(3, dim=-1)
|
||||
h = h * (1 + scale) + shift
|
||||
h = self.linear1(h)
|
||||
h = self.act(h)
|
||||
h = self.dropout(h)
|
||||
h = self.linear2(h)
|
||||
return x + gate * h
|
||||
@@ -0,0 +1,695 @@
|
||||
"""Top-level stage models: `Stage1Model`, `Stage2OneShot`, `Stage2Autoregressive`,
|
||||
`CriticModel` — composed from encoders/trunks/history (issues.md Issue 8)."""
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
from giant.config import ConditioningAxisConfig, HeadConfig, ParticleTypeConfig
|
||||
from giant.constants import CONT_SLOT_DIM, K_MAX, PARTICLE_PHYS_DIM, SEC_DIM, SEC_SLOT_DIM, X_DIM
|
||||
from giant.model.encoders import ConditionEncoder
|
||||
from giant.model.history import HistoryEncoder, build_history
|
||||
from giant.model.layers import ContextAdapter, ResBlock, SinusoidalEmbedding, build_mlp_head
|
||||
from giant.model.objectives import build_objective
|
||||
from giant.model.routers import Router
|
||||
from giant.model.trunks import build_trunk
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage models
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def resolve_type_n_classes(particle_type_cfg: ParticleTypeConfig, particle_emb_dim: int) -> int:
|
||||
"""Effective width fed to `stage2_type_dim`/`stage2_trunk_sec_dim` in
|
||||
place of a bare `conditioning.particle.emb_dim` read. Under
|
||||
`target = "onehot"` this is `stage2_model.particle_type.n_classes` (0 =
|
||||
inherit `conditioning.particle.emb_dim`) — see gitea #29, which decoupled
|
||||
the secondary-species vocabulary size from the unrelated
|
||||
physical-conditioning MLP's output width. Under `target = "embedding"`
|
||||
(or `"physical"`, which ignores this value entirely) `n_classes` doesn't
|
||||
apply — the width stays `conditioning.particle.emb_dim`, the embedding
|
||||
table's own dimensionality (`validate_config` requires
|
||||
`conditioning.particle.type = "embedding"` here)."""
|
||||
if particle_type_cfg.target == "onehot":
|
||||
return particle_type_cfg.n_classes or particle_emb_dim
|
||||
return particle_emb_dim
|
||||
|
||||
|
||||
def stage2_type_dim(particle_type_cfg: ParticleTypeConfig, emb_dim: int) -> int:
|
||||
"""Width of a single secondary slot's type slice —
|
||||
`PARTICLE_PHYS_DIM` (log_mass, charge) for `target = "physical"`, else
|
||||
`emb_dim` (both `"onehot"` class logits and `"embedding"` vectors are
|
||||
this many classes/dims wide — callers resolve `emb_dim` via
|
||||
`resolve_type_n_classes` first)."""
|
||||
return PARTICLE_PHYS_DIM if particle_type_cfg.target == "physical" else emb_dim
|
||||
|
||||
|
||||
def stage2_trunk_sec_dim(particle_type_cfg: ParticleTypeConfig, generator: str, k_max: int, emb_dim: int) -> int:
|
||||
"""`Stage2OneShot`'s trunk output width.
|
||||
|
||||
`target = "physical"` is untouched from v0.2/today:
|
||||
`k_max * SEC_SLOT_DIM`, the type slice folded into the same
|
||||
flow-matched/WGAN vector as the continuous stick/dir slots.
|
||||
|
||||
`target` in `("onehot", "embedding")`: under an objective with
|
||||
`folds_type_slice` (currently just wgan) the type slice is still folded
|
||||
in (adversarial for onehot via ST-Gumbel, already-continuous for
|
||||
embedding), just `emb_dim` wide instead of `PARTICLE_PHYS_DIM` wide:
|
||||
`k_max * (CONT_SLOT_DIM + emb_dim)`. Otherwise (flow/ddpm) the type slice
|
||||
isn't part of this vector at all — it's `Stage2OneShot.type_head`'s job
|
||||
instead — so the trunk only covers `k_max * CONT_SLOT_DIM`.
|
||||
"""
|
||||
if particle_type_cfg.target == "physical":
|
||||
return k_max * SEC_SLOT_DIM
|
||||
if build_objective(generator).folds_type_slice:
|
||||
return k_max * (CONT_SLOT_DIM + emb_dim)
|
||||
return k_max * CONT_SLOT_DIM
|
||||
|
||||
|
||||
class StageModel(nn.Module):
|
||||
"""Base owning the scaffolding common to `Stage1Model`, `Stage2OneShot`,
|
||||
`Stage2Autoregressive` (gitea #39): build-or-share `cond_enc`,
|
||||
`particle_type_cfg` normalisation, and — via `_build_trunk_and_heads`,
|
||||
called by each subclass's `__init__` once its own conditioning-assembly
|
||||
modules exist — the objective/time-embedding/trunk construction and the
|
||||
`n_sec_head`/`type_head` classifier heads. A subclass supplies only its
|
||||
own conditioning assembly (`Stage1Model` uses `cond_enc` directly;
|
||||
`Stage2OneShot`/`Stage2Autoregressive` add a context-fusion path) and its
|
||||
trunk's output width.
|
||||
|
||||
`cond_enc`, if given, is used in place of building a fresh
|
||||
`ConditionEncoder` — `conditioning.share_stages = true`: `build_models`
|
||||
constructs one shared instance and passes it to both stages, halving the
|
||||
conditioning parameter count and forcing a common representation."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
cond_out_dim: int,
|
||||
generator: str,
|
||||
noise_dim: int,
|
||||
k_max: int | None = None,
|
||||
particle_type_cfg: ParticleTypeConfig | None = None,
|
||||
cond_enc: ConditionEncoder | None = None,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.generator_kind = generator
|
||||
self.noise_dim = noise_dim
|
||||
self.k_max = k_max
|
||||
# `ParticleTypeConfig()`'s own dataclass default is target="onehot"
|
||||
# (the config.toml default when [stage2_model.particle_type] is
|
||||
# omitted) — a different question from "nobody passed anything to
|
||||
# this constructor", which direct/test construction relies on
|
||||
# defaulting to "physical" (build_models/build_critics always pass
|
||||
# particle_type_cfg explicitly, so this sentinel is never hit there).
|
||||
self.particle_type_cfg = (
|
||||
particle_type_cfg if particle_type_cfg is not None else ParticleTypeConfig(target="physical")
|
||||
)
|
||||
self.type_dim = stage2_type_dim(
|
||||
self.particle_type_cfg, resolve_type_n_classes(self.particle_type_cfg, particle_cfg.emb_dim)
|
||||
)
|
||||
self.cond_enc = (
|
||||
cond_enc
|
||||
if cond_enc is not None
|
||||
else ConditionEncoder(pdg_vocab, mat_vocab, particle_cfg, material_cfg, out_dim=cond_out_dim)
|
||||
)
|
||||
|
||||
def _build_trunk_and_heads(
|
||||
self,
|
||||
*,
|
||||
trunk_out_dim: int,
|
||||
hidden_dim: int,
|
||||
n_res_blocks: int,
|
||||
cond_out_dim: int,
|
||||
time_dim: int,
|
||||
router: Router | None,
|
||||
trunk_type: str,
|
||||
block_conditioning: str,
|
||||
dropout: float,
|
||||
n_sec_head_k_max: int | None,
|
||||
n_sec_head_cfg: dict | None,
|
||||
type_head_out_dim: int | None,
|
||||
type_head_cfg: dict | None,
|
||||
) -> None:
|
||||
"""Builds `self.time_emb`, `self.trunk`, `self.n_sec_head`,
|
||||
`self.type_head`. Called by a subclass's `__init__` after it has set
|
||||
up its own conditioning-assembly modules — `merged_cond_dim` below
|
||||
must match the width that assembly (`_cond_embed`/`_base_cond`/
|
||||
`_token_cond`, or plain `cond_enc` for `Stage1Model`) actually
|
||||
produces.
|
||||
|
||||
`n_sec_head` is built iff `n_sec_head_k_max is not None` (output
|
||||
width `n_sec_head_k_max + 1`) — `Stage1Model` passes this only for a
|
||||
migrated v0.2 checkpoint, `Stage2OneShot`/`Stage2Autoregressive` pass
|
||||
it whenever `build_n_sec_head=True`. `type_head` is built iff
|
||||
`type_head_out_dim is not None` (the caller — only the two Stage2
|
||||
classes — passes `None` exactly when `particle_type_cfg.target ==
|
||||
"physical"`) *and* the objective doesn't fold the type slice into its
|
||||
own trunk output (checked here, since `objective` is already needed
|
||||
for the trunk itself).
|
||||
"""
|
||||
objective = build_objective(self.generator_kind)
|
||||
has_time = objective.needs_time
|
||||
self.time_emb = SinusoidalEmbedding(time_dim) if has_time else None
|
||||
merged_cond_dim = (time_dim if has_time else 0) + cond_out_dim
|
||||
in_dim = objective.trunk_in_dim(trunk_out_dim, self.noise_dim)
|
||||
self.trunk = build_trunk(
|
||||
router,
|
||||
trunk_type,
|
||||
in_dim,
|
||||
trunk_out_dim,
|
||||
hidden_dim,
|
||||
n_res_blocks,
|
||||
merged_cond_dim,
|
||||
dropout,
|
||||
block_conditioning,
|
||||
)
|
||||
self.n_sec_head = None
|
||||
if n_sec_head_k_max is not None:
|
||||
head_cfg = HeadConfig.from_dict(n_sec_head_cfg)
|
||||
hidden = max(1, round(hidden_dim * head_cfg.hidden_ratio))
|
||||
self.n_sec_head = build_mlp_head(cond_out_dim, n_sec_head_k_max + 1, hidden, head_cfg.depth)
|
||||
self.type_head = None
|
||||
if type_head_out_dim is not None and not objective.folds_type_slice:
|
||||
head_cfg = HeadConfig.from_dict(type_head_cfg)
|
||||
hidden = max(1, round(hidden_dim * head_cfg.hidden_ratio))
|
||||
self.type_head = build_mlp_head(cond_out_dim, type_head_out_dim, hidden, head_cfg.depth)
|
||||
|
||||
def _require_n_sec_head(self) -> None:
|
||||
if self.n_sec_head is None:
|
||||
raise RuntimeError(
|
||||
f"this {type(self).__name__} has no n_sec_head — it belongs to "
|
||||
"a migrated v0.2 checkpoint (n_sec.owner='stage1'); call "
|
||||
"stage1.predict_n_sec(cond_cont, cond_cat) instead"
|
||||
)
|
||||
|
||||
def _require_type_head(self) -> None:
|
||||
if self.type_head is None:
|
||||
raise RuntimeError(
|
||||
f"this {type(self).__name__} has no type_head — either "
|
||||
"particle_type.target='physical' (the type slice is part of "
|
||||
"forward()'s own output) or generator='wgan' (the WGAN "
|
||||
"trainer reads the type slice out of forward()'s output "
|
||||
"directly instead)"
|
||||
)
|
||||
|
||||
|
||||
class Stage1Model(StageModel):
|
||||
"""Predicts the 9D primary post-step vector. No `n_sec_head` — fresh runs
|
||||
move it to stage 2, except for a migrated v0.2 checkpoint
|
||||
(`n_sec_head_k_max` given), where it stays attached here
|
||||
since that's where its weights live and what conditioning it was trained
|
||||
against (see `_migrate_legacy_model_config`)."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
hidden_dim: int = 256,
|
||||
n_res_blocks: int = 6,
|
||||
cond_out_dim: int = 128,
|
||||
x_dim: int = X_DIM,
|
||||
dropout: float = 0.0,
|
||||
generator: str = "flow",
|
||||
time_dim: int = 64,
|
||||
noise_dim: int = 64,
|
||||
router: Router | None = None,
|
||||
trunk_type: str = "resmlp",
|
||||
block_conditioning: str = "add",
|
||||
n_sec_head_k_max: int | None = None,
|
||||
cond_enc: ConditionEncoder | None = None,
|
||||
n_sec_head_cfg: dict | None = None,
|
||||
) -> None:
|
||||
super().__init__(
|
||||
pdg_vocab,
|
||||
mat_vocab,
|
||||
particle_cfg,
|
||||
material_cfg,
|
||||
cond_out_dim=cond_out_dim,
|
||||
generator=generator,
|
||||
noise_dim=noise_dim,
|
||||
cond_enc=cond_enc,
|
||||
)
|
||||
self._build_trunk_and_heads(
|
||||
trunk_out_dim=x_dim,
|
||||
hidden_dim=hidden_dim,
|
||||
n_res_blocks=n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
time_dim=time_dim,
|
||||
router=router,
|
||||
trunk_type=trunk_type,
|
||||
block_conditioning=block_conditioning,
|
||||
dropout=dropout,
|
||||
n_sec_head_k_max=n_sec_head_k_max,
|
||||
n_sec_head_cfg=n_sec_head_cfg,
|
||||
type_head_out_dim=None,
|
||||
type_head_cfg=None,
|
||||
)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x_t: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
t: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
c_emb = self.cond_enc(cond_cont, cond_cat)
|
||||
cond = torch.cat([self.time_emb(t), c_emb], dim=-1) if self.time_emb is not None else c_emb
|
||||
return self.trunk(x_t, cond, cond_cont, cond_cat)
|
||||
|
||||
def _require_n_sec_head(self) -> None:
|
||||
"""Overrides `StageModel`'s guard — a `Stage1Model` with no
|
||||
`n_sec_head` points the caller to stage 2 (n_sec's default owner),
|
||||
not to `stage1` as the base's message would."""
|
||||
if self.n_sec_head is None:
|
||||
raise RuntimeError(
|
||||
"this Stage1Model has no n_sec_head — n_sec now lives on "
|
||||
"stage 2 by default; this method only exists "
|
||||
"for a migrated v0.2 checkpoint (n_sec.owner='stage1')"
|
||||
)
|
||||
|
||||
def predict_n_sec(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""Return n_sec logits (B, K_MAX+1) from conditioning alone. Only
|
||||
valid on a migrated v0.2 checkpoint's Stage1Model — fresh v0.3.0
|
||||
configs predict n_sec from Stage2OneShot instead."""
|
||||
self._require_n_sec_head()
|
||||
assert self.n_sec_head is not None
|
||||
c_emb = self.cond_enc(cond_cont, cond_cat)
|
||||
return self.n_sec_head(c_emb)
|
||||
|
||||
|
||||
class Stage2OneShot(StageModel):
|
||||
"""Predicts all `k_max` secondary slots simultaneously — v0.2 behaviour,
|
||||
reproduced exactly (`decoder = "autoregressive"` is `Stage2Autoregressive`,
|
||||
step 4/5, not implemented yet).
|
||||
|
||||
Owns `n_sec_head` by default unless `build_n_sec_head=False`
|
||||
(a migrated v0.2 checkpoint, whose n_sec_head instead attaches to
|
||||
Stage1Model — see `_migrate_legacy_model_config`).
|
||||
|
||||
`particle_type_cfg.target` (default `"physical"`) selects the
|
||||
secondary-type mechanism: `"physical"` keeps the type slice folded into
|
||||
the trunk's own
|
||||
flow-matched/WGAN output, unchanged from v0.2 (`sec_dim` — computed by
|
||||
the caller via `stage2_trunk_sec_dim` — already reflects this). Under
|
||||
`"onehot"`/`"embedding"` with an objective (`giant.model.objectives`) that
|
||||
doesn't fold the type slice (flow/ddpm), the type
|
||||
slice is predicted by a separate `type_head` instead (same shape pattern
|
||||
as `n_sec_head`) — `sec_dim` then covers only the continuous
|
||||
stick/dir slots, `type_head` covers `k_max * emb_dim` type logits/vectors.
|
||||
Under a folding objective (wgan) the type slice stays folded into `sec_dim`
|
||||
(just `emb_dim` instead of `PARTICLE_PHYS_DIM` wide) and `type_head` is
|
||||
unused (`None`) — the WGAN trainer handles the ST-Gumbel relaxation.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
hidden_dim: int = 256,
|
||||
n_res_blocks: int = 6,
|
||||
cond_out_dim: int = 128,
|
||||
context_dim: int = 64,
|
||||
sec_dim: int = SEC_DIM,
|
||||
x_dim: int = X_DIM,
|
||||
dropout: float = 0.0,
|
||||
generator: str = "wgan",
|
||||
time_dim: int = 64,
|
||||
noise_dim: int = 64,
|
||||
k_max: int = K_MAX,
|
||||
router: Router | None = None,
|
||||
trunk_type: str = "resmlp",
|
||||
block_conditioning: str = "add",
|
||||
build_n_sec_head: bool = True,
|
||||
particle_type_cfg: ParticleTypeConfig | None = None,
|
||||
cond_enc: ConditionEncoder | None = None,
|
||||
n_sec_head_cfg: dict | None = None,
|
||||
type_head_cfg: dict | None = None,
|
||||
) -> None:
|
||||
super().__init__(
|
||||
pdg_vocab,
|
||||
mat_vocab,
|
||||
particle_cfg,
|
||||
material_cfg,
|
||||
cond_out_dim=cond_out_dim,
|
||||
generator=generator,
|
||||
noise_dim=noise_dim,
|
||||
k_max=k_max,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
cond_enc=cond_enc,
|
||||
)
|
||||
self.context_adapter = ContextAdapter(x_dim, context_dim)
|
||||
self.fuse = nn.Sequential(
|
||||
nn.Linear(cond_out_dim + context_dim, cond_out_dim),
|
||||
nn.SiLU(),
|
||||
)
|
||||
target = self.particle_type_cfg.target
|
||||
type_head_out_dim = None if target == "physical" else k_max * self.type_dim
|
||||
self._build_trunk_and_heads(
|
||||
trunk_out_dim=sec_dim,
|
||||
hidden_dim=hidden_dim,
|
||||
n_res_blocks=n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
time_dim=time_dim,
|
||||
router=router,
|
||||
trunk_type=trunk_type,
|
||||
block_conditioning=block_conditioning,
|
||||
dropout=dropout,
|
||||
n_sec_head_k_max=k_max if build_n_sec_head else None,
|
||||
n_sec_head_cfg=n_sec_head_cfg,
|
||||
type_head_out_dim=type_head_out_dim,
|
||||
type_head_cfg=type_head_cfg,
|
||||
)
|
||||
|
||||
def _cond_embed(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, stage1_out: torch.Tensor) -> torch.Tensor:
|
||||
base = self.cond_enc(cond_cont, cond_cat)
|
||||
ctx = self.context_adapter(stage1_out)
|
||||
return self.fuse(torch.cat([base, ctx], dim=-1))
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x_t: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
t: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
c_emb = self._cond_embed(cond_cont, cond_cat, stage1_out)
|
||||
cond = torch.cat([self.time_emb(t), c_emb], dim=-1) if self.time_emb is not None else c_emb
|
||||
return self.trunk(x_t, cond, cond_cont, cond_cat)
|
||||
|
||||
def predict_n_sec(
|
||||
self,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
self._require_n_sec_head()
|
||||
assert self.n_sec_head is not None
|
||||
c_emb = self._cond_embed(cond_cont, cond_cat, stage1_out)
|
||||
return self.n_sec_head(c_emb)
|
||||
|
||||
def predict_type(
|
||||
self,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
"""`(B, k_max, emb_dim)` per-slot type logits (`target="onehot"`) or
|
||||
vectors (`target="embedding"`) — only under `generator in ("flow",
|
||||
"ddpm")`; `generator == "wgan"` folds the type slice into `forward`'s
|
||||
own output instead (see class docstring)."""
|
||||
self._require_type_head()
|
||||
assert self.type_head is not None
|
||||
c_emb = self._cond_embed(cond_cont, cond_cat, stage1_out)
|
||||
return self.type_head(c_emb).view(-1, self.k_max, self.type_dim)
|
||||
|
||||
|
||||
class Stage2Autoregressive(StageModel):
|
||||
"""Emits secondaries one at a time in descending-energy order, instead
|
||||
of `Stage2OneShot`'s simultaneous
|
||||
k_max-slot prediction. `history` selects `MarkovHistory` or
|
||||
`AttentionHistory` (`attn_n_heads`/`attn_n_layers`, attention only).
|
||||
`teacher_forcing` handling lives entirely in the trainer
|
||||
(`giant/train.py`), since it only affects how training inputs are
|
||||
assembled, not this module's architecture.
|
||||
|
||||
Under teacher forcing every token's conditioning is built from ground
|
||||
truth, so a whole K-token sequence trains in one parallel batched pass:
|
||||
`forward` accepts `(B, K, ...)` tensors for an arbitrary K (not hardcoded
|
||||
to `k_max`) — this also means a future one-token-at-a-time inference loop
|
||||
(`K=1` per call, step 6) needs no interface change here.
|
||||
|
||||
Two independent conditioning paths, mirroring `Stage2OneShot`'s
|
||||
`_cond_embed` but split in two: `_base_cond` (`cond_enc` +
|
||||
`context_adapter` only) feeds `predict_n_sec`, since n_sec doesn't depend
|
||||
on token position; `_token_cond` additionally fuses in the history
|
||||
encoding and two running scalars (remaining energy-budget fraction,
|
||||
normalized slot index), and feeds `forward`/`predict_type`/the trunk.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
hidden_dim: int = 256,
|
||||
n_res_blocks: int = 6,
|
||||
cond_out_dim: int = 128,
|
||||
context_dim: int = 64,
|
||||
x_dim: int = X_DIM,
|
||||
dropout: float = 0.0,
|
||||
generator: str = "wgan",
|
||||
time_dim: int = 64,
|
||||
noise_dim: int = 64,
|
||||
k_max: int = K_MAX,
|
||||
router: Router | None = None,
|
||||
trunk_type: str = "resmlp",
|
||||
block_conditioning: str = "add",
|
||||
build_n_sec_head: bool = True,
|
||||
particle_type_cfg: ParticleTypeConfig | None = None,
|
||||
history: str = "markov",
|
||||
attn_n_heads: int = 4,
|
||||
attn_n_layers: int = 2,
|
||||
cond_enc: ConditionEncoder | None = None,
|
||||
n_sec_head_cfg: dict | None = None,
|
||||
type_head_cfg: dict | None = None,
|
||||
) -> None:
|
||||
super().__init__(
|
||||
pdg_vocab,
|
||||
mat_vocab,
|
||||
particle_cfg,
|
||||
material_cfg,
|
||||
cond_out_dim=cond_out_dim,
|
||||
generator=generator,
|
||||
noise_dim=noise_dim,
|
||||
k_max=k_max,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
cond_enc=cond_enc,
|
||||
)
|
||||
self.history_kind = history
|
||||
self.context_adapter = ContextAdapter(x_dim, context_dim)
|
||||
self.base_fuse = nn.Sequential(
|
||||
nn.Linear(cond_out_dim + context_dim, cond_out_dim),
|
||||
nn.SiLU(),
|
||||
)
|
||||
|
||||
# Reuses conditioning.out_dim for the history encoder's own output
|
||||
# width — there's no dedicated stage2_model.autoregressive key for
|
||||
# this, a reasonable default rather than a design-doc-specified value.
|
||||
history_dim = cond_out_dim
|
||||
hist_in_dim = CONT_SLOT_DIM + self.type_dim
|
||||
self.history_encoder: HistoryEncoder = build_history(
|
||||
history, hist_in_dim, history_dim, n_heads=attn_n_heads, n_layers=attn_n_layers
|
||||
)
|
||||
token_fuse_in = cond_out_dim + context_dim + history_dim + 2 # +2: remaining_frac, slot_idx
|
||||
self.token_fuse = nn.Sequential(
|
||||
nn.Linear(token_fuse_in, cond_out_dim),
|
||||
nn.SiLU(),
|
||||
)
|
||||
|
||||
# `self.type_dim` (set by StageModel.__init__) doubles as the raw
|
||||
# `emb_dim` `stage2_trunk_sec_dim` wants: for a non-"physical" target
|
||||
# `stage2_type_dim` already resolved `type_dim` to exactly that value;
|
||||
# for "physical" the emb_dim argument goes unused anyway.
|
||||
token_dim = stage2_trunk_sec_dim(self.particle_type_cfg, generator, 1, self.type_dim)
|
||||
target = self.particle_type_cfg.target
|
||||
type_head_out_dim = None if target == "physical" else self.type_dim
|
||||
self._build_trunk_and_heads(
|
||||
trunk_out_dim=token_dim,
|
||||
hidden_dim=hidden_dim,
|
||||
n_res_blocks=n_res_blocks,
|
||||
cond_out_dim=cond_out_dim,
|
||||
time_dim=time_dim,
|
||||
router=router,
|
||||
trunk_type=trunk_type,
|
||||
block_conditioning=block_conditioning,
|
||||
dropout=dropout,
|
||||
n_sec_head_k_max=k_max if build_n_sec_head else None,
|
||||
n_sec_head_cfg=n_sec_head_cfg,
|
||||
type_head_out_dim=type_head_out_dim,
|
||||
type_head_cfg=type_head_cfg,
|
||||
)
|
||||
|
||||
def _base_cond(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, stage1_out: torch.Tensor) -> torch.Tensor:
|
||||
base = self.cond_enc(cond_cont, cond_cat)
|
||||
ctx = self.context_adapter(stage1_out)
|
||||
return self.base_fuse(torch.cat([base, ctx], dim=-1))
|
||||
|
||||
def _token_cond(
|
||||
self,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
history_feat: torch.Tensor,
|
||||
has_prev: torch.Tensor,
|
||||
remaining_frac: torch.Tensor,
|
||||
slot_idx: torch.Tensor,
|
||||
hist: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
"""`hist`, if given, overrides recomputing `self.history_encoder`
|
||||
from `history_feat`/`has_prev` — the inference-time KV-cache path
|
||||
(`Stage2Autoregressive.history_step`) precomputes it once per slot and
|
||||
passes it in here so a slot's (possibly several) model calls — an ODE
|
||||
loop's substeps, or a separate `predict_type` call — read the same
|
||||
cached history instead of each re-deriving (and, under attention,
|
||||
re-appending to the cache — see `AttentionHistory.step`'s docstring)."""
|
||||
K = history_feat.size(1)
|
||||
base = self.cond_enc(cond_cont, cond_cat).unsqueeze(1).expand(-1, K, -1)
|
||||
ctx = self.context_adapter(stage1_out).unsqueeze(1).expand(-1, K, -1)
|
||||
if hist is None:
|
||||
hist = self.history_encoder(history_feat, has_prev)
|
||||
scalars = torch.stack([remaining_frac, slot_idx], dim=-1)
|
||||
return self.token_fuse(torch.cat([base, ctx, hist, scalars], dim=-1))
|
||||
|
||||
def init_history_cache(self):
|
||||
"""Inference-only incremental-decoding state for `self.history_encoder`
|
||||
(`giant/sample.py`'s AR loop) — whatever `self.history_encoder.init_cache()`
|
||||
returns for the configured `history` type: `None` under `history="markov"`
|
||||
(its per-step cost is already O(1) — see `HistoryEncoder`'s docstring),
|
||||
or `AttentionHistory.init_cache()`'s real per-block KV cache under
|
||||
`history="attention"`."""
|
||||
return self.history_encoder.init_cache()
|
||||
|
||||
def history_step(self, token_feat: torch.Tensor, has_prev: torch.Tensor, cache) -> tuple[torch.Tensor, object]:
|
||||
"""One inference slot's worth of history encoding: advances `cache`
|
||||
(from `init_history_cache`, or a previous `history_step` call) by
|
||||
`token_feat`/`has_prev` (`(B, 1, ...)` — the just-emitted previous
|
||||
token, same convention `giant.sample.sample_secondaries_ar` already
|
||||
threads as `prev_repr`), and returns `(hist, new_cache)` — `hist` is
|
||||
this slot's history summary (pass it as `_token_cond`'s `hist=` to
|
||||
every model call made for this slot), `new_cache` is what to pass into
|
||||
the *next* slot's `history_step`. Must be called exactly once per
|
||||
slot — see `AttentionHistory.step`'s docstring."""
|
||||
return self.history_encoder.step(token_feat, has_prev, cache)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x_t: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
history_feat: torch.Tensor,
|
||||
has_prev: torch.Tensor,
|
||||
remaining_frac: torch.Tensor,
|
||||
slot_idx: torch.Tensor,
|
||||
t: torch.Tensor | None = None,
|
||||
hist: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
B, K = x_t.shape[0], x_t.shape[1]
|
||||
c_emb = self._token_cond(
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
stage1_out,
|
||||
history_feat,
|
||||
has_prev,
|
||||
remaining_frac,
|
||||
slot_idx,
|
||||
hist=hist,
|
||||
)
|
||||
if self.time_emb is not None:
|
||||
assert t is not None
|
||||
t_emb = self.time_emb(t.reshape(-1)).view(B, K, -1)
|
||||
cond = torch.cat([t_emb, c_emb], dim=-1)
|
||||
else:
|
||||
cond = c_emb
|
||||
x_flat = x_t.reshape(B * K, -1)
|
||||
cond_flat = cond.reshape(B * K, -1)
|
||||
cond_cont_flat = cond_cont.unsqueeze(1).expand(-1, K, -1).reshape(B * K, -1)
|
||||
cond_cat_flat = cond_cat.unsqueeze(1).expand(-1, K, -1).reshape(B * K, -1)
|
||||
out = self.trunk(x_flat, cond_flat, cond_cont_flat, cond_cat_flat)
|
||||
return out.view(B, K, -1)
|
||||
|
||||
def predict_n_sec(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, stage1_out: torch.Tensor) -> torch.Tensor:
|
||||
self._require_n_sec_head()
|
||||
assert self.n_sec_head is not None
|
||||
return self.n_sec_head(self._base_cond(cond_cont, cond_cat, stage1_out))
|
||||
|
||||
def predict_type(
|
||||
self,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor,
|
||||
history_feat: torch.Tensor,
|
||||
has_prev: torch.Tensor,
|
||||
remaining_frac: torch.Tensor,
|
||||
slot_idx: torch.Tensor,
|
||||
hist: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
self._require_type_head()
|
||||
assert self.type_head is not None
|
||||
c_emb = self._token_cond(
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
stage1_out,
|
||||
history_feat,
|
||||
has_prev,
|
||||
remaining_frac,
|
||||
slot_idx,
|
||||
hist=hist,
|
||||
)
|
||||
B, K, _ = c_emb.shape
|
||||
return self.type_head(c_emb.reshape(B * K, -1)).view(B, K, self.type_dim)
|
||||
|
||||
|
||||
class CriticModel(nn.Module):
|
||||
"""Generator-agnostic WGAN-GP critic body: a scalar realism score, for
|
||||
either stage (`stage="stage1"` mirrors v0.2 `Critic`; `stage="stage2"`
|
||||
mirrors v0.2 `SecondaryCritic`, adding the same context-fusion path as
|
||||
`Stage2OneShot`). Used only when that stage's `generator == "wgan"`."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_cfg: ConditioningAxisConfig,
|
||||
material_cfg: ConditioningAxisConfig,
|
||||
in_dim: int,
|
||||
hidden_dim: int = 256,
|
||||
n_res_blocks: int = 6,
|
||||
cond_out_dim: int = 128,
|
||||
dropout: float = 0.0,
|
||||
stage: str = "stage1",
|
||||
context_dim: int = 64,
|
||||
context_in_dim: int = X_DIM,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
if stage not in ("stage1", "stage2"):
|
||||
raise ValueError(f"stage must be 'stage1' or 'stage2', got {stage!r}")
|
||||
self.stage = stage
|
||||
self.cond_enc = ConditionEncoder(pdg_vocab, mat_vocab, particle_cfg, material_cfg, out_dim=cond_out_dim)
|
||||
if stage == "stage2":
|
||||
self.context_adapter = ContextAdapter(context_in_dim, context_dim)
|
||||
self.fuse = nn.Sequential(
|
||||
nn.Linear(cond_out_dim + context_dim, cond_out_dim),
|
||||
nn.SiLU(),
|
||||
)
|
||||
self.input_proj = nn.Linear(in_dim, hidden_dim)
|
||||
self.blocks = nn.ModuleList([ResBlock(hidden_dim, cond_out_dim, dropout=dropout) for _ in range(n_res_blocks)])
|
||||
self.out_norm = nn.LayerNorm(hidden_dim)
|
||||
self.out_proj = nn.Linear(hidden_dim, 1)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_out: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
base = self.cond_enc(cond_cont, cond_cat)
|
||||
if self.stage == "stage2":
|
||||
ctx = self.context_adapter(stage1_out)
|
||||
cond = self.fuse(torch.cat([base, ctx], dim=-1))
|
||||
else:
|
||||
cond = base
|
||||
h = self.input_proj(x)
|
||||
for block in self.blocks:
|
||||
h = block(h, cond)
|
||||
return self.out_proj(self.out_norm(h)).squeeze(-1)
|
||||
+132
-1749
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,202 @@
|
||||
"""Generative objectives (flow/ddpm/wgan): `Objective` base + registry,
|
||||
mirroring `giant.model.routers`'s `Router` pattern (gitea #32). Each objective
|
||||
answers, in one place, the handful of questions every stage model/sampler/
|
||||
trainer used to re-derive independently from a bare `generator` string: does
|
||||
this stage need a time embedding, is it adversarial, does it fold the
|
||||
secondary type slice into its own trunk output, what does the trunk take as
|
||||
input, which stage-1/stage-2 loss does it train against.
|
||||
|
||||
Self-contained (no dependency on `giant.model.models`, unlike `Router` which
|
||||
`giant.model.trunks` depends on) — `Objective` never needs to construct a
|
||||
stage model or critic itself, only describe one. This also sidesteps a
|
||||
`models.py` <-> `objectives.py` import cycle, since `models.py` calls
|
||||
`build_objective`.
|
||||
"""
|
||||
|
||||
import inspect
|
||||
|
||||
import torch
|
||||
|
||||
from giant.model.schedule import (
|
||||
CosineSchedule,
|
||||
flow_matching_loss,
|
||||
flow_matching_loss_secondary,
|
||||
flow_matching_loss_secondary_ar,
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Objective contract
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class Objective:
|
||||
"""Contract for a pluggable generative objective. Not an `nn.Module` —
|
||||
unlike `Router`, no objective owns learnable parameters, so a plain
|
||||
strategy object is the honest fit.
|
||||
|
||||
`needs_time`/`is_adversarial`/`folds_type_slice`/`supports_stage2_decoder`
|
||||
are set by each concrete subclass (no defaults here — a new objective
|
||||
should have to state all four, not silently inherit one that happens to
|
||||
be wrong for it). See `FlowObjective`/`DdpmObjective`/`WganObjective`.
|
||||
"""
|
||||
|
||||
needs_time: bool
|
||||
is_adversarial: bool
|
||||
folds_type_slice: bool
|
||||
supports_stage2_decoder: bool = True
|
||||
|
||||
def trunk_in_dim(self, out_dim: int, noise_dim: int) -> int:
|
||||
"""Width of the trunk's own input — `out_dim` (denoising/flow-matching
|
||||
a same-shape vector) for every non-adversarial objective;
|
||||
`WganObjective` overrides to `noise_dim` (a single-pass noise-to-output
|
||||
generator)."""
|
||||
return out_dim
|
||||
|
||||
def build_schedule(self, n_steps: int, device: torch.device) -> CosineSchedule | None:
|
||||
"""Objective-owned auxiliary state a stage trainer must build once
|
||||
and hold onto (device-placed) across its training loop. `None` for
|
||||
every objective except `DdpmObjective` (its noise schedule)."""
|
||||
return None
|
||||
|
||||
def stage1_loss(
|
||||
self,
|
||||
model: torch.nn.Module,
|
||||
x1: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
*,
|
||||
schedule: object | None = None,
|
||||
) -> torch.Tensor:
|
||||
"""Stage-1 training loss. Only implemented by non-adversarial
|
||||
objectives — `WganObjective` is unused here, `WGANStageTrainer` has
|
||||
its own G/D step instead."""
|
||||
raise NotImplementedError(f"{type(self).__name__} has no stage1_loss")
|
||||
|
||||
def stage2_loss(
|
||||
self,
|
||||
model: torch.nn.Module,
|
||||
x1_s2: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
stage1_ctx: torch.Tensor,
|
||||
sec_mask: torch.Tensor,
|
||||
*,
|
||||
type_dim: int | None,
|
||||
ar_inputs: dict[str, torch.Tensor] | None = None,
|
||||
) -> torch.Tensor:
|
||||
"""Stage-2 secondary-decoder training loss, one-shot or
|
||||
autoregressive depending on whether `ar_inputs` is given. Same
|
||||
adversarial caveat as `stage1_loss`."""
|
||||
raise NotImplementedError(f"{type(self).__name__} has no stage2_loss")
|
||||
|
||||
|
||||
OBJECTIVE_REGISTRY: dict[str, type[Objective]] = {}
|
||||
|
||||
|
||||
def register_objective(name: str):
|
||||
def decorator(cls: type[Objective]) -> type[Objective]:
|
||||
OBJECTIVE_REGISTRY[name] = cls
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def build_objective(name: str, **kwargs) -> Objective:
|
||||
"""Factory: look up an `Objective` subclass by name (a `generator`
|
||||
config value) from the registry.
|
||||
|
||||
Every registered objective is fed the same kwargs; kwargs not declared by
|
||||
that type's constructor are silently dropped, so per-type hyperparameters
|
||||
(e.g. `DdpmObjective`'s `n_steps`) can coexist in one call without
|
||||
special-casing — same convention as `giant.model.routers.build_router`.
|
||||
"""
|
||||
if name not in OBJECTIVE_REGISTRY:
|
||||
raise ValueError(f"unknown generator/objective {name!r}; available: {sorted(OBJECTIVE_REGISTRY)}")
|
||||
cls = OBJECTIVE_REGISTRY[name]
|
||||
accepted = set(inspect.signature(cls.__init__).parameters) - {"self"}
|
||||
filtered = {k: v for k, v in kwargs.items() if k in accepted}
|
||||
return cls(**filtered)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Concrete objectives
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@register_objective("flow")
|
||||
class FlowObjective(Objective):
|
||||
"""Conditional flow matching (Lipman et al. 2022) — the primary
|
||||
objective. ~10 ODE steps at inference (`giant.sample.sample_flow`)."""
|
||||
|
||||
needs_time = True
|
||||
is_adversarial = False
|
||||
folds_type_slice = False
|
||||
|
||||
def stage1_loss(self, model, x1, cond_cont, cond_cat, *, schedule=None) -> torch.Tensor:
|
||||
return flow_matching_loss(model, x1, cond_cont, cond_cat)
|
||||
|
||||
def stage2_loss(
|
||||
self,
|
||||
model,
|
||||
x1_s2,
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
stage1_ctx,
|
||||
sec_mask,
|
||||
*,
|
||||
type_dim=None,
|
||||
ar_inputs=None,
|
||||
) -> torch.Tensor:
|
||||
if ar_inputs is not None:
|
||||
return flow_matching_loss_secondary_ar(
|
||||
model,
|
||||
x1_s2,
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
stage1_ctx,
|
||||
ar_inputs["history_feat"],
|
||||
ar_inputs["has_prev"],
|
||||
ar_inputs["remaining_frac"],
|
||||
ar_inputs["slot_idx"],
|
||||
sec_mask,
|
||||
type_dim=type_dim,
|
||||
)
|
||||
return flow_matching_loss_secondary(model, x1_s2, cond_cont, cond_cat, stage1_ctx, sec_mask, type_dim=type_dim)
|
||||
|
||||
|
||||
@register_objective("ddpm")
|
||||
class DdpmObjective(Objective):
|
||||
"""Full DDPM ancestral sampling (Nichol & Dhariwal 2021 cosine schedule)
|
||||
— the throwaway baseline. Stage-1 only: no `Stage2*` class has ever been
|
||||
trained with `generator="ddpm"` in practice, so there's no stage-2 ddpm
|
||||
loss to dispatch to (matches `FlowDDPMStageTrainer`'s pre-existing
|
||||
stage-2 guard)."""
|
||||
|
||||
needs_time = True
|
||||
is_adversarial = False
|
||||
folds_type_slice = False
|
||||
supports_stage2_decoder = False
|
||||
|
||||
def __init__(self, n_steps: int = 1000) -> None:
|
||||
self.n_steps = n_steps
|
||||
|
||||
def build_schedule(self, n_steps: int, device: torch.device) -> CosineSchedule:
|
||||
return CosineSchedule(T=n_steps).to(device)
|
||||
|
||||
def stage1_loss(self, model, x1, cond_cont, cond_cat, *, schedule=None) -> torch.Tensor:
|
||||
assert schedule is not None, "DdpmObjective.stage1_loss needs a schedule (see build_schedule)"
|
||||
return schedule.loss(model, x1, cond_cont, cond_cat)
|
||||
|
||||
|
||||
@register_objective("wgan")
|
||||
class WganObjective(Objective):
|
||||
"""WGAN-GP (Gulrajani et al. 2017) — single forward pass instead of an
|
||||
ODE loop. `stage1_loss`/`stage2_loss` are unused: `WGANStageTrainer` owns
|
||||
its own dual generator/critic step instead of a single scalar loss."""
|
||||
|
||||
needs_time = False
|
||||
is_adversarial = True
|
||||
folds_type_slice = True
|
||||
|
||||
def trunk_in_dim(self, out_dim: int, noise_dim: int) -> int:
|
||||
return noise_dim
|
||||
@@ -0,0 +1,376 @@
|
||||
"""Mixture-of-experts routing: `Router` base + registry, the four concrete
|
||||
router types, and composed/config-driven construction — self-contained, no
|
||||
dependency on any other `giant.model` submodule (issues.md Issue 8)."""
|
||||
|
||||
import inspect
|
||||
import math
|
||||
import re
|
||||
from collections.abc import Sequence
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
|
||||
from giant.cond_layout import CondLayout
|
||||
from giant.constants import COND_DIM
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Routers — carried over unchanged from v0.2
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class Router(nn.Module):
|
||||
"""Contract for a pluggable mixture-of-experts routing axis.
|
||||
|
||||
Subclasses implement `gate` (soft partition-of-unity weights over
|
||||
experts, used in train mode for a fully differentiable mixture);
|
||||
`top1` and `balance_loss` have working defaults so a new routing axis
|
||||
is usually a one-method add. See `ROUTER_REGISTRY` / `build_router`.
|
||||
"""
|
||||
|
||||
def __init__(self, n_experts: int) -> None:
|
||||
super().__init__()
|
||||
self.n_experts = n_experts
|
||||
self.gumbel = False
|
||||
self.gumbel_tau = 1.0
|
||||
|
||||
def gate(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""(B, n_experts) soft weights, rows summing to 1."""
|
||||
raise NotImplementedError
|
||||
|
||||
def combine_weights(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""(B, n_experts) train-time expert-combination weights.
|
||||
|
||||
Default (`gumbel=False`): identical to `gate()`. Opt-in
|
||||
straight-through Gumbel-softmax (`gumbel=True`, train mode only):
|
||||
hardens the forward pass to a one-hot sample (matching eval-time
|
||||
top-1 dispatch) while keeping the soft sample's gradient on backward.
|
||||
"""
|
||||
probs = self.gate(cond_cont, cond_cat)
|
||||
if not (self.gumbel and self.training):
|
||||
return probs
|
||||
log_probs = torch.log(probs.clamp_min(1e-8))
|
||||
return F.gumbel_softmax(log_probs, tau=self.gumbel_tau, hard=True, dim=-1)
|
||||
|
||||
def top1(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""(B,) hard expert index, used for eval-time grouped dispatch."""
|
||||
return self.gate(cond_cont, cond_cat).argmax(dim=-1)
|
||||
|
||||
def balance_loss(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""Importance CV^2 load-balancing auxiliary loss (Shazeer et al. 2017)."""
|
||||
importance = self.gate(cond_cont, cond_cat).sum(dim=0) # (n_experts,)
|
||||
return (importance.std() / (importance.mean() + 1e-8)) ** 2
|
||||
|
||||
def classify_loss(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, labels: torch.Tensor) -> torch.Tensor:
|
||||
"""Optional supervised auxiliary loss shaping the router's own belief.
|
||||
|
||||
Default: none (a scalar 0). Routers gating on an unobservable
|
||||
pre-step quantity (e.g. ProcessRouter) override this.
|
||||
"""
|
||||
return torch.zeros((), device=cond_cont.device)
|
||||
|
||||
def entropy_loss(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
"""Optional auxiliary loss rewarding sharper (lower-entropy) routing."""
|
||||
norm_entropy, _ = self.gate_stats(cond_cont, cond_cat)
|
||||
return norm_entropy
|
||||
|
||||
def gate_stats(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
"""Diagnostics: `(norm_entropy, importance)` — see v0.2 docstring for
|
||||
the full explanation, unchanged in v0.3.0."""
|
||||
gate = self.gate(cond_cont, cond_cat) # (B, n_experts)
|
||||
row_entropy = -(gate * (gate + 1e-8).log()).sum(dim=-1) # (B,)
|
||||
norm_entropy = row_entropy.mean() / math.log(self.n_experts)
|
||||
importance = gate.sum(dim=0) # (n_experts,)
|
||||
return norm_entropy, importance
|
||||
|
||||
|
||||
ROUTER_REGISTRY: dict[str, type[Router]] = {}
|
||||
|
||||
|
||||
def register_router(name: str):
|
||||
def decorator(cls: type[Router]) -> type[Router]:
|
||||
ROUTER_REGISTRY[name] = cls
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def build_router(name: str, n_experts: int, **kwargs) -> Router:
|
||||
"""Factory: look up a `Router` subclass by name from the registry.
|
||||
|
||||
Every registered router type is fed the same `router` config dict;
|
||||
kwargs not declared by that type's constructor are silently dropped, so
|
||||
per-type hyperparameters (e.g. EnergyRouter's `temperature`) can coexist
|
||||
in one config without special-casing.
|
||||
"""
|
||||
if name not in ROUTER_REGISTRY:
|
||||
raise ValueError(f"unknown router type {name!r}; available: {sorted(ROUTER_REGISTRY)}")
|
||||
cls = ROUTER_REGISTRY[name]
|
||||
accepted = set(inspect.signature(cls.__init__).parameters) - {"self", "n_experts"}
|
||||
filtered = {k: v for k, v in kwargs.items() if k in accepted}
|
||||
return cls(n_experts=n_experts, **filtered)
|
||||
|
||||
|
||||
def _bounded_interp(raw: torch.Tensor, lo: float, hi: float) -> torch.Tensor:
|
||||
"""Sigmoid interpolation into `[lo, hi]` — smooth, always-positive-gradient
|
||||
bound used for EnergyRouter's `learn_width`/`learn_temperature` modes."""
|
||||
return lo + (hi - lo) * torch.sigmoid(raw)
|
||||
|
||||
|
||||
def _inverse_bounded_interp(value: float, lo: float, hi: float) -> float:
|
||||
"""Inverse of `_bounded_interp`, used once at construction to warm-start
|
||||
`raw` so the initial effective width/temperature exactly equals `value`."""
|
||||
p = min(max((value - lo) / (hi - lo), 1e-6), 1 - 1e-6)
|
||||
return math.log(p / (1 - p))
|
||||
|
||||
|
||||
@register_router("energy")
|
||||
class EnergyRouter(Router):
|
||||
"""Soft turn-on gate over normalized pre-step log-energy.
|
||||
|
||||
Reads `cond_cont[:, energy_idx]` (ignores cond_cat). `gate(e) =
|
||||
softmax_i(-(e - c_i)^2 / tau)`; as tau -> 0 this hardens to
|
||||
nearest-center (Voronoi) selection, exactly what `top1` uses at eval.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
n_experts: int = 4,
|
||||
temperature: float = 0.5,
|
||||
learn_centers: bool = True,
|
||||
energy_idx: int = 3,
|
||||
centers_init: Sequence[float] | None = None,
|
||||
learn_width: bool = False,
|
||||
learn_temperature: bool = False,
|
||||
width_min_ratio: float = 0.1,
|
||||
width_max_ratio: float = 10.0,
|
||||
) -> None:
|
||||
super().__init__(n_experts)
|
||||
if learn_width and learn_temperature:
|
||||
raise ValueError("learn_width and learn_temperature are mutually exclusive")
|
||||
self.temperature = temperature
|
||||
self.energy_idx = energy_idx
|
||||
self.learn_width = learn_width
|
||||
self.learn_temperature = learn_temperature
|
||||
if learn_width or learn_temperature:
|
||||
if not (width_min_ratio < 1.0 < width_max_ratio):
|
||||
raise ValueError(
|
||||
f"width_min_ratio ({width_min_ratio}) and width_max_ratio ({width_max_ratio}) must bracket 1.0"
|
||||
)
|
||||
self._width_lo = width_min_ratio * temperature
|
||||
self._width_hi = width_max_ratio * temperature
|
||||
raw0 = _inverse_bounded_interp(temperature, self._width_lo, self._width_hi)
|
||||
if learn_width:
|
||||
self.raw_width = nn.Parameter(torch.full((n_experts,), raw0))
|
||||
else:
|
||||
self.raw_temperature = nn.Parameter(torch.tensor(raw0))
|
||||
if centers_init is None:
|
||||
centers = torch.linspace(-2.0, 2.0, n_experts)
|
||||
else:
|
||||
if len(centers_init) != n_experts:
|
||||
raise ValueError(f"centers_init has {len(centers_init)} values, expected n_experts={n_experts}")
|
||||
centers = torch.tensor(list(centers_init), dtype=torch.float32)
|
||||
if learn_centers:
|
||||
self.centers = nn.Parameter(centers)
|
||||
else:
|
||||
self.register_buffer("centers", centers)
|
||||
|
||||
def effective_width(self) -> torch.Tensor | float:
|
||||
if self.learn_width:
|
||||
return _bounded_interp(self.raw_width, self._width_lo, self._width_hi)
|
||||
if self.learn_temperature:
|
||||
return _bounded_interp(self.raw_temperature, self._width_lo, self._width_hi)
|
||||
return self.temperature
|
||||
|
||||
def gate(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
e = cond_cont[:, self.energy_idx].unsqueeze(-1) # (B, 1)
|
||||
d2 = (e - self.centers.unsqueeze(0)) ** 2 # (B, n_experts)
|
||||
return torch.softmax(-d2 / self.effective_width(), dim=-1)
|
||||
|
||||
|
||||
@register_router("pdg")
|
||||
class PdgRouter(Router):
|
||||
"""Soft turn-on gate over a learned PDG embedding (own table, separate
|
||||
from the trunk's `ConditionEncoder`). No supervision needed — PDG code
|
||||
is already known at pre-step time."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
n_experts: int,
|
||||
pdg_vocab: int,
|
||||
emb_dim: int = 8,
|
||||
temperature: float = 0.5,
|
||||
learn_centers: bool = True,
|
||||
) -> None:
|
||||
super().__init__(n_experts)
|
||||
self.temperature = temperature
|
||||
self.pdg_emb = nn.Embedding(pdg_vocab, emb_dim)
|
||||
centers = torch.randn(n_experts, emb_dim) * 0.1
|
||||
if learn_centers:
|
||||
self.centers = nn.Parameter(centers)
|
||||
else:
|
||||
self.register_buffer("centers", centers)
|
||||
|
||||
def gate(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
e = self.pdg_emb(cond_cat[:, CondLayout.PDG_COL]) # (B, emb_dim)
|
||||
d2 = ((e.unsqueeze(1) - self.centers.unsqueeze(0)) ** 2).sum(-1) # (B, n_experts)
|
||||
return torch.softmax(-d2 / self.temperature, dim=-1)
|
||||
|
||||
|
||||
@register_router("process")
|
||||
class ProcessRouter(Router):
|
||||
"""Routes on the physics process expected to end the step — a post-step
|
||||
outcome, so a small classifier over pre-step conditioning predicts it
|
||||
(own pdg/material embeddings, separate from the trunk's ConditionEncoder).
|
||||
`n_experts` doubles as the number of process classes. Supervised via
|
||||
`classify_loss` against the true `process` label at train time only;
|
||||
`gate`/`top1` never see it."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
n_experts: int,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
emb_dim: int = 8,
|
||||
hidden_dim: int = 64,
|
||||
) -> None:
|
||||
super().__init__(n_experts)
|
||||
self.pdg_emb = nn.Embedding(pdg_vocab, emb_dim)
|
||||
self.mat_emb = nn.Embedding(mat_vocab, emb_dim)
|
||||
self.classifier = nn.Sequential(
|
||||
nn.Linear(COND_DIM + 2 * emb_dim, hidden_dim),
|
||||
nn.SiLU(),
|
||||
nn.Linear(hidden_dim, n_experts),
|
||||
)
|
||||
|
||||
def logits(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
pdg_e = self.pdg_emb(cond_cat[:, CondLayout.PDG_COL])
|
||||
mat_e = self.mat_emb(cond_cat[:, CondLayout.MAT_COL])
|
||||
h = torch.cat([cond_cont, pdg_e, mat_e], dim=-1)
|
||||
return self.classifier(h)
|
||||
|
||||
def gate(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
return torch.softmax(self.logits(cond_cont, cond_cat), dim=-1)
|
||||
|
||||
def classify_loss(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, labels: torch.Tensor) -> torch.Tensor:
|
||||
return F.cross_entropy(self.logits(cond_cont, cond_cat), labels)
|
||||
|
||||
|
||||
class ComposedRouter(Router):
|
||||
"""Joint router over independent axes (e.g. energy x pdg), outer-product
|
||||
gated. Not registered in `ROUTER_REGISTRY`; use `build_composed_router`."""
|
||||
|
||||
def __init__(self, routers: list[Router]) -> None:
|
||||
if not routers:
|
||||
raise ValueError("ComposedRouter needs at least one sub-router")
|
||||
n_experts = 1
|
||||
for r in routers:
|
||||
n_experts *= r.n_experts
|
||||
super().__init__(n_experts)
|
||||
self.routers = nn.ModuleList(routers)
|
||||
|
||||
def gate(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor) -> torch.Tensor:
|
||||
joint = self.routers[0].gate(cond_cont, cond_cat) # (B, n_0)
|
||||
for router in self.routers[1:]:
|
||||
g = router.gate(cond_cont, cond_cat) # (B, n_i)
|
||||
joint = (joint.unsqueeze(-1) * g.unsqueeze(1)).flatten(1) # (B, prod so far)
|
||||
return joint
|
||||
|
||||
def classify_loss(self, cond_cont: torch.Tensor, cond_cat: torch.Tensor, labels: torch.Tensor) -> torch.Tensor:
|
||||
total = torch.zeros((), device=cond_cont.device)
|
||||
for router in self.routers:
|
||||
total = total + router.classify_loss(cond_cont, cond_cat, labels)
|
||||
return total
|
||||
|
||||
|
||||
def build_composed_router(specs: list[dict], **shared_kwargs) -> ComposedRouter:
|
||||
"""Build a `ComposedRouter` from a list of per-axis router specs — see
|
||||
`_parse_composed_axes`."""
|
||||
routers = [
|
||||
build_router(
|
||||
spec["type"],
|
||||
spec["n_experts"],
|
||||
**{
|
||||
**shared_kwargs,
|
||||
**{k: v for k, v in spec.items() if k not in ("type", "n_experts")},
|
||||
},
|
||||
)
|
||||
for spec in specs
|
||||
]
|
||||
return ComposedRouter(routers)
|
||||
|
||||
|
||||
_AXIS_KEY_RE = re.compile(r"^axis(\d+)_(.+)$")
|
||||
|
||||
|
||||
def _parse_composed_axes(router_cfg: dict) -> list[dict]:
|
||||
"""Regroup `axis{i}_{field}` flat keys into a list of per-axis spec dicts.
|
||||
|
||||
e.g. `axis0_type = "energy"`, `axis0_n_experts = 4`, `axis1_type = "pdg"`,
|
||||
`axis1_n_experts = 3`, `axis1_emb_dim = 8`. Axis indices must be
|
||||
contiguous from 0.
|
||||
"""
|
||||
axes: dict[int, dict] = {}
|
||||
for key, value in router_cfg.items():
|
||||
m = _AXIS_KEY_RE.match(key)
|
||||
if m is None:
|
||||
continue
|
||||
idx, field = int(m.group(1)), m.group(2)
|
||||
axes.setdefault(idx, {})[field] = value
|
||||
missing = set(range(len(axes))) - axes.keys()
|
||||
if missing:
|
||||
raise ValueError(f"composed router config has gaps at axis indices {missing}")
|
||||
return [axes[i] for i in range(len(axes))]
|
||||
|
||||
|
||||
# Router types that read cond_cat's pdg index through their own
|
||||
# nn.Embedding(pdg_vocab, ...), regardless of the trunk's particle
|
||||
# conditioning mode — see _check_router_conditioning_compat.
|
||||
_VOCAB_SCOPED_ROUTER_TYPES = ("pdg", "process")
|
||||
|
||||
|
||||
def _check_router_conditioning_compat(router_types: list[str], particle_conditioning: str) -> None:
|
||||
"""Reject a router axis that reintroduces a training-vocab PDG lookup
|
||||
under `conditioning.particle.type = "physical"`.
|
||||
|
||||
`PdgRouter`/`ProcessRouter` always build their own dataset-scoped
|
||||
`nn.Embedding(pdg_vocab, ...)`, independent of `ConditionEncoder`'s
|
||||
particle mode. Pairing either with `"physical"` would silently
|
||||
reintroduce a training-menu-scoped lookup at the routing layer,
|
||||
defeating the point of physical-property conditioning. Raised loudly at
|
||||
model-build time.
|
||||
"""
|
||||
bad = sorted(set(router_types) & set(_VOCAB_SCOPED_ROUTER_TYPES))
|
||||
if bad and particle_conditioning == "physical":
|
||||
raise ValueError(
|
||||
f"router type(s) {bad} always use a training-vocab PDG embedding, "
|
||||
"which is incompatible with conditioning.particle.type='physical' "
|
||||
"(whose whole point is generalizing beyond that vocab) — pick a "
|
||||
"different router type (e.g. 'energy') or use "
|
||||
"conditioning.particle.type='embedding'."
|
||||
)
|
||||
|
||||
|
||||
def _build_router_from_cfg(
|
||||
router_cfg: dict,
|
||||
pdg_vocab: int,
|
||||
mat_vocab: int,
|
||||
particle_conditioning: str = "embedding",
|
||||
) -> Router:
|
||||
"""Resolve one stage's `router` config into a `Router`, single-axis or
|
||||
composed. `gumbel` is set as a post-construction attribute (shared by
|
||||
every router type, not a per-type constructor kwarg)."""
|
||||
shared_vocab = dict(pdg_vocab=pdg_vocab, mat_vocab=mat_vocab)
|
||||
if router_cfg["type"] == "composed":
|
||||
axes = _parse_composed_axes(router_cfg)
|
||||
_check_router_conditioning_compat([a["type"] for a in axes], particle_conditioning)
|
||||
router = build_composed_router(axes, **shared_vocab)
|
||||
router.gumbel = bool(router_cfg.get("gumbel", False))
|
||||
return router
|
||||
_check_router_conditioning_compat([router_cfg["type"]], particle_conditioning)
|
||||
router_kwargs = {k: v for k, v in router_cfg.items() if k not in ("enabled", "type", "n_experts")}
|
||||
router_kwargs.setdefault("pdg_vocab", pdg_vocab)
|
||||
router_kwargs.setdefault("mat_vocab", mat_vocab)
|
||||
router = build_router(router_cfg["type"], router_cfg["n_experts"], **router_kwargs)
|
||||
router.gumbel = bool(router_cfg.get("gumbel", False))
|
||||
return router
|
||||
@@ -0,0 +1,214 @@
|
||||
"""Trunks: everything downstream of the fused conditioning vector — a
|
||||
registrable expert *body* architecture (`TRUNK_REGISTRY`/`register_trunk`),
|
||||
used standalone or mixed by a `Router` (issues.md Issue 8; trunk-selectability
|
||||
gitea #33).
|
||||
|
||||
Whether a body is mixed is orthogonal to which body it is: `RoutedTrunk`
|
||||
builds `router.n_experts` instances of whichever body `trunk_type` names, so
|
||||
a future body (e.g. a transformer) automatically gets a mixture variant for
|
||||
free — no separate "routed transformer trunk" class needed.
|
||||
"""
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
from giant.model.layers import build_block
|
||||
from giant.model.routers import Router
|
||||
|
||||
TRUNK_REGISTRY: dict[str, type[nn.Module]] = {}
|
||||
|
||||
|
||||
def register_trunk(name: str):
|
||||
def decorator(cls: type[nn.Module]) -> type[nn.Module]:
|
||||
TRUNK_REGISTRY[name] = cls
|
||||
return cls
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
def build_expert_body(
|
||||
name: str,
|
||||
in_dim: int,
|
||||
out_dim: int,
|
||||
hidden_dim: int,
|
||||
n_blocks: int,
|
||||
cond_dim: int,
|
||||
dropout: float = 0.0,
|
||||
block_conditioning: str = "add",
|
||||
) -> nn.Module:
|
||||
"""Factory: look up a registered trunk body by name and construct one
|
||||
instance of it — used both for a standalone (unrouted) trunk and for each
|
||||
expert inside a `RoutedTrunk`. `block_conditioning` selects the
|
||||
`BLOCK_REGISTRY` entry each body's internal `ResBlock`-family blocks use
|
||||
(`trunk.block_conditioning`, gitea #34) — an optional trailing kwarg a
|
||||
future non-`ResBlock`-based body can simply ignore, same idiom as
|
||||
`Trunk.forward`'s accept-and-ignore `cond_cont`/`cond_cat`."""
|
||||
if name not in TRUNK_REGISTRY:
|
||||
raise ValueError(f"unknown trunk type {name!r}; available: {sorted(TRUNK_REGISTRY)}")
|
||||
cls = TRUNK_REGISTRY[name]
|
||||
return cls(in_dim, out_dim, hidden_dim, n_blocks, cond_dim, dropout, block_conditioning=block_conditioning)
|
||||
|
||||
|
||||
@register_trunk("resmlp")
|
||||
class ExpertTrunk(nn.Module):
|
||||
"""`input_proj -> ResBlock stack -> out_proj` — the registered `"resmlp"`
|
||||
trunk body. Used both standalone (no router: `forward`'s `cond_cont`/
|
||||
`cond_cat` are accepted and ignored, satisfying the `Trunk` interface
|
||||
directly with no wrapper class) and as one expert inside a `RoutedTrunk`
|
||||
(`_route_forward` calls it with just `(x, cond)`).
|
||||
|
||||
Unlike v0.2, `out_dim` is independent of `in_dim` — needed by stage-2 AR
|
||||
tokens later (`noise_dim` in, `4 + type_dim` out), even though every
|
||||
step-2/3 caller still has `in_dim == out_dim`.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_dim: int,
|
||||
out_dim: int,
|
||||
hidden_dim: int,
|
||||
n_blocks: int,
|
||||
cond_dim: int,
|
||||
dropout: float = 0.0,
|
||||
block_conditioning: str = "add",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.out_dim = out_dim
|
||||
self.input_proj = nn.Linear(in_dim, hidden_dim)
|
||||
self.blocks = nn.ModuleList(
|
||||
[build_block(block_conditioning, hidden_dim, cond_dim, dropout) for _ in range(n_blocks)]
|
||||
)
|
||||
self.out_proj = nn.Linear(hidden_dim, out_dim)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cond: torch.Tensor,
|
||||
cond_cont: torch.Tensor | None = None,
|
||||
cond_cat: torch.Tensor | None = None,
|
||||
) -> torch.Tensor:
|
||||
x = self.input_proj(x)
|
||||
for block in self.blocks:
|
||||
x = block(x, cond)
|
||||
return self.out_proj(x)
|
||||
|
||||
|
||||
def _route_forward(
|
||||
experts: nn.ModuleList,
|
||||
router: Router,
|
||||
x: torch.Tensor,
|
||||
cond: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
training: bool,
|
||||
) -> torch.Tensor:
|
||||
"""Shared dispatch for `RoutedTrunk`.
|
||||
|
||||
Train mode: full mixture `sum_i weight_i * expert_i(x)` — always
|
||||
N-expert dense compute, fully differentiable (`weight` is
|
||||
`router.combine_weights`). Eval mode: grouped top-1 dispatch — each row
|
||||
runs exactly one expert, the actual source of the per-call speedup.
|
||||
"""
|
||||
if training:
|
||||
weights = router.combine_weights(cond_cont, cond_cat) # (B, n_experts)
|
||||
out = torch.zeros(x.shape[0], experts[0].out_dim, device=x.device)
|
||||
for i, expert in enumerate(experts):
|
||||
out = out + weights[:, i : i + 1] * expert(x, cond)
|
||||
return out
|
||||
|
||||
idx = router.top1(cond_cont, cond_cat) # (B,)
|
||||
out_dim = experts[0].out_dim
|
||||
out = torch.zeros(x.shape[0], out_dim, device=x.device)
|
||||
for i, expert in enumerate(experts):
|
||||
mask = idx == i
|
||||
if mask.any():
|
||||
out[mask] = expert(x[mask], cond[mask])
|
||||
return out
|
||||
|
||||
|
||||
class Trunk(nn.Module):
|
||||
"""Interface implemented by a standalone trunk body (any `TRUNK_REGISTRY`
|
||||
entry, e.g. `ExpertTrunk`) and by `RoutedTrunk`: everything downstream of
|
||||
the fused conditioning vector, i.e. the actual generative trunk of a
|
||||
stage."""
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cond: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class RoutedTrunk(Trunk):
|
||||
def __init__(
|
||||
self,
|
||||
router: Router,
|
||||
trunk_type: str,
|
||||
in_dim: int,
|
||||
out_dim: int,
|
||||
hidden_dim: int,
|
||||
n_res_blocks: int,
|
||||
cond_dim: int,
|
||||
dropout: float = 0.0,
|
||||
block_conditioning: str = "add",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.router = router
|
||||
self.experts = nn.ModuleList(
|
||||
[
|
||||
build_expert_body(
|
||||
trunk_type,
|
||||
in_dim,
|
||||
out_dim,
|
||||
hidden_dim,
|
||||
n_res_blocks,
|
||||
cond_dim,
|
||||
dropout,
|
||||
block_conditioning=block_conditioning,
|
||||
)
|
||||
for _ in range(router.n_experts)
|
||||
]
|
||||
)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
x: torch.Tensor,
|
||||
cond: torch.Tensor,
|
||||
cond_cont: torch.Tensor,
|
||||
cond_cat: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
return _route_forward(self.experts, self.router, x, cond, cond_cont, cond_cat, self.training)
|
||||
|
||||
|
||||
def build_trunk(
|
||||
router: Router | None,
|
||||
trunk_type: str,
|
||||
in_dim: int,
|
||||
out_dim: int,
|
||||
hidden_dim: int,
|
||||
n_res_blocks: int,
|
||||
cond_dim: int,
|
||||
dropout: float = 0.0,
|
||||
block_conditioning: str = "add",
|
||||
) -> nn.Module:
|
||||
"""Build a stage's trunk: `trunk_type` (a `TRUNK_REGISTRY` key, e.g.
|
||||
`"resmlp"`) selects the expert body architecture; `router`, if given,
|
||||
wraps `router.n_experts` instances of that body in a `RoutedTrunk`
|
||||
mixture — otherwise a single body is returned directly (no wrapper
|
||||
class), which is what makes an unrouted trunk's state-dict keys land
|
||||
directly under `trunk.*` instead of `trunk.experts.0.*` (see
|
||||
`giant.model._legacy.migrate_legacy_state_dict`, which assumes exactly
|
||||
this flat layout for a v0.2 monolithic checkpoint). `block_conditioning`
|
||||
(a `BLOCK_REGISTRY` key, e.g. `"add"`/`"film"`/`"adaln"`) selects each
|
||||
body's conditioning-injection mechanism (gitea #34).
|
||||
"""
|
||||
if router is not None:
|
||||
return RoutedTrunk(
|
||||
router, trunk_type, in_dim, out_dim, hidden_dim, n_res_blocks, cond_dim, dropout, block_conditioning
|
||||
)
|
||||
return build_expert_body(
|
||||
trunk_type, in_dim, out_dim, hidden_dim, n_res_blocks, cond_dim, dropout, block_conditioning
|
||||
)
|
||||
+7
-5
@@ -140,9 +140,8 @@ def decode_topn_class(
|
||||
other_policy: str = "sample",
|
||||
rng: np.random.Generator | None = None,
|
||||
) -> np.ndarray:
|
||||
"""`conditioning.particle.type` / `stage2_model.particle_type.target =
|
||||
"onehot"` inference decode: per-row top-N class index -> concrete PDG
|
||||
code.
|
||||
"""`stage2_model.particle_type.target = "onehot"` inference decode:
|
||||
per-row top-N class index -> concrete secondary-species PDG code.
|
||||
|
||||
class_idx: int array, any shape, values in `[0, n_classes)`.
|
||||
topn_map: the `TopNMap` (`giant.data.loader.build_pdg_topn_map_from_files`)
|
||||
@@ -150,8 +149,11 @@ def decode_topn_class(
|
||||
except at the shared "other" index) plus `other_members` (the
|
||||
empirical within-"other" distribution, needed for `other_policy =
|
||||
"sample"`/`"modal"`).
|
||||
n_classes: `conditioning.particle.emb_dim` — the class count; the "other"
|
||||
bucket is index `n_classes - 1` by construction
|
||||
n_classes: the resolved secondary-species class count
|
||||
(`giant.model.models.resolve_type_n_classes` —
|
||||
`stage2_model.particle_type.n_classes`, 0 = inherit
|
||||
`conditioning.particle.emb_dim`; see gitea #29); the "other" bucket
|
||||
is index `n_classes - 1` by construction
|
||||
(`giant.data.loader._topn_plus_other_map`).
|
||||
other_policy: `"sample"` draws from `other_members`' empirical frequency;
|
||||
`"modal"` always the single most common "other" member; `"drop"`
|
||||
|
||||
+48
-21
@@ -31,7 +31,7 @@ from giant.data.transforms import (
|
||||
sorted_membership,
|
||||
)
|
||||
from giant.data.dataset import make_event_split, StreamingStepsDataset
|
||||
from giant.model.network import build_models, build_critics
|
||||
from giant.model.network import build_models, build_critics, resolve_type_n_classes
|
||||
from giant.training import train as run_training
|
||||
|
||||
|
||||
@@ -49,6 +49,7 @@ class SetupStageResult:
|
||||
mat_map: dict[str, int]
|
||||
proc_map: dict[str, int] | None
|
||||
pdg_topn_map: TopNMap | None
|
||||
sec_type_topn_map: TopNMap | None
|
||||
mat_topn_map: TopNMap | None
|
||||
cond_norm: Normalizer
|
||||
tgt_norm: Normalizer
|
||||
@@ -175,29 +176,42 @@ def run_setup_stage(
|
||||
cache.proc_maps[n_experts] = proc_map
|
||||
|
||||
# Top-N-plus-other maps for onehot conditioning/type axes.
|
||||
# The PDG axis is shared by
|
||||
# conditioning.particle.type="onehot" and
|
||||
# stage2_model.particle_type.target="onehot" (both key off
|
||||
# conditioning.particle.emb_dim), so at most one PDG scan is needed even
|
||||
# if both consumers are active. The material axis is independent.
|
||||
# The PDG axis is used independently by conditioning.particle.type="onehot"
|
||||
# (cond_cat's onehot feature) and stage2_model.particle_type.target="onehot"
|
||||
# (secondary-species decode) — their class counts can now differ (gitea
|
||||
# #29: stage2_model.particle_type.n_classes, 0 = inherit
|
||||
# conditioning.particle.emb_dim), so each is resolved and built
|
||||
# independently via _pdg_topn below. cache.topn_maps is keyed by
|
||||
# (axis, n_classes) (setup_cache.topn_key), so when the two resolve to
|
||||
# the same N the second call is a cache hit against the first — no extra
|
||||
# scan in the common case where they still match. The material axis is
|
||||
# independent of both.
|
||||
particle_cfg = cfg["conditioning"]["particle"]
|
||||
material_cfg = cfg["conditioning"]["material"]
|
||||
particle_type_target = config.ParticleTypeConfig.from_dict(cfg["stage2_model"].get("particle_type")).target
|
||||
particle_type_cfg = config.ParticleTypeConfig.from_dict(cfg["stage2_model"].get("particle_type"))
|
||||
particle_type_target = particle_type_cfg.target
|
||||
|
||||
pdg_topn_map: TopNMap | None = None
|
||||
if particle_cfg["type"] == "onehot" or particle_type_target == "onehot":
|
||||
n_classes = particle_cfg["emb_dim"]
|
||||
def _pdg_topn(n_classes: int) -> TopNMap:
|
||||
cache_key = setup_cache.topn_key("pdg", n_classes)
|
||||
cached = cache.topn_maps.get(cache_key) if cache is not None else None
|
||||
if cached is not None:
|
||||
pdg_topn_map = cached
|
||||
echo(f"pdg top-N map: cache hit ({len(pdg_topn_map.class_map)} codes, {n_classes} classes)")
|
||||
else:
|
||||
echo("building pdg top-N map …")
|
||||
pdg_topn_map = build_pdg_topn_map_from_files(files, n_classes=n_classes)
|
||||
echo(f" {len(pdg_topn_map.class_map)} pdg codes mapped to {n_classes} classes")
|
||||
if cache is not None:
|
||||
cache.topn_maps[cache_key] = pdg_topn_map
|
||||
echo(f"pdg top-N map: cache hit ({len(cached.class_map)} codes, {n_classes} classes)")
|
||||
return cached
|
||||
echo("building pdg top-N map …")
|
||||
topn_map = build_pdg_topn_map_from_files(files, n_classes=n_classes)
|
||||
echo(f" {len(topn_map.class_map)} pdg codes mapped to {n_classes} classes")
|
||||
if cache is not None:
|
||||
cache.topn_maps[cache_key] = topn_map
|
||||
return topn_map
|
||||
|
||||
pdg_topn_map: TopNMap | None = None
|
||||
if particle_cfg["type"] == "onehot":
|
||||
pdg_topn_map = _pdg_topn(particle_cfg["emb_dim"])
|
||||
|
||||
sec_type_topn_map: TopNMap | None = None
|
||||
if particle_type_target == "onehot":
|
||||
sec_type_n_classes = resolve_type_n_classes(particle_type_cfg, particle_cfg["emb_dim"])
|
||||
sec_type_topn_map = _pdg_topn(sec_type_n_classes)
|
||||
|
||||
mat_topn_map: TopNMap | None = None
|
||||
if material_cfg["type"] == "onehot":
|
||||
@@ -248,7 +262,7 @@ def run_setup_stage(
|
||||
if not mask.any():
|
||||
continue
|
||||
chunk_tr = {k: v[mask] for k, v in chunk.items()}
|
||||
cond_cont, _, target_s1, n_sec, sec_cont, _proc, _, _, _ = build_features(
|
||||
feats = build_features(
|
||||
chunk_tr,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
@@ -257,8 +271,19 @@ def run_setup_stage(
|
||||
particle_conditioning=particle_conditioning,
|
||||
material_conditioning=material_conditioning,
|
||||
sec_phys_only=True,
|
||||
# This pass reads only cond_cont/sec_cont, never cond_cat —
|
||||
# but cond_cat's width is the conditioning modes' call
|
||||
# (giant.cond_layout.CondLayout), so an "onehot" axis still
|
||||
# has to be handed its map rather than silently yielding a
|
||||
# narrower array.
|
||||
pdg_topn_map=pdg_topn_map.class_map if pdg_topn_map is not None else None,
|
||||
mat_topn_map=mat_topn_map.class_map if mat_topn_map is not None else None,
|
||||
k_max=k_max,
|
||||
)
|
||||
cond_cont = feats.cond_cont
|
||||
target_s1 = feats.target_s1
|
||||
n_sec = feats.n_sec
|
||||
sec_cont = feats.sec_cont
|
||||
cond_acc.update(cond_cont)
|
||||
tgt_acc.update(target_s1)
|
||||
if energy_sampler is not None:
|
||||
@@ -292,6 +317,7 @@ def run_setup_stage(
|
||||
mat_map=mat_map,
|
||||
proc_map=proc_map,
|
||||
pdg_topn_map=pdg_topn_map,
|
||||
sec_type_topn_map=sec_type_topn_map,
|
||||
mat_topn_map=mat_topn_map,
|
||||
cond_norm=cond_norm,
|
||||
tgt_norm=tgt_norm,
|
||||
@@ -381,8 +407,8 @@ def run_train_job(
|
||||
# (physical stays untouched/None).
|
||||
particle_type_target = config.ParticleTypeConfig.from_dict(cfg["stage2_model"].get("particle_type")).target
|
||||
if particle_type_target == "onehot":
|
||||
assert setup.pdg_topn_map is not None
|
||||
sec_type_class_map = setup.pdg_topn_map.class_map
|
||||
assert setup.sec_type_topn_map is not None
|
||||
sec_type_class_map = setup.sec_type_topn_map.class_map
|
||||
elif particle_type_target == "embedding":
|
||||
sec_type_class_map = pdg_map
|
||||
else:
|
||||
@@ -486,6 +512,7 @@ def run_train_job(
|
||||
mat_map={str(k): v for k, v in mat_map.items()},
|
||||
proc_map=proc_map,
|
||||
pdg_topn_map=setup.pdg_topn_map,
|
||||
sec_type_topn_map=setup.sec_type_topn_map,
|
||||
mat_topn_map=setup.mat_topn_map,
|
||||
model_config=model_config,
|
||||
resume_path=resume,
|
||||
|
||||
+25
-14
@@ -117,7 +117,7 @@ def decode_secondary_identity(
|
||||
pre_dir: np.ndarray,
|
||||
sec_phys_norm: Normalizer,
|
||||
pdg_map: dict[int, int],
|
||||
pdg_topn_map: "TopNMap | None",
|
||||
sec_type_topn_map: "TopNMap | None",
|
||||
other_policy: str,
|
||||
rng: np.random.Generator | None,
|
||||
) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray, np.ndarray, np.ndarray | None]:
|
||||
@@ -141,7 +141,7 @@ def decode_secondary_identity(
|
||||
Returns (sec_E, sec_dir_world, sec_mass, sec_charge, sec_pdg,
|
||||
sec_type_l1_dist) — the last is `None` except under `"embedding"`.
|
||||
"""
|
||||
target = sec_decoder.particle_type_cfg.get("target", "physical")
|
||||
target = sec_decoder.particle_type_cfg.target
|
||||
|
||||
if target == "physical":
|
||||
sec_full = torch.cat([sec_cont, sec_type], dim=-1).cpu().numpy()
|
||||
@@ -158,15 +158,15 @@ def decode_secondary_identity(
|
||||
l1_dist = None
|
||||
|
||||
if target == "onehot":
|
||||
if pdg_topn_map is None:
|
||||
if sec_type_topn_map is None:
|
||||
raise RuntimeError(
|
||||
"particle_type.target='onehot' rollout needs pdg_topn_map "
|
||||
"(the checkpoint's saved top-N map) — see ckpt['pdg_topn_map']"
|
||||
"particle_type.target='onehot' rollout needs sec_type_topn_map "
|
||||
"(the checkpoint's saved top-N map) — see ckpt['sec_type_topn_map']"
|
||||
)
|
||||
class_idx = sec_type_np.argmax(axis=-1)
|
||||
sec_pdg = decode_topn_class(
|
||||
class_idx,
|
||||
pdg_topn_map,
|
||||
sec_type_topn_map,
|
||||
n_classes=sec_decoder.type_dim,
|
||||
other_policy=other_policy,
|
||||
rng=rng,
|
||||
@@ -444,6 +444,7 @@ def rollout(
|
||||
material_conditioning: str = "embedding",
|
||||
pdg_topn_map: "TopNMap | None" = None,
|
||||
mat_topn_map: "TopNMap | None" = None,
|
||||
sec_type_topn_map: "TopNMap | None" = None,
|
||||
other_policy: str = "sample",
|
||||
seed: int | None = None,
|
||||
stage1_ddpm_steps: int = 1000,
|
||||
@@ -469,14 +470,17 @@ def rollout(
|
||||
autoregressive) is inferred from `sec_decoder`'s own class — see
|
||||
`sample_stage1`/`sample_stage2` (giant.sample).
|
||||
|
||||
`pdg_topn_map`/`mat_topn_map` serve two independent purposes that happen
|
||||
to share `pdg_topn_map` (one PDG map, not two): they're required
|
||||
whenever `particle_conditioning`/`material_conditioning` is `"onehot"`
|
||||
(feeds `build_cond_features`'s extra `cond_cat` top-N columns), and
|
||||
`pdg_topn_map`/`other_policy` are additionally read under
|
||||
`pdg_topn_map`/`mat_topn_map`/`sec_type_topn_map` serve three independent
|
||||
purposes, no longer required to share one map (see gitea #29):
|
||||
`pdg_topn_map`/`mat_topn_map` are required whenever
|
||||
`particle_conditioning`/`material_conditioning` is `"onehot"` (feeds
|
||||
`build_cond_features`'s extra `cond_cat` top-N columns); `sec_type_topn_map`/
|
||||
`other_policy` are required instead under
|
||||
`stage2_model.particle_type.target = "onehot"` (secondary-species
|
||||
decode). `seed` seeds the `other_policy = "sample"` draw only
|
||||
(torch/numpy sampling itself is seeded by the caller, same as today).
|
||||
decode) — its class count (`stage2_model.particle_type.n_classes`) may
|
||||
differ from `pdg_topn_map`'s. `seed` seeds the `other_policy = "sample"`
|
||||
draw only (torch/numpy sampling itself is seeded by the caller, same as
|
||||
today).
|
||||
|
||||
`l1_dist_collector`, if given, accumulates the embedding-distance
|
||||
diagnostic across the whole run — see `L1DistCollector`. Only populated
|
||||
@@ -487,6 +491,11 @@ def rollout(
|
||||
"conditioning.particle.type='onehot' rollout needs pdg_topn_map "
|
||||
"(the checkpoint's saved top-N map) — see ckpt['pdg_topn_map']"
|
||||
)
|
||||
if sec_decoder.particle_type_cfg.target == "onehot" and sec_type_topn_map is None:
|
||||
raise RuntimeError(
|
||||
"stage2_model.particle_type.target='onehot' rollout needs sec_type_topn_map "
|
||||
"(the checkpoint's saved top-N map) — see ckpt['sec_type_topn_map']"
|
||||
)
|
||||
if material_conditioning == "onehot" and mat_topn_map is None:
|
||||
raise RuntimeError(
|
||||
"conditioning.material.type='onehot' rollout needs mat_topn_map "
|
||||
@@ -536,6 +545,7 @@ def rollout(
|
||||
material_conditioning,
|
||||
pdg_topn_map,
|
||||
mat_topn_map,
|
||||
sec_type_topn_map,
|
||||
other_policy,
|
||||
rng,
|
||||
stage1_ddpm_steps,
|
||||
@@ -574,6 +584,7 @@ def _step_chunk(
|
||||
material_conditioning,
|
||||
pdg_topn_map,
|
||||
mat_topn_map,
|
||||
sec_type_topn_map,
|
||||
other_policy,
|
||||
rng,
|
||||
stage1_ddpm_steps,
|
||||
@@ -690,7 +701,7 @@ def _step_chunk(
|
||||
tr["pre_dir"],
|
||||
sec_phys_norm,
|
||||
pdg_map,
|
||||
pdg_topn_map,
|
||||
sec_type_topn_map,
|
||||
other_policy,
|
||||
rng,
|
||||
)
|
||||
|
||||
+10
-10
@@ -2,7 +2,7 @@ import torch
|
||||
import torch.nn.functional as F
|
||||
|
||||
from giant.constants import CONT_SLOT_DIM, X_DIM
|
||||
from giant.model.network import Stage2Autoregressive, stage2_trunk_sec_dim
|
||||
from giant.model.network import DdpmObjective, Stage2Autoregressive, build_objective, stage2_trunk_sec_dim
|
||||
from giant.model.schedule import CosineSchedule
|
||||
|
||||
|
||||
@@ -131,8 +131,8 @@ def _stage2_flat_width(sec_decoder: torch.nn.Module) -> int:
|
||||
|
||||
|
||||
def _type_folded(sec_decoder: torch.nn.Module) -> bool:
|
||||
target = sec_decoder.particle_type_cfg.get("target", "physical")
|
||||
return target == "physical" or sec_decoder.generator_kind == "wgan"
|
||||
target = sec_decoder.particle_type_cfg.target
|
||||
return target == "physical" or build_objective(sec_decoder.generator_kind).folds_type_slice
|
||||
|
||||
|
||||
def _decode_stage2_flat(
|
||||
@@ -281,8 +281,8 @@ def sample_secondaries_ar(
|
||||
device = cond_cont.device
|
||||
k_max = sec_decoder.k_max
|
||||
type_dim = sec_decoder.type_dim
|
||||
generator = sec_decoder.generator_kind
|
||||
target = sec_decoder.particle_type_cfg.get("target", "physical")
|
||||
objective = build_objective(sec_decoder.generator_kind)
|
||||
target = sec_decoder.particle_type_cfg.target
|
||||
type_folded = _type_folded(sec_decoder)
|
||||
token_dim = CONT_SLOT_DIM + type_dim if type_folded else CONT_SLOT_DIM
|
||||
|
||||
@@ -301,7 +301,7 @@ def sample_secondaries_ar(
|
||||
slot_idx = torch.full((B, 1), k / max(k_max - 1, 1), device=device, dtype=torch.float32)
|
||||
hist, history_cache = sec_decoder.history_step(history_feat, has_prev, history_cache)
|
||||
|
||||
if generator == "wgan":
|
||||
if objective.is_adversarial:
|
||||
z = torch.randn(B, 1, sec_decoder.noise_dim, device=device)
|
||||
token = sec_decoder(
|
||||
z,
|
||||
@@ -383,10 +383,10 @@ def sample_stage1(
|
||||
ddpm_steps: int = 1000,
|
||||
) -> tuple[torch.Tensor, torch.Tensor | None]:
|
||||
"""Dispatches on `stage1_model.generator_kind`."""
|
||||
kind = stage1_model.generator_kind
|
||||
if kind == "wgan":
|
||||
objective = build_objective(stage1_model.generator_kind)
|
||||
if objective.is_adversarial:
|
||||
return sample_wgan(stage1_model, cond_cont, cond_cat)
|
||||
if kind == "ddpm":
|
||||
if isinstance(objective, DdpmObjective):
|
||||
schedule = CosineSchedule(T=ddpm_steps).to(cond_cont.device)
|
||||
return sample_ddpm(stage1_model, cond_cont, cond_cat, schedule)
|
||||
return sample_flow(stage1_model, cond_cont, cond_cat, steps=steps)
|
||||
@@ -409,7 +409,7 @@ def sample_stage2(
|
||||
"""
|
||||
if isinstance(sec_decoder, Stage2Autoregressive):
|
||||
return sample_secondaries_ar(sec_decoder, cond_cont, cond_cat, stage1_out, n_sec_pred, steps=steps)
|
||||
if sec_decoder.generator_kind == "wgan":
|
||||
if build_objective(sec_decoder.generator_kind).is_adversarial:
|
||||
return sample_secondaries_wgan(sec_decoder, cond_cont, cond_cat, stage1_out, n_sec_pred)
|
||||
return sample_secondaries(sec_decoder, cond_cont, cond_cat, stage1_out, n_sec_pred, steps=steps)
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
"""Cut a new raw generation or processed schema version for the geant_steps
|
||||
dataset tree (see scripts/migrate_geant_steps.py for the layout):
|
||||
dataset tree (see giant/tools/migrate_geant_steps.py for the layout):
|
||||
|
||||
raw/<kind>/<gen>/<detector>/shard-NNN.root
|
||||
processed/<kind>/<gen>/<schema>/<detector>/shard-NNN.parquet
|
||||
@@ -579,7 +579,7 @@ def apply_create_manifest(output_path: Path, lines: list[str]) -> None:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CLI entry points (called from scripts/dwarf.py)
|
||||
# CLI entry points (called from giant/tools/dwarf.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@@ -10,9 +10,9 @@ machine against an actual trained checkpoint before merging
|
||||
|
||||
Usage (from the repo root, on a portal machine):
|
||||
|
||||
uv run python scripts/check_migration_v02_v03.py /ceph/lbogner/.../best.pt
|
||||
uv run python scripts/check_migration_v02_v03.py /ceph/lbogner/.../best.pt --ema
|
||||
uv run python scripts/check_migration_v02_v03.py /ceph/lbogner/.../best.pt --batch 32 --seed 1
|
||||
uv run python giant/tools/check_migration_v02_v03.py /ceph/lbogner/.../best.pt
|
||||
uv run python giant/tools/check_migration_v02_v03.py /ceph/lbogner/.../best.pt --ema
|
||||
uv run python giant/tools/check_migration_v02_v03.py /ceph/lbogner/.../best.pt --batch 32 --seed 1
|
||||
|
||||
Run it once against a flow (or ddpm) checkpoint and once against a wgan
|
||||
checkpoint ("one flow checkpoint and one WGAN checkpoint").
|
||||
@@ -32,7 +32,7 @@ from dataclasses import dataclass
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
|
||||
# Must match scripts/bump_dataset_version.py's GEN_RE.
|
||||
# Must match giant/tools/bump_dataset_version.py's GEN_RE.
|
||||
GEN_RE = re.compile(r"^gen\d+$")
|
||||
SHARD_RE = re.compile(r"^shard-(\d+)\.root$")
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""dwarf — little helper to `giant`: dataset/tooling CLI for the geant_steps pipeline.
|
||||
|
||||
Unifies the standalone scripts/*.py conversion, migration, versioning, and
|
||||
Unifies the standalone giant/tools/*.py conversion, migration, versioning, and
|
||||
simulation-fanout tools into one Typer app so there's a single command name
|
||||
(and `--help`) to remember instead of five differently-hyphenated ones.
|
||||
"""
|
||||
@@ -14,20 +14,20 @@ import typer
|
||||
from typing_extensions import Annotated
|
||||
|
||||
from giant.config import Conditioning
|
||||
from scripts.bump_dataset_version import (
|
||||
from giant.tools.bump_dataset_version import (
|
||||
run_bump_gen,
|
||||
run_bump_schema,
|
||||
run_create_manifest,
|
||||
run_status,
|
||||
run_update_manifest,
|
||||
)
|
||||
from scripts.create_root_files import run_make_root
|
||||
from scripts.geometry_oracle import run_build_geometry_oracle
|
||||
from scripts.hparam_scan import DATA_DEFAULT, SCAN_DIR_DEFAULT, run_hparam_scan
|
||||
from scripts.migrate_geant_steps import run_migration
|
||||
from scripts.steps_to_parquet import convert_steps_to_parquet
|
||||
from scripts.steps_to_parquet_parallel import run_parallel_job
|
||||
from scripts.warm_setup_cache import run_warm_setup_cache
|
||||
from giant.tools.create_root_files import run_make_root
|
||||
from giant.tools.geometry_oracle import run_build_geometry_oracle
|
||||
from giant.tools.hparam_scan import DATA_DEFAULT, SCAN_DIR_DEFAULT, run_hparam_scan
|
||||
from giant.tools.migrate_geant_steps import run_migration
|
||||
from giant.tools.steps_to_parquet import convert_steps_to_parquet
|
||||
from giant.tools.steps_to_parquet_parallel import run_parallel_job
|
||||
from giant.tools.warm_setup_cache import run_warm_setup_cache
|
||||
|
||||
app = typer.Typer(no_args_is_help=True)
|
||||
|
||||
@@ -13,7 +13,7 @@ real checkpoint) they short-circuit almost instantly and are excluded here —
|
||||
see `runtime_estimate.py`'s `_ROUTER_FIXED_S` for how those are handled
|
||||
instead.
|
||||
|
||||
Usage: ``uv run python scripts/profile_analysis_costs.py``
|
||||
Usage: ``uv run python giant/tools/profile_analysis_costs.py``
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -2,11 +2,11 @@
|
||||
|
||||
A single `dwarf convert` call converts a list of files one at a time; this
|
||||
module runs up to --jobs conversions concurrently, each as its own `dwarf
|
||||
convert` subprocess (invoked via `python -m scripts.dwarf`, so it picks up
|
||||
convert` subprocess (invoked via `python -m giant.tools.dwarf`, so it picks up
|
||||
the active venv/uv environment automatically).
|
||||
|
||||
Inputs must live under <dataset-root>/raw/<kind>/<gen>/<detector>/<file>.root
|
||||
(see scripts/migrate_geant_steps.py) — each is written to the matching
|
||||
(see giant/tools/migrate_geant_steps.py) — each is written to the matching
|
||||
processed/<kind>/<gen>/<schema>/<detector>/<file>.parquet, where <schema>
|
||||
defaults to the highest schemaN already under processed/<kind>/<gen>/ (pass
|
||||
--schema to pick a specific one, e.g. one just created by `dwarf bump-schema`).
|
||||
@@ -22,7 +22,7 @@ import sys
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
|
||||
# Must match scripts/bump_dataset_version.py's GEN_RE / SCHEMA_RE.
|
||||
# Must match giant/tools/bump_dataset_version.py's GEN_RE / SCHEMA_RE.
|
||||
GEN_RE = re.compile(r"^gen\d+$")
|
||||
SCHEMA_RE = re.compile(r"^schema(\d+)$")
|
||||
|
||||
@@ -82,7 +82,7 @@ def resolve_destination(root_file: Path, dataset_root: Path, schema_override: st
|
||||
return processed_gen_dir / schema_tag / detector / f"{shard_stem}.parquet"
|
||||
|
||||
|
||||
_DWARF_CONVERT_CMD = [sys.executable, "-m", "scripts.dwarf", "convert"]
|
||||
_DWARF_CONVERT_CMD = [sys.executable, "-m", "giant.tools.dwarf", "convert"]
|
||||
|
||||
|
||||
def _convert_one(
|
||||
@@ -127,7 +127,7 @@ def run_parallel(
|
||||
written next to the input .root).
|
||||
|
||||
*cmd_prefix* overrides the subprocess command run per file (defaults to
|
||||
`python -m scripts.dwarf convert`) — used by tests to substitute a fake
|
||||
`python -m giant.tools.dwarf convert`) — used by tests to substitute a fake
|
||||
conversion script.
|
||||
|
||||
Returns one (root_file, returncode, stdout, stderr) tuple per file, in
|
||||
@@ -109,6 +109,7 @@ def train(
|
||||
mat_map: dict | None = None,
|
||||
proc_map: dict | None = None,
|
||||
pdg_topn_map: TopNMap | None = None,
|
||||
sec_type_topn_map: TopNMap | None = None,
|
||||
mat_topn_map: TopNMap | None = None,
|
||||
model_config: dict | None = None,
|
||||
resume_path: str | Path | None = None,
|
||||
@@ -142,6 +143,7 @@ def train(
|
||||
"mat_map": mat_map,
|
||||
"proc_map": proc_map,
|
||||
"pdg_topn_map": topnmap_to_json(pdg_topn_map) if pdg_topn_map is not None else None,
|
||||
"sec_type_topn_map": topnmap_to_json(sec_type_topn_map) if sec_type_topn_map is not None else None,
|
||||
"mat_topn_map": topnmap_to_json(mat_topn_map) if mat_topn_map is not None else None,
|
||||
"model_config": model_config,
|
||||
}
|
||||
|
||||
@@ -11,7 +11,9 @@ live in one place and stay unit-testable on their own.
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
|
||||
from giant.config import ParticleTypeConfig
|
||||
from giant.constants import CONT_SLOT_DIM, PARTICLE_PHYS_DIM
|
||||
from giant.model.objectives import build_objective
|
||||
from giant.sample import sample_secondaries_ar
|
||||
|
||||
|
||||
@@ -30,7 +32,7 @@ def _gumbel_tau(step: int, total_steps: int, tau_start: float, tau_end: float) -
|
||||
def _type_repr(
|
||||
sec_type_idx: torch.Tensor,
|
||||
sec_cont: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
cond_enc: torch.nn.Module,
|
||||
emb_dim: int,
|
||||
) -> torch.Tensor:
|
||||
@@ -45,7 +47,7 @@ def _type_repr(
|
||||
latter must always reflect the true physical secondary that came before,
|
||||
regardless of what the *current* token's own training objective is.
|
||||
"""
|
||||
target = particle_type_cfg.get("target", "physical")
|
||||
target = particle_type_cfg.target
|
||||
if target == "physical":
|
||||
return sec_cont[..., CONT_SLOT_DIM : CONT_SLOT_DIM + PARTICLE_PHYS_DIM]
|
||||
if target == "onehot":
|
||||
@@ -56,7 +58,7 @@ def _type_repr(
|
||||
def _assemble_stage2_ar_target(
|
||||
sec_cont: torch.Tensor,
|
||||
sec_type_idx: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
generator: str,
|
||||
cond_enc: torch.nn.Module,
|
||||
emb_dim: int,
|
||||
@@ -69,19 +71,20 @@ def _assemble_stage2_ar_target(
|
||||
|
||||
- `target = "physical"`: unchanged from v0.2 — `sec_cont` (stick_logit,
|
||||
dir, log_mass, charge) as-is.
|
||||
- `target` in `("onehot", "embedding")` + `generator in ("flow", "ddpm")`:
|
||||
just the continuous stick/dir slots — the type slice isn't part of
|
||||
this tensor at all (`type_head` handles it separately).
|
||||
- `target` in `("onehot", "embedding")` + `generator == "wgan"`: stick/dir
|
||||
slots concatenated with the per-slot type representation (a one-hot of
|
||||
the true class, relaxed on the *generated* side only, by the caller;
|
||||
or the conditioning's own detached embedding-table row).
|
||||
- `target` in `("onehot", "embedding")` + an objective that doesn't fold
|
||||
the type slice (flow/ddpm): just the continuous stick/dir slots — the
|
||||
type slice isn't part of this tensor at all (`type_head` handles it
|
||||
separately).
|
||||
- `target` in `("onehot", "embedding")` + a folding objective (wgan):
|
||||
stick/dir slots concatenated with the per-slot type representation (a
|
||||
one-hot of the true class, relaxed on the *generated* side only, by the
|
||||
caller; or the conditioning's own detached embedding-table row).
|
||||
"""
|
||||
target = particle_type_cfg.get("target", "physical")
|
||||
target = particle_type_cfg.target
|
||||
if target == "physical":
|
||||
return sec_cont
|
||||
cont = sec_cont[..., :CONT_SLOT_DIM]
|
||||
if generator != "wgan":
|
||||
if not build_objective(generator).folds_type_slice:
|
||||
return cont
|
||||
type_repr = _type_repr(sec_type_idx, sec_cont, particle_type_cfg, cond_enc, emb_dim)
|
||||
return torch.cat([cont, type_repr], dim=-1)
|
||||
@@ -90,7 +93,7 @@ def _assemble_stage2_ar_target(
|
||||
def _assemble_stage2_real(
|
||||
sec_cont: torch.Tensor,
|
||||
sec_type_idx: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
generator: str,
|
||||
cond_enc: torch.nn.Module,
|
||||
emb_dim: int,
|
||||
@@ -155,7 +158,7 @@ def _ar_meta(k_max: int, batch: int, device: torch.device, fraction: torch.Tenso
|
||||
def _assemble_stage2_ar_inputs(
|
||||
sec_cont: torch.Tensor,
|
||||
sec_type_idx: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
cond_enc: torch.nn.Module,
|
||||
emb_dim: int,
|
||||
) -> dict[str, torch.Tensor]:
|
||||
@@ -198,7 +201,7 @@ def _stage2_tf_prob(mode: str, p_start: float, p_end: float, epoch: int, total_e
|
||||
def _history_repr_from_ar_sample(
|
||||
sec_cont_pred: torch.Tensor,
|
||||
sec_type_pred: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""`(fraction, direction, type_repr)` — the same triple `_type_repr` /
|
||||
`_stick_fraction` derive from ground truth, but from a free-running
|
||||
@@ -211,7 +214,7 @@ def _history_repr_from_ar_sample(
|
||||
representation."""
|
||||
fraction = torch.sigmoid(sec_cont_pred[..., 0])
|
||||
direction = sec_cont_pred[..., 1:4]
|
||||
if particle_type_cfg.get("target", "physical") == "onehot":
|
||||
if particle_type_cfg.target == "onehot":
|
||||
type_dim = sec_type_pred.size(-1)
|
||||
type_repr = F.one_hot(sec_type_pred.argmax(-1), num_classes=type_dim).float()
|
||||
else:
|
||||
@@ -227,7 +230,7 @@ def _assemble_stage2_ar_inputs_scheduled(
|
||||
sec_cont: torch.Tensor,
|
||||
sec_type_idx: torch.Tensor,
|
||||
n_sec: torch.Tensor,
|
||||
particle_type_cfg: dict,
|
||||
particle_type_cfg: ParticleTypeConfig,
|
||||
cond_enc: torch.nn.Module,
|
||||
emb_dim: int,
|
||||
p_tf: float,
|
||||
|
||||
+64
-67
@@ -16,6 +16,7 @@ adversarial and non-adversarial stages identically.
|
||||
import copy
|
||||
import math
|
||||
from dataclasses import dataclass, field
|
||||
from typing import NamedTuple
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
@@ -23,13 +24,8 @@ import torch.optim as optim
|
||||
|
||||
from giant.config import ParticleTypeConfig, Stage1ModelConfig, Stage2ModelConfig, TrainConfig
|
||||
from giant.constants import CONT_SLOT_DIM
|
||||
from giant.model.network import Router, stage2_type_dim
|
||||
from giant.model.schedule import (
|
||||
CosineSchedule,
|
||||
flow_matching_loss,
|
||||
flow_matching_loss_secondary,
|
||||
flow_matching_loss_secondary_ar,
|
||||
)
|
||||
from giant.data.dataset import StepBatch
|
||||
from giant.model.network import Router, build_objective, resolve_type_n_classes, stage2_type_dim
|
||||
from giant.model.wgan import generator_loss, gradient_penalty
|
||||
from giant.training.metrics import MetricSpec, stage_metric, train_metric, val_metric
|
||||
from giant.training.stage2_inputs import (
|
||||
@@ -72,8 +68,8 @@ def _cosine_warmup_lambda(warmup_steps: int, total_steps: int):
|
||||
return _lr_lambda
|
||||
|
||||
|
||||
def _batch_to_device(batch: tuple, device: torch.device) -> tuple:
|
||||
return tuple(t.to(device) for t in batch)
|
||||
def _batch_to_device(batch: StepBatch, device: torch.device) -> StepBatch:
|
||||
return type(batch)(*(t.to(device) for t in batch))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -96,7 +92,7 @@ class StageSpec:
|
||||
|
||||
# particle-type target (stage 2 only)
|
||||
particle_type: ParticleTypeConfig = field(default_factory=ParticleTypeConfig)
|
||||
particle_type_emb_dim: int = 16
|
||||
particle_type_n_classes: int = 16
|
||||
|
||||
# optimization
|
||||
lr: float = 3e-4
|
||||
@@ -149,7 +145,9 @@ class StageSpec:
|
||||
lambda_weight=stage_spec.lambda_weight,
|
||||
n_sec_lambda=s2_spec.n_sec.lambda_weight,
|
||||
particle_type=s2_spec.particle_type,
|
||||
particle_type_emb_dim=cfg["conditioning"]["particle"]["emb_dim"],
|
||||
particle_type_n_classes=resolve_type_n_classes(
|
||||
s2_spec.particle_type, cfg["conditioning"]["particle"]["emb_dim"]
|
||||
),
|
||||
# train.* keys are all guaranteed by DEFAULT_CONFIG's deep-merge
|
||||
# (giant/config.py), so TrainConfig.from_dict never has to fall
|
||||
# back to a literal here; the field defaults below exist only
|
||||
@@ -185,11 +183,10 @@ class StageSpec:
|
||||
class StageTrainer:
|
||||
"""One active stage's optimizer(s), EMA, and per-batch step.
|
||||
|
||||
Reads only the shared batch tuple `(cond_cont, cond_cat, x1_s1, n_sec,
|
||||
sec_cont, proc_idx, sec_type_idx)` — stage 2 always conditions on the
|
||||
ground-truth `x1_s1` (`stage2_model.stage1_context = "truth"`,
|
||||
stage-level teacher forcing; `"sampled"` is not implemented), so stage
|
||||
trainers never need each other's output at train time. This means
|
||||
Reads only the shared `StepBatch` (`giant.data.dataset`) — stage 2 always
|
||||
conditions on the ground-truth `x1_s1` (`stage2_model.stage1_context =
|
||||
"truth"`, stage-level teacher forcing; `"sampled"` is not implemented),
|
||||
so stage trainers never need each other's output at train time. This means
|
||||
"stage-2-only training is a cheap ablation, not new plumbing" falls out
|
||||
for free: a trainer only exists for active stages, and inactive stages
|
||||
are simply never constructed.
|
||||
@@ -236,8 +233,8 @@ class StageTrainer:
|
||||
self.router = _stage_router(self.model)
|
||||
self._modules = (self.model, *extra_modules)
|
||||
|
||||
self.particle_type_cfg = spec.particle_type.to_dict()
|
||||
self.particle_type_emb_dim = spec.particle_type_emb_dim
|
||||
self.particle_type_cfg = spec.particle_type
|
||||
self.particle_type_n_classes = spec.particle_type_n_classes
|
||||
self.ema_decay = spec.ema_decay
|
||||
|
||||
self.ema_model: torch.nn.Module | None = None
|
||||
@@ -255,10 +252,10 @@ class StageTrainer:
|
||||
|
||||
# --- per-batch (subclass responsibility) ----------------------------
|
||||
|
||||
def step(self, batch: tuple, device: torch.device, global_step: int) -> dict:
|
||||
def step(self, batch: StepBatch, device: torch.device, global_step: int) -> dict:
|
||||
raise NotImplementedError
|
||||
|
||||
def val_loss(self, batch: tuple, device: torch.device) -> dict:
|
||||
def val_loss(self, batch: StepBatch, device: torch.device) -> dict:
|
||||
raise NotImplementedError
|
||||
|
||||
# --- reporting hooks ------------------------------------------------
|
||||
@@ -328,7 +325,7 @@ class StageTrainer:
|
||||
n_sec,
|
||||
self.particle_type_cfg,
|
||||
self.model.cond_enc,
|
||||
self.particle_type_emb_dim,
|
||||
self.particle_type_n_classes,
|
||||
p_tf,
|
||||
self.spec.ar_sample_steps,
|
||||
)
|
||||
@@ -354,7 +351,7 @@ class StageTrainer:
|
||||
self.particle_type_cfg,
|
||||
generator,
|
||||
self.model.cond_enc,
|
||||
self.particle_type_emb_dim,
|
||||
self.particle_type_n_classes,
|
||||
)
|
||||
return target.flatten(1) if flatten else target
|
||||
|
||||
@@ -447,20 +444,23 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
"""flow or ddpm generator for a single stage."""
|
||||
|
||||
def __init__(self, spec: StageSpec, model: torch.nn.Module, device: torch.device) -> None:
|
||||
if spec.is_stage2 and spec.generator not in ("flow",):
|
||||
objective = build_objective(spec.generator, n_steps=spec.ddpm_n_steps)
|
||||
if spec.is_stage2 and not objective.supports_stage2_decoder:
|
||||
raise NotImplementedError(
|
||||
f"stage2_model.generator={spec.generator!r} is accepted by the "
|
||||
"schema but not implemented in v0.3.0 for stage 2 (only "
|
||||
"'flow' and 'wgan' have a stage-2 secondary-decoder loss)"
|
||||
)
|
||||
super().__init__(spec, model, device)
|
||||
self.particle_type_lambda = self.particle_type_cfg.get("lambda", 1.0)
|
||||
self.objective = objective
|
||||
self.particle_type_lambda = self.particle_type_cfg.lambda_weight
|
||||
# Width of the type slice actually folded into x1_s2 by _sec_target,
|
||||
# under this trainer's generator (flow/ddpm only — see the
|
||||
# NotImplementedError above): "physical" keeps it folded in
|
||||
# (PARTICLE_PHYS_DIM wide, unchanged from v0.2); "onehot"/"embedding"
|
||||
# pull it out into model.type_head instead (0 here).
|
||||
self._flow_type_dim = None if self.particle_type_cfg.get("target", "physical") == "physical" else 0
|
||||
# under this trainer's objective (flow/ddpm only — see the
|
||||
# NotImplementedError above, neither folds the type slice): "physical"
|
||||
# keeps it folded in (PARTICLE_PHYS_DIM wide, unchanged from v0.2);
|
||||
# "onehot"/"embedding" pull it out into model.type_head instead (0
|
||||
# here).
|
||||
self._flow_type_dim = None if self.particle_type_cfg.target == "physical" else 0
|
||||
|
||||
self.params = list(self.model.parameters())
|
||||
self.optimizer = optim.AdamW(self.params, lr=spec.lr, weight_decay=spec.weight_decay)
|
||||
@@ -469,7 +469,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
warmup_steps=spec.warmup_epochs * spec.steps_per_epoch,
|
||||
total_steps=max(spec.epochs * spec.steps_per_epoch, 1),
|
||||
)
|
||||
self.ddpm_schedule = CosineSchedule(T=spec.ddpm_n_steps).to(device) if spec.generator == "ddpm" else None
|
||||
self.ddpm_schedule = self.objective.build_schedule(spec.ddpm_n_steps, device)
|
||||
|
||||
self.train_metrics = [
|
||||
train_metric(key)
|
||||
@@ -501,26 +501,9 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
|
||||
def _generator_loss(self, cond_cont, cond_cat, x1_s1, x1_s2, sec_mask, stage1_ctx, ar_inputs=None):
|
||||
if not self.is_stage2:
|
||||
if self.generator == "flow":
|
||||
return flow_matching_loss(self.model, x1_s1, cond_cont, cond_cat)
|
||||
assert self.ddpm_schedule is not None
|
||||
return self.ddpm_schedule.loss(self.model, x1_s1, cond_cont, cond_cat)
|
||||
if self.decoder == "autoregressive":
|
||||
assert ar_inputs is not None
|
||||
return flow_matching_loss_secondary_ar(
|
||||
self.model,
|
||||
x1_s2,
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
stage1_ctx,
|
||||
ar_inputs["history_feat"],
|
||||
ar_inputs["has_prev"],
|
||||
ar_inputs["remaining_frac"],
|
||||
ar_inputs["slot_idx"],
|
||||
sec_mask,
|
||||
type_dim=self._flow_type_dim,
|
||||
)
|
||||
return flow_matching_loss_secondary(
|
||||
return self.objective.stage1_loss(self.model, x1_s1, cond_cont, cond_cat, schedule=self.ddpm_schedule)
|
||||
assert self.decoder != "autoregressive" or ar_inputs is not None
|
||||
return self.objective.stage2_loss(
|
||||
self.model,
|
||||
x1_s2,
|
||||
cond_cont,
|
||||
@@ -528,6 +511,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
stage1_ctx,
|
||||
sec_mask,
|
||||
type_dim=self._flow_type_dim,
|
||||
ar_inputs=ar_inputs,
|
||||
)
|
||||
|
||||
def _type_loss(
|
||||
@@ -565,7 +549,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
type_out = self.model.predict_type(cond_cont, cond_cat, stage1_ctx)
|
||||
mask = sec_mask.float()
|
||||
denom = mask.sum().clamp(min=1)
|
||||
if self.particle_type_cfg.get("target") == "onehot":
|
||||
if self.particle_type_cfg.target == "onehot":
|
||||
ce = F.cross_entropy(type_out.transpose(1, 2), sec_type_idx, reduction="none")
|
||||
l_type = (ce * mask).sum() / denom
|
||||
type_acc = ((type_out.argmax(-1) == sec_type_idx).float() * mask).sum() / denom
|
||||
@@ -575,7 +559,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
l_type = (se * mask).sum() / denom
|
||||
return l_type, type_acc
|
||||
|
||||
def _compute(self, batch: tuple, device: torch.device, epoch: int | None = None) -> dict:
|
||||
def _compute(self, batch: StepBatch, device: torch.device, epoch: int | None = None) -> dict:
|
||||
"""`epoch=None` (the `val_loss` path) always uses full teacher
|
||||
forcing (`p_tf=1.0`) regardless of `spec.teacher_forcing` — validation
|
||||
should stay a stable, non-stochastic ground-truth comparison; only
|
||||
@@ -615,9 +599,12 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
|
||||
l_balance = l_proc = l_entropy = torch.zeros((), device=device)
|
||||
if self.router is not None:
|
||||
l_balance = self.router.balance_loss(cond_cont, cond_cat)
|
||||
l_proc = self.router.classify_loss(cond_cont, cond_cat, proc_idx)
|
||||
l_entropy = self.router.entropy_loss(cond_cont, cond_cat)
|
||||
if self.spec.lambda_balance > 0:
|
||||
l_balance = self.router.balance_loss(cond_cont, cond_cat)
|
||||
if self.spec.lambda_proc > 0:
|
||||
l_proc = self.router.classify_loss(cond_cont, cond_cat, proc_idx)
|
||||
if self.spec.lambda_entropy > 0:
|
||||
l_entropy = self.router.entropy_loss(cond_cont, cond_cat)
|
||||
|
||||
total = self.spec.lambda_weight * l_gen + self.spec.n_sec_lambda * l_nsec + self.particle_type_lambda * l_type
|
||||
if self.spec.lambda_balance > 0:
|
||||
@@ -639,7 +626,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
"nsec_acc": nsec_acc,
|
||||
}
|
||||
|
||||
def step(self, batch: tuple, device: torch.device, global_step: int) -> dict:
|
||||
def step(self, batch: StepBatch, device: torch.device, global_step: int) -> dict:
|
||||
if self.router is not None:
|
||||
self.router.gumbel_tau = _gumbel_tau(
|
||||
global_step,
|
||||
@@ -659,7 +646,7 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
return stats
|
||||
|
||||
@torch.no_grad()
|
||||
def val_loss(self, batch: tuple, device: torch.device) -> dict:
|
||||
def val_loss(self, batch: StepBatch, device: torch.device) -> dict:
|
||||
return {key: value.item() for key, value in self._compute(batch, device).items()}
|
||||
|
||||
# --- reporting ------------------------------------------------------
|
||||
@@ -674,6 +661,16 @@ class FlowDDPMStageTrainer(StageTrainer):
|
||||
return val_means.get("loss", 0.0)
|
||||
|
||||
|
||||
class _Stage2RealFakeBatch(NamedTuple):
|
||||
"""Subset of `StepBatch` that `_stage2_real_and_fake` needs."""
|
||||
|
||||
cond_cont: torch.Tensor
|
||||
cond_cat: torch.Tensor
|
||||
n_sec: torch.Tensor
|
||||
sec_cont: torch.Tensor
|
||||
sec_type_idx: torch.Tensor
|
||||
|
||||
|
||||
class WGANStageTrainer(StageTrainer):
|
||||
"""WGAN-GP generator+critic for a single stage (see giant/model/wgan.py).
|
||||
|
||||
@@ -729,7 +726,7 @@ class WGANStageTrainer(StageTrainer):
|
||||
"grad_norm_d",
|
||||
"grad_norm_g",
|
||||
]
|
||||
if self.is_stage2 and self.particle_type_cfg.get("target") == "onehot":
|
||||
if self.is_stage2 and self.particle_type_cfg.target == "onehot":
|
||||
# Differentiability instrumentation — only meaningful when the
|
||||
# type slice is a straight-through Gumbel relaxation.
|
||||
train_keys += ["grad_norm_type_slice", "grad_norm_cont_slice"]
|
||||
@@ -737,7 +734,7 @@ class WGANStageTrainer(StageTrainer):
|
||||
self.val_metrics = []
|
||||
self.stage_metrics = [stage_metric("lr"), stage_metric("critic_lr")]
|
||||
|
||||
def _stage2_real_and_fake(self, batch_tensors, stage1_ctx, global_step, device):
|
||||
def _stage2_real_and_fake(self, batch_tensors: _Stage2RealFakeBatch, stage1_ctx, global_step, device):
|
||||
"""Build `(real, fake_raw, mask, critic_fn)` for stage 2, covering
|
||||
both decoders and all three particle-type targets. `fake_raw` still
|
||||
needs the caller's straight-through relaxation under
|
||||
@@ -745,7 +742,7 @@ class WGANStageTrainer(StageTrainer):
|
||||
multiplied on the fake side yet."""
|
||||
cond_cont, cond_cat, n_sec, sec_cont, sec_type_idx = batch_tensors
|
||||
B = cond_cont.size(0)
|
||||
type_dim = stage2_type_dim(self.particle_type_cfg, self.particle_type_emb_dim)
|
||||
type_dim = stage2_type_dim(self.particle_type_cfg, self.particle_type_n_classes)
|
||||
slot_width = CONT_SLOT_DIM + type_dim
|
||||
k_max = sec_cont.size(1)
|
||||
|
||||
@@ -758,7 +755,7 @@ class WGANStageTrainer(StageTrainer):
|
||||
if self.decoder == "autoregressive":
|
||||
epoch = global_step // self.spec.steps_per_epoch
|
||||
ar = self._ar_inputs(cond_cont, cond_cat, stage1_ctx, sec_cont, sec_type_idx, n_sec, epoch)
|
||||
real = self._sec_target(sec_cont, sec_type_idx, "wgan", flatten=False).reshape(B, -1) * mask
|
||||
real = self._sec_target(sec_cont, sec_type_idx, self.generator, flatten=False).reshape(B, -1) * mask
|
||||
z = torch.randn(B, k_max, self.model.noise_dim, device=device)
|
||||
fake_raw = self.model(
|
||||
z,
|
||||
@@ -771,13 +768,13 @@ class WGANStageTrainer(StageTrainer):
|
||||
ar["slot_idx"],
|
||||
).reshape(B, -1)
|
||||
else:
|
||||
real = self._sec_target(sec_cont, sec_type_idx, "wgan", flatten=True) * mask
|
||||
real = self._sec_target(sec_cont, sec_type_idx, self.generator, flatten=True) * mask
|
||||
z = torch.randn(B, self.model.noise_dim, device=device)
|
||||
fake_raw = self.model(z, cond_cont, cond_cat, stage1_ctx)
|
||||
|
||||
return real, fake_raw, mask, critic_fn
|
||||
|
||||
def step(self, batch: tuple, device: torch.device, global_step: int) -> dict:
|
||||
def step(self, batch: StepBatch, device: torch.device, global_step: int) -> dict:
|
||||
(
|
||||
cond_cont,
|
||||
cond_cat,
|
||||
@@ -802,12 +799,12 @@ class WGANStageTrainer(StageTrainer):
|
||||
mask = None
|
||||
else:
|
||||
real, fake_raw, mask, critic_fn = self._stage2_real_and_fake(
|
||||
(cond_cont, cond_cat, n_sec, sec_cont, sec_type_idx),
|
||||
_Stage2RealFakeBatch(cond_cont, cond_cat, n_sec, sec_cont, sec_type_idx),
|
||||
stage1_ctx,
|
||||
global_step,
|
||||
device,
|
||||
)
|
||||
if self.particle_type_cfg.get("target", "physical") == "onehot":
|
||||
if self.particle_type_cfg.target == "onehot":
|
||||
# Straight-through Gumbel-softmax relaxation of the type
|
||||
# slice only — the critic must see a hard one-hot forward
|
||||
# (matching what "real" data looks like) while gradient
|
||||
@@ -824,7 +821,7 @@ class WGANStageTrainer(StageTrainer):
|
||||
fake_raw,
|
||||
sec_cont.size(1),
|
||||
CONT_SLOT_DIM,
|
||||
stage2_type_dim(self.particle_type_cfg, self.particle_type_emb_dim),
|
||||
stage2_type_dim(self.particle_type_cfg, self.particle_type_n_classes),
|
||||
tau,
|
||||
grad_probe=grad_probe,
|
||||
)
|
||||
@@ -941,7 +938,7 @@ def build_stage_trainers(
|
||||
if model is None:
|
||||
continue
|
||||
spec = StageSpec.from_config(cfg, name, is_stage2, max(total_train_batches, 1))
|
||||
if spec.generator == "wgan":
|
||||
if build_objective(spec.generator).is_adversarial:
|
||||
critic = critics.get(name)
|
||||
assert critic is not None, (
|
||||
f"{name}_model.generator='wgan' requires a critic (see giant.model.network.build_critics)"
|
||||
|
||||
+6
-7
@@ -82,7 +82,7 @@ def validate_marginals(
|
||||
one-shot-vs-autoregressive-agnostic): n_sec
|
||||
distribution (+ classification accuracy), per-slot energy-fraction
|
||||
marginals, and a particle-type marginal whose shape depends on
|
||||
`sec_decoder.particle_type_cfg["target"]` — restricted to each side's own
|
||||
`sec_decoder.particle_type_cfg.target` — restricted to each side's own
|
||||
valid slots (real: `n_sec`; generated: the resolved `n_sec_pred`), since
|
||||
the two need not agree on how many slots are valid. Adds {"n_sec_real",
|
||||
"n_sec_pred", "n_sec_accuracy", "energy_fraction_kl"} plus, under
|
||||
@@ -103,7 +103,7 @@ def validate_marginals(
|
||||
sec_decoder.eval()
|
||||
|
||||
k_max = sec_decoder.k_max if sec_decoder is not None else 0
|
||||
target = sec_decoder.particle_type_cfg.get("target", "physical") if sec_decoder is not None else "physical"
|
||||
target = sec_decoder.particle_type_cfg.target if sec_decoder is not None else "physical"
|
||||
|
||||
all_real, all_gen = [], []
|
||||
all_n_sec_real, all_n_sec_pred = [], []
|
||||
@@ -115,11 +115,10 @@ def validate_marginals(
|
||||
for i, batch in enumerate(val_loader):
|
||||
if n_batches is not None and i >= n_batches:
|
||||
break
|
||||
# Batch is (cond_cont, cond_cat, target_s1, n_sec, sec_cont, proc_idx,
|
||||
# sec_type_idx).
|
||||
cond_cont, cond_cat, x1, n_sec, sec_cont, _proc_idx, sec_type_idx = batch
|
||||
cond_cont = cond_cont.to(device)
|
||||
cond_cat = cond_cat.to(device)
|
||||
# batch is a StepBatch (giant.data.dataset).
|
||||
x1, n_sec, sec_cont, sec_type_idx = batch.target_s1, batch.n_sec, batch.sec_cont, batch.sec_type_idx
|
||||
cond_cont = batch.cond_cont.to(device)
|
||||
cond_cat = batch.cond_cat.to(device)
|
||||
|
||||
gen, n_sec_pred = sample_stage1(stage1_model, cond_cont, cond_cat, steps=steps, ddpm_steps=ddpm_steps)
|
||||
|
||||
|
||||
+4
-4
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "giant"
|
||||
version = "0.3.0"
|
||||
version = "0.3.1"
|
||||
description = "Geant4 step-function surrogate via conditional flow matching"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
@@ -50,13 +50,13 @@ analysis = [
|
||||
|
||||
[project.scripts]
|
||||
giant = "giant.cli:app"
|
||||
dwarf = "scripts.dwarf:app"
|
||||
dwarf = "giant.tools.dwarf:app"
|
||||
|
||||
[tool.ruff]
|
||||
line-length = 120
|
||||
|
||||
[tool.coverage.run]
|
||||
source = ["giant", "scripts"]
|
||||
source = ["giant"]
|
||||
omit = ["*/legacy/*"]
|
||||
|
||||
[tool.coverage.report]
|
||||
@@ -70,7 +70,7 @@ requires = ["hatchling"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[tool.hatch.build.targets.wheel]
|
||||
packages = ["giant", "scripts"]
|
||||
packages = ["giant"]
|
||||
|
||||
[tool.uv]
|
||||
conflicts = [
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from scripts import bump_dataset_version
|
||||
from giant.tools import bump_dataset_version
|
||||
|
||||
plan_bump_gen = bump_dataset_version.plan_bump_gen
|
||||
plan_bump_schema = bump_dataset_version.plan_bump_schema
|
||||
|
||||
@@ -0,0 +1,280 @@
|
||||
"""Tests for giant.checkpoint_io.load_for_inference (issues.md Issue 5) —
|
||||
the shared bootstrap `giant predict`/`giant rollout` use to go from a
|
||||
checkpoint path to ready-to-run models."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant import config as gconfig
|
||||
from giant.checkpoint_io import (
|
||||
CheckpointCompatibilityError,
|
||||
InferenceContext,
|
||||
conditioning_axes,
|
||||
load_for_inference,
|
||||
stage_cfg,
|
||||
)
|
||||
from giant.data.loader import TopNMap
|
||||
from giant.data.setup_cache import topnmap_to_json
|
||||
from giant.data.transforms import Normalizer
|
||||
from giant.model.network import build_models
|
||||
|
||||
PDG_MAP = {11: 0, 22: 1, -11: 2}
|
||||
MAT_MAP = {"G4_PbWO4": 0, "G4_AIR": 1}
|
||||
|
||||
|
||||
def _model_cfg(stage2_active: bool = True) -> dict:
|
||||
"""DEFAULT_CONFIG-derived, shrunk for speed — same pattern as
|
||||
tests/test_network.py::_minimal_model_config. Default `conditioning`
|
||||
(both axes "physical") needs no top-N vocab map, so this is a cheap,
|
||||
fully self-contained happy-path config."""
|
||||
cfg = copy.deepcopy(gconfig.DEFAULT_CONFIG)
|
||||
cfg["conditioning"]["particle"]["emb_dim"] = 4
|
||||
cfg["conditioning"]["material"]["emb_dim"] = 4
|
||||
cfg["stage1_model"].update({"hidden_dim": 8, "n_res_blocks": 1})
|
||||
cfg["stage2_model"].update({"hidden_dim": 8, "n_res_blocks": 1, "k_max": 3})
|
||||
cfg["stage2_model"]["active"] = stage2_active
|
||||
return {
|
||||
"pdg_vocab": len(PDG_MAP),
|
||||
"mat_vocab": len(MAT_MAP),
|
||||
"conditioning": cfg["conditioning"],
|
||||
"stage1_model": cfg["stage1_model"],
|
||||
"stage2_model": cfg["stage2_model"],
|
||||
}
|
||||
|
||||
|
||||
def _norms() -> tuple[Normalizer, Normalizer, Normalizer]:
|
||||
rng = np.random.default_rng(0)
|
||||
cond = Normalizer().fit(rng.standard_normal((100, 15)).astype(np.float32))
|
||||
tgt = Normalizer().fit(rng.standard_normal((100, 9)).astype(np.float32))
|
||||
sec_phys = Normalizer().fit(rng.standard_normal((100, 2)).astype(np.float32))
|
||||
return cond, tgt, sec_phys
|
||||
|
||||
|
||||
def _write_checkpoint(tmp_path, model_cfg=None, ema: bool = False, **ckpt_overrides):
|
||||
cfg = model_cfg if model_cfg is not None else _model_cfg()
|
||||
built = build_models(cfg)
|
||||
stage1, stage2 = built["stage1"], built["stage2"]
|
||||
cond, tgt, sec_phys = _norms()
|
||||
ckpt: dict = {
|
||||
"model_config": cfg,
|
||||
"model": stage1.state_dict() if stage1 is not None else {},
|
||||
"sec_decoder": stage2.state_dict() if stage2 is not None else {},
|
||||
"pdg_map": PDG_MAP,
|
||||
"mat_map": MAT_MAP,
|
||||
"normalizer": {"cond": cond.to_dict(), "target": tgt.to_dict(), "sec_phys": sec_phys.to_dict()},
|
||||
"epoch": 3,
|
||||
"best_val_loss": 0.5,
|
||||
}
|
||||
if ema:
|
||||
ckpt["model_ema"] = stage1.state_dict() if stage1 is not None else {}
|
||||
ckpt["sec_decoder_ema"] = stage2.state_dict() if stage2 is not None else {}
|
||||
# DEFAULT_CONFIG's stage2_model.particle_type.target defaults to
|
||||
# "onehot", and giant train's pipeline (gitea #29) now always writes a
|
||||
# sec_type_topn_map in that case — default one in here too, unless a
|
||||
# test explicitly overrides it, so fixtures represent a real, loadable
|
||||
# checkpoint by default rather than exercising the "missing" guard by
|
||||
# accident.
|
||||
particle_type_target = cfg.get("stage2_model", {}).get("particle_type", {}).get("target", "onehot")
|
||||
if particle_type_target == "onehot" and "sec_type_topn_map" not in ckpt_overrides:
|
||||
default_sec_type_topn = TopNMap(class_map=dict(zip(PDG_MAP, range(len(PDG_MAP)))), other_members={})
|
||||
ckpt["sec_type_topn_map"] = topnmap_to_json(default_sec_type_topn)
|
||||
ckpt.update(ckpt_overrides)
|
||||
path = tmp_path / "ckpt.pt"
|
||||
torch.save(ckpt, path)
|
||||
return path
|
||||
|
||||
|
||||
def _onehot_model_cfg() -> dict:
|
||||
cfg = _model_cfg()
|
||||
cfg["conditioning"]["particle"]["type"] = "onehot"
|
||||
return cfg
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Happy path
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_happy_path_returns_populated_context(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path)
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
assert isinstance(ctx, InferenceContext)
|
||||
assert ctx.stage1 is not None and ctx.stage2 is not None
|
||||
assert not ctx.stage1.training
|
||||
assert not ctx.stage2.training
|
||||
assert next(ctx.stage1.parameters()).device == torch.device("cpu")
|
||||
assert ctx.pdg_map == PDG_MAP
|
||||
assert ctx.mat_map == MAT_MAP
|
||||
assert all(isinstance(k, int) for k in ctx.pdg_map)
|
||||
assert all(isinstance(k, str) for k in ctx.mat_map)
|
||||
assert ctx.particle_conditioning == "physical"
|
||||
assert ctx.material_conditioning == "physical"
|
||||
assert ctx.k_max == 3
|
||||
assert ctx.epoch == 3
|
||||
assert ctx.best_val_loss == 0.5
|
||||
assert ctx.model_config["stage1_model"]["hidden_dim"] == 8
|
||||
|
||||
|
||||
def test_happy_path_normalizer_values_round_trip(tmp_path):
|
||||
cond, tgt, sec_phys = _norms()
|
||||
checkpoint = _write_checkpoint(tmp_path)
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
assert ctx.cond_norm.mean is not None and cond.mean is not None
|
||||
assert ctx.tgt_norm.mean is not None and tgt.mean is not None
|
||||
assert ctx.sec_phys_norm.mean is not None and sec_phys.mean is not None
|
||||
np.testing.assert_allclose(ctx.cond_norm.mean, cond.mean)
|
||||
np.testing.assert_allclose(ctx.tgt_norm.mean, tgt.mean)
|
||||
np.testing.assert_allclose(ctx.sec_phys_norm.mean, sec_phys.mean)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Guards
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_missing_model_config_raises(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path)
|
||||
ckpt = torch.load(checkpoint, weights_only=False)
|
||||
del ckpt["model_config"]
|
||||
torch.save(ckpt, checkpoint)
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="no model_config"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
|
||||
def test_missing_sec_decoder_raises(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path)
|
||||
ckpt = torch.load(checkpoint, weights_only=False)
|
||||
del ckpt["sec_decoder"]
|
||||
torch.save(ckpt, checkpoint)
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="no sec_decoder"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
|
||||
def test_missing_sec_phys_normalizer_raises(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path)
|
||||
ckpt = torch.load(checkpoint, weights_only=False)
|
||||
del ckpt["normalizer"]["sec_phys"]
|
||||
torch.save(ckpt, checkpoint)
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="no normalizer.sec_phys"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
|
||||
def test_onehot_particle_conditioning_without_topn_map_raises(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path, model_cfg=_onehot_model_cfg())
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="pdg_topn_map"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
|
||||
def test_onehot_particle_conditioning_with_topn_map_succeeds(tmp_path):
|
||||
topn = TopNMap(class_map={11: 0, 22: 1}, other_members={})
|
||||
checkpoint = _write_checkpoint(
|
||||
tmp_path,
|
||||
model_cfg=_onehot_model_cfg(),
|
||||
pdg_topn_map=topnmap_to_json(topn),
|
||||
)
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
assert ctx.particle_conditioning == "onehot"
|
||||
assert ctx.pdg_topn_map is not None
|
||||
assert ctx.pdg_topn_map.class_map == {11: 0, 22: 1}
|
||||
|
||||
|
||||
def test_onehot_particle_type_target_without_sec_type_topn_map_raises(tmp_path):
|
||||
"""DEFAULT_CONFIG's stage2_model.particle_type.target="onehot" needs a
|
||||
sec_type_topn_map (gitea #29) — a checkpoint with neither key at all
|
||||
(not even the pre-#29 pdg_topn_map to fall back to) must fail loudly."""
|
||||
checkpoint = _write_checkpoint(tmp_path, sec_type_topn_map=None)
|
||||
ckpt = torch.load(checkpoint, weights_only=False)
|
||||
del ckpt["sec_type_topn_map"]
|
||||
torch.save(ckpt, checkpoint)
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="sec_type_topn_map"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
|
||||
|
||||
def test_pre_gitea_29_checkpoint_falls_back_to_pdg_topn_map_for_sec_type(tmp_path):
|
||||
"""A checkpoint written before gitea #29 has no sec_type_topn_map key at
|
||||
all — conditioning and secondary-type onehot maps were always the same
|
||||
map, saved once under pdg_topn_map. load_for_inference must reproduce
|
||||
that exact pre-#29 behavior for such a checkpoint."""
|
||||
topn = TopNMap(class_map={11: 0, 22: 1, -11: 2}, other_members={})
|
||||
checkpoint = _write_checkpoint(
|
||||
tmp_path,
|
||||
model_cfg=_onehot_model_cfg(),
|
||||
pdg_topn_map=topnmap_to_json(topn),
|
||||
sec_type_topn_map=None,
|
||||
)
|
||||
ckpt = torch.load(checkpoint, weights_only=False)
|
||||
del ckpt["sec_type_topn_map"]
|
||||
torch.save(ckpt, checkpoint)
|
||||
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict")
|
||||
assert ctx.sec_type_topn_map is not None
|
||||
assert ctx.sec_type_topn_map.class_map == {11: 0, 22: 1, -11: 2}
|
||||
|
||||
|
||||
def test_ema_weights_requested_but_missing_raises(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path, ema=False)
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match="no EMA weights"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), "predict", weights="ema")
|
||||
|
||||
|
||||
def test_ema_weights_requested_and_present_succeeds(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path, ema=True)
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict", weights="ema")
|
||||
assert ctx.stage1 is not None and ctx.stage2 is not None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("command_name", ["predict", "rollout"])
|
||||
def test_inactive_stage_with_require_stage2_raises_with_command_name(tmp_path, command_name):
|
||||
checkpoint = _write_checkpoint(tmp_path, model_cfg=_model_cfg(stage2_active=False))
|
||||
|
||||
with pytest.raises(CheckpointCompatibilityError, match=f"{command_name} needs both"):
|
||||
load_for_inference(checkpoint, torch.device("cpu"), command_name)
|
||||
|
||||
|
||||
def test_inactive_stage_with_require_stage2_false_succeeds_with_stage2_none(tmp_path):
|
||||
checkpoint = _write_checkpoint(tmp_path, model_cfg=_model_cfg(stage2_active=False))
|
||||
|
||||
ctx = load_for_inference(checkpoint, torch.device("cpu"), "predict", require_stage2=False)
|
||||
assert ctx.stage1 is not None
|
||||
assert ctx.stage2 is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# conditioning_axes / stage_cfg
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_conditioning_axes_v02_flat_string_applies_to_both_axes():
|
||||
assert conditioning_axes({"conditioning": "embedding"}) == ("embedding", "embedding")
|
||||
|
||||
|
||||
def test_conditioning_axes_v03_nested_dict_independent_per_axis():
|
||||
model_cfg = {"conditioning": {"particle": {"type": "onehot"}, "material": {"type": "physical"}}}
|
||||
assert conditioning_axes(model_cfg) == ("onehot", "physical")
|
||||
|
||||
|
||||
def test_conditioning_axes_missing_key_uses_default():
|
||||
assert conditioning_axes({}, default="embedding") == ("embedding", "embedding")
|
||||
|
||||
|
||||
def test_stage_cfg_new_shape_returns_subdict():
|
||||
model_cfg = {"stage2_model": {"k_max": 7}}
|
||||
assert stage_cfg(model_cfg, "stage2") == {"k_max": 7}
|
||||
|
||||
|
||||
def test_stage_cfg_v02_flat_shape_returns_empty_dict():
|
||||
model_cfg = {"hidden_dim": 32, "n_blocks": 4}
|
||||
assert stage_cfg(model_cfg, "stage2") == {}
|
||||
@@ -1,13 +1,18 @@
|
||||
import uuid
|
||||
|
||||
import torch
|
||||
import yaml
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from giant.cli import (
|
||||
_CEPH_PREDICTIONS,
|
||||
_resolve_prediction_output,
|
||||
_write_prediction_ref,
|
||||
app,
|
||||
)
|
||||
|
||||
runner = CliRunner()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _resolve_prediction_output
|
||||
@@ -150,3 +155,20 @@ def test_ref_checkpoint_path_is_absolute(tmp_path):
|
||||
data = yaml.safe_load(ref_path.read_text())
|
||||
|
||||
assert data["checkpoint"].startswith("/")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bootstrap failure surfaces via the CLI (issues.md Issue 5 — confirms
|
||||
# CheckpointCompatibilityError -> typer.Exit(1) actually wires up end-to-end,
|
||||
# not just at the giant.checkpoint_io unit level).
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_predict_exits_1_on_checkpoint_missing_model_config(tmp_path):
|
||||
checkpoint = tmp_path / "bad.pt"
|
||||
torch.save({"sec_decoder": {}, "normalizer": {"sec_phys": {}}}, checkpoint)
|
||||
|
||||
result = runner.invoke(app, ["predict", "dummy.parquet", "--checkpoint", str(checkpoint)])
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert "checkpoint has no model_config" in result.output
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
"""Thin CLI smoke coverage for `giant rollout` (issues.md Issue 5) — confirms
|
||||
the CheckpointCompatibilityError raised by giant.checkpoint_io.load_for_inference
|
||||
surfaces as a clean typer.Exit(1) with the expected message, end-to-end
|
||||
through the CLI, not just at the giant.checkpoint_io unit level."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import torch
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from giant.cli import app
|
||||
|
||||
runner = CliRunner()
|
||||
|
||||
|
||||
def test_rollout_exits_1_on_checkpoint_missing_model_config(tmp_path):
|
||||
checkpoint = tmp_path / "bad.pt"
|
||||
torch.save({"sec_decoder": {}, "normalizer": {"sec_phys": {}}}, checkpoint)
|
||||
|
||||
result = runner.invoke(
|
||||
app,
|
||||
[
|
||||
"rollout",
|
||||
"dummy.parquet",
|
||||
"--checkpoint",
|
||||
str(checkpoint),
|
||||
"--geometry",
|
||||
"dummy_geometry.pkl",
|
||||
],
|
||||
)
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert "checkpoint has no model_config" in result.output
|
||||
@@ -41,6 +41,10 @@ def test_stage_prefixed_generator_overrides_shared_mode(monkeypatch, tmp_path):
|
||||
|
||||
|
||||
def test_stage2_only_knobs(monkeypatch, tmp_path):
|
||||
# --stage2-stage1-context is exercised separately at the overrides-dict
|
||||
# level (test_overrides_from_flags_stage2_only_knobs in test_config.py):
|
||||
# its only non-default value, "sampled", is rejected by validate_config
|
||||
# (issues.md Issue 1), so it can't appear in a full CLI invocation here.
|
||||
cfg = _invoke_and_capture_cfg(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
@@ -53,15 +57,12 @@ def test_stage2_only_knobs(monkeypatch, tmp_path):
|
||||
"32",
|
||||
"--stage2-context-dim",
|
||||
"16",
|
||||
"--stage2-stage1-context",
|
||||
"sampled",
|
||||
],
|
||||
)
|
||||
assert cfg["stage2_model"]["decoder"] == "one_shot"
|
||||
assert cfg["stage2_model"]["k_max"] == 8
|
||||
assert cfg["stage2_model"]["hidden_dim"] == 32
|
||||
assert cfg["stage2_model"]["context_dim"] == 16
|
||||
assert cfg["stage2_model"]["stage1_context"] == "sampled"
|
||||
# untouched stage1 defaults
|
||||
assert cfg["stage1_model"]["hidden_dim"] == 256
|
||||
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
import pytest
|
||||
from giant.cond_layout import AXIS_TYPES, CondLayout
|
||||
from giant.constants import COND_DIM, COND_DIM_BASE, MATERIAL_PHYS_DIM, PARTICLE_PHYS_DIM
|
||||
|
||||
# ── cond_cat column layout ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_topn_cols_neither_onehot():
|
||||
layout = CondLayout.from_types("physical", "embedding")
|
||||
assert (layout.particle_topn_col, layout.material_topn_col) == (None, None)
|
||||
assert layout.cat_dim == 2
|
||||
|
||||
|
||||
def test_topn_cols_particle_only():
|
||||
layout = CondLayout.from_types("onehot", "physical")
|
||||
assert (layout.particle_topn_col, layout.material_topn_col) == (2, None)
|
||||
assert layout.cat_dim == 3
|
||||
|
||||
|
||||
def test_topn_cols_material_only():
|
||||
layout = CondLayout.from_types("physical", "onehot")
|
||||
assert (layout.particle_topn_col, layout.material_topn_col) == (None, 2)
|
||||
assert layout.cat_dim == 3
|
||||
|
||||
|
||||
def test_topn_cols_both_onehot_particle_then_material():
|
||||
layout = CondLayout.from_types("onehot", "onehot")
|
||||
assert (layout.particle_topn_col, layout.material_topn_col) == (2, 3)
|
||||
assert layout.cat_dim == 4
|
||||
|
||||
|
||||
def test_dense_vocab_cols_are_mode_independent():
|
||||
"""Columns 0/1 are always the dense pdg/material index — giant.model.routers
|
||||
reads them without knowing the conditioning mode."""
|
||||
assert (CondLayout.PDG_COL, CondLayout.MAT_COL) == (0, 1)
|
||||
for particle in AXIS_TYPES:
|
||||
for material in AXIS_TYPES:
|
||||
layout = CondLayout.from_types(particle, material)
|
||||
assert layout.particle_topn_col not in (layout.PDG_COL, layout.MAT_COL)
|
||||
assert layout.material_topn_col not in (layout.PDG_COL, layout.MAT_COL)
|
||||
|
||||
|
||||
# ── cond_cont slice layout ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_cont_slices_tile_cond_cont_exactly():
|
||||
"""base / particle_phys / material_phys must partition cond_cont with no
|
||||
gap and no overlap — a gap or overlap is exactly the silent
|
||||
mis-indexing this object exists to prevent."""
|
||||
layout = CondLayout.from_types("physical", "physical")
|
||||
covered = list(range(*layout.base.indices(COND_DIM)))
|
||||
covered += list(range(*layout.particle_phys.indices(COND_DIM)))
|
||||
covered += list(range(*layout.material_phys.indices(COND_DIM)))
|
||||
assert covered == list(range(COND_DIM))
|
||||
|
||||
|
||||
def test_cont_slice_widths_match_constants():
|
||||
layout = CondLayout.from_types("embedding", "embedding")
|
||||
assert layout.base == slice(0, COND_DIM_BASE)
|
||||
assert layout.particle_phys.stop - layout.particle_phys.start == PARTICLE_PHYS_DIM
|
||||
assert layout.material_phys.stop - layout.material_phys.start == MATERIAL_PHYS_DIM
|
||||
assert layout.cont_dim == COND_DIM
|
||||
|
||||
|
||||
def test_cont_slices_are_mode_independent():
|
||||
"""cond_cont is COND_DIM wide in every mode — a non-"physical" axis gets
|
||||
its block zero-filled rather than dropped, so the slices never move."""
|
||||
physical = CondLayout.from_types("physical", "physical")
|
||||
for particle in AXIS_TYPES:
|
||||
for material in AXIS_TYPES:
|
||||
layout = CondLayout.from_types(particle, material)
|
||||
assert layout.base == physical.base
|
||||
assert layout.particle_phys == physical.particle_phys
|
||||
assert layout.material_phys == physical.material_phys
|
||||
|
||||
|
||||
# ── validation ───────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_unknown_particle_type_raises():
|
||||
with pytest.raises(ValueError, match="unknown conditioning.particle.type 'bogus'"):
|
||||
CondLayout.from_types("bogus", "physical")
|
||||
|
||||
|
||||
def test_unknown_material_type_raises():
|
||||
with pytest.raises(ValueError, match="unknown conditioning.material.type 'bogus'"):
|
||||
CondLayout.from_types("physical", "bogus")
|
||||
+145
-6
@@ -51,9 +51,13 @@ def test_giant_config_to_dict_matches_default_config():
|
||||
gconfig.Stage2WganConfig,
|
||||
gconfig.RouterConfig,
|
||||
gconfig.Stage2RouterConfig,
|
||||
gconfig.TrunkConfig,
|
||||
gconfig.NSecConfig,
|
||||
gconfig.ParticleTypeConfig,
|
||||
gconfig.AutoregressiveConfig,
|
||||
gconfig.HeadConfig,
|
||||
gconfig.Stage1HeadsConfig,
|
||||
gconfig.Stage2HeadsConfig,
|
||||
gconfig.Stage1ModelConfig,
|
||||
gconfig.Stage2ModelConfig,
|
||||
gconfig.TrainConfig,
|
||||
@@ -75,6 +79,49 @@ def test_stage2_model_config_defaults_match_documented_v030_intent():
|
||||
assert spec.particle_type.target == "onehot"
|
||||
|
||||
|
||||
def test_trunk_config_defaults_to_resmlp_for_both_stages():
|
||||
"""gitea #33: a v0.2-migrated / pre-existing config with no `trunk` key
|
||||
at all must reproduce today's behaviour exactly."""
|
||||
assert gconfig.Stage1ModelConfig().trunk.type == "resmlp"
|
||||
assert gconfig.Stage2ModelConfig().trunk.type == "resmlp"
|
||||
assert gconfig.DEFAULT_CONFIG["stage1_model"]["trunk"]["type"] == "resmlp"
|
||||
assert gconfig.DEFAULT_CONFIG["stage2_model"]["trunk"]["type"] == "resmlp"
|
||||
|
||||
|
||||
def test_trunk_config_defaults_block_conditioning_to_add_for_both_stages():
|
||||
"""gitea #34: a pre-existing config with no `block_conditioning` key
|
||||
must reproduce today's additive-bias behaviour exactly."""
|
||||
assert gconfig.Stage1ModelConfig().trunk.block_conditioning == "add"
|
||||
assert gconfig.Stage2ModelConfig().trunk.block_conditioning == "add"
|
||||
assert gconfig.DEFAULT_CONFIG["stage1_model"]["trunk"]["block_conditioning"] == "add"
|
||||
assert gconfig.DEFAULT_CONFIG["stage2_model"]["trunk"]["block_conditioning"] == "add"
|
||||
|
||||
|
||||
def test_heads_config_defaults_reproduce_pre_gitea_36_hardcoded_shape():
|
||||
"""gitea #36: a pre-existing config with no `heads` key must reproduce
|
||||
today's hardcoded `hidden_dim // 2`, one-hidden-layer architecture
|
||||
exactly."""
|
||||
assert gconfig.Stage1ModelConfig().heads.n_sec.hidden_ratio == 0.5
|
||||
assert gconfig.Stage1ModelConfig().heads.n_sec.depth == 2
|
||||
assert gconfig.Stage2ModelConfig().heads.n_sec.hidden_ratio == 0.5
|
||||
assert gconfig.Stage2ModelConfig().heads.n_sec.depth == 2
|
||||
assert gconfig.Stage2ModelConfig().heads.type.hidden_ratio == 0.5
|
||||
assert gconfig.Stage2ModelConfig().heads.type.depth == 2
|
||||
assert gconfig.DEFAULT_CONFIG["stage1_model"]["heads"]["n_sec"] == {"hidden_ratio": 0.5, "depth": 2}
|
||||
assert gconfig.DEFAULT_CONFIG["stage2_model"]["heads"]["n_sec"] == {"hidden_ratio": 0.5, "depth": 2}
|
||||
assert gconfig.DEFAULT_CONFIG["stage2_model"]["heads"]["type"] == {"hidden_ratio": 0.5, "depth": 2}
|
||||
|
||||
|
||||
def test_particle_type_config_n_classes_defaults_to_zero_and_round_trips():
|
||||
"""gitea #29: n_classes=0 means "inherit conditioning.particle.emb_dim"
|
||||
— the default must stay 0 so an existing config.toml with no
|
||||
stage2_model.particle_type.n_classes key reproduces pre-#29 behavior."""
|
||||
assert gconfig.ParticleTypeConfig().n_classes == 0
|
||||
spec = gconfig.ParticleTypeConfig.from_dict({"n_classes": 32})
|
||||
assert spec.n_classes == 32
|
||||
assert spec.to_dict()["n_classes"] == 32
|
||||
|
||||
|
||||
def test_router_config_extra_round_trips_composed_axis_keys():
|
||||
d = {"enabled": True, "type": "composed", "axis0_type": "energy", "axis0_n_experts": 4}
|
||||
router = gconfig.RouterConfig.from_dict(d)
|
||||
@@ -95,10 +142,15 @@ def test_stage1_router_config_has_no_tie_to_stage1_key():
|
||||
assert "tie_to_stage1" not in gconfig.RouterConfig().to_dict()
|
||||
|
||||
|
||||
def test_n_sec_config_extra_round_trips_legacy_owner():
|
||||
n_sec = gconfig.NSecConfig.from_dict({"mode": "head", "legacy_owner": "stage1"})
|
||||
assert n_sec.legacy_owner == "stage1"
|
||||
assert n_sec.to_dict() == {"mode": "head", "lambda": 0.1, "legacy_owner": "stage1"}
|
||||
def test_n_sec_config_owner_defaults_to_stage2():
|
||||
n_sec = gconfig.NSecConfig()
|
||||
assert n_sec.owner == "stage2"
|
||||
|
||||
|
||||
def test_n_sec_config_owner_round_trips():
|
||||
n_sec = gconfig.NSecConfig.from_dict({"mode": "head", "owner": "stage1"})
|
||||
assert n_sec.owner == "stage1"
|
||||
assert n_sec.to_dict() == {"mode": "head", "lambda": 0.1, "owner": "stage1"}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -682,6 +734,15 @@ def test_validate_config_stop_token_not_implemented():
|
||||
assert "stop_token" in str(e)
|
||||
|
||||
|
||||
def test_validate_config_stage1_context_sampled_not_implemented():
|
||||
cfg = _cfg_with(**{"stage2_model.stage1_context": "sampled"})
|
||||
try:
|
||||
gconfig.validate_config(cfg)
|
||||
assert False, "expected ValueError"
|
||||
except ValueError as e:
|
||||
assert "sampled" in str(e)
|
||||
|
||||
|
||||
def test_validate_config_n_sec_truth_rejected_for_rollout_capable_checkpoint():
|
||||
"""'n_sec.mode = "truth" is invalid for a rollout-capable checkpoint' —
|
||||
both stages active means giant rollout
|
||||
@@ -733,6 +794,31 @@ def test_validate_config_ar_default_markov_always_passes():
|
||||
gconfig.validate_config(cfg) # must not raise
|
||||
|
||||
|
||||
def test_validate_config_ar_order_energy_desc_passes():
|
||||
"""'energy_desc' is the only implemented order — must not raise."""
|
||||
cfg = _cfg_with(
|
||||
**{
|
||||
"stage2_model.decoder": "autoregressive",
|
||||
"stage2_model.autoregressive.order": "energy_desc",
|
||||
}
|
||||
)
|
||||
gconfig.validate_config(cfg) # must not raise
|
||||
|
||||
|
||||
def test_validate_config_ar_order_invalid_value_rejected():
|
||||
cfg = _cfg_with(
|
||||
**{
|
||||
"stage2_model.decoder": "autoregressive",
|
||||
"stage2_model.autoregressive.order": "energy_asc",
|
||||
}
|
||||
)
|
||||
try:
|
||||
gconfig.validate_config(cfg)
|
||||
assert False, "expected ValueError"
|
||||
except ValueError as e:
|
||||
assert "order" in str(e)
|
||||
|
||||
|
||||
def test_validate_config_ar_history_attention_passes():
|
||||
"""v0.3.0 step 7 implements history='attention' — must not raise."""
|
||||
cfg = _cfg_with(
|
||||
@@ -786,11 +872,12 @@ def test_validate_config_ar_teacher_forcing_invalid_value_rejected():
|
||||
|
||||
|
||||
def test_validate_config_ar_checks_skipped_under_one_shot():
|
||||
"""history/teacher_forcing values that would fail under AR are irrelevant
|
||||
(and unchecked) when decoder='one_shot'."""
|
||||
"""order/history/teacher_forcing values that would fail under AR are
|
||||
irrelevant (and unchecked) when decoder='one_shot'."""
|
||||
cfg = _cfg_with(
|
||||
**{
|
||||
"stage2_model.decoder": "one_shot",
|
||||
"stage2_model.autoregressive.order": "bogus",
|
||||
"stage2_model.autoregressive.history": "attention",
|
||||
"stage2_model.autoregressive.teacher_forcing": "scheduled",
|
||||
}
|
||||
@@ -861,6 +948,41 @@ def test_validate_config_keys_skips_meta_section():
|
||||
gconfig.validate_config_keys(cfg) # must not raise
|
||||
|
||||
|
||||
def test_validate_config_keys_allows_trunk_type():
|
||||
cfg = _cfg_with(**{"stage1_model.trunk.type": "resmlp", "stage2_model.trunk.type": "resmlp"})
|
||||
gconfig.validate_config_keys(cfg) # must not raise
|
||||
|
||||
|
||||
def test_validate_config_keys_rejects_unknown_trunk_key():
|
||||
cfg = _cfg_with(**{"stage1_model.trunk.type_o": "resmlp"}) # typo for type
|
||||
try:
|
||||
gconfig.validate_config_keys(cfg)
|
||||
assert False, "expected ValueError"
|
||||
except ValueError as e:
|
||||
assert "stage1_model.trunk.type_o" in str(e)
|
||||
assert "type" in str(e)
|
||||
|
||||
|
||||
def test_validate_config_keys_allows_block_conditioning():
|
||||
cfg = _cfg_with(
|
||||
**{
|
||||
"stage1_model.trunk.block_conditioning": "film",
|
||||
"stage2_model.trunk.block_conditioning": "adaln",
|
||||
}
|
||||
)
|
||||
gconfig.validate_config_keys(cfg) # must not raise
|
||||
|
||||
|
||||
def test_validate_config_keys_rejects_unknown_block_conditioning_key():
|
||||
cfg = _cfg_with(**{"stage1_model.trunk.block_conditioning_o": "film"}) # typo
|
||||
try:
|
||||
gconfig.validate_config_keys(cfg)
|
||||
assert False, "expected ValueError"
|
||||
except ValueError as e:
|
||||
assert "stage1_model.trunk.block_conditioning_o" in str(e)
|
||||
assert "block_conditioning" in str(e)
|
||||
|
||||
|
||||
def test_merge_cli_overrides_rejects_typo_in_toml_file(tmp_path):
|
||||
path = tmp_path / "config.toml"
|
||||
path.write_text("[meta]\nconfig_version = 3\n\n[stage1_model]\nn_res_block = 12\n")
|
||||
@@ -1008,6 +1130,23 @@ def test_overrides_from_flags_wgan_knobs_split_per_stage(shared, stage1_specific
|
||||
assert overrides["stage2_model"]["wgan"][path_key] == 2.5
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("stage_flag", "stage_model", "path_key"),
|
||||
[
|
||||
("stage1_critic_hidden_dim", "stage1_model", "critic_hidden_dim"),
|
||||
("stage1_critic_n_res_blocks", "stage1_model", "critic_n_res_blocks"),
|
||||
("stage2_critic_hidden_dim", "stage2_model", "critic_hidden_dim"),
|
||||
("stage2_critic_n_res_blocks", "stage2_model", "critic_n_res_blocks"),
|
||||
],
|
||||
)
|
||||
def test_overrides_from_flags_critic_sizing_is_stage_scoped_only(stage_flag, stage_model, path_key):
|
||||
"""critic_hidden_dim/critic_n_res_blocks are architectural per-stage
|
||||
knobs (gitea #28) — unlike n_critic/gp_weight/noise_dim/critic_lr above,
|
||||
there is deliberately no shared alias that fans out to both stages."""
|
||||
overrides = gconfig.overrides_from_flags({stage_flag: 32})
|
||||
assert overrides == {stage_model: {"wgan": {path_key: 32}}}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# checkpoint config-mismatch warnings (unchanged surface, still exercised)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
"""Consumed-keys audit (issues.md Issue 5).
|
||||
|
||||
`validate_config_keys` (`giant/config.py`) only checks that a config key is
|
||||
*declared* — present somewhere in `DEFAULT_CONFIG`, which is generated from
|
||||
the frozen dataclasses. It says nothing about whether anything actually
|
||||
*reads* the value once parsed. Issues 1, 2 and 4 are three keys that slipped
|
||||
through exactly that gap: declared, round-tripped, silently ignored. This
|
||||
module walks every leaf path in `DEFAULT_CONFIG` and asserts each is either
|
||||
genuinely consumed by the model-building/training/rollout code, or explicitly
|
||||
recorded in `_KNOWN_UNUSED` with a reason.
|
||||
|
||||
"Consumed" is approximated by static analysis rather than true call-graph
|
||||
reachability: for each leaf path's field name, does it appear anywhere in a
|
||||
fixed whitelist of source files as a real attribute access, a dict-key-shaped
|
||||
string constant, or a function/constructor parameter name (the last of these
|
||||
because `Router` subclasses receive their config via `**kwargs` filtered by
|
||||
signature — see `giant.model.routers.build_router`)? Docstrings are excluded
|
||||
from the string-constant scan so prose mentioning a dotted config path in
|
||||
passing can't masquerade as a read of it. This whitelist-based approach is
|
||||
deliberately narrower than "anywhere in `giant/`": scanning the whole package
|
||||
produces false negatives from unrelated identifier collisions (e.g.
|
||||
`giant/analysis/router_gating.py`'s `_top1_shares(..., order: list, ...)`
|
||||
parameter would otherwise make `stage2_model.autoregressive.order` read as
|
||||
"consumed").
|
||||
"""
|
||||
|
||||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
from giant.config import DEFAULT_CONFIG
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
|
||||
# Files that legitimately consume model_config / training config at
|
||||
# build/train/rollout time. Not `giant/cli.py` (a CLI flag existing is not
|
||||
# consumption — that's precisely how Issue 1 slipped through), not
|
||||
# `giant/config.py` itself (declaring/parsing a field is not reading it), and
|
||||
# not `giant/model/_legacy.py` (the protected v0.2 migration surface, which
|
||||
# intentionally re-derives old flat keys under old names).
|
||||
_CONSUMER_ROOTS = ("giant/model", "giant/training")
|
||||
_CONSUMER_FILES = (
|
||||
"giant/sample.py",
|
||||
"giant/pipeline.py",
|
||||
"giant/rollout.py",
|
||||
"giant/checkpoint_io.py",
|
||||
"giant/particles.py",
|
||||
"giant/materials.py",
|
||||
)
|
||||
_EXCLUDED_FILES = ("giant/model/_legacy.py",)
|
||||
|
||||
# Leaf DEFAULT_CONFIG paths that are declared but not (yet) read anywhere in
|
||||
# the consumer whitelist above. Each entry must name the issue that tracks
|
||||
# it. If a key here starts showing up as consumed, the fix landed and this
|
||||
# entry is stale — see test_known_unused_allow_list_has_no_stale_entries.
|
||||
_KNOWN_UNUSED = {
|
||||
"stage2_model.stage1_context": (
|
||||
"issues.md Issue 1 — trainers.py hardcodes stage1_ctx to the "
|
||||
"ground-truth stage-1 output; 'sampled' is now rejected loudly by "
|
||||
"validate_config (not silently accepted), but the key still isn't "
|
||||
"read by any build/train consumer file since only 'truth' can pass "
|
||||
"validation — see Issue 16 for the real implementation"
|
||||
),
|
||||
"stage2_model.autoregressive.order": (
|
||||
"gitea #30 — validate_config now checks order is 'energy_desc', but "
|
||||
"nothing in the build/train/rollout consumer whitelist reads the "
|
||||
"value itself since it's still single-valued"
|
||||
),
|
||||
}
|
||||
|
||||
# "lambda" is a Python keyword, so the dataclasses expose the dict key
|
||||
# "lambda" as the field `lambda_weight` (giant/config.py:49-50).
|
||||
_FIELD_NAME_OVERRIDES = {"lambda": "lambda_weight"}
|
||||
|
||||
|
||||
def _leaf_paths(node: dict, prefix: str = "") -> list[str]:
|
||||
paths = []
|
||||
for key, value in node.items():
|
||||
if prefix == "" and key == "meta":
|
||||
continue
|
||||
path = f"{prefix}.{key}" if prefix else key
|
||||
if isinstance(value, dict):
|
||||
paths.extend(_leaf_paths(value, path))
|
||||
else:
|
||||
paths.append(path)
|
||||
return paths
|
||||
|
||||
|
||||
def _field_name(leaf_path: str) -> str:
|
||||
name = leaf_path.rsplit(".", 1)[-1]
|
||||
return _FIELD_NAME_OVERRIDES.get(name, name)
|
||||
|
||||
|
||||
def _is_docstring_expr(expr: ast.Expr) -> bool:
|
||||
return isinstance(expr.value, ast.Constant) and isinstance(expr.value.value, str)
|
||||
|
||||
|
||||
def _collect_names(source: str, filename: str) -> set[str]:
|
||||
tree = ast.parse(source, filename=filename)
|
||||
docstring_ids = set()
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, (ast.Module, ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
|
||||
body = getattr(node, "body", [])
|
||||
if body and isinstance(body[0], ast.Expr) and _is_docstring_expr(body[0]):
|
||||
docstring_ids.add(id(body[0].value))
|
||||
|
||||
names: set[str] = set()
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Attribute):
|
||||
names.add(node.attr)
|
||||
elif isinstance(node, ast.Constant) and isinstance(node.value, str) and id(node) not in docstring_ids:
|
||||
names.add(node.value)
|
||||
elif isinstance(node, ast.arg):
|
||||
names.add(node.arg)
|
||||
elif isinstance(node, ast.keyword) and node.arg is not None:
|
||||
names.add(node.arg)
|
||||
return names
|
||||
|
||||
|
||||
def _consumer_files() -> list[Path]:
|
||||
files: set[Path] = {_REPO_ROOT / f for f in _CONSUMER_FILES}
|
||||
for root in _CONSUMER_ROOTS:
|
||||
files |= set((_REPO_ROOT / root).rglob("*.py"))
|
||||
files -= {_REPO_ROOT / f for f in _EXCLUDED_FILES}
|
||||
return sorted(files)
|
||||
|
||||
|
||||
def _consumed_names() -> set[str]:
|
||||
names: set[str] = set()
|
||||
for path in _consumer_files():
|
||||
names |= _collect_names(path.read_text(), str(path))
|
||||
return names
|
||||
|
||||
|
||||
def test_every_config_key_is_consumed_or_allow_listed():
|
||||
consumed = _consumed_names()
|
||||
unconsumed = {p for p in _leaf_paths(DEFAULT_CONFIG) if _field_name(p) not in consumed}
|
||||
unexplained = unconsumed - _KNOWN_UNUSED.keys()
|
||||
assert not unexplained, (
|
||||
f"config key(s) {sorted(unexplained)} are declared in DEFAULT_CONFIG "
|
||||
"but not read anywhere in the build/train/rollout consumer files "
|
||||
f"({[str(f.relative_to(_REPO_ROOT)) for f in _consumer_files()]}) — "
|
||||
"either wire the key up, or add it to _KNOWN_UNUSED with a reason "
|
||||
"(see issues.md Issue 5)"
|
||||
)
|
||||
|
||||
|
||||
def test_known_unused_allow_list_has_no_stale_entries():
|
||||
consumed = _consumed_names()
|
||||
all_paths = set(_leaf_paths(DEFAULT_CONFIG))
|
||||
stale = {p for p in _KNOWN_UNUSED if p not in all_paths or _field_name(p) in consumed}
|
||||
assert not stale, (
|
||||
f"_KNOWN_UNUSED entry/entries {sorted(stale)} no longer belong on the "
|
||||
"allow-list — either the key was removed from DEFAULT_CONFIG, or it "
|
||||
"is now consumed (the underlying issue was fixed). Remove the stale "
|
||||
"entry/entries."
|
||||
)
|
||||
@@ -4,7 +4,7 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts import create_root_files
|
||||
from giant.tools import create_root_files
|
||||
|
||||
parse_detector_spec = create_root_files.parse_detector_spec
|
||||
next_shard_index = create_root_files.next_shard_index
|
||||
|
||||
@@ -95,7 +95,7 @@ def _dummy_normalizer(width):
|
||||
|
||||
def test_streaming_dataset_offsets_colliding_event_ids_across_files(tmp_path):
|
||||
"""Two files that each restart event_id from 0 (one Geant4 job per file,
|
||||
see scripts/steps_to_parquet.py) must not have their same-numbered events
|
||||
see giant/tools/steps_to_parquet.py) must not have their same-numbered events
|
||||
collapsed together: every row from every file must show up in exactly one
|
||||
of train/val, and the number of distinct events must be the sum across
|
||||
files, not the union of raw ids."""
|
||||
|
||||
+3
-3
@@ -3,15 +3,15 @@ from typer.testing import CliRunner
|
||||
from giant import cli as giant_cli
|
||||
from giant.config import Conditioning
|
||||
from giant.data import setup_cache
|
||||
from scripts import dwarf
|
||||
from scripts.dwarf import app
|
||||
from giant.tools import dwarf
|
||||
from giant.tools.dwarf import app
|
||||
from test_pipeline import _make_synthetic_steps
|
||||
|
||||
runner = CliRunner()
|
||||
|
||||
|
||||
def test_conditioning_enum_shared_across_both_clis():
|
||||
"""giant.cli and scripts.dwarf must use the one giant.config.Conditioning
|
||||
"""giant.cli and giant.tools.dwarf must use the one giant.config.Conditioning
|
||||
enum, not independently redefined copies that could silently drift apart
|
||||
on valid --conditioning values."""
|
||||
assert dwarf.Conditioning is Conditioning
|
||||
|
||||
+3
-2
@@ -1,11 +1,12 @@
|
||||
import torch
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import COND_DIM
|
||||
from giant.model.network import Stage1Model
|
||||
from giant.model.schedule import CosineSchedule, flow_matching_loss
|
||||
from giant.sample import sample_flow, sample_ddim
|
||||
|
||||
PARTICLE_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
MATERIAL_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
PARTICLE_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
MATERIAL_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
|
||||
|
||||
def _small_model():
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.model.layers import build_mlp_head
|
||||
|
||||
|
||||
def test_build_mlp_head_depth_1_is_bare_linear():
|
||||
head = build_mlp_head(8, 4, hidden=16, depth=1)
|
||||
assert len(head) == 1
|
||||
assert isinstance(head[0], torch.nn.Linear)
|
||||
assert head[0].in_features == 8
|
||||
assert head[0].out_features == 4
|
||||
out = head(torch.randn(3, 8))
|
||||
assert out.shape == (3, 4)
|
||||
|
||||
|
||||
def test_build_mlp_head_depth_2_matches_pre_gitea_36_shape():
|
||||
head = build_mlp_head(8, 4, hidden=16, depth=2)
|
||||
assert len(head) == 3
|
||||
assert isinstance(head[0], torch.nn.Linear)
|
||||
assert head[0].in_features == 8
|
||||
assert head[0].out_features == 16
|
||||
assert isinstance(head[1], torch.nn.SiLU)
|
||||
assert isinstance(head[2], torch.nn.Linear)
|
||||
assert head[2].in_features == 16
|
||||
assert head[2].out_features == 4
|
||||
out = head(torch.randn(5, 8))
|
||||
assert out.shape == (5, 4)
|
||||
|
||||
|
||||
def test_build_mlp_head_depth_3_has_extra_hidden_layer():
|
||||
head = build_mlp_head(8, 4, hidden=16, depth=3)
|
||||
assert len(head) == 5
|
||||
widths = [(m.in_features, m.out_features) for m in head if isinstance(m, torch.nn.Linear)]
|
||||
assert widths == [(8, 16), (16, 16), (16, 4)]
|
||||
out = head(torch.randn(2, 8))
|
||||
assert out.shape == (2, 4)
|
||||
|
||||
|
||||
def test_build_mlp_head_depth_0_raises():
|
||||
with pytest.raises(ValueError, match="depth"):
|
||||
build_mlp_head(8, 4, hidden=16, depth=0)
|
||||
@@ -130,7 +130,7 @@ def _run_migration_check(mode: str, conditioning: str) -> None:
|
||||
new_stage1, new_stage2 = new_models["stage1"], new_models["stage2"]
|
||||
assert isinstance(new_stage1, net.Stage1Model)
|
||||
assert isinstance(new_stage2, net.Stage2OneShot)
|
||||
# legacy_owner="stage1": n_sec lives on stage1, not stage2, for a
|
||||
# n_sec.owner="stage1": n_sec lives on stage1, not stage2, for a
|
||||
# migrated v0.2 checkpoint.
|
||||
assert new_stage1.n_sec_head is not None
|
||||
assert new_stage2.n_sec_head is None
|
||||
@@ -177,7 +177,7 @@ def test_migration_wgan_physical():
|
||||
|
||||
def test_migrate_legacy_model_config_shape():
|
||||
"""_migrate_legacy_model_config produces the nested shape build_models
|
||||
expects, with the legacy_owner marker set so build_models routes the
|
||||
expects, with the n_sec.owner marker set so build_models routes the
|
||||
n_sec head back onto stage 1."""
|
||||
legacy_cfg = _legacy_model_config(mode="flow", conditioning="physical")
|
||||
migrated = net._migrate_legacy_model_config(legacy_cfg)
|
||||
@@ -187,7 +187,7 @@ def test_migrate_legacy_model_config_shape():
|
||||
assert migrated["conditioning"]["particle"]["n_layers"] == 2
|
||||
assert migrated["conditioning"]["material"]["n_layers"] == 2
|
||||
assert migrated["stage1_model"]["hidden_dim"] == HIDDEN_DIM
|
||||
assert migrated["stage2_model"]["n_sec"]["legacy_owner"] == "stage1"
|
||||
assert migrated["stage2_model"]["n_sec"]["owner"] == "stage1"
|
||||
assert migrated["stage2_model"]["decoder"] == "one_shot"
|
||||
|
||||
|
||||
@@ -278,6 +278,6 @@ def test_build_models_accepts_new_nested_shape_unchanged():
|
||||
models = net.build_models(cfg)
|
||||
assert isinstance(models["stage1"], net.Stage1Model)
|
||||
assert isinstance(models["stage2"], net.Stage2OneShot)
|
||||
# Fresh v0.3.0 config, no legacy_owner: n_sec lives on stage 2.
|
||||
# Fresh v0.3.0 config, n_sec.owner defaults to "stage2": n_sec lives on stage 2.
|
||||
assert models["stage1"].n_sec_head is None
|
||||
assert models["stage2"].n_sec_head is not None
|
||||
|
||||
+538
-49
@@ -5,24 +5,28 @@ import torch
|
||||
from giant import config as gconfig
|
||||
from giant.constants import CONT_SLOT_DIM, COND_DIM, PARTICLE_PHYS_DIM, SEC_SLOT_DIM
|
||||
from giant.model.network import (
|
||||
HISTORY_REGISTRY,
|
||||
AttentionHistory,
|
||||
ConditionEncoder,
|
||||
HistoryEncoder,
|
||||
MarkovHistory,
|
||||
SinusoidalEmbedding,
|
||||
Stage1Model,
|
||||
Stage2Autoregressive,
|
||||
Stage2OneShot,
|
||||
StageModel,
|
||||
build_critics,
|
||||
build_history,
|
||||
build_models,
|
||||
cat_col_layout,
|
||||
build_objective,
|
||||
stage2_trunk_sec_dim,
|
||||
stage2_type_dim,
|
||||
)
|
||||
|
||||
PARTICLE_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
MATERIAL_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
ONEHOT_PARTICLE_CFG = {"type": "onehot", "emb_dim": 6, "n_layers": 1}
|
||||
ONEHOT_MATERIAL_CFG = {"type": "onehot", "emb_dim": 4, "n_layers": 1}
|
||||
PARTICLE_CFG = gconfig.ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
MATERIAL_CFG = gconfig.ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
ONEHOT_PARTICLE_CFG = gconfig.ConditioningAxisConfig(type="onehot", emb_dim=6, n_layers=1)
|
||||
ONEHOT_MATERIAL_CFG = gconfig.ConditioningAxisConfig(type="onehot", emb_dim=4, n_layers=1)
|
||||
|
||||
|
||||
def test_sinusoidal_embedding_shape():
|
||||
@@ -90,48 +94,72 @@ def test_stage1_model_no_n_sec_head_by_default():
|
||||
assert model.n_sec_head is None
|
||||
|
||||
|
||||
# --- cat_col_layout / stage2_type_dim / stage2_trunk_sec_dim ---------------
|
||||
def test_stage1_model_n_sec_head_default_cfg_matches_pre_gitea_36_shape():
|
||||
"""No n_sec_head_cfg given must reproduce the old hardcoded
|
||||
hidden_dim // 2, one-hidden-layer architecture exactly (gitea #36)."""
|
||||
model = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=40,
|
||||
cond_out_dim=12,
|
||||
n_sec_head_k_max=15,
|
||||
)
|
||||
assert model.n_sec_head is not None
|
||||
assert len(model.n_sec_head) == 3
|
||||
assert model.n_sec_head[0].in_features == 12
|
||||
assert model.n_sec_head[0].out_features == 20 # hidden_dim // 2
|
||||
assert model.n_sec_head[2].out_features == 16 # k_max + 1
|
||||
|
||||
|
||||
def test_cat_col_layout_neither_onehot():
|
||||
assert cat_col_layout("physical", "embedding") == (None, None)
|
||||
def test_stage1_model_n_sec_head_cfg_controls_hidden_width_and_depth():
|
||||
model = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=32,
|
||||
cond_out_dim=16,
|
||||
n_sec_head_k_max=15,
|
||||
n_sec_head_cfg={"hidden_ratio": 0.25, "depth": 1},
|
||||
)
|
||||
assert model.n_sec_head is not None
|
||||
assert len(model.n_sec_head) == 1
|
||||
assert model.n_sec_head[0].in_features == 16
|
||||
assert model.n_sec_head[0].out_features == 16
|
||||
|
||||
|
||||
def test_cat_col_layout_particle_only():
|
||||
assert cat_col_layout("onehot", "physical") == (2, None)
|
||||
|
||||
|
||||
def test_cat_col_layout_material_only():
|
||||
assert cat_col_layout("physical", "onehot") == (None, 2)
|
||||
|
||||
|
||||
def test_cat_col_layout_both_onehot_particle_then_material():
|
||||
assert cat_col_layout("onehot", "onehot") == (2, 3)
|
||||
# --- stage2_type_dim / stage2_trunk_sec_dim --------------------------------
|
||||
# (the cond_cat column-layout tests live in tests/test_cond_layout.py)
|
||||
|
||||
|
||||
def test_stage2_type_dim_physical_is_particle_phys_dim():
|
||||
assert stage2_type_dim({"target": "physical"}, emb_dim=16) == PARTICLE_PHYS_DIM
|
||||
assert stage2_type_dim(gconfig.ParticleTypeConfig(target="physical"), emb_dim=16) == PARTICLE_PHYS_DIM
|
||||
|
||||
|
||||
def test_stage2_type_dim_onehot_and_embedding_are_emb_dim():
|
||||
assert stage2_type_dim({"target": "onehot"}, emb_dim=16) == 16
|
||||
assert stage2_type_dim({"target": "embedding"}, emb_dim=16) == 16
|
||||
assert stage2_type_dim(gconfig.ParticleTypeConfig(target="onehot"), emb_dim=16) == 16
|
||||
assert stage2_type_dim(gconfig.ParticleTypeConfig(target="embedding"), emb_dim=16) == 16
|
||||
|
||||
|
||||
def test_stage2_trunk_sec_dim_physical_matches_v02_sec_dim():
|
||||
k_max = 15
|
||||
assert stage2_trunk_sec_dim({"target": "physical"}, "flow", k_max, emb_dim=16) == k_max * SEC_SLOT_DIM
|
||||
assert stage2_trunk_sec_dim({"target": "physical"}, "wgan", k_max, emb_dim=16) == k_max * SEC_SLOT_DIM
|
||||
physical = gconfig.ParticleTypeConfig(target="physical")
|
||||
assert stage2_trunk_sec_dim(physical, "flow", k_max, emb_dim=16) == k_max * SEC_SLOT_DIM
|
||||
assert stage2_trunk_sec_dim(physical, "wgan", k_max, emb_dim=16) == k_max * SEC_SLOT_DIM
|
||||
|
||||
|
||||
def test_stage2_trunk_sec_dim_onehot_wgan_folds_type_in():
|
||||
k_max = 15
|
||||
assert stage2_trunk_sec_dim({"target": "onehot"}, "wgan", k_max, emb_dim=16) == k_max * (CONT_SLOT_DIM + 16)
|
||||
onehot = gconfig.ParticleTypeConfig(target="onehot")
|
||||
assert stage2_trunk_sec_dim(onehot, "wgan", k_max, emb_dim=16) == k_max * (CONT_SLOT_DIM + 16)
|
||||
|
||||
|
||||
def test_stage2_trunk_sec_dim_onehot_flow_excludes_type():
|
||||
k_max = 15
|
||||
assert stage2_trunk_sec_dim({"target": "onehot"}, "flow", k_max, emb_dim=16) == k_max * CONT_SLOT_DIM
|
||||
onehot = gconfig.ParticleTypeConfig(target="onehot")
|
||||
assert stage2_trunk_sec_dim(onehot, "flow", k_max, emb_dim=16) == k_max * CONT_SLOT_DIM
|
||||
|
||||
|
||||
# --- ConditionEncoder onehot mode -------------------------------------------
|
||||
@@ -139,8 +167,8 @@ def test_stage2_trunk_sec_dim_onehot_flow_excludes_type():
|
||||
|
||||
def test_condition_encoder_onehot_forward_shape_and_gradients():
|
||||
B = 8
|
||||
particle_emb_dim = int(ONEHOT_PARTICLE_CFG["emb_dim"])
|
||||
material_emb_dim = int(ONEHOT_MATERIAL_CFG["emb_dim"])
|
||||
particle_emb_dim = ONEHOT_PARTICLE_CFG.emb_dim
|
||||
material_emb_dim = ONEHOT_MATERIAL_CFG.emb_dim
|
||||
enc = ConditionEncoder(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
@@ -171,12 +199,12 @@ def test_condition_encoder_onehot_is_a_true_one_hot_vector():
|
||||
verify the concatenated input segment really is one-hot, not e.g. an
|
||||
accidentally-learned embedding."""
|
||||
B = 4
|
||||
particle_emb_dim = int(ONEHOT_PARTICLE_CFG["emb_dim"])
|
||||
particle_emb_dim = ONEHOT_PARTICLE_CFG.emb_dim
|
||||
enc = ConditionEncoder(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=ONEHOT_PARTICLE_CFG,
|
||||
material_cfg={"type": "physical", "emb_dim": 4, "n_layers": 1},
|
||||
material_cfg=gconfig.ConditioningAxisConfig(type="physical", emb_dim=4, n_layers=1),
|
||||
out_dim=16,
|
||||
)
|
||||
cond_cont = torch.zeros(B, COND_DIM)
|
||||
@@ -198,13 +226,11 @@ def test_condition_encoder_onehot_is_a_true_one_hot_vector():
|
||||
|
||||
|
||||
def _build_stage2(target: str, generator: str, emb_dim: int = 6) -> Stage2OneShot:
|
||||
particle_cfg = {"type": "physical", "emb_dim": emb_dim, "n_layers": 1}
|
||||
if target != "physical":
|
||||
particle_cfg = dict(particle_cfg)
|
||||
if target == "embedding":
|
||||
particle_cfg["type"] = "embedding"
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=emb_dim, n_layers=1)
|
||||
if target == "embedding":
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="embedding", emb_dim=emb_dim, n_layers=1)
|
||||
k_max = 5
|
||||
sec_dim = stage2_trunk_sec_dim({"target": target}, generator, k_max, emb_dim)
|
||||
sec_dim = stage2_trunk_sec_dim(gconfig.ParticleTypeConfig(target=target), generator, k_max, emb_dim)
|
||||
return Stage2OneShot(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
@@ -217,7 +243,7 @@ def _build_stage2(target: str, generator: str, emb_dim: int = 6) -> Stage2OneSho
|
||||
sec_dim=sec_dim,
|
||||
generator=generator,
|
||||
k_max=k_max,
|
||||
particle_type_cfg={"target": target, "lambda": 1.0},
|
||||
particle_type_cfg=gconfig.ParticleTypeConfig(target=target),
|
||||
)
|
||||
|
||||
|
||||
@@ -265,6 +291,38 @@ def test_stage2_oneshot_predict_type_raises_when_no_type_head():
|
||||
pass
|
||||
|
||||
|
||||
def test_stage2_oneshot_n_sec_head_and_type_head_cfg_control_hidden_width_and_depth():
|
||||
"""gitea #36: n_sec_head_cfg/type_head_cfg are independently tunable."""
|
||||
k_max, emb_dim = 5, 6
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="onehot", emb_dim=emb_dim, n_layers=1)
|
||||
sec_dim = stage2_trunk_sec_dim(gconfig.ParticleTypeConfig(target="onehot"), "flow", k_max, emb_dim)
|
||||
model = Stage2OneShot(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=40,
|
||||
n_res_blocks=1,
|
||||
cond_out_dim=12,
|
||||
context_dim=8,
|
||||
sec_dim=sec_dim,
|
||||
generator="flow",
|
||||
k_max=k_max,
|
||||
particle_type_cfg=gconfig.ParticleTypeConfig(target="onehot"),
|
||||
n_sec_head_cfg={"hidden_ratio": 0.25, "depth": 1},
|
||||
type_head_cfg={"hidden_ratio": 0.75, "depth": 2},
|
||||
)
|
||||
assert model.n_sec_head is not None
|
||||
assert len(model.n_sec_head) == 1
|
||||
assert model.n_sec_head[0].in_features == 12
|
||||
assert model.n_sec_head[0].out_features == k_max + 1
|
||||
|
||||
assert model.type_head is not None
|
||||
assert len(model.type_head) == 3
|
||||
assert model.type_head[0].out_features == 30 # round(40 * 0.75)
|
||||
assert model.type_head[2].out_features == k_max * emb_dim
|
||||
|
||||
|
||||
def test_stage2_oneshot_forward_shape_onehot_wgan():
|
||||
B, k_max, emb_dim = 4, 5, 6
|
||||
model = _build_stage2("onehot", "wgan", emb_dim=emb_dim)
|
||||
@@ -288,6 +346,33 @@ def test_stage2_oneshot_forward_shape_onehot_flow_excludes_type():
|
||||
assert out.shape == (B, k_max * CONT_SLOT_DIM)
|
||||
|
||||
|
||||
def test_stage2_oneshot_particle_type_n_classes_overrides_conditioning_emb_dim():
|
||||
"""gitea #29: stage2_model.particle_type.n_classes, not
|
||||
conditioning.particle.emb_dim, sizes the onehot type_head/type_dim when
|
||||
explicitly set — the two used to be silently the same number."""
|
||||
k_max = 5
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=6, n_layers=1)
|
||||
particle_type_cfg = gconfig.ParticleTypeConfig(target="onehot", n_classes=20)
|
||||
sec_dim = stage2_trunk_sec_dim(particle_type_cfg, "flow", k_max, 20)
|
||||
model = Stage2OneShot(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
cond_out_dim=16,
|
||||
context_dim=8,
|
||||
sec_dim=sec_dim,
|
||||
generator="flow",
|
||||
k_max=k_max,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
)
|
||||
assert model.type_dim == 20 # not particle_cfg.emb_dim == 6
|
||||
assert model.type_head is not None
|
||||
assert model.type_head[-1].out_features == k_max * 20
|
||||
|
||||
|
||||
# --- MarkovHistory -----------------------------------------------------------
|
||||
|
||||
|
||||
@@ -380,6 +465,48 @@ def test_attention_history_step_matches_forward():
|
||||
assert torch.allclose(stepped, expected, atol=1e-5)
|
||||
|
||||
|
||||
# --- HISTORY_REGISTRY / build_history (gitea #35) ----------------------------
|
||||
|
||||
|
||||
def test_history_registry_has_exactly_the_two_known_histories():
|
||||
assert set(HISTORY_REGISTRY) == {"markov", "attention"}
|
||||
|
||||
|
||||
def test_build_history_returns_correct_concrete_type():
|
||||
assert isinstance(build_history("markov", 4, 6), MarkovHistory)
|
||||
assert isinstance(build_history("attention", 4, 8), AttentionHistory)
|
||||
|
||||
|
||||
def test_build_history_unknown_name_raises():
|
||||
with pytest.raises(ValueError):
|
||||
build_history("bogus", 4, 6)
|
||||
|
||||
|
||||
def test_build_history_filters_kwargs_by_signature():
|
||||
"""Attention-only kwargs (n_heads/n_layers) must be silently dropped when
|
||||
building a MarkovHistory, matching build_router's documented behavior for
|
||||
per-type hyperparameters coexisting in one config."""
|
||||
hist = build_history("markov", 4, 6, n_heads=2, n_layers=1)
|
||||
assert isinstance(hist, MarkovHistory)
|
||||
|
||||
|
||||
def test_history_encoder_base_default_init_cache_and_step():
|
||||
"""A HistoryEncoder subclass implementing only forward() must still get
|
||||
working O(1) init_cache/step defaults from the base class."""
|
||||
|
||||
class _StubHistory(HistoryEncoder):
|
||||
def forward(self, feat, has_prev):
|
||||
return feat * 2
|
||||
|
||||
hist = _StubHistory()
|
||||
assert hist.init_cache() is None
|
||||
feat = torch.randn(2, 1, 4)
|
||||
has_prev = torch.ones(2, 1, dtype=torch.bool)
|
||||
out, cache = hist.step(feat, has_prev, "unused-cache")
|
||||
assert torch.equal(out, hist.forward(feat, has_prev))
|
||||
assert cache == "unused-cache"
|
||||
|
||||
|
||||
# --- Stage2Autoregressive (v0.3.0 step 5) -----------------------------------
|
||||
|
||||
|
||||
@@ -390,10 +517,9 @@ def _build_stage2_ar(
|
||||
k_max: int = 5,
|
||||
history: str = "markov",
|
||||
) -> Stage2Autoregressive:
|
||||
particle_cfg = {"type": "physical", "emb_dim": emb_dim, "n_layers": 1}
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=emb_dim, n_layers=1)
|
||||
if target == "embedding":
|
||||
particle_cfg = dict(particle_cfg)
|
||||
particle_cfg["type"] = "embedding"
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="embedding", emb_dim=emb_dim, n_layers=1)
|
||||
return Stage2Autoregressive(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
@@ -405,7 +531,7 @@ def _build_stage2_ar(
|
||||
context_dim=8,
|
||||
generator=generator,
|
||||
k_max=k_max,
|
||||
particle_type_cfg={"target": target, "lambda": 1.0},
|
||||
particle_type_cfg=gconfig.ParticleTypeConfig(target=target),
|
||||
history=history,
|
||||
)
|
||||
|
||||
@@ -423,6 +549,60 @@ def test_stage2_autoregressive_history_invalid_raises():
|
||||
_build_stage2_ar("onehot", "wgan", history="bogus")
|
||||
|
||||
|
||||
def test_stage2_autoregressive_particle_type_n_classes_overrides_conditioning_emb_dim():
|
||||
"""gitea #29, Stage2Autoregressive side — see the Stage2OneShot version
|
||||
of this test for the full rationale."""
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=6, n_layers=1)
|
||||
particle_type_cfg = gconfig.ParticleTypeConfig(target="onehot", n_classes=20)
|
||||
model = Stage2Autoregressive(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
cond_out_dim=16,
|
||||
context_dim=8,
|
||||
generator="flow",
|
||||
k_max=5,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
)
|
||||
assert model.type_dim == 20 # not particle_cfg.emb_dim == 6
|
||||
assert model.type_head is not None
|
||||
assert model.type_head[-1].out_features == 20
|
||||
|
||||
|
||||
def test_stage2_autoregressive_n_sec_head_and_type_head_cfg_control_hidden_width_and_depth():
|
||||
"""gitea #36, Stage2Autoregressive side — see the Stage2OneShot version
|
||||
of this test for the full rationale."""
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=6, n_layers=1)
|
||||
particle_type_cfg = gconfig.ParticleTypeConfig(target="onehot")
|
||||
model = Stage2Autoregressive(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=40,
|
||||
n_res_blocks=1,
|
||||
cond_out_dim=12,
|
||||
context_dim=8,
|
||||
generator="flow",
|
||||
k_max=5,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
n_sec_head_cfg={"hidden_ratio": 0.25, "depth": 1},
|
||||
type_head_cfg={"hidden_ratio": 0.75, "depth": 2},
|
||||
)
|
||||
assert model.n_sec_head is not None
|
||||
assert len(model.n_sec_head) == 1
|
||||
assert model.n_sec_head[0].in_features == 12
|
||||
assert model.n_sec_head[0].out_features == 6 # k_max + 1
|
||||
|
||||
assert model.type_head is not None
|
||||
assert len(model.type_head) == 3
|
||||
assert model.type_head[0].out_features == 30 # round(40 * 0.75)
|
||||
assert model.type_head[2].out_features == model.type_dim
|
||||
|
||||
|
||||
@pytest.mark.parametrize("target", ["physical", "onehot", "embedding"])
|
||||
@pytest.mark.parametrize("generator", ["wgan", "flow"])
|
||||
@pytest.mark.parametrize("history", ["markov", "attention"])
|
||||
@@ -432,9 +612,9 @@ def test_stage2_autoregressive_forward_shape(target, generator, history):
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.zeros(B, 2, dtype=torch.long)
|
||||
stage1_out = torch.randn(B, 9)
|
||||
type_dim = stage2_type_dim({"target": target}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target=target), emb_dim)
|
||||
history_feat, has_prev, remaining_frac, slot_idx = _ar_inputs(B, K, CONT_SLOT_DIM + type_dim)
|
||||
token_dim = stage2_trunk_sec_dim({"target": target}, generator, 1, emb_dim)
|
||||
token_dim = stage2_trunk_sec_dim(gconfig.ParticleTypeConfig(target=target), generator, 1, emb_dim)
|
||||
if generator == "wgan":
|
||||
x_t = torch.randn(B, K, model.noise_dim)
|
||||
t = None
|
||||
@@ -471,7 +651,7 @@ def test_stage2_autoregressive_predict_type_shape():
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.zeros(B, 2, dtype=torch.long)
|
||||
stage1_out = torch.randn(B, 9)
|
||||
type_dim = stage2_type_dim({"target": "onehot"}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target="onehot"), emb_dim)
|
||||
history_feat, has_prev, remaining_frac, slot_idx = _ar_inputs(B, K, CONT_SLOT_DIM + type_dim)
|
||||
out = model.predict_type(
|
||||
cond_cont,
|
||||
@@ -492,7 +672,7 @@ def test_stage2_autoregressive_predict_type_raises_when_no_type_head(target, gen
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.zeros(B, 2, dtype=torch.long)
|
||||
stage1_out = torch.randn(B, 9)
|
||||
type_dim = stage2_type_dim({"target": target}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target=target), emb_dim)
|
||||
history_feat, has_prev, remaining_frac, slot_idx = _ar_inputs(B, K, CONT_SLOT_DIM + type_dim)
|
||||
with pytest.raises(RuntimeError):
|
||||
model.predict_type(
|
||||
@@ -512,7 +692,7 @@ def test_stage2_autoregressive_gradients_flow_wgan_onehot():
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.zeros(B, 2, dtype=torch.long)
|
||||
stage1_out = torch.randn(B, 9)
|
||||
type_dim = stage2_type_dim({"target": "onehot"}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target="onehot"), emb_dim)
|
||||
history_feat, has_prev, remaining_frac, slot_idx = _ar_inputs(B, K, CONT_SLOT_DIM + type_dim)
|
||||
z = torch.randn(B, K, model.noise_dim)
|
||||
gen_out = model(
|
||||
@@ -537,9 +717,9 @@ def test_stage2_autoregressive_gradients_flow_onehot():
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.zeros(B, 2, dtype=torch.long)
|
||||
stage1_out = torch.randn(B, 9)
|
||||
type_dim = stage2_type_dim({"target": "onehot"}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target="onehot"), emb_dim)
|
||||
history_feat, has_prev, remaining_frac, slot_idx = _ar_inputs(B, K, CONT_SLOT_DIM + type_dim)
|
||||
token_dim = stage2_trunk_sec_dim({"target": "onehot"}, "flow", 1, emb_dim)
|
||||
token_dim = stage2_trunk_sec_dim(gconfig.ParticleTypeConfig(target="onehot"), "flow", 1, emb_dim)
|
||||
x_t = torch.randn(B, K, token_dim)
|
||||
t = torch.rand(B, K)
|
||||
flow_out = model(
|
||||
@@ -579,7 +759,7 @@ def test_stage2_autoregressive_history_step_matches_parallel_history_encoder():
|
||||
B, K, emb_dim = 3, 6, 6
|
||||
model = _build_stage2_ar("physical", "wgan", emb_dim=emb_dim, k_max=K, history="attention")
|
||||
model.eval()
|
||||
type_dim = stage2_type_dim({"target": "physical"}, emb_dim)
|
||||
type_dim = stage2_type_dim(gconfig.ParticleTypeConfig(target="physical"), emb_dim)
|
||||
hist_in_dim = CONT_SLOT_DIM + type_dim
|
||||
own_feat = torch.randn(B, K, hist_in_dim) # token i's own raw feature
|
||||
has_prev_full = (torch.arange(K) >= 1).unsqueeze(0).expand(B, -1)
|
||||
@@ -655,6 +835,81 @@ def test_build_models_share_stages_true_shared_params_are_in_both_stage_paramete
|
||||
assert shared_ids <= {id(p) for p in stage2.parameters()}
|
||||
|
||||
|
||||
def test_condition_encoder_stores_the_exact_particle_and_material_cfg_instances_passed_in():
|
||||
"""gitea #38: ConditionEncoder must not round-trip particle_cfg/
|
||||
material_cfg through a dict — the exact ConditioningAxisConfig instance
|
||||
passed in is what `.particle_cfg`/`.material_cfg` hold afterward."""
|
||||
particle_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
material_cfg = gconfig.ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
enc = ConditionEncoder(pdg_vocab=3, mat_vocab=2, particle_cfg=particle_cfg, material_cfg=material_cfg)
|
||||
assert enc.particle_cfg is particle_cfg
|
||||
assert enc.material_cfg is material_cfg
|
||||
|
||||
|
||||
def test_stagemodel_stores_the_exact_particle_type_cfg_instance_passed_in():
|
||||
"""gitea #38: a StageModel subclass must not round-trip particle_type_cfg
|
||||
through a dict — the exact ParticleTypeConfig instance passed in is what
|
||||
`.particle_type_cfg` holds afterward."""
|
||||
particle_type_cfg = gconfig.ParticleTypeConfig(target="onehot", n_classes=11)
|
||||
sec_dim = stage2_trunk_sec_dim(particle_type_cfg, "flow", 5, 11)
|
||||
model = Stage2OneShot(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
k_max=5,
|
||||
sec_dim=sec_dim,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
)
|
||||
assert model.particle_type_cfg is particle_type_cfg
|
||||
|
||||
|
||||
def test_build_models_particle_type_cfg_and_conditioning_axes_are_dataclasses_not_dicts():
|
||||
"""gitea #38: build_models must pass the parsed ConditioningAxisConfig/
|
||||
ParticleTypeConfig dataclasses themselves down to the model constructors,
|
||||
not re-serialize them to a dict first (the inversion the issue names) —
|
||||
before the fix, .particle_type_cfg was a plain dict (s2_spec.particle_type
|
||||
.to_dict()) and .cond_enc.particle_cfg came from the raw, unparsed
|
||||
conditioning["particle"] dict."""
|
||||
cfg = _minimal_model_config(share_stages=False)
|
||||
cfg["stage2_model"]["particle_type"] = {"target": "onehot", "lambda": 1.0, "n_classes": 11}
|
||||
built = build_models(cfg)
|
||||
stage1, stage2 = built["stage1"], built["stage2"]
|
||||
assert stage1 is not None and stage2 is not None
|
||||
assert isinstance(stage2.particle_type_cfg, gconfig.ParticleTypeConfig)
|
||||
assert isinstance(stage1.cond_enc.particle_cfg, gconfig.ConditioningAxisConfig)
|
||||
assert isinstance(stage1.cond_enc.material_cfg, gconfig.ConditioningAxisConfig)
|
||||
|
||||
|
||||
def test_build_models_particle_type_n_classes_overrides_conditioning_emb_dim():
|
||||
"""gitea #29 end-to-end through build_models: setting
|
||||
stage2_model.particle_type.n_classes independently of
|
||||
conditioning.particle.emb_dim actually resizes the built stage2 model,
|
||||
not just the two lower-level unit tests above."""
|
||||
cfg = _minimal_model_config(share_stages=False) # conditioning.particle.emb_dim = 4
|
||||
cfg["stage2_model"]["particle_type"] = {"target": "onehot", "lambda": 1.0, "n_classes": 11}
|
||||
built = build_models(cfg)
|
||||
assert built["stage2"] is not None
|
||||
assert built["stage2"].type_dim == 11
|
||||
|
||||
|
||||
def test_build_critics_particle_type_n_classes_overrides_conditioning_emb_dim():
|
||||
cfg = _minimal_model_config(share_stages=False) # conditioning.particle.emb_dim = 4
|
||||
cfg["stage2_model"]["generator"] = "wgan"
|
||||
cfg["stage2_model"]["particle_type"] = {"target": "onehot", "lambda": 1.0, "n_classes": 4}
|
||||
default_n_classes_critic = build_critics(cfg)["stage2"]
|
||||
assert default_n_classes_critic is not None
|
||||
|
||||
cfg["stage2_model"]["particle_type"]["n_classes"] = 11
|
||||
wider_critic = build_critics(cfg)["stage2"]
|
||||
assert wider_critic is not None
|
||||
# k_max=3 slots, each CONT_SLOT_DIM + n_classes wide under wgan folding —
|
||||
# widening n_classes alone (emb_dim stays 4) must widen the critic input.
|
||||
assert wider_critic.input_proj.in_features > default_n_classes_critic.input_proj.in_features
|
||||
|
||||
|
||||
# ── build_models/build_critics: DEFAULT_CONFIG fallback drift (issues.md #1) ─
|
||||
|
||||
|
||||
@@ -688,7 +943,40 @@ def _partial_model_config() -> dict:
|
||||
def test_build_models_omitted_decoder_and_particle_type_match_default_config():
|
||||
built = build_models(_partial_model_config())
|
||||
assert isinstance(built["stage2"], Stage2Autoregressive)
|
||||
assert built["stage2"].particle_type_cfg["target"] == "onehot"
|
||||
assert built["stage2"].particle_type_cfg.target == "onehot"
|
||||
|
||||
|
||||
def test_build_models_custom_heads_block_controls_head_shapes():
|
||||
"""gitea #36: stage{1,2}_model.heads flows all the way from config dict
|
||||
through build_models to the actual constructed head shapes."""
|
||||
cfg = _partial_model_config()
|
||||
cfg["stage1_model"] = {
|
||||
"active": True,
|
||||
"hidden_dim": 40,
|
||||
"n_res_blocks": 1,
|
||||
"heads": {"n_sec": {"hidden_ratio": 0.25, "depth": 1}},
|
||||
}
|
||||
cfg["stage2_model"]["decoder"] = "one_shot"
|
||||
cfg["stage2_model"]["generator"] = "flow" # wgan folds the type slice; no separate type_head
|
||||
cfg["stage2_model"]["n_sec"] = {"owner": "stage1"}
|
||||
cfg["stage2_model"]["heads"] = {
|
||||
"n_sec": {"hidden_ratio": 0.25, "depth": 1},
|
||||
"type": {"hidden_ratio": 0.75, "depth": 2},
|
||||
}
|
||||
|
||||
built = build_models(cfg)
|
||||
stage1, stage2 = built["stage1"], built["stage2"]
|
||||
assert stage1 is not None
|
||||
assert stage2 is not None
|
||||
|
||||
assert stage1.n_sec_head is not None
|
||||
assert len(stage1.n_sec_head) == 1 # owner=stage1, so stage1 builds it
|
||||
assert stage2.n_sec_head is None # owner=stage1, so stage2 doesn't
|
||||
|
||||
assert isinstance(stage2, Stage2OneShot)
|
||||
assert stage2.type_head is not None
|
||||
assert len(stage2.type_head) == 3
|
||||
assert stage2.type_head[0].out_features == 6 # round(8 * 0.75)
|
||||
|
||||
|
||||
def test_build_critics_omitted_particle_type_matches_default_config():
|
||||
@@ -708,3 +996,204 @@ def test_build_critics_omitted_particle_type_matches_default_config():
|
||||
# this also confirms the critic was actually built in onehot mode by
|
||||
# default, not silently falling back to physical.
|
||||
assert onehot_in_dim != physical_in_dim
|
||||
|
||||
|
||||
# ── build_critics: critic_hidden_dim/critic_n_res_blocks honoured (gitea #28) ─
|
||||
|
||||
|
||||
def test_build_critics_stage1_critic_hidden_dim_and_n_res_blocks_override_generator_size():
|
||||
cfg = _minimal_model_config(share_stages=False)
|
||||
cfg["stage1_model"]["generator"] = "wgan"
|
||||
cfg["stage1_model"]["hidden_dim"] = 8
|
||||
cfg["stage1_model"]["n_res_blocks"] = 1
|
||||
|
||||
inherited = build_critics(cfg)["stage1"]
|
||||
assert inherited is not None
|
||||
assert inherited.input_proj.out_features == 8
|
||||
assert len(inherited.blocks) == 1
|
||||
|
||||
cfg["stage1_model"]["wgan"]["critic_hidden_dim"] = 16
|
||||
cfg["stage1_model"]["wgan"]["critic_n_res_blocks"] = 3
|
||||
overridden = build_critics(cfg)["stage1"]
|
||||
assert overridden is not None
|
||||
assert overridden.input_proj.out_features == 16
|
||||
assert len(overridden.blocks) == 3
|
||||
|
||||
|
||||
def test_build_critics_stage2_critic_hidden_dim_and_n_res_blocks_override_generator_size():
|
||||
cfg = _minimal_model_config(share_stages=False)
|
||||
cfg["stage2_model"]["generator"] = "wgan"
|
||||
cfg["stage2_model"]["hidden_dim"] = 8
|
||||
cfg["stage2_model"]["n_res_blocks"] = 1
|
||||
|
||||
inherited = build_critics(cfg)["stage2"]
|
||||
assert inherited is not None
|
||||
assert inherited.input_proj.out_features == 8
|
||||
assert len(inherited.blocks) == 1
|
||||
|
||||
cfg["stage2_model"]["wgan"]["critic_hidden_dim"] = 16
|
||||
cfg["stage2_model"]["wgan"]["critic_n_res_blocks"] = 3
|
||||
overridden = build_critics(cfg)["stage2"]
|
||||
assert overridden is not None
|
||||
assert overridden.input_proj.out_features == 16
|
||||
assert len(overridden.blocks) == 3
|
||||
|
||||
|
||||
# ── StageModel base (gitea #39): Stage1Model/Stage2OneShot/Stage2Autoregressive
|
||||
# scaffolding — construction order, and therefore fresh-init RNG draw order and
|
||||
# state_dict key set, must stay byte-for-byte what it was before the base class
|
||||
# existed. ------------------------------------------------------------------
|
||||
|
||||
_STAGE_HIDDEN_DIM = 32
|
||||
_STAGE_N_BLOCKS = 2
|
||||
_STAGE_COND_OUT_DIM = 16
|
||||
|
||||
|
||||
def _resblock_keys(prefix: str) -> set[str]:
|
||||
return {
|
||||
f"{prefix}.norm.weight",
|
||||
f"{prefix}.norm.bias",
|
||||
f"{prefix}.linear1.weight",
|
||||
f"{prefix}.linear1.bias",
|
||||
f"{prefix}.cond_proj.weight",
|
||||
f"{prefix}.linear2.weight",
|
||||
f"{prefix}.linear2.bias",
|
||||
}
|
||||
|
||||
|
||||
def _trunk_keys(prefix: str = "trunk") -> set[str]:
|
||||
keys = {
|
||||
f"{prefix}.input_proj.weight",
|
||||
f"{prefix}.input_proj.bias",
|
||||
f"{prefix}.out_proj.weight",
|
||||
f"{prefix}.out_proj.bias",
|
||||
}
|
||||
for i in range(_STAGE_N_BLOCKS):
|
||||
keys |= _resblock_keys(f"{prefix}.blocks.{i}")
|
||||
return keys
|
||||
|
||||
|
||||
def _cond_enc_keys() -> set[str]:
|
||||
return {
|
||||
"cond_enc.mlp.0.weight",
|
||||
"cond_enc.mlp.0.bias",
|
||||
"cond_enc.mlp.2.weight",
|
||||
"cond_enc.mlp.2.bias",
|
||||
"cond_enc.particle_mlp.0.weight",
|
||||
"cond_enc.particle_mlp.0.bias",
|
||||
"cond_enc.material_mlp.0.weight",
|
||||
"cond_enc.material_mlp.0.bias",
|
||||
}
|
||||
|
||||
|
||||
def _fuse_keys(name: str) -> set[str]:
|
||||
return {f"{name}.0.weight", f"{name}.0.bias"}
|
||||
|
||||
|
||||
def _head_keys(name: str) -> set[str]:
|
||||
return {f"{name}.0.weight", f"{name}.0.bias", f"{name}.2.weight", f"{name}.2.bias"}
|
||||
|
||||
|
||||
def _expected_stage_keys(*, has_time: bool, extra: set[str]) -> set[str]:
|
||||
keys = _cond_enc_keys() | _trunk_keys() | extra
|
||||
if has_time:
|
||||
keys.add("time_emb.freqs")
|
||||
return keys
|
||||
|
||||
|
||||
@pytest.mark.parametrize("generator", ["flow", "wgan"])
|
||||
def test_stage1_model_state_dict_keys_unchanged_by_stagemodel_refactor(generator):
|
||||
model = Stage1Model(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=_STAGE_HIDDEN_DIM,
|
||||
n_res_blocks=_STAGE_N_BLOCKS,
|
||||
cond_out_dim=_STAGE_COND_OUT_DIM,
|
||||
generator=generator,
|
||||
time_dim=8,
|
||||
noise_dim=8,
|
||||
n_sec_head_k_max=15,
|
||||
)
|
||||
expected = _expected_stage_keys(
|
||||
has_time=build_objective(generator).needs_time,
|
||||
extra=_head_keys("n_sec_head"),
|
||||
)
|
||||
assert set(model.state_dict().keys()) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("generator", ["flow", "wgan"])
|
||||
def test_stage2_oneshot_state_dict_keys_unchanged_by_stagemodel_refactor(generator):
|
||||
model = Stage2OneShot(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=_STAGE_HIDDEN_DIM,
|
||||
n_res_blocks=_STAGE_N_BLOCKS,
|
||||
cond_out_dim=_STAGE_COND_OUT_DIM,
|
||||
generator=generator,
|
||||
time_dim=8,
|
||||
noise_dim=8,
|
||||
k_max=15,
|
||||
)
|
||||
extra = _head_keys("n_sec_head") | {"context_adapter.proj.weight", "context_adapter.proj.bias"} | _fuse_keys("fuse")
|
||||
expected = _expected_stage_keys(has_time=build_objective(generator).needs_time, extra=extra)
|
||||
assert set(model.state_dict().keys()) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("generator", ["flow", "wgan"])
|
||||
def test_stage2_autoregressive_state_dict_keys_unchanged_by_stagemodel_refactor(generator):
|
||||
model = Stage2Autoregressive(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=_STAGE_HIDDEN_DIM,
|
||||
n_res_blocks=_STAGE_N_BLOCKS,
|
||||
cond_out_dim=_STAGE_COND_OUT_DIM,
|
||||
generator=generator,
|
||||
time_dim=8,
|
||||
noise_dim=8,
|
||||
k_max=15,
|
||||
)
|
||||
extra = (
|
||||
_head_keys("n_sec_head")
|
||||
| {"context_adapter.proj.weight", "context_adapter.proj.bias"}
|
||||
| _fuse_keys("base_fuse")
|
||||
| _fuse_keys("token_fuse")
|
||||
| {"history_encoder.start", "history_encoder.mlp.0.weight", "history_encoder.mlp.0.bias"}
|
||||
)
|
||||
expected = _expected_stage_keys(has_time=build_objective(generator).needs_time, extra=extra)
|
||||
assert set(model.state_dict().keys()) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cls", [Stage1Model, Stage2OneShot, Stage2Autoregressive])
|
||||
def test_stage_classes_are_stagemodel_subclasses(cls):
|
||||
assert issubclass(cls, StageModel)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cls", [Stage1Model, Stage2OneShot, Stage2Autoregressive])
|
||||
@pytest.mark.parametrize("generator", ["flow", "ddpm", "wgan"])
|
||||
def test_stagemodel_time_emb_matches_objective_needs_time(cls, generator):
|
||||
kwargs = dict(
|
||||
pdg_vocab=5,
|
||||
mat_vocab=3,
|
||||
particle_cfg=PARTICLE_CFG,
|
||||
material_cfg=MATERIAL_CFG,
|
||||
hidden_dim=_STAGE_HIDDEN_DIM,
|
||||
n_res_blocks=_STAGE_N_BLOCKS,
|
||||
cond_out_dim=_STAGE_COND_OUT_DIM,
|
||||
generator=generator,
|
||||
time_dim=8,
|
||||
noise_dim=8,
|
||||
)
|
||||
if cls is Stage1Model:
|
||||
kwargs["n_sec_head_k_max"] = 15
|
||||
else:
|
||||
kwargs["k_max"] = 15
|
||||
model = cls(**kwargs)
|
||||
assert model.generator_kind == generator
|
||||
assert model.noise_dim == 8
|
||||
assert (model.time_emb is not None) == build_objective(generator).needs_time
|
||||
|
||||
@@ -0,0 +1,251 @@
|
||||
"""Tests for `giant/model/objectives.py` — the generator/objective registry
|
||||
(gitea #32) that replaced bare `generator in ("flow", "ddpm", "wgan")`
|
||||
string checks scattered across models.py/sample.py/builders.py/
|
||||
stage2_inputs.py/trainers.py."""
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import COND_DIM, CONT_SLOT_DIM, PARTICLE_PHYS_DIM, SEC_SLOT_DIM, X_DIM
|
||||
from giant.model.network import (
|
||||
OBJECTIVE_REGISTRY,
|
||||
DdpmObjective,
|
||||
FlowObjective,
|
||||
Stage1Model,
|
||||
Stage2Autoregressive,
|
||||
Stage2OneShot,
|
||||
WganObjective,
|
||||
build_objective,
|
||||
)
|
||||
from giant.model.schedule import CosineSchedule, flow_matching_loss, flow_matching_loss_secondary
|
||||
|
||||
_PHYS_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
|
||||
|
||||
def _cond(B: int, pdg: int = 3, mat: int = 2) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
cond_cat = torch.stack([torch.randint(0, pdg, (B,)), torch.randint(0, mat, (B,))], dim=1)
|
||||
return cond_cont, cond_cat
|
||||
|
||||
|
||||
# ── registry ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_registry_has_exactly_the_three_known_objectives():
|
||||
assert set(OBJECTIVE_REGISTRY) == {"flow", "ddpm", "wgan"}
|
||||
|
||||
|
||||
def test_build_objective_returns_correct_concrete_type():
|
||||
assert isinstance(build_objective("flow"), FlowObjective)
|
||||
assert isinstance(build_objective("ddpm"), DdpmObjective)
|
||||
assert isinstance(build_objective("wgan"), WganObjective)
|
||||
|
||||
|
||||
def test_build_objective_unknown_name_raises():
|
||||
with pytest.raises(ValueError, match="unknown generator/objective"):
|
||||
build_objective("bogus")
|
||||
|
||||
|
||||
def test_build_objective_filters_kwargs_by_signature():
|
||||
# FlowObjective takes no constructor args — n_steps (a DdpmObjective-only
|
||||
# kwarg) must be silently dropped, not raise a TypeError.
|
||||
build_objective("flow", n_steps=500)
|
||||
ddpm = build_objective("ddpm", n_steps=250)
|
||||
assert isinstance(ddpm, DdpmObjective)
|
||||
assert ddpm.n_steps == 250
|
||||
|
||||
|
||||
# ── flags ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_flow_objective_flags():
|
||||
obj = build_objective("flow")
|
||||
assert obj.needs_time is True
|
||||
assert obj.is_adversarial is False
|
||||
assert obj.folds_type_slice is False
|
||||
assert obj.supports_stage2_decoder is True
|
||||
|
||||
|
||||
def test_ddpm_objective_flags():
|
||||
obj = build_objective("ddpm")
|
||||
assert obj.needs_time is True
|
||||
assert obj.is_adversarial is False
|
||||
assert obj.folds_type_slice is False
|
||||
assert obj.supports_stage2_decoder is False
|
||||
|
||||
|
||||
def test_wgan_objective_flags():
|
||||
obj = build_objective("wgan")
|
||||
assert obj.needs_time is False
|
||||
assert obj.is_adversarial is True
|
||||
assert obj.folds_type_slice is True
|
||||
assert obj.supports_stage2_decoder is True
|
||||
|
||||
|
||||
# ── trunk_in_dim ─────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_trunk_in_dim_flow_and_ddpm_pass_through_out_dim():
|
||||
assert build_objective("flow").trunk_in_dim(out_dim=9, noise_dim=8) == 9
|
||||
assert build_objective("ddpm").trunk_in_dim(out_dim=9, noise_dim=8) == 9
|
||||
|
||||
|
||||
def test_trunk_in_dim_wgan_uses_noise_dim():
|
||||
assert build_objective("wgan").trunk_in_dim(out_dim=9, noise_dim=8) == 8
|
||||
|
||||
|
||||
# ── ddpm schedule ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_ddpm_build_schedule_has_requested_length():
|
||||
schedule = build_objective("ddpm").build_schedule(n_steps=17, device=torch.device("cpu"))
|
||||
assert isinstance(schedule, CosineSchedule)
|
||||
assert schedule.T == 17
|
||||
|
||||
|
||||
def test_flow_and_wgan_build_schedule_is_none():
|
||||
assert build_objective("flow").build_schedule(100, torch.device("cpu")) is None
|
||||
assert build_objective("wgan").build_schedule(100, torch.device("cpu")) is None
|
||||
|
||||
|
||||
# ── stage1_loss parity ──────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_flow_objective_stage1_loss_matches_direct_call():
|
||||
torch.manual_seed(0)
|
||||
model = Stage1Model(
|
||||
pdg_vocab=3, mat_vocab=2, particle_cfg=_PHYS_CFG, material_cfg=_PHYS_CFG, hidden_dim=16, n_res_blocks=1
|
||||
)
|
||||
cond_cont, cond_cat = _cond(4)
|
||||
x1 = torch.randn(4, X_DIM)
|
||||
|
||||
torch.manual_seed(1)
|
||||
expected = flow_matching_loss(model, x1, cond_cont, cond_cat)
|
||||
torch.manual_seed(1)
|
||||
actual = build_objective("flow").stage1_loss(model, x1, cond_cont, cond_cat)
|
||||
assert torch.allclose(actual, expected)
|
||||
|
||||
|
||||
def test_ddpm_objective_stage1_loss_matches_direct_call():
|
||||
torch.manual_seed(0)
|
||||
model = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=_PHYS_CFG,
|
||||
material_cfg=_PHYS_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
generator="ddpm",
|
||||
)
|
||||
cond_cont, cond_cat = _cond(4)
|
||||
x1 = torch.randn(4, X_DIM)
|
||||
objective = build_objective("ddpm", n_steps=50)
|
||||
schedule = objective.build_schedule(50, torch.device("cpu"))
|
||||
assert isinstance(schedule, CosineSchedule)
|
||||
|
||||
torch.manual_seed(1)
|
||||
expected = schedule.loss(model, x1, cond_cont, cond_cat)
|
||||
torch.manual_seed(1)
|
||||
actual = objective.stage1_loss(model, x1, cond_cont, cond_cat, schedule=schedule)
|
||||
assert torch.allclose(actual, expected)
|
||||
|
||||
|
||||
def test_ddpm_objective_stage1_loss_requires_a_schedule():
|
||||
model = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=_PHYS_CFG,
|
||||
material_cfg=_PHYS_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
generator="ddpm",
|
||||
)
|
||||
cond_cont, cond_cat = _cond(4)
|
||||
with pytest.raises(AssertionError):
|
||||
build_objective("ddpm").stage1_loss(model, torch.randn(4, X_DIM), cond_cont, cond_cat, schedule=None)
|
||||
|
||||
|
||||
# ── stage2_loss dispatch ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_flow_objective_stage2_loss_one_shot_matches_direct_call():
|
||||
torch.manual_seed(0)
|
||||
B, k_max = 4, 5
|
||||
sec_dim = k_max * SEC_SLOT_DIM
|
||||
model = Stage2OneShot(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=_PHYS_CFG,
|
||||
material_cfg=_PHYS_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
generator="flow",
|
||||
sec_dim=sec_dim,
|
||||
k_max=k_max,
|
||||
)
|
||||
cond_cont, cond_cat = _cond(B)
|
||||
stage1_ctx = torch.randn(B, X_DIM)
|
||||
x1_s2 = torch.randn(B, sec_dim)
|
||||
sec_mask = torch.ones(B, k_max, dtype=torch.bool)
|
||||
|
||||
torch.manual_seed(1)
|
||||
expected = flow_matching_loss_secondary(model, x1_s2, cond_cont, cond_cat, stage1_ctx, sec_mask, type_dim=None)
|
||||
torch.manual_seed(1)
|
||||
actual = build_objective("flow").stage2_loss(
|
||||
model, x1_s2, cond_cont, cond_cat, stage1_ctx, sec_mask, type_dim=None, ar_inputs=None
|
||||
)
|
||||
assert torch.allclose(actual, expected)
|
||||
|
||||
|
||||
def test_flow_objective_stage2_loss_dispatches_to_ar_when_ar_inputs_given():
|
||||
torch.manual_seed(0)
|
||||
B, k_max = 4, 5
|
||||
model = Stage2Autoregressive(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=_PHYS_CFG,
|
||||
material_cfg=_PHYS_CFG,
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
generator="flow",
|
||||
k_max=k_max,
|
||||
)
|
||||
cond_cont, cond_cat = _cond(B)
|
||||
stage1_ctx = torch.randn(B, X_DIM)
|
||||
token_dim = CONT_SLOT_DIM + PARTICLE_PHYS_DIM
|
||||
x1_s2 = torch.randn(B, k_max, token_dim)
|
||||
sec_mask = torch.ones(B, k_max, dtype=torch.bool)
|
||||
ar_inputs = {
|
||||
"history_feat": torch.randn(B, k_max, token_dim),
|
||||
"has_prev": torch.ones(B, k_max, dtype=torch.bool),
|
||||
"remaining_frac": torch.rand(B, k_max),
|
||||
"slot_idx": torch.linspace(0, 1, k_max).unsqueeze(0).expand(B, -1),
|
||||
}
|
||||
|
||||
loss = build_objective("flow").stage2_loss(
|
||||
model, x1_s2, cond_cont, cond_cat, stage1_ctx, sec_mask, type_dim=None, ar_inputs=ar_inputs
|
||||
)
|
||||
assert loss.dim() == 0
|
||||
assert torch.isfinite(loss)
|
||||
|
||||
|
||||
def test_ddpm_objective_stage2_loss_not_implemented():
|
||||
dummy_model = torch.nn.Module()
|
||||
dummy = torch.zeros(1)
|
||||
with pytest.raises(NotImplementedError):
|
||||
build_objective("ddpm").stage2_loss(
|
||||
dummy_model, dummy, dummy, dummy, dummy, torch.ones(1, 1, dtype=torch.bool), type_dim=None
|
||||
)
|
||||
|
||||
|
||||
def test_wgan_objective_has_no_loss_methods():
|
||||
dummy_model = torch.nn.Module()
|
||||
dummy = torch.zeros(1)
|
||||
objective = build_objective("wgan")
|
||||
with pytest.raises(NotImplementedError):
|
||||
objective.stage1_loss(dummy_model, dummy, dummy, dummy)
|
||||
with pytest.raises(NotImplementedError):
|
||||
objective.stage2_loss(
|
||||
dummy_model, dummy, dummy, dummy, dummy, torch.ones(1, 1, dtype=torch.bool), type_dim=None
|
||||
)
|
||||
@@ -4,6 +4,7 @@ import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import (
|
||||
COND_DIM,
|
||||
CONT_SLOT_DIM,
|
||||
@@ -23,9 +24,9 @@ from giant.sample import sample_secondaries
|
||||
# ── helpers ──────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _particle_material_cfg(conditioning: str) -> tuple[dict, dict]:
|
||||
cfg = {"type": conditioning, "emb_dim": 16, "n_layers": 1}
|
||||
return dict(cfg), dict(cfg)
|
||||
def _particle_material_cfg(conditioning: str) -> tuple[ConditioningAxisConfig, ConditioningAxisConfig]:
|
||||
cfg = ConditioningAxisConfig(type=conditioning, emb_dim=16, n_layers=1)
|
||||
return cfg, cfg
|
||||
|
||||
|
||||
def _stage1(pdg=3, mat=2, conditioning="embedding"):
|
||||
|
||||
+39
-7
@@ -152,28 +152,60 @@ def test_run_train_job_second_run_hits_cache(tmp_path, data, monkeypatch):
|
||||
assert "normalizer: cache hit" in joined
|
||||
|
||||
|
||||
def test_run_train_job_builds_caches_and_persists_pdg_topn_map(tmp_path, data):
|
||||
def test_run_train_job_builds_caches_and_persists_sec_type_topn_map(tmp_path, data):
|
||||
"""DEFAULT_CONFIG's stage2_model.particle_type.target defaults to
|
||||
"onehot" — a plain _tiny_cfg() run must
|
||||
build the shared pdg top-N map, cache it in the setup-cache sidecar, and
|
||||
persist it into the checkpoint, with no extra config needed."""
|
||||
"onehot" while conditioning.particle.type stays "physical" — a plain
|
||||
_tiny_cfg() run must build the secondary-type-only pdg top-N map (gitea
|
||||
#29: no longer shared with any conditioning-side onehot map), cache it in
|
||||
the setup-cache sidecar, and persist it into the checkpoint's
|
||||
sec_type_topn_map key, with no extra config needed. pdg_topn_map
|
||||
(conditioning-only) stays unbuilt since conditioning.particle.type is
|
||||
"physical" here."""
|
||||
echo1 = _run(data, tmp_path / "out1")
|
||||
assert any("building pdg top-N map" in m for m in echo1)
|
||||
|
||||
loaded = setup_cache.load(data, [data])
|
||||
assert loaded is not None
|
||||
key = setup_cache.topn_key("pdg", 4) # conditioning.particle.emb_dim = 4
|
||||
# stage2_model.particle_type.n_classes = 0 -> conditioning.particle.emb_dim = 4
|
||||
key = setup_cache.topn_key("pdg", 4)
|
||||
assert key in loaded.topn_maps
|
||||
assert set(loaded.topn_maps[key].class_map.keys()) >= {11, 22}
|
||||
|
||||
ckpt = torch.load(tmp_path / "out1" / "last.pt", weights_only=False)
|
||||
assert "pdg_topn_map" in ckpt
|
||||
assert set(ckpt["pdg_topn_map"]["class_map"].keys()) >= {"11", "22"}
|
||||
assert ckpt.get("pdg_topn_map") is None
|
||||
assert "sec_type_topn_map" in ckpt
|
||||
assert set(ckpt["sec_type_topn_map"]["class_map"].keys()) >= {"11", "22"}
|
||||
|
||||
echo2 = _run(data, tmp_path / "out2")
|
||||
assert any("pdg top-N map: cache hit" in m for m in echo2)
|
||||
|
||||
|
||||
def test_run_train_job_independent_cond_and_sec_type_topn_maps(tmp_path, data):
|
||||
"""conditioning.particle.type="onehot" and
|
||||
stage2_model.particle_type.target="onehot" with different class counts
|
||||
(gitea #29's fix: stage2_model.particle_type.n_classes decouples the two)
|
||||
build two distinct top-N maps, cached under their own (axis, n_classes)
|
||||
key and persisted under two distinct checkpoint keys — no longer forced
|
||||
to share conditioning.particle.emb_dim."""
|
||||
cfg = _tiny_cfg()
|
||||
cfg["conditioning"]["particle"]["type"] = "onehot" # emb_dim = 4, from _tiny_cfg
|
||||
cfg["stage2_model"]["particle_type"]["n_classes"] = 3
|
||||
echo = _run(data, tmp_path / "out", cfg=cfg)
|
||||
assert any("mapped to 4 classes" in m for m in echo)
|
||||
assert any("mapped to 3 classes" in m for m in echo)
|
||||
|
||||
loaded = setup_cache.load(data, [data])
|
||||
assert loaded is not None
|
||||
cond_key = setup_cache.topn_key("pdg", 4)
|
||||
type_key = setup_cache.topn_key("pdg", 3)
|
||||
assert cond_key in loaded.topn_maps
|
||||
assert type_key in loaded.topn_maps
|
||||
|
||||
ckpt = torch.load(tmp_path / "out" / "last.pt", weights_only=False)
|
||||
assert ckpt.get("pdg_topn_map") is not None
|
||||
assert ckpt.get("sec_type_topn_map") is not None
|
||||
|
||||
|
||||
def test_run_train_job_builds_caches_and_persists_material_topn_map(tmp_path, data):
|
||||
"""conditioning.material.type="onehot" is an independent axis from the
|
||||
pdg one above, with its own build/cache-hit branch in run_setup_stage —
|
||||
|
||||
+94
-13
@@ -8,6 +8,7 @@ import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig, ParticleTypeConfig
|
||||
from giant.constants import TERM_ESCAPED, TERM_MAX_STEPS, TERM_UNKNOWN_PDG, K_MAX
|
||||
from giant.data.loader import TopNMap
|
||||
from giant.data.transforms import Normalizer
|
||||
@@ -27,8 +28,8 @@ MAT_MAP = {"G4_AIR": 0, "G4_PbWO4": 1}
|
||||
|
||||
|
||||
def _models(conditioning="embedding"):
|
||||
particle_cfg = {"type": conditioning, "emb_dim": 16, "n_layers": 1}
|
||||
material_cfg = {"type": conditioning, "emb_dim": 16, "n_layers": 1}
|
||||
particle_cfg = ConditioningAxisConfig(type=conditioning, emb_dim=16, n_layers=1)
|
||||
material_cfg = ConditioningAxisConfig(type=conditioning, emb_dim=16, n_layers=1)
|
||||
s1 = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
@@ -366,8 +367,8 @@ def _models_v3(
|
||||
emb_dim=4,
|
||||
stage2_has_n_sec_head=True,
|
||||
):
|
||||
particle_cfg = {"type": conditioning, "emb_dim": emb_dim, "n_layers": 1}
|
||||
material_cfg = {"type": conditioning, "emb_dim": emb_dim, "n_layers": 1}
|
||||
particle_cfg = ConditioningAxisConfig(type=conditioning, emb_dim=emb_dim, n_layers=1)
|
||||
material_cfg = ConditioningAxisConfig(type=conditioning, emb_dim=emb_dim, n_layers=1)
|
||||
# A fresh v0.3.0 Stage1Model — no n_sec_head_k_max, unlike _models() above
|
||||
# (n_sec ownership moves to stage 2 by default).
|
||||
s1 = Stage1Model(
|
||||
@@ -380,7 +381,7 @@ def _models_v3(
|
||||
generator=generator1,
|
||||
noise_dim=8,
|
||||
)
|
||||
particle_type_cfg = {"target": target}
|
||||
particle_type_cfg = ParticleTypeConfig(target=target)
|
||||
# Explicit kwargs rather than a shared **common dict: a dict() call whose
|
||||
# values have heterogeneous types (str/int/dict/bool) widens under static
|
||||
# analysis to dict[str, <big union>], which then makes every constructor
|
||||
@@ -430,7 +431,7 @@ def _run_v3(
|
||||
max_tracks_per_event=100,
|
||||
seeds=None,
|
||||
conditioning="physical",
|
||||
pdg_topn_map=None,
|
||||
sec_type_topn_map=None,
|
||||
other_policy="sample",
|
||||
seed=0,
|
||||
stage1_ddpm_steps=1000,
|
||||
@@ -457,7 +458,7 @@ def _run_v3(
|
||||
escape_threshold=escape_threshold,
|
||||
particle_conditioning=conditioning,
|
||||
material_conditioning=conditioning,
|
||||
pdg_topn_map=pdg_topn_map,
|
||||
sec_type_topn_map=sec_type_topn_map,
|
||||
other_policy=other_policy,
|
||||
seed=seed,
|
||||
stage1_ddpm_steps=stage1_ddpm_steps,
|
||||
@@ -532,7 +533,7 @@ def test_rollout_onehot_target_end_to_end(fake_material_props, decoder):
|
||||
(giant.particles.particle_phys_array) become the secondary's identity —
|
||||
unlike "physical", not just a reporting label."""
|
||||
s1, s2 = _models_v3(decoder=decoder, target="onehot", emb_dim=3)
|
||||
rec = _run_v3(s1, s2, pdg_topn_map=PDG_TOPN_MAP, other_policy="modal")
|
||||
rec = _run_v3(s1, s2, sec_type_topn_map=PDG_TOPN_MAP, other_policy="modal")
|
||||
assert len(rec["event_id"]) > 0
|
||||
# Every spawned secondary's nominal pdg must be one decode_topn_class can
|
||||
# actually produce (the topn map's known classes + its "other" members).
|
||||
@@ -543,8 +544,8 @@ def test_rollout_onehot_target_end_to_end(fake_material_props, decoder):
|
||||
|
||||
def test_rollout_onehot_target_missing_topn_map_raises(fake_material_props):
|
||||
s1, s2 = _models_v3(target="onehot", emb_dim=3)
|
||||
with pytest.raises(RuntimeError, match="pdg_topn_map"):
|
||||
_run_v3(s1, s2, pdg_topn_map=None)
|
||||
with pytest.raises(RuntimeError, match="sec_type_topn_map"):
|
||||
_run_v3(s1, s2, sec_type_topn_map=None)
|
||||
|
||||
|
||||
# --- conditioning.{particle,material}.type = "onehot" — a separate axis from
|
||||
@@ -557,8 +558,8 @@ COND_MAT_TOPN_MAP = TopNMap(class_map={"G4_AIR": 0, "G4_PbWO4": 1}, other_member
|
||||
|
||||
|
||||
def _onehot_conditioning_models():
|
||||
particle_cfg = {"type": "onehot", "emb_dim": len(PDG_MAP), "n_layers": 1}
|
||||
material_cfg = {"type": "onehot", "emb_dim": len(MAT_MAP), "n_layers": 1}
|
||||
particle_cfg = ConditioningAxisConfig(type="onehot", emb_dim=len(PDG_MAP), n_layers=1)
|
||||
material_cfg = ConditioningAxisConfig(type="onehot", emb_dim=len(MAT_MAP), n_layers=1)
|
||||
s1 = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
@@ -574,7 +575,7 @@ def _onehot_conditioning_models():
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=32,
|
||||
n_res_blocks=2,
|
||||
sec_dim=stage2_trunk_sec_dim({"target": "physical"}, "flow", K_MAX, 3),
|
||||
sec_dim=stage2_trunk_sec_dim(ParticleTypeConfig(target="physical"), "flow", K_MAX, 3),
|
||||
generator="flow",
|
||||
time_dim=16,
|
||||
)
|
||||
@@ -627,6 +628,86 @@ def test_rollout_conditioning_onehot_material_missing_topn_map_raises(
|
||||
_run_onehot_conditioning(mat_topn_map=None)
|
||||
|
||||
|
||||
SEC_TYPE_TOPN_MAP_DIFFERENT_N = TopNMap(class_map={22: 0, 11: 1, -11: 2, 13: 3}, other_members={2112: 3, 2212: 1})
|
||||
|
||||
|
||||
def _run_conditioning_and_type_onehot_different_n_classes():
|
||||
"""Both conditioning.particle.type="onehot" and
|
||||
stage2_model.particle_type.target="onehot" active at once, with
|
||||
stage2_model.particle_type.n_classes deliberately different from
|
||||
conditioning.particle.emb_dim (gitea #29)."""
|
||||
cond_emb_dim = len(PDG_MAP) # 3
|
||||
type_n_classes = 5 # deliberately different from cond_emb_dim
|
||||
particle_cfg = ConditioningAxisConfig(type="onehot", emb_dim=cond_emb_dim, n_layers=1)
|
||||
material_cfg = ConditioningAxisConfig(type="onehot", emb_dim=len(MAT_MAP), n_layers=1)
|
||||
particle_type_cfg = ParticleTypeConfig(target="onehot", n_classes=type_n_classes)
|
||||
|
||||
s1 = Stage1Model(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=32,
|
||||
n_res_blocks=2,
|
||||
).eval()
|
||||
sec_dim = stage2_trunk_sec_dim(particle_type_cfg, "flow", K_MAX, type_n_classes)
|
||||
s2 = Stage2OneShot(
|
||||
pdg_vocab=3,
|
||||
mat_vocab=2,
|
||||
particle_cfg=particle_cfg,
|
||||
material_cfg=material_cfg,
|
||||
hidden_dim=32,
|
||||
n_res_blocks=2,
|
||||
sec_dim=sec_dim,
|
||||
generator="flow",
|
||||
time_dim=16,
|
||||
k_max=K_MAX,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
).eval()
|
||||
# Sanity: the model's own type_dim followed n_classes, not cond_emb_dim.
|
||||
assert s2.type_dim == type_n_classes
|
||||
|
||||
cond, tgt, sec_phys = _norms()
|
||||
return rollout(
|
||||
s1,
|
||||
s2,
|
||||
_oracle(),
|
||||
_seeds(),
|
||||
cond,
|
||||
tgt,
|
||||
sec_phys,
|
||||
PDG_MAP,
|
||||
MAT_MAP,
|
||||
energy_cutoff=1.0,
|
||||
max_steps=15,
|
||||
steps=3,
|
||||
batch_size=128,
|
||||
max_tracks_per_event=100,
|
||||
escape_threshold=1e9,
|
||||
particle_conditioning="onehot",
|
||||
material_conditioning="onehot",
|
||||
pdg_topn_map=COND_PDG_TOPN_MAP,
|
||||
mat_topn_map=COND_MAT_TOPN_MAP,
|
||||
sec_type_topn_map=SEC_TYPE_TOPN_MAP_DIFFERENT_N,
|
||||
other_policy="modal",
|
||||
)
|
||||
|
||||
|
||||
def test_rollout_conditioning_and_type_onehot_with_different_n_classes(fake_material_props):
|
||||
"""gitea #29 end-to-end: conditioning.particle.type="onehot" and
|
||||
stage2_model.particle_type.target="onehot" now use independently sized
|
||||
top-N maps (stage2_model.particle_type.n_classes != conditioning.particle
|
||||
.emb_dim), and rollout must decode secondaries using the type-side map,
|
||||
not silently reuse the conditioning-side one (the pre-#29 bug)."""
|
||||
rec = _run_conditioning_and_type_onehot_different_n_classes()
|
||||
assert len(rec["event_id"]) > 0
|
||||
possible = set(SEC_TYPE_TOPN_MAP_DIFFERENT_N.class_map.keys()) | set(
|
||||
SEC_TYPE_TOPN_MAP_DIFFERENT_N.other_members.keys()
|
||||
)
|
||||
secondary_pdgs = set(rec["pdg"][rec["generation"] > 0].tolist())
|
||||
assert secondary_pdgs <= possible
|
||||
|
||||
|
||||
@pytest.mark.parametrize("decoder", ["one_shot", "autoregressive"])
|
||||
def test_rollout_embedding_target_end_to_end(decoder):
|
||||
"""particle_type.target="embedding" L1-snaps to the nearest row of the
|
||||
|
||||
+134
-7
@@ -3,24 +3,32 @@
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import COND_DIM, K_MAX, SEC_DIM, X_DIM
|
||||
from giant.model.network import (
|
||||
BLOCK_REGISTRY,
|
||||
TRUNK_REGISTRY,
|
||||
AdaLNResBlock,
|
||||
ComposedRouter,
|
||||
EnergyRouter,
|
||||
MonolithicTrunk,
|
||||
ExpertTrunk,
|
||||
FilmResBlock,
|
||||
PdgRouter,
|
||||
ProcessRouter,
|
||||
ROUTER_REGISTRY,
|
||||
ResBlock,
|
||||
RoutedTrunk,
|
||||
Stage1Model,
|
||||
Stage2OneShot,
|
||||
build_block,
|
||||
build_composed_router,
|
||||
build_expert_body,
|
||||
build_models,
|
||||
build_router,
|
||||
)
|
||||
|
||||
PARTICLE_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
MATERIAL_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
PARTICLE_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
MATERIAL_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
|
||||
|
||||
def _cond(B=8, pdg=3, mat=2):
|
||||
@@ -144,6 +152,68 @@ def test_build_router_unknown_type_raises():
|
||||
raise AssertionError("expected ValueError for unknown router type")
|
||||
|
||||
|
||||
# ── TRUNK_REGISTRY / build_expert_body ──────────────────────────────────────
|
||||
|
||||
|
||||
def test_trunk_registry_has_resmlp():
|
||||
assert "resmlp" in TRUNK_REGISTRY
|
||||
assert TRUNK_REGISTRY["resmlp"] is ExpertTrunk
|
||||
|
||||
|
||||
def test_build_expert_body_unknown_type_raises():
|
||||
try:
|
||||
build_expert_body("nonexistent", in_dim=4, out_dim=4, hidden_dim=8, n_blocks=1, cond_dim=4)
|
||||
except ValueError:
|
||||
return
|
||||
raise AssertionError("expected ValueError for unknown trunk type")
|
||||
|
||||
|
||||
# ── BLOCK_REGISTRY / build_block (gitea #34) ────────────────────────────────
|
||||
|
||||
|
||||
def test_block_registry_has_add_film_adaln():
|
||||
assert BLOCK_REGISTRY["add"] is ResBlock
|
||||
assert BLOCK_REGISTRY["film"] is FilmResBlock
|
||||
assert BLOCK_REGISTRY["adaln"] is AdaLNResBlock
|
||||
|
||||
|
||||
def test_build_block_unknown_type_raises():
|
||||
try:
|
||||
build_block("nonexistent", dim=8, cond_dim=4)
|
||||
except ValueError:
|
||||
return
|
||||
raise AssertionError("expected ValueError for unknown block conditioning type")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("block_type", ["add", "film", "adaln"])
|
||||
def test_block_forward_shape(block_type):
|
||||
block = build_block(block_type, dim=8, cond_dim=4)
|
||||
x = torch.randn(5, 8)
|
||||
cond = torch.randn(5, 4)
|
||||
out = block(x, cond)
|
||||
assert out.shape == (5, 8)
|
||||
|
||||
|
||||
def test_film_res_block_output_invariant_to_cond_at_init():
|
||||
"""Zero-initialized film_proj means gamma=beta=0 at construction, so the
|
||||
output must not depend on which cond is passed in."""
|
||||
block = FilmResBlock(dim=8, cond_dim=4)
|
||||
x = torch.randn(5, 8)
|
||||
cond_a = torch.randn(5, 4)
|
||||
cond_b = torch.randn(5, 4)
|
||||
torch.testing.assert_close(block(x, cond_a), block(x, cond_b))
|
||||
|
||||
|
||||
def test_adaln_res_block_is_identity_at_init():
|
||||
"""Zero-initialized adaln_proj means scale=shift=gate=0 at construction,
|
||||
so the block must be the exact identity function (the 'Zero' in
|
||||
AdaLN-Zero)."""
|
||||
block = AdaLNResBlock(dim=8, cond_dim=4)
|
||||
x = torch.randn(5, 8)
|
||||
cond = torch.randn(5, 4)
|
||||
torch.testing.assert_close(block(x, cond), x)
|
||||
|
||||
|
||||
# ── EnergyRouter learn_width / learn_temperature ────────────────────────────
|
||||
|
||||
|
||||
@@ -993,8 +1063,10 @@ def test_build_models_monolith_when_router_absent():
|
||||
stage1, stage2 = models["stage1"], models["stage2"]
|
||||
assert isinstance(stage1, Stage1Model)
|
||||
assert isinstance(stage2, Stage2OneShot)
|
||||
assert isinstance(stage1.trunk, MonolithicTrunk)
|
||||
assert isinstance(stage2.trunk, MonolithicTrunk)
|
||||
assert not isinstance(stage1.trunk, RoutedTrunk)
|
||||
assert not isinstance(stage2.trunk, RoutedTrunk)
|
||||
assert isinstance(stage1.trunk, ExpertTrunk)
|
||||
assert isinstance(stage2.trunk, ExpertTrunk)
|
||||
|
||||
|
||||
def test_build_models_monolith_when_router_disabled():
|
||||
@@ -1006,8 +1078,10 @@ def test_build_models_monolith_when_router_disabled():
|
||||
models = build_models(cfg)
|
||||
stage1, stage2 = models["stage1"], models["stage2"]
|
||||
assert stage1 is not None and stage2 is not None
|
||||
assert isinstance(stage1.trunk, MonolithicTrunk)
|
||||
assert isinstance(stage2.trunk, MonolithicTrunk)
|
||||
assert not isinstance(stage1.trunk, RoutedTrunk)
|
||||
assert not isinstance(stage2.trunk, RoutedTrunk)
|
||||
assert isinstance(stage1.trunk, ExpertTrunk)
|
||||
assert isinstance(stage2.trunk, ExpertTrunk)
|
||||
|
||||
|
||||
def test_build_models_routed_when_enabled():
|
||||
@@ -1040,6 +1114,59 @@ def test_build_models_routed_when_enabled():
|
||||
assert len(stage2.trunk.experts) == 4
|
||||
|
||||
|
||||
def test_build_models_explicit_resmlp_trunk_type_matches_default():
|
||||
"""stage1_model.trunk.type = 'resmlp' is the default's spelled-out
|
||||
equivalent, not a behaviour change — gitea #33."""
|
||||
default_cfg = _nested_cfg(pdg_vocab=4, mat_vocab=2)
|
||||
explicit_cfg = _nested_cfg(pdg_vocab=4, mat_vocab=2, trunk={"type": "resmlp"})
|
||||
default_stage1 = build_models(default_cfg)["stage1"]
|
||||
explicit_stage1 = build_models(explicit_cfg)["stage1"]
|
||||
assert default_stage1 is not None and explicit_stage1 is not None
|
||||
assert type(default_stage1.trunk) is type(explicit_stage1.trunk) is ExpertTrunk
|
||||
assert default_stage1.trunk.input_proj.weight.shape == explicit_stage1.trunk.input_proj.weight.shape
|
||||
default_params = sum(p.numel() for p in default_stage1.parameters())
|
||||
explicit_params = sum(p.numel() for p in explicit_stage1.parameters())
|
||||
assert default_params == explicit_params
|
||||
|
||||
|
||||
def test_build_models_explicit_add_block_conditioning_matches_default():
|
||||
"""stage1_model.trunk.block_conditioning = 'add' is the default's
|
||||
spelled-out equivalent, not a behaviour change — gitea #34."""
|
||||
default_cfg = _nested_cfg(pdg_vocab=4, mat_vocab=2)
|
||||
explicit_cfg = _nested_cfg(pdg_vocab=4, mat_vocab=2, trunk={"type": "resmlp", "block_conditioning": "add"})
|
||||
default_stage1 = build_models(default_cfg)["stage1"]
|
||||
explicit_stage1 = build_models(explicit_cfg)["stage1"]
|
||||
assert default_stage1 is not None and explicit_stage1 is not None
|
||||
assert type(default_stage1.trunk.blocks[0]) is type(explicit_stage1.trunk.blocks[0]) is ResBlock
|
||||
default_params = sum(p.numel() for p in default_stage1.parameters())
|
||||
explicit_params = sum(p.numel() for p in explicit_stage1.parameters())
|
||||
assert default_params == explicit_params
|
||||
|
||||
|
||||
@pytest.mark.parametrize("block_type,cls", [("film", FilmResBlock), ("adaln", AdaLNResBlock)])
|
||||
def test_build_models_selects_block_conditioning(block_type, cls):
|
||||
cfg = _nested_cfg(pdg_vocab=4, mat_vocab=2, trunk={"type": "resmlp", "block_conditioning": block_type})
|
||||
stage1 = build_models(cfg)["stage1"]
|
||||
assert stage1 is not None
|
||||
assert isinstance(stage1.trunk, ExpertTrunk)
|
||||
assert all(isinstance(b, cls) for b in stage1.trunk.blocks)
|
||||
|
||||
|
||||
def test_build_models_routed_trunk_uses_block_conditioning_for_every_expert():
|
||||
cfg = _nested_cfg(
|
||||
pdg_vocab=4,
|
||||
mat_vocab=2,
|
||||
trunk={"type": "resmlp", "block_conditioning": "film"},
|
||||
stage1_router={"enabled": True, "type": "energy", "n_experts": 3},
|
||||
)
|
||||
stage1 = build_models(cfg)["stage1"]
|
||||
assert stage1 is not None
|
||||
assert isinstance(stage1.trunk, RoutedTrunk)
|
||||
assert len(stage1.trunk.experts) == 3
|
||||
for expert in stage1.trunk.experts:
|
||||
assert all(isinstance(b, FilmResBlock) for b in expert.blocks)
|
||||
|
||||
|
||||
def test_build_models_routed_pair_is_drop_in_for_sample_flow():
|
||||
"""Exercise the exact calling convention giant/sample.py uses."""
|
||||
from giant.sample import sample_flow, sample_secondaries
|
||||
|
||||
@@ -5,6 +5,7 @@ for the one-shot samplers."""
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig, ParticleTypeConfig
|
||||
from giant.constants import COND_DIM, CONT_SLOT_DIM, K_MAX, PARTICLE_PHYS_DIM, X_DIM
|
||||
from giant.model.network import (
|
||||
Stage1Model,
|
||||
@@ -20,12 +21,12 @@ from giant.sample import (
|
||||
sample_wgan,
|
||||
)
|
||||
|
||||
_PHYS_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
_PHYS_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
|
||||
|
||||
def _particle_material_cfg(conditioning: str, emb_dim: int) -> tuple[dict, dict]:
|
||||
cfg = {"type": conditioning, "emb_dim": emb_dim, "n_layers": 1}
|
||||
return dict(cfg), dict(cfg)
|
||||
def _particle_material_cfg(conditioning: str, emb_dim: int) -> tuple[ConditioningAxisConfig, ConditioningAxisConfig]:
|
||||
cfg = ConditioningAxisConfig(type=conditioning, emb_dim=emb_dim, n_layers=1)
|
||||
return cfg, cfg
|
||||
|
||||
|
||||
def _cond(B: int, pdg: int = 3, mat: int = 2) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
@@ -43,7 +44,7 @@ def _conditioning_for(target: str) -> str:
|
||||
|
||||
def _stage2_oneshot(target: str, generator: str, emb_dim: int = 6, pdg: int = 3, mat: int = 2) -> Stage2OneShot:
|
||||
particle_cfg, material_cfg = _particle_material_cfg(_conditioning_for(target), emb_dim)
|
||||
particle_type_cfg = {"target": target}
|
||||
particle_type_cfg = ParticleTypeConfig(target=target)
|
||||
# build_models (giant/model/network.py) computes sec_dim this same way
|
||||
# before constructing Stage2OneShot — its own default (SEC_DIM, the
|
||||
# "physical" width) is only correct for target="physical".
|
||||
@@ -84,7 +85,7 @@ def _stage2_ar(
|
||||
time_dim=16,
|
||||
noise_dim=8,
|
||||
k_max=k_max,
|
||||
particle_type_cfg={"target": target},
|
||||
particle_type_cfg=ParticleTypeConfig(target=target),
|
||||
history=history,
|
||||
attn_n_heads=2,
|
||||
attn_n_layers=1,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import polars as pl
|
||||
|
||||
from scripts import steps_to_parquet
|
||||
from giant.tools import steps_to_parquet
|
||||
|
||||
|
||||
def _frame() -> pl.DataFrame:
|
||||
|
||||
@@ -2,7 +2,7 @@ import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts import steps_to_parquet_parallel
|
||||
from giant.tools import steps_to_parquet_parallel
|
||||
|
||||
run_parallel = steps_to_parquet_parallel.run_parallel
|
||||
resolve_destination = steps_to_parquet_parallel.resolve_destination
|
||||
|
||||
+61
-4
@@ -5,10 +5,12 @@ import csv
|
||||
import math
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from giant.config import ParticleTypeConfig
|
||||
from giant.constants import (
|
||||
COND_DIM,
|
||||
CONT_SLOT_DIM,
|
||||
@@ -17,6 +19,7 @@ from giant.constants import (
|
||||
SEC_SLOT_DIM,
|
||||
X_DIM,
|
||||
)
|
||||
from giant.data.dataset import StepBatch
|
||||
from giant.model.network import build_critics, build_models
|
||||
from giant.training import (
|
||||
FlowDDPMStageTrainer,
|
||||
@@ -156,7 +159,7 @@ def test_type_repr_shapes_and_values(target):
|
||||
cond_enc = torch.nn.Module()
|
||||
if target == "embedding":
|
||||
cond_enc.pdg_emb = torch.nn.Embedding(emb_dim, emb_dim)
|
||||
repr_ = _type_repr(sec_type_idx, sec_cont, {"target": target}, cond_enc, emb_dim)
|
||||
repr_ = _type_repr(sec_type_idx, sec_cont, ParticleTypeConfig(target=target), cond_enc, emb_dim)
|
||||
expected_width = PARTICLE_PHYS_DIM if target == "physical" else emb_dim
|
||||
assert repr_.shape == (B, K, expected_width)
|
||||
if target == "physical":
|
||||
@@ -185,7 +188,7 @@ def test_assemble_stage2_ar_target_matches_assemble_stage2_real_flattened(target
|
||||
cond_enc = torch.nn.Module()
|
||||
if target == "embedding":
|
||||
cond_enc.pdg_emb = torch.nn.Embedding(emb_dim, emb_dim)
|
||||
particle_type_cfg = {"target": target}
|
||||
particle_type_cfg = ParticleTypeConfig(target=target)
|
||||
flat = _assemble_stage2_real(sec_cont, sec_type_idx, particle_type_cfg, generator, cond_enc, emb_dim)
|
||||
unflat = _assemble_stage2_ar_target(sec_cont, sec_type_idx, particle_type_cfg, generator, cond_enc, emb_dim)
|
||||
assert torch.equal(unflat.flatten(1), flat)
|
||||
@@ -196,7 +199,7 @@ def test_assemble_stage2_ar_inputs_shapes_and_history_feat_width():
|
||||
sec_cont = torch.randn(B, K_MAX, SEC_SLOT_DIM)
|
||||
sec_type_idx = torch.randint(0, emb_dim, (B, K_MAX))
|
||||
cond_enc = torch.nn.Module()
|
||||
out = _assemble_stage2_ar_inputs(sec_cont, sec_type_idx, {"target": "physical"}, cond_enc, emb_dim)
|
||||
out = _assemble_stage2_ar_inputs(sec_cont, sec_type_idx, ParticleTypeConfig(target="physical"), cond_enc, emb_dim)
|
||||
assert out["history_feat"].shape == (B, K_MAX, CONT_SLOT_DIM + PARTICLE_PHYS_DIM)
|
||||
assert out["has_prev"].shape == (B, K_MAX)
|
||||
assert out["remaining_frac"].shape == (B, K_MAX)
|
||||
@@ -316,7 +319,7 @@ def _fake_batches(n_batches, batch_size, seed=0):
|
||||
sec_cont = torch.randn(batch_size, K_MAX, SEC_SLOT_DIM, generator=g)
|
||||
proc_idx = torch.zeros(batch_size, dtype=torch.long)
|
||||
sec_type_idx = torch.zeros(batch_size, K_MAX, dtype=torch.long)
|
||||
batches.append((cond_cont, cond_cat, x1, n_sec, sec_cont, proc_idx, sec_type_idx))
|
||||
batches.append(StepBatch(cond_cont, cond_cat, x1, n_sec, sec_cont, proc_idx, sec_type_idx))
|
||||
return batches
|
||||
|
||||
|
||||
@@ -586,6 +589,60 @@ def test_stage_spec_from_config_omitted_decoder_and_particle_type_match_default_
|
||||
assert spec.particle_type.target == "onehot"
|
||||
|
||||
|
||||
def _routed_stage1_trainer(lambda_balance, lambda_proc, lambda_entropy):
|
||||
cfg = _base_cfg()
|
||||
cfg["stage1_model"]["router"] = {
|
||||
"enabled": True,
|
||||
"type": "energy",
|
||||
"n_experts": 3,
|
||||
"temperature": 0.5,
|
||||
"learn_centers": True,
|
||||
"lambda_balance": lambda_balance,
|
||||
"lambda_proc": lambda_proc,
|
||||
"lambda_entropy": lambda_entropy,
|
||||
}
|
||||
model_config = _model_config(cfg)
|
||||
models = build_models(model_config)
|
||||
critics = build_critics(model_config)
|
||||
trainers = build_stage_trainers(cfg, models, critics, torch.device("cpu"), total_train_batches=4)
|
||||
return trainers["stage1"]
|
||||
|
||||
|
||||
def test_router_aux_losses_skipped_when_lambda_zero_but_run_when_positive():
|
||||
"""Gitea #31: FlowDDPMStageTrainer._compute must not call
|
||||
router.balance_loss/classify_loss/entropy_loss when the corresponding
|
||||
lambda is 0 (the default) -- those calls do their own router.gate(...)
|
||||
forward pass that is wasted once the term is masked out of the total
|
||||
loss anyway. Checked both ways: zero lambdas must skip all three calls,
|
||||
positive lambdas must still make them (the guard must not accidentally
|
||||
suppress the real path)."""
|
||||
batch = _fake_batches(1, 4)[0]
|
||||
device = torch.device("cpu")
|
||||
|
||||
trainer_zero = _routed_stage1_trainer(0.0, 0.0, 0.0)
|
||||
router_zero = trainer_zero.router
|
||||
router_zero.balance_loss = MagicMock(wraps=router_zero.balance_loss)
|
||||
router_zero.classify_loss = MagicMock(wraps=router_zero.classify_loss)
|
||||
router_zero.entropy_loss = MagicMock(wraps=router_zero.entropy_loss)
|
||||
stats_zero = trainer_zero.step(batch, device, global_step=1)
|
||||
assert router_zero.balance_loss.call_count == 0
|
||||
assert router_zero.classify_loss.call_count == 0
|
||||
assert router_zero.entropy_loss.call_count == 0
|
||||
assert stats_zero["loss_balance"] == 0.0
|
||||
assert stats_zero["loss_proc"] == 0.0
|
||||
assert stats_zero["loss_entropy"] == 0.0
|
||||
|
||||
trainer_pos = _routed_stage1_trainer(0.1, 0.1, 0.01)
|
||||
router_pos = trainer_pos.router
|
||||
router_pos.balance_loss = MagicMock(wraps=router_pos.balance_loss)
|
||||
router_pos.classify_loss = MagicMock(wraps=router_pos.classify_loss)
|
||||
router_pos.entropy_loss = MagicMock(wraps=router_pos.entropy_loss)
|
||||
trainer_pos.step(batch, device, global_step=1)
|
||||
assert router_pos.balance_loss.call_count == 1
|
||||
assert router_pos.classify_loss.call_count == 1
|
||||
assert router_pos.entropy_loss.call_count == 1
|
||||
|
||||
|
||||
# --- AR trainer wiring (v0.3.0 step 5) --------------------------------------
|
||||
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ import warnings
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from giant.cond_layout import AXIS_TYPES, CondLayout
|
||||
from giant.constants import COND_DIM, COND_DIM_BASE, K_MAX
|
||||
from giant.data.transforms import (
|
||||
build_cond_features,
|
||||
@@ -531,6 +532,129 @@ def test_build_cond_features_rejects_legacy_normalizer_in_physical_mode(
|
||||
)
|
||||
|
||||
|
||||
# ── build_cond_features / build_features share one column layout (gitea #37) ──
|
||||
|
||||
|
||||
@pytest.mark.parametrize("particle_type", AXIS_TYPES)
|
||||
@pytest.mark.parametrize("material_type", AXIS_TYPES)
|
||||
def test_both_builders_agree_column_for_column(particle_type, material_type, fake_material_props):
|
||||
"""The two builders used to lay out cond_cont/cond_cat independently and
|
||||
drift apart silently. They now share `_build_cond_arrays`, so for every
|
||||
mode pair they must produce identical arrays."""
|
||||
data = _minimal_step_data(3)
|
||||
pdg_map, mat_map = {11: 0}, {"PbWO4": 0}
|
||||
pdg_topn = {11: 0} if particle_type == "onehot" else None
|
||||
mat_topn = {"PbWO4": 0} if material_type == "onehot" else None
|
||||
|
||||
cond_cont, cond_cat = build_cond_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
particle_conditioning=particle_type,
|
||||
material_conditioning=material_type,
|
||||
pdg_topn_map=pdg_topn,
|
||||
mat_topn_map=mat_topn,
|
||||
)
|
||||
feats = build_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
particle_conditioning=particle_type,
|
||||
material_conditioning=material_type,
|
||||
pdg_topn_map=pdg_topn,
|
||||
mat_topn_map=mat_topn,
|
||||
)
|
||||
|
||||
layout = CondLayout.from_types(particle_type, material_type)
|
||||
assert cond_cat.shape[1] == layout.cat_dim
|
||||
np.testing.assert_array_equal(feats.cond_cont, cond_cont)
|
||||
np.testing.assert_array_equal(feats.cond_cat, cond_cat)
|
||||
|
||||
|
||||
def test_build_features_physical_mode_tolerates_out_of_vocab_pdg_and_material():
|
||||
"""The permissive vocab lookup added for "physical"/"onehot" mode (see
|
||||
build_cond_features) applies to build_features too — `giant predict` on a
|
||||
file whose pdg/material aren't in the checkpoint's dense vocab must not
|
||||
KeyError when nothing reads those indices."""
|
||||
pdg_map = {11: 0, 22: 1}
|
||||
mat_map = {"G4_AIR": 0}
|
||||
data = _minimal_step_data(2)
|
||||
data["pdg"] = np.full(2, 13, dtype=np.int64) # not in pdg_map
|
||||
data["material"] = np.full(2, "G4_Pb", dtype=object) # not in mat_map
|
||||
|
||||
_, cond_cat, *_ = build_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
particle_conditioning="physical",
|
||||
material_conditioning="physical",
|
||||
)
|
||||
np.testing.assert_array_equal(cond_cat, [[0, 0], [0, 0]]) # dummy indices, no raise
|
||||
|
||||
with pytest.raises(KeyError):
|
||||
build_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
particle_conditioning="embedding",
|
||||
material_conditioning="embedding",
|
||||
)
|
||||
|
||||
|
||||
def test_build_features_pads_legacy_normalizer_in_embedding_mode():
|
||||
"""The legacy-normalizer padding (a pre-physical-conditioning checkpoint's
|
||||
cond normalizer is COND_DIM_BASE wide) applies to build_features too —
|
||||
`giant predict` reaches build_features, not build_cond_features."""
|
||||
data = _minimal_step_data(3)
|
||||
pdg_map, mat_map = {11: 0}, {"PbWO4": 0}
|
||||
legacy_norm = Normalizer()
|
||||
legacy_norm.mean = np.zeros(COND_DIM_BASE, dtype=np.float32)
|
||||
legacy_norm.std = np.ones(COND_DIM_BASE, dtype=np.float32)
|
||||
|
||||
cond_cont, *_ = build_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
cond_normalizer=legacy_norm,
|
||||
particle_conditioning="embedding",
|
||||
material_conditioning="embedding",
|
||||
)
|
||||
|
||||
assert cond_cont.shape[-1] == COND_DIM
|
||||
np.testing.assert_allclose(cond_cont[:, COND_DIM_BASE:], 0.0)
|
||||
|
||||
|
||||
def test_build_features_rejects_legacy_normalizer_in_physical_mode(fake_material_props):
|
||||
data = _minimal_step_data(3)
|
||||
pdg_map, mat_map = {11: 0}, {"PbWO4": 0}
|
||||
legacy_norm = Normalizer()
|
||||
legacy_norm.mean = np.zeros(COND_DIM_BASE, dtype=np.float32)
|
||||
legacy_norm.std = np.ones(COND_DIM_BASE, dtype=np.float32)
|
||||
|
||||
with pytest.raises(ValueError, match="predates physical-property conditioning"):
|
||||
build_features(
|
||||
data,
|
||||
pdg_map,
|
||||
mat_map,
|
||||
cond_normalizer=legacy_norm,
|
||||
particle_conditioning="physical",
|
||||
material_conditioning="physical",
|
||||
)
|
||||
|
||||
|
||||
def test_onehot_axis_without_its_topn_map_raises():
|
||||
"""`cond_cat`'s width is the layout's call, so a "onehot" axis with no
|
||||
top-N map is a hard error rather than a silently-narrower array that
|
||||
ConditionEncoder would then index out of bounds."""
|
||||
data = _minimal_step_data(2)
|
||||
pdg_map, mat_map = {11: 0}, {"PbWO4": 0}
|
||||
|
||||
with pytest.raises(ValueError, match="needs pdg_topn_map"):
|
||||
build_cond_features(data, pdg_map, mat_map, particle_conditioning="onehot")
|
||||
with pytest.raises(ValueError, match="needs mat_topn_map"):
|
||||
build_cond_features(data, pdg_map, mat_map, material_conditioning="onehot")
|
||||
|
||||
|
||||
# ── sorted_membership / _vectorized_map_lookup ──────────────────────────────
|
||||
|
||||
|
||||
|
||||
+13
-11
@@ -1,16 +1,18 @@
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig, ParticleTypeConfig
|
||||
from giant.constants import COND_DIM, SEC_SLOT_DIM, X_DIM
|
||||
from giant.data.dataset import StepBatch
|
||||
from giant.model.network import Stage1Model, Stage2OneShot, stage2_trunk_sec_dim
|
||||
from giant.validate import validate_marginals
|
||||
|
||||
_PARTICLE_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
_MATERIAL_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
_PARTICLE_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
_MATERIAL_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
_K_MAX = 5
|
||||
|
||||
|
||||
def _tiny_models(particle_type_cfg: dict | None = None):
|
||||
def _tiny_models(particle_type_cfg: ParticleTypeConfig | None = None):
|
||||
"""A fresh v0.3.0 pair: Stage1Model owns no n_sec_head, so n_sec always
|
||||
comes from Stage2OneShot."""
|
||||
s1 = Stage1Model(
|
||||
@@ -21,12 +23,13 @@ def _tiny_models(particle_type_cfg: dict | None = None):
|
||||
hidden_dim=16,
|
||||
n_res_blocks=1,
|
||||
)
|
||||
target = (particle_type_cfg or {}).get("target", "physical")
|
||||
resolved_type_cfg = particle_type_cfg or ParticleTypeConfig(target="physical")
|
||||
target = resolved_type_cfg.target
|
||||
sec_dim = stage2_trunk_sec_dim(
|
||||
particle_type_cfg or {"target": "physical"},
|
||||
resolved_type_cfg,
|
||||
"flow",
|
||||
_K_MAX,
|
||||
int(_PARTICLE_CFG["emb_dim"]),
|
||||
_PARTICLE_CFG.emb_dim,
|
||||
)
|
||||
s2 = Stage2OneShot(
|
||||
pdg_vocab=3,
|
||||
@@ -41,13 +44,12 @@ def _tiny_models(particle_type_cfg: dict | None = None):
|
||||
sec_dim=sec_dim,
|
||||
particle_type_cfg=particle_type_cfg,
|
||||
)
|
||||
assert s2.particle_type_cfg.get("target", "physical") == target
|
||||
assert s2.particle_type_cfg.target == target
|
||||
return s1.eval(), s2.eval()
|
||||
|
||||
|
||||
def _loader(B: int = 4, n_batches: int = 2, n_sec_value: int = 0, n_classes: int = 8):
|
||||
"""A val_loader matching StreamingStepsDataset's 7-tuple batch shape:
|
||||
(cond_cont, cond_cat, target_s1, n_sec, sec_cont, proc_idx, sec_type_idx)."""
|
||||
"""A val_loader matching StreamingStepsDataset's StepBatch shape."""
|
||||
batches = []
|
||||
for _ in range(n_batches):
|
||||
cond_cont = torch.randn(B, COND_DIM)
|
||||
@@ -57,7 +59,7 @@ def _loader(B: int = 4, n_batches: int = 2, n_sec_value: int = 0, n_classes: int
|
||||
sec_cont = torch.randn(B, _K_MAX, SEC_SLOT_DIM)
|
||||
proc_idx = torch.zeros(B, dtype=torch.long)
|
||||
sec_type_idx = torch.randint(0, n_classes, (B, _K_MAX), dtype=torch.long)
|
||||
batches.append((cond_cont, cond_cat, x1, n_sec, sec_cont, proc_idx, sec_type_idx))
|
||||
batches.append(StepBatch(cond_cont, cond_cat, x1, n_sec, sec_cont, proc_idx, sec_type_idx))
|
||||
return batches
|
||||
|
||||
|
||||
@@ -95,7 +97,7 @@ def test_validate_marginals_physical_target_shapes():
|
||||
|
||||
|
||||
def test_validate_marginals_onehot_type_class_marginal():
|
||||
particle_type_cfg = {"target": "onehot"}
|
||||
particle_type_cfg = ParticleTypeConfig(target="onehot")
|
||||
s1, s2 = _tiny_models(particle_type_cfg)
|
||||
loader = _loader(n_sec_value=2, n_classes=s2.type_dim)
|
||||
|
||||
|
||||
+3
-2
@@ -1,12 +1,13 @@
|
||||
import torch
|
||||
|
||||
from giant.config import ConditioningAxisConfig
|
||||
from giant.constants import COND_DIM, K_MAX, SEC_DIM, SEC_SLOT_DIM, X_DIM
|
||||
from giant.model.network import CriticModel, Stage1Model, Stage2OneShot
|
||||
from giant.model.wgan import critic_loss, generator_loss, gradient_penalty
|
||||
from giant.sample import sample_secondaries_wgan, sample_wgan
|
||||
|
||||
PARTICLE_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
MATERIAL_CFG = {"type": "physical", "emb_dim": 8, "n_layers": 1}
|
||||
PARTICLE_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
MATERIAL_CFG = ConditioningAxisConfig(type="physical", emb_dim=8, n_layers=1)
|
||||
|
||||
|
||||
def _cond(B=8):
|
||||
|
||||
Reference in New Issue
Block a user