Expose dataset/conversion scripts as uv entry points

scripts/ is now a proper package (scripts/__init__.py, added to the wheel's
packages), with each script registered under [project.scripts] using its
bare dashed name (e.g. `uv run migrate-geant-steps`). Tests now import these
modules normally instead of loading them by file path.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-06-25 16:51:47 +02:00
parent 320365606a
commit 5be91e0d17
7 changed files with 18 additions and 41 deletions
+8 -4
View File
@@ -34,7 +34,7 @@ The model is developed in two phases:
## Data
Input: parquet files produced by [miniCaloSim](https://gitlab.etp.kit.edu/lbogner/minicalosim), or converted from a ROOT file via `scripts/steps_to_parquet.py`. Each row is one Geant4 step. Train/val split is by `event_id` (not row shuffle) to avoid leaking correlated steps from the same shower.
Input: parquet files produced by [miniCaloSim](https://gitlab.etp.kit.edu/lbogner/minicalosim), or converted from a ROOT file via `uv run steps-to-parquet`. Each row is one Geant4 step. Train/val split is by `event_id` (not row shuffle) to avoid leaking correlated steps from the same shower.
## Project structure
@@ -55,9 +55,13 @@ giant/
│ ├── validate.py # step-level marginal + KL-divergence validation
│ ├── analysis.py # notebook diagnostics: marginals, correlations, constraint checks
│ └── cli.py # `giant train` / `giant predict` Typer app
├── scripts/
│ ├── train.py # argparse training entry point
── steps_to_parquet.py # ROOT → parquet conversion (uproot/awkward/polars)
├── scripts/ # also exposed as uv entry points, e.g. `uv run steps-to-parquet`
│ ├── steps_to_parquet.py # ROOT → parquet conversion (uproot/awkward/polars)
── steps_to_parquet_parallel.py # fan out steps_to_parquet.py over several ROOT files
│ ├── migrate_geant_steps.py # one-time move into the raw/processed/pools/derived layout
│ ├── bump_dataset_version.py # cut a new raw gen or parquet schema, with a logged reason
│ ├── create_root_files.py # generate new ROOT shards via a minicalosim executable
│ └── hparam_scan.py # hyperparameter grid scan over `giant train` runs
└── tests/
```
+6 -1
View File
@@ -37,13 +37,18 @@ analysis = [
[project.scripts]
giant = "giant.cli:app"
steps-to-parquet = "scripts.steps_to_parquet:main"
steps-to-parquet-parallel = "scripts.steps_to_parquet_parallel:main"
migrate-geant-steps = "scripts.migrate_geant_steps:main"
bump-dataset-version = "scripts.bump_dataset_version:main"
create-root-files = "scripts.create_root_files:main"
[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"
[tool.hatch.build.targets.wheel]
packages = ["giant"]
packages = ["giant", "scripts"]
[tool.uv]
conflicts = [
View File
+1 -10
View File
@@ -1,13 +1,4 @@
import importlib.util
from pathlib import Path
# scripts/ is not an installed package — load the module straight from its path.
_SPEC = importlib.util.spec_from_file_location(
"bump_dataset_version",
Path(__file__).resolve().parents[1] / "scripts" / "bump_dataset_version.py",
)
bump_dataset_version = importlib.util.module_from_spec(_SPEC)
_SPEC.loader.exec_module(bump_dataset_version)
from scripts import bump_dataset_version
plan_bump_gen = bump_dataset_version.plan_bump_gen
plan_bump_schema = bump_dataset_version.plan_bump_schema
+1 -8
View File
@@ -1,17 +1,10 @@
import importlib.util
import json
import stat
from pathlib import Path
import pytest
# scripts/ is not an installed package — load the module straight from its path.
_SPEC = importlib.util.spec_from_file_location(
"create_root_files",
Path(__file__).resolve().parents[1] / "scripts" / "create_root_files.py",
)
create_root_files = importlib.util.module_from_spec(_SPEC)
_SPEC.loader.exec_module(create_root_files)
from scripts import create_root_files
parse_detector_spec = create_root_files.parse_detector_spec
next_shard_index = create_root_files.next_shard_index
+1 -10
View File
@@ -1,15 +1,6 @@
import importlib.util
from pathlib import Path
import polars as pl
# scripts/ is not an installed package — load the module straight from its path.
_SPEC = importlib.util.spec_from_file_location(
"steps_to_parquet",
Path(__file__).resolve().parents[1] / "scripts" / "steps_to_parquet.py",
)
steps_to_parquet = importlib.util.module_from_spec(_SPEC)
_SPEC.loader.exec_module(steps_to_parquet)
from scripts import steps_to_parquet
def _frame() -> pl.DataFrame:
+1 -8
View File
@@ -1,14 +1,7 @@
import importlib.util
import json
from pathlib import Path
# scripts/ is not an installed package — load the module straight from its path.
_SPEC = importlib.util.spec_from_file_location(
"steps_to_parquet_parallel",
Path(__file__).resolve().parents[1] / "scripts" / "steps_to_parquet_parallel.py",
)
steps_to_parquet_parallel = importlib.util.module_from_spec(_SPEC)
_SPEC.loader.exec_module(steps_to_parquet_parallel)
from scripts import steps_to_parquet_parallel
run_parallel = steps_to_parquet_parallel.run_parallel
resolve_destination = steps_to_parquet_parallel.resolve_destination