Update and refactor

This commit is contained in:
Kylian Schmidt
2025-07-07 14:44:09 +02:00
parent 976d90f26b
commit 9041abb72f
4 changed files with 1510 additions and 128 deletions
+134 -83
View File
@@ -1,6 +1,20 @@
"""
Scientific Gallery Generator
This module generates static HTML galleries from scientific plot collections.
It converts PDF plots to PNG thumbnails, creates responsive web interfaces,
and organizes plots into hierarchical directory structures.
Features:
- PDF to PNG conversion with configurable DPI
- Incremental updates (only converts when source is newer)
- Jinja2 templating for consistent HTML generation
- Support for nested folder structures
- Responsive grid layout with search and navigation
"""
import subprocess
import shutil
from pathlib import Path
from jinja2 import Environment, FileSystemLoader
from config import Config
@@ -12,15 +26,25 @@ env = Environment(loader=FileSystemLoader("."))
template = env.get_template("template.html")
def convert_pdf_to_png(pdf_path: Path):
def convert_pdf_to_png(pdf_path: Path) -> None:
"""
Convert a PDF file to PNG format using ImageMagick.
Only converts if the PNG doesn't exist or if the PDF is newer than
the PNG (with a 30-second buffer to handle filesystem timing issues).
Args:
pdf_path: Path to the source PDF file
Raises:
subprocess.CalledProcessError: If ImageMagick conversion fails
"""
png_path = pdf_path.with_suffix(".png")
# Check if PNG exists and is newer than PDF (with 30 second buffer)
if png_path.exists():
pdf_mtime = pdf_path.stat().st_mtime
png_mtime = png_path.stat().st_mtime
if png_mtime >= (pdf_mtime + 30): # 30 second buffer
# PNG is up to date, no need to convert
if png_mtime >= (pdf_mtime + 30):
return
else:
print(f"PDF {pdf_path.name} is newer than PNG, reconverting...")
@@ -36,72 +60,85 @@ def convert_pdf_to_png(pdf_path: Path):
def needs_update(source_file: Path, target_file: Path) -> bool:
"""Check if target file needs to be updated based on source file modification time"""
"""
Check if target file needs updating based on source modification time.
Args:
source_file: Path to the source file
target_file: Path to the target file
Returns:
True if target needs update, False otherwise
"""
if not target_file.exists():
return True
source_mtime = source_file.stat().st_mtime
target_mtime = target_file.stat().st_mtime
# Return True if source is newer than target (with 30 second buffer)
return source_mtime > (target_mtime + 30)
def build_gallery(source_dir: Path, web_dir: Path, relative_path: Path = None):
def build_gallery(source_dir: Path, web_dir: Path,
relative_path: Path = None) -> None:
"""
Build gallery for a directory, copying files to web directory and generating index.html
Recursively build gallery structure from source directory.
Processes all PDF files in the source directory, converts them to PNG,
copies both to the web directory, and generates index.html files with
navigation and thumbnails.
Args:
source_dir: Source directory containing PDFs
web_dir: Target web directory
relative_path: Relative path for navigation (None for root)
source_dir: Source directory containing PDF files
web_dir: Target web directory for gallery output
relative_path: Relative path from gallery root (for navigation)
"""
if relative_path is None:
relative_path = Path(".")
# Ensure web directory exists
target_dir = web_dir / relative_path
target_dir.mkdir(parents=True, exist_ok=True)
pdf_files = list(source_dir.glob("*.pdf"))
subdirs = [d for d in source_dir.iterdir() if d.is_dir()]
# Find all PDFs in source directory
pdfs = sorted(source_dir.glob("*.pdf"))
items = []
for pdf in pdfs:
# Convert PDF to PNG in source directory
convert_pdf_to_png(pdf)
png_path = pdf.with_suffix(".png")
for pdf_file in pdf_files:
png_file = pdf_file.with_suffix(".png")
# Copy both PDF and PNG to target directory only if needed
target_pdf = target_dir / pdf.name
target_png = target_dir / png_path.name
web_pdf = web_dir / pdf_file.name
web_png = web_dir / png_file.name
if needs_update(pdf, target_pdf):
print(f"Copying {pdf.name} to {target_pdf}")
shutil.copy2(pdf, target_pdf)
if needs_update(pdf_file, web_pdf):
print(f"Copying {pdf_file} to {web_pdf}")
shutil.copy2(pdf_file, web_pdf)
else:
print(f"Skipping {pdf.name} (up to date)")
print(f"Skipping {pdf_file.name} (up to date)")
if png_path.exists() and needs_update(png_path, target_png):
print(f"Copying {png_path.name} to {target_png}")
shutil.copy2(png_path, target_png)
elif png_path.exists():
print(f"Skipping {png_path.name} (up to date)")
if not png_file.exists():
convert_pdf_to_png(pdf_file)
if needs_update(png_file, web_png):
print(f"Copying {png_file} to {web_png}")
shutil.copy2(png_file, web_png)
else:
print(f"Skipping {png_file.name} (up to date)")
items.append({
"name": pdf.name,
"pdf_href": pdf.name,
"png_href": png_path.name
"name": pdf_file.stem,
"pdf_href": pdf_file.name,
"png_href": png_file.name
})
# Find subdirectories in source
subdirs = sorted([d.name for d in source_dir.iterdir() if d.is_dir()])
subdir_names = []
for subdir in subdirs:
subdir_web = web_dir / subdir.name
subdir_web.mkdir(exist_ok=True)
subdir_relative = relative_path / subdir.name
build_gallery(subdir, subdir_web, subdir_relative)
subdir_names.append(subdir.name)
# Generate index.html in target directory
output_html = target_dir / "index.html"
output_html = web_dir / "index.html"
# Create a meaningful title
if str(relative_path) == ".":
if relative_path == Path("."):
title = "Gallery"
else:
title = f"Gallery: {relative_path}"
@@ -110,75 +147,89 @@ def build_gallery(source_dir: Path, web_dir: Path, relative_path: Path = None):
f.write(template.render(
title=title,
items=items,
subdirs=subdirs,
relpath=str(relative_path)
subdirs=subdir_names,
relpath=str(relative_path),
ui=CONFIG.ui
))
print(f"Generated {output_html}")
# Recursively process subdirectories
for subdir in subdirs:
source_subdir = source_dir / subdir
new_relative_path = relative_path / subdir
build_gallery(source_subdir, web_dir, new_relative_path)
def main() -> None:
"""
Main entry point for gallery generation.
def loop_over_sources():
web_root = Path(CONFIG.web_folder)
Processes all configured sources and generates the complete gallery
structure in the web directory. Cleans up the gallery directory
on first run to ensure consistency.
"""
gallery_root = Path(CONFIG.web_folder) / CONFIG.plot_root
if gallery_root.exists():
print(f"Gallery directory {gallery_root} exists, cleaning up...")
shutil.rmtree(gallery_root)
gallery_root.mkdir(parents=True, exist_ok=True)
for source in CONFIG.sources:
print(f"Processing {source.name}: {source.path}")
if source.path.is_dir():
# Create a subdirectory based on plot_root and source name
source_web_dir = web_root / CONFIG.plot_root / source.name
build_gallery(source.path, source_web_dir)
print(f"Processed {source.name}: {source.path}")
elif source.path.suffix == ".pdf":
# For single PDF files, copy to web root with plot_root and source name as directory
source_web_dir = web_root / CONFIG.plot_root / source.name
convert_pdf_to_png(source.path)
png_path = source.path.with_suffix(".png")
source_path = Path(source.path)
# Ensure web directory exists
if source_path.is_file() and source_path.suffix == '.pdf':
source_web_dir = gallery_root / source.name
source_web_dir.mkdir(parents=True, exist_ok=True)
# Copy PDF and PNG to web directory only if needed
target_pdf = source_web_dir / source.path.name
target_png = source_web_dir / png_path.name
pdf_name = source_path.name
png_name = source_path.with_suffix('.png').name
if needs_update(source.path, target_pdf):
print(f"Copying {source.path.name} to {target_pdf}")
shutil.copy2(source.path, target_pdf)
web_pdf_path = source_web_dir / pdf_name
web_png_path = source_web_dir / png_name
source_png_path = source_path.with_suffix('.png')
if needs_update(source_path, web_pdf_path):
print(f"Copying {source_path} to {web_pdf_path}")
shutil.copy2(source_path, web_pdf_path)
else:
print(f"Skipping {source.path.name} (up to date)")
print(f"Skipping {source_path.name} (up to date)")
if png_path.exists() and needs_update(png_path, target_png):
print(f"Copying {png_path.name} to {target_png}")
shutil.copy2(png_path, target_png)
elif png_path.exists():
print(f"Skipping {png_path.name} (up to date)")
if not source_png_path.exists():
convert_pdf_to_png(source_path)
if needs_update(source_png_path, web_png_path):
print(f"Copying {source_png_path} to {web_png_path}")
shutil.copy2(source_png_path, web_png_path)
else:
print(f"Skipping {source_png_path.name} (up to date)")
# Create items list for template
items = [{
"name": source.path.name,
"pdf_href": source.path.name,
"png_href": png_path.name
"name": source_path.stem,
"pdf_href": pdf_name,
"png_href": png_name
}]
# Generate index.html for single PDF
output_html = source_web_dir / "index.html"
with output_html.open("w") as f:
f.write(template.render(
title=source.name,
items=items,
subdirs=[],
relpath=source.name
relpath=source.name,
ui=CONFIG.ui
))
print(f"Generated {output_html}")
print(f"Processed {source.name}: {source.path}")
elif source_path.is_dir():
source_web_dir = gallery_root / source.name
source_web_dir.mkdir(parents=True, exist_ok=True)
build_gallery(source_path, source_web_dir, Path(source.name))
print(f"Processed {source.name}: {source.path}")
else:
print(f"Warning: Source {source.path} is neither a "
f"directory nor a PDF file")
print("Done")
if __name__ == "__main__":
loop_over_sources()
print("Done")
main()