From 0a8345bd141b5937d65e145a44f44f6cace3dad5 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 9 Sep 2026 10:31:57 +1200 Subject: [PATCH 01/25] Saving. --- .github/workflows/ci.yml | 10 + .github/workflows/claude-review.yml | 19 + .gitignore | 181 +++ README.md | 92 ++ imdb/__init__.py | 5 + imdb/imdb.py | 1627 +++++++++++++++++++++++++++ imdb/schema.py | 220 ++++ pyproject.toml | 95 ++ tests/__init__.py | 0 tests/conftest.py | 117 ++ tests/test_imdb.py | 311 +++++ uv.lock | 960 ++++++++++++++++ 12 files changed, 3637 insertions(+) create mode 100644 .github/workflows/ci.yml create mode 100644 .github/workflows/claude-review.yml create mode 100644 .gitignore create mode 100644 README.md create mode 100644 imdb/__init__.py create mode 100644 imdb/imdb.py create mode 100644 imdb/schema.py create mode 100644 pyproject.toml create mode 100644 tests/__init__.py create mode 100644 tests/conftest.py create mode 100644 tests/test_imdb.py create mode 100644 uv.lock diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..cfd40c8 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,10 @@ +name: CI +on: [pull_request] +jobs: + ci: + uses: ucgmsim/meta-ci-action/.github/workflows/ci.yml@main + with: + package-dir: imdb + uv-extra-args: "--all-groups" + enable-coverage: true + cov-package: imdb diff --git a/.github/workflows/claude-review.yml b/.github/workflows/claude-review.yml new file mode 100644 index 0000000..14b2c86 --- /dev/null +++ b/.github/workflows/claude-review.yml @@ -0,0 +1,19 @@ +name: Claude PR Review + +# Triggered by a "@claude review" comment on a pull request. +# +# Why issue_comment and not pull_request: GitHub withholds secrets from runs +# triggered by fork pull requests, so a pull_request trigger would silently +# fail on exactly the PRs that most need review. issue_comment runs in the base +# repository's context with secrets available. The action independently checks +# that the commenting user has write access before doing anything, so the gate +# is "a member of this org asked for this review". +on: + issue_comment: + types: [created] + +jobs: + review: + uses: ucgmsim/meta-ci-action/.github/workflows/claude-review.yml@main + secrets: + claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..68d7a63 --- /dev/null +++ b/.gitignore @@ -0,0 +1,181 @@ +.DS_Store +.idea +*.log +tmp/ +# Created by https://www.toptal.com/developers/gitignore/api/python +# Edit at https://www.toptal.com/developers/gitignore?templates=python + +### Python ### +# Byte-compiled / optimized / DLL files +__pycache__/ + +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ +cover/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +.pybuilder/ +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +# For a library or package, you might want to ignore these files since the code is +# intended to run in multiple environments; otherwise, check them in: +# .python-version + +# pipenv +# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. +# However, in case of collaboration, if having platform-specific dependencies or dependencies +# having no cross-platform support, pipenv may install dependencies that don't work, or not +# install all needed dependencies. +#Pipfile.lock + +# poetry +# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control. +# This is especially recommended for binary packages to ensure reproducibility, and is more +# commonly ignored for libraries. +# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control +#poetry.lock + +# pdm +# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control. +#pdm.lock +# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it +# in version control. +# https://pdm.fming.dev/#use-with-ide +.pdm.toml + +# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# pytype static type analyzer +.pytype/ + +# Cython debug symbols +cython_debug/ + +# PyCharm +# JetBrains specific template is maintained in a separate JetBrains.gitignore that can +# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore +# and can be added to the global gitignore or merged into this file. For a more nuclear +# option (not recommended) you can uncomment the following to ignore the entire idea folder. +#.idea/ + +### Python Patch ### +# Poetry local configuration file - https://python-poetry.org/docs/configuration/#local-configuration +poetry.toml + +# ruff +.ruff_cache/ + +# LSP config files +pyrightconfig.json + +# End of https://www.toptal.com/developers/gitignore/api/python diff --git a/README.md b/README.md new file mode 100644 index 0000000..2e806de --- /dev/null +++ b/README.md @@ -0,0 +1,92 @@ +# imdb + +Reading and writing intensity measure databases (IMDBs): DuckDB files holding +intensity measures from physics-based ground-motion simulation. + +One file holds one simulation run set. Every file uses the same schema and is +self-contained; combine several with `ATTACH` and `UNION ALL`. + +## Schema + +Thirteen tables: four dimensions (`events`, `realisations`, `sites`, +`site_event`), one identity table (`records`), three IM tables (`psa_ims`, +`fas_ims`, `scalars_ims`), two vocabulary tables (`periods`, `frequencies`) and +three documentation tables (`db_meta`, `im_units`, `notes`). + +A ground motion is identified by `(rel_id, site_id, component)`. pSA and FAS are +stored as one `FLOAT[]` per record, indexed by `periods.period_index` and +`frequencies.freq_index` (both 1-based). Scalar IMs are named columns. + +`imdb/schema.py` holds the DDL and is the single source of truth. Every database +also documents itself: read the `notes` and `db_meta` tables. + +## Reading + +```python +from imdb import IMDB + +with IMDB("cs200.duckdb") as db: + df = db.get_im_df( + ["PGA", "pSA_0.1", "pSA_1.0", "FAS_5.0"], + events=["AlpineF2K"], component="rotd50", max_rrup=200, + ) +``` + +`get_im_df` returns one DataFrame indexed by `record_id`, columns named as +requested. Underneath it are `get_psa(periods=...)`, `get_fas(frequencies=...)` +and `get_scalars(ims=...)`, which label their columns with the period in +seconds, the frequency in Hz, and the IM name respectively. + +Every read takes the same keyword filters: `events`, `rels`, `sites`, +`component`, `max_rrup`, `record_ids`. Anything more specific is a raw query +through `db.sql(...)` or `db.conn`. + +Dimension tables come back whole: `get_events`, `get_realisations`, `get_sites`, +`get_site_event`. Pass `expand_metadata=True` to unpack their JSON `metadata` +column into columns. + +## Writing + +```python +with IMDB.create("new.duckdb", periods=[0.01, 0.1, 1.0], frequencies=[1.0, 10.0], + components=["rotd50"], db_meta={"source": "CyberShake v24p1"}) as db: + db.add_events(event_df) # event_id, magnitude, tect_type, ... + db.add_realisations(rel_df) # rel_id, event_id, rake, hypo_*, ... + db.add_sites(site_df) # site_id, lat, lon, vs30, ... + db.add_site_event(site_event_df) # site_id, event_id, rrup, rjb, ... + db.add_records(record_df) # rel_id, site_id, component, pSA, FAS, PGA, ... + db.finalise() +``` + +Integer surrogates (`event_int_id`, `rel_int_id`, `site_int_id`, `record_id`) are +assigned here and are file-local: they change on rebuild, so nothing outside the +file may reference them. Callers work in string IDs throughout. + +`add_records` writes `records` plus whichever IM tables the input covers; a row +with no `pSA` array simply gets no `psa_ims` row. `CAV`, `AI`, `Ds575` and +`Ds595` are undefined for `rotd*` components and are stored as NULL there. + +Extra source-specific fields go in a `metadata` column, passed as a dict. The +first write of a table's metadata declares its permitted keys in `db_meta`; +later writes are validated against that declaration. + +Statements autocommit, so writes are not rolled back on error. To re-ingest an +event, `delete_event(event_id)` first: it removes the event and everything +derived from it, leaving shared sites alone. + +## Validation + +The large tables carry no `PRIMARY KEY`, `UNIQUE` or `FOREIGN KEY`, because at +these row counts each one is an ART index loaded into memory on open and buys no +lookup speed. `validate()` checks what they would have enforced (orphan keys, +duplicated logical keys, array lengths, undeclared components and metadata keys) +and returns the problems as a list. `finalise()` runs it and raises. + +## Development + +```bash +uv sync --all-groups +uv run pytest +uv run ruff check imdb tests && uv run ruff format --check imdb tests +uv run ty check imdb +``` diff --git a/imdb/__init__.py b/imdb/__init__.py new file mode 100644 index 0000000..3331754 --- /dev/null +++ b/imdb/__init__.py @@ -0,0 +1,5 @@ +"""Reading and writing intensity measure databases.""" + +from imdb.imdb import IMDB + +__all__ = ["IMDB"] diff --git a/imdb/imdb.py b/imdb/imdb.py new file mode 100644 index 0000000..d8a392f --- /dev/null +++ b/imdb/imdb.py @@ -0,0 +1,1627 @@ +""" +Read and write intensity measure databases. + +An intensity measure database (IMDB) is a DuckDB file holding intensity +measures (IMs) from physics-based ground-motion simulation, laid out according +to :mod:`imdb.schema`. One file holds one simulation run set. + +Reading: + +>>> with IMDB("path/to/ims.duckdb") as db: +... df = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"], component="rotd50") + +Writing: + +>>> with IMDB.create("new.duckdb", periods=[0.1, 1.0], frequencies=[1.0]) as db: +... db.add_events(event_df) +... db.add_realisations(rel_df) +... db.add_sites(site_df) +... db.add_site_event(site_event_df) +... db.add_records(record_df) +... db.finalise() +""" + +import contextlib +import getpass +import json +import logging +from collections.abc import Iterable, Mapping, Sequence +from datetime import UTC, datetime +from functools import cached_property +from importlib.metadata import PackageNotFoundError, version +from pathlib import Path +from types import TracebackType +from typing import Any, Self + +import duckdb +import numpy as np +import pandas as pd + +from imdb import schema + +logger = logging.getLogger(__name__) + +_CACHED = ( + "db_meta", + "notes", + "im_units", + "components", + "periods", + "frequencies", + "event_ids", + "rel_ids", + "site_ids", + "rel_to_event", +) + +_DIM_JOINS = { + "e": "JOIN events e ON e.event_int_id = r.event_int_id", + "rl": "JOIN realisations rl ON rl.rel_int_id = r.rel_int_id", + "s": "JOIN sites s ON s.site_int_id = r.site_int_id", + "se": ( + "JOIN site_event se ON se.site_int_id = r.site_int_id " + "AND se.event_int_id = r.event_int_id" + ), +} + +_MATCH_RTOL = 1e-6 +"""Relative tolerance when matching a requested period or frequency to the grid.""" + + +def _imdb_version() -> str: + """Return the installed version of this library, or ``unknown``. + + Returns + ------- + str + Version string. + """ + try: + return version("imdb") + except PackageNotFoundError: + return "unknown" + + +def _parse_number(text: str) -> float: + """Parse a period or frequency written with either ``.`` or ``p`` as the point. + + Parameters + ---------- + text : str + Number to parse, for example ``1.0`` or ``0p1``. + + Returns + ------- + float + The parsed value. + """ + try: + return float(text) + except ValueError: + return float(text.replace("p", ".", 1)) + + +class IMDB(contextlib.AbstractContextManager): + """An intensity measure database. + + Parameters + ---------- + db_path : Path or str + Path to the DuckDB file. + read_only : bool, optional + Open read-only. Write methods raise when true. + memory_limit : str, optional + DuckDB memory limit applied on open. + """ + + def __init__( + self, + db_path: Path | str, + read_only: bool = True, + memory_limit: str = "8GB", + ) -> None: + """Create a handle. The connection is opened lazily, or by :meth:`open`.""" + self.db_path = Path(db_path) + self.read_only = read_only + self.memory_limit = memory_limit + self._conn: duckdb.DuckDBPyConnection | None = None + + def open(self) -> Self: + """Open the database connection. + + Returns + ------- + IMDB + This database, for chaining. + """ + if self._conn is not None: + return self + if self.read_only and not self.db_path.exists(): + raise FileNotFoundError(self.db_path) + self._conn = duckdb.connect(str(self.db_path), read_only=self.read_only) + self._conn.execute(f"SET memory_limit='{self.memory_limit}'") + self._conn.execute("SET enable_progress_bar = false") + return self + + def close(self) -> None: + """Close the database connection.""" + if self._conn is not None: + self._conn.close() + self._conn = None + self._invalidate() + + @property + def conn(self) -> duckdb.DuckDBPyConnection: + """The open DuckDB connection. + + Returns + ------- + duckdb.DuckDBPyConnection + The connection. + """ + if self._conn is None: + self.open() + assert self._conn is not None + return self._conn + + def sql( + self, query: str, params: Sequence | None = None + ) -> duckdb.DuckDBPyRelation: + """Run an arbitrary query against the database. + + The escape hatch for reads the keyword filters cannot express. + + Parameters + ---------- + query : str + SQL to execute. + params : sequence, optional + Prepared-statement parameters. + + Returns + ------- + duckdb.DuckDBPyRelation + The query result. + """ + return self.conn.sql(query, params=params) + + def __enter__(self) -> Self: + """Open the connection and return this database.""" + return self.open() + + def __exit__( + self, + exc_type: type[BaseException] | None, + exc_val: BaseException | None, + exc_tb: TracebackType | None, + ) -> None: + """Close the connection. + + Statements autocommit, so a failed write is not rolled back here. Repair + a partial ingest with :meth:`delete_event`, which is what makes a + re-ingest idempotent. + """ + self.close() + + def _invalidate(self) -> None: + """Drop every cached lookup, after a write or a close.""" + for name in _CACHED: + self.__dict__.pop(name, None) + + def _require_write(self) -> None: + """Raise if the database was opened read-only.""" + if self.read_only: + raise PermissionError(f"{self.db_path} is open read-only") + + @contextlib.contextmanager + def _temp_frames(self, frames: Mapping[str, pd.DataFrame]): + """Register DataFrames as views for the duration of a query. + + Parameters + ---------- + frames : mapping of str to pandas.DataFrame + View name to DataFrame. + + Yields + ------ + None + """ + for name, df in frames.items(): + self.conn.register(name, df) + try: + yield + finally: + for name in frames: + self.conn.unregister(name) + + @contextlib.contextmanager + def _transaction(self): + """Group several statements into one atomic write. + + Yields + ------ + None + """ + self.conn.execute("BEGIN TRANSACTION") + try: + yield + except Exception: + self.conn.execute("ROLLBACK") + raise + else: + self.conn.execute("COMMIT") + + # ------------------------------------------------------------------ + # cached vocabulary + # ------------------------------------------------------------------ + + @cached_property + def db_meta(self) -> dict[str, str]: + """Contents of the ``db_meta`` table. + + Returns + ------- + dict of str to str + Key to value. + """ + return dict(self.conn.execute("SELECT key, value FROM db_meta").fetchall()) + + @cached_property + def notes(self) -> dict[str, str]: + """Contents of the ``notes`` table. + + Returns + ------- + dict of str to str + Topic to note. + """ + return dict(self.conn.execute("SELECT topic, note FROM notes").fetchall()) + + @cached_property + def im_units(self) -> dict[str, str]: + """Contents of the ``im_units`` table. + + Returns + ------- + dict of str to str + IM name to unit. + """ + return dict(self.conn.execute("SELECT im, unit FROM im_units").fetchall()) + + @cached_property + def components(self) -> list[str]: + """Components held by this database, from ``db_meta``. + + Returns + ------- + list of str + Component names. + """ + value = self.db_meta.get("components", "") + return [c for c in (part.strip() for part in value.split(",")) if c] + + @cached_property + def periods(self) -> pd.Series: + """The pSA period grid. + + Returns + ------- + pandas.Series + Period in seconds, indexed by 1-based ``period_index``. + """ + return ( + self.conn.execute("SELECT period_index, period FROM periods ORDER BY 1") + .df() + .set_index("period_index")["period"] + ) + + @cached_property + def frequencies(self) -> pd.Series: + """The FAS frequency grid. + + Returns + ------- + pandas.Series + Frequency in Hz, indexed by 1-based ``freq_index``. + """ + return ( + self.conn.execute( + "SELECT freq_index, frequency FROM frequencies ORDER BY 1" + ) + .df() + .set_index("freq_index")["frequency"] + ) + + @cached_property + def event_ids(self) -> pd.Series: + """Mapping of ``event_id`` to ``event_int_id``. + + Returns + ------- + pandas.Series + Integer id indexed by string id. + """ + return self._id_map("events", "event_id", "event_int_id") + + @cached_property + def rel_ids(self) -> pd.Series: + """Mapping of ``rel_id`` to ``rel_int_id``. + + Returns + ------- + pandas.Series + Integer id indexed by string id. + """ + return self._id_map("realisations", "rel_id", "rel_int_id") + + @cached_property + def site_ids(self) -> pd.Series: + """Mapping of ``site_id`` to ``site_int_id``. + + Returns + ------- + pandas.Series + Integer id indexed by string id. + """ + return self._id_map("sites", "site_id", "site_int_id") + + @cached_property + def rel_to_event(self) -> pd.Series: + """Mapping of ``rel_int_id`` to ``event_int_id``. + + Returns + ------- + pandas.Series + Event integer id indexed by realisation integer id. + """ + return self._id_map("realisations", "rel_int_id", "event_int_id") + + def _id_map(self, table: str, key: str, value: str) -> pd.Series: + """Read a two-column mapping out of a dimension table. + + Parameters + ---------- + table : str + Table to read. + key : str + Column to index by. + value : str + Column to map to. + + Returns + ------- + pandas.Series + ``value`` indexed by ``key``. + """ + return ( + self.conn.execute(f"SELECT {key}, {value} FROM {table}") + .df() + .set_index(key)[value] + ) + + def _resolve(self, ids: Iterable[str], mapping: pd.Series, what: str) -> np.ndarray: + """Resolve string ids to integer surrogates. + + Parameters + ---------- + ids : iterable of str + String ids to resolve. + mapping : pandas.Series + Mapping to resolve against. + what : str + Name used in the error message. + + Returns + ------- + numpy.ndarray + Integer ids, in the order given. + """ + # the id columns are VARCHAR, so a numeric id from the source reaches the + # database as its string form and must be looked up that way + ids = np.asarray([str(value) for value in ids], dtype=object) + unknown = pd.unique(ids[~pd.Index(ids).isin(mapping.index)]) + if len(unknown): + raise KeyError(f"unknown {what}: {sorted(unknown)[:10]}") + return mapping.loc[ids].to_numpy(dtype=np.int64) + + def _scalar(self, query: str, params: Sequence | None = None) -> Any: + """Run a query returning exactly one row and one column. + + Parameters + ---------- + query : str + SQL to execute. + params : sequence, optional + Prepared-statement parameters. + + Returns + ------- + Any + The single value. + """ + row = self.conn.execute(query, params).fetchone() + if row is None: + raise RuntimeError(f"query returned no rows: {query}") + return row[0] + + def _table_columns(self, table: str) -> list[str]: + """Column names of a table, in declaration order. + + Parameters + ---------- + table : str + Table name. + + Returns + ------- + list of str + Column names. + """ + rows = self.conn.execute( + "SELECT column_name FROM information_schema.columns " + "WHERE table_name = ? ORDER BY ordinal_position", + [table], + ).fetchall() + if not rows: + raise KeyError(f"no such table: {table}") + return [row[0] for row in rows] + + def _grid_indices( + self, values: Iterable[float] | None, grid: pd.Series, what: str + ) -> tuple[list[int], list[float]]: + """Resolve requested grid values to 1-based array indices. + + Parameters + ---------- + values : iterable of float, optional + Values to resolve. ``None`` returns the whole grid. + grid : pandas.Series + Grid to resolve against, indexed by array index. + what : str + Name used in the error message. + + Returns + ------- + tuple of (list of int, list of float) + Array indices and the matched grid values, in the order requested. + """ + if grid.empty: + raise KeyError(f"this database has no {what} grid") + if values is None: + return list(grid.index.astype(int)), list(grid.to_numpy(dtype=float)) + + available = grid.to_numpy(dtype=float) + indices, matched = [], [] + for value in values: + value = float(value) + close = np.flatnonzero( + np.isclose(available, value, rtol=_MATCH_RTOL, atol=0.0) + ) + if close.size == 0: + raise KeyError( + f"{what} {value} is not on this database's grid; " + f"available: {np.array2string(available, threshold=20)}" + ) + indices.append(int(grid.index[close[0]])) + matched.append(float(available[close[0]])) + return indices, matched + + # ------------------------------------------------------------------ + # record filtering + # ------------------------------------------------------------------ + + def _record_filter( + self, + need: Iterable[str] = (), + events: Iterable[str] | None = None, + rels: Iterable[str] | None = None, + sites: Iterable[str] | None = None, + component: str | Iterable[str] | None = None, + max_rrup: float | None = None, + record_ids: Iterable[int] | None = None, + ) -> tuple[str, str, dict[str, pd.DataFrame]]: + """Build the FROM and WHERE clauses shared by every record query. + + Parameters + ---------- + need : iterable of str, optional + Dimension aliases the caller's SELECT needs, from ``e``, ``rl``, ``s``, ``se``. + events : iterable of str, optional + Keep only these ``event_id`` values. + rels : iterable of str, optional + Keep only these ``rel_id`` values. + sites : iterable of str, optional + Keep only these ``site_id`` values. + component : str or iterable of str, optional + Keep only these components. + max_rrup : float, optional + Keep only records whose ``site_event.rrup`` is at most this. + record_ids : iterable of int, optional + Keep only these ``record_id`` values. + + Returns + ------- + tuple of (str, str, dict of str to pandas.DataFrame) + FROM clause, WHERE clause and the frames to register while querying. + """ + dims, joins, wheres, frames = set(need), [], [], {} + + for alias, column, table, values, mapping in ( + ("e", "event_id", "events", events, self.event_ids), + ("rl", "rel_id", "realisations", rels, self.rel_ids), + ("s", "site_id", "sites", sites, self.site_ids), + ): + if values is None: + continue + dims.add(alias) + values = [str(value) for value in values] + self._resolve(values, mapping, column) + view = f"_f_{table}" + frames[view] = pd.DataFrame({"_key": np.asarray(values, dtype=object)}) + joins.append(f"JOIN {view} ON {view}._key = {alias}.{column}") + + if record_ids is not None: + frames["_f_records"] = pd.DataFrame( + {"_key": np.asarray(list(record_ids), dtype=np.int64)} + ) + joins.append("JOIN _f_records ON _f_records._key = r.record_id") + + if max_rrup is not None: + dims.add("se") + wheres.append(f"se.rrup <= {float(max_rrup)}") + + if component is not None: + wanted = [component] if isinstance(component, str) else list(component) + unknown = set(wanted) - set(self.components) + if unknown: + raise ValueError( + f"components {sorted(unknown)} are not in this database; " + f"it holds {self.components}" + ) + wheres.append( + "r.component IN (" + ", ".join(f"'{c}'" for c in wanted) + ")" + ) + + from_sql = "\n".join( + ["FROM records r"] + + [_DIM_JOINS[alias] for alias in ("e", "rl", "s", "se") if alias in dims] + + joins + ) + where_sql = ("WHERE " + " AND ".join(wheres)) if wheres else "" + return from_sql, where_sql, frames + + def _record_query( + self, + select: str, + need: Iterable[str] = (), + joins: Iterable[str] = (), + **filters, + ) -> pd.DataFrame: + """Run a query over ``records``, indexed by ``record_id``. + + Parameters + ---------- + select : str + SELECT list, without the leading ``record_id``. + need : iterable of str, optional + Dimension aliases the SELECT needs. + joins : iterable of str, optional + Extra join clauses, for example onto an IM table. + **filters + Passed to :meth:`_record_filter`. + + Returns + ------- + pandas.DataFrame + Query result indexed by ``record_id``. + """ + from_sql, where_sql, frames = self._record_filter(need=need, **filters) + from_sql = "\n".join([from_sql, *joins]) + query = f"SELECT r.record_id, {select}\n{from_sql}\n{where_sql}" + logger.debug("query: %s", query) + with self._temp_frames(frames): + return self.conn.execute(query).df().set_index("record_id") + + # ------------------------------------------------------------------ + # dimension reads + # ------------------------------------------------------------------ + + def get_events(self, expand_metadata: bool = False) -> pd.DataFrame: + """Read the ``events`` table. + + Parameters + ---------- + expand_metadata : bool, optional + Expand the JSON ``metadata`` column into columns. + + Returns + ------- + pandas.DataFrame + Events indexed by ``event_int_id``. + """ + df = self.conn.execute("SELECT * FROM events").df().set_index("event_int_id") + return _expand_metadata(df) if expand_metadata else df + + def get_realisations(self, expand_metadata: bool = False) -> pd.DataFrame: + """Read the ``realisations`` table. + + Parameters + ---------- + expand_metadata : bool, optional + Expand the JSON ``metadata`` column into columns. + + Returns + ------- + pandas.DataFrame + Realisations indexed by ``rel_int_id``. + """ + df = ( + self.conn.execute("SELECT * FROM realisations").df().set_index("rel_int_id") + ) + return _expand_metadata(df) if expand_metadata else df + + def get_sites(self, expand_metadata: bool = False) -> pd.DataFrame: + """Read the ``sites`` table. + + Parameters + ---------- + expand_metadata : bool, optional + Expand the JSON ``metadata`` column into columns. + + Returns + ------- + pandas.DataFrame + Sites indexed by ``site_int_id``. + """ + df = self.conn.execute("SELECT * FROM sites").df().set_index("site_int_id") + return _expand_metadata(df) if expand_metadata else df + + def get_site_event( + self, + sites: Iterable[str] | None = None, + events: Iterable[str] | None = None, + max_rrup: float | None = None, + expand_metadata: bool = False, + ) -> pd.DataFrame: + """Read the ``site_event`` table. + + Parameters + ---------- + sites : iterable of str, optional + Keep only these ``site_id`` values. + events : iterable of str, optional + Keep only these ``event_id`` values. + max_rrup : float, optional + Keep only pairs whose ``rrup`` is at most this. + expand_metadata : bool, optional + Expand the JSON ``metadata`` column into columns. + + Returns + ------- + pandas.DataFrame + Site-event pairs, with ``site_id`` and ``event_id`` added. + """ + wheres, frames = [], {} + if sites is not None: + sites = [str(value) for value in sites] + self._resolve(sites, self.site_ids, "site_id") + frames["_f_sites"] = pd.DataFrame({"_key": np.asarray(sites, dtype=object)}) + wheres.append("s.site_id IN (SELECT _key FROM _f_sites)") + if events is not None: + events = [str(value) for value in events] + self._resolve(events, self.event_ids, "event_id") + frames["_f_events"] = pd.DataFrame( + {"_key": np.asarray(events, dtype=object)} + ) + wheres.append("e.event_id IN (SELECT _key FROM _f_events)") + if max_rrup is not None: + wheres.append(f"se.rrup <= {float(max_rrup)}") + where_sql = ("WHERE " + " AND ".join(wheres)) if wheres else "" + + query = f""" + SELECT se.*, s.site_id, e.event_id + FROM site_event se + JOIN sites s ON s.site_int_id = se.site_int_id + JOIN events e ON e.event_int_id = se.event_int_id + {where_sql} + """ + with self._temp_frames(frames): + df = self.conn.execute(query).df() + return _expand_metadata(df) if expand_metadata else df + + # ------------------------------------------------------------------ + # record and IM reads + # ------------------------------------------------------------------ + + def get_records(self, **filters) -> pd.DataFrame: + """Read record identities. + + Parameters + ---------- + **filters + Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, + ``record_ids``. + + Returns + ------- + pandas.DataFrame + ``event_id``, ``rel_id``, ``site_id`` and ``component``, indexed by + ``record_id``. + """ + return self._record_query( + "e.event_id, rl.rel_id, s.site_id, r.component", + need=("e", "rl", "s"), + **filters, + ) + + def get_psa( + self, periods: Iterable[float] | None = None, **filters + ) -> pd.DataFrame: + """Read pSA values. + + Parameters + ---------- + periods : iterable of float, optional + Periods in seconds. ``None`` reads the whole grid. + **filters + Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, + ``record_ids``. + + Returns + ------- + pandas.DataFrame + pSA in g, indexed by ``record_id``, one column per period, labelled + with the period in seconds. Records with no ``psa_ims`` row are absent. + """ + return self._spectral_read("psa_ims", "pSA", self.periods, periods, **filters) + + def get_fas( + self, frequencies: Iterable[float] | None = None, **filters + ) -> pd.DataFrame: + """Read FAS values. + + Parameters + ---------- + frequencies : iterable of float, optional + Frequencies in Hz. ``None`` reads the whole grid. + **filters + Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, + ``record_ids``. + + Returns + ------- + pandas.DataFrame + FAS in g.s, indexed by ``record_id``, one column per frequency, + labelled with the frequency in Hz. Records with no ``fas_ims`` row + are absent. + """ + return self._spectral_read( + "fas_ims", "FAS", self.frequencies, frequencies, **filters + ) + + def _spectral_read( + self, + table: str, + column: str, + grid: pd.Series, + values: Iterable[float] | None, + **filters, + ) -> pd.DataFrame: + """Project selected array elements out of a spectral IM table. + + Parameters + ---------- + table : str + IM table to read. + column : str + Array column in that table. + grid : pandas.Series + The database's grid for that column. + values : iterable of float, optional + Grid values to read. ``None`` reads all of them. + **filters + Passed to :meth:`_record_filter`. + + Returns + ------- + pandas.DataFrame + One column per requested grid value, indexed by ``record_id``. + """ + indices, matched = self._grid_indices(values, grid, column) + select = ", ".join(f'im.{column}[{i}] AS "c{n}"' for n, i in enumerate(indices)) + df = self._record_query( + select, joins=[f"JOIN {table} im USING (record_id)"], **filters + ) + df.columns = pd.Index(matched, name=column) + return df + + def get_scalars(self, ims: Iterable[str] | None = None, **filters) -> pd.DataFrame: + """Read scalar IM values. + + Parameters + ---------- + ims : iterable of str, optional + Scalar IM names. ``None`` reads all of them. + **filters + Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, + ``record_ids``. + + Returns + ------- + pandas.DataFrame + One column per requested IM, indexed by ``record_id``. Records with + no ``scalars_ims`` row are absent. ``CAV``, ``AI``, ``Ds575`` and + ``Ds595`` are NULL for ``rotd*`` components. + """ + wanted = list(schema.SCALAR_IMS) if ims is None else list(ims) + unknown = set(wanted) - set(schema.SCALAR_IMS) + if unknown: + raise ValueError( + f"unknown scalar IMs {sorted(unknown)}; known: {list(schema.SCALAR_IMS)}" + ) + select = ", ".join(f"im.{im}" for im in wanted) + return self._record_query( + select, joins=["JOIN scalars_ims im USING (record_id)"], **filters + ) + + def get_im_df(self, ims: Iterable[str], **filters) -> pd.DataFrame: + """Read named intensity measures into one DataFrame. + + Names are ``PGA``, ``PGV``, ``PGD``, ``CAV``, ``AI``, ``Ds575``, + ``Ds595`` for scalars, ``pSA_`` for response spectra and + ``FAS_`` for Fourier spectra. The point may be written as + ``.`` or ``p``, so ``pSA_0.1`` and ``pSA_0p1`` are the same. + + Parameters + ---------- + ims : iterable of str + IM names to read. + **filters + Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, + ``record_ids``. + + Returns + ------- + pandas.DataFrame + One column per requested name, in the order requested, indexed by + ``record_id``. IM tables are joined outer, so a record missing from + one of them reads as NaN in its columns. + """ + ims = list(ims) + scalars, psa, fas = [], [], [] + for name in ims: + if name in schema.SCALAR_IMS: + scalars.append(name) + elif name.startswith("pSA_"): + psa.append((name, _parse_number(name[4:]))) + elif name.startswith("FAS_"): + fas.append((name, _parse_number(name[4:]))) + else: + raise ValueError( + f"cannot parse IM name {name!r}; expected one of " + f"{list(schema.SCALAR_IMS)}, pSA_ or FAS_" + ) + + parts = [] + if scalars: + parts.append(self.get_scalars(ims=scalars, **filters)) + for getter, requested in ((self.get_psa, psa), (self.get_fas, fas)): + if not requested: + continue + part = getter([value for _, value in requested], **filters) + part.columns = pd.Index([name for name, _ in requested]) + parts.append(part) + + df = parts[0] if len(parts) == 1 else pd.concat(parts, axis=1, join="outer") + return df[ims] + + # ------------------------------------------------------------------ + # creation + # ------------------------------------------------------------------ + + @classmethod + def create( + cls, + db_path: Path | str, + periods: Iterable[float], + frequencies: Iterable[float] = (), + components: Iterable[str] = schema.COMPONENTS, + db_meta: Mapping[str, object] | None = None, + overwrite: bool = False, + **kwargs, + ) -> Self: + """Create an empty database and return it open for writing. + + Parameters + ---------- + db_path : Path or str + Path of the file to create. + periods : iterable of float + pSA period grid in seconds. Sorted ascending on write. + frequencies : iterable of float, optional + FAS frequency grid in Hz. Sorted ascending on write. + components : iterable of str, optional + Components this database will hold. + db_meta : mapping, optional + Values merged over the defaults, for example ``dataset_description``, + ``source`` and the ``*_metadata_keys`` declarations. + overwrite : bool, optional + Replace an existing file. + **kwargs + Passed to the constructor. + + Returns + ------- + IMDB + The new database, open for writing. + """ + db_path = Path(db_path) + if db_path.exists(): + if not overwrite: + raise FileExistsError(db_path) + db_path.unlink() + + unknown = set(components) - set(schema.COMPONENTS) + if unknown: + raise ValueError( + f"unknown components {sorted(unknown)}; known: {list(schema.COMPONENTS)}" + ) + + db = cls(db_path, read_only=False, **kwargs).open() + db.conn.execute(schema.DDL) + + periods = np.unique(np.asarray(list(periods), dtype=float)) + frequencies = np.unique(np.asarray(list(frequencies), dtype=float)) + db._insert( + "periods", + pd.DataFrame( + {"period_index": np.arange(1, len(periods) + 1), "period": periods} + ), + ) + db._insert( + "frequencies", + pd.DataFrame( + { + "freq_index": np.arange(1, len(frequencies) + 1), + "frequency": frequencies, + } + ), + ) + db._insert( + "im_units", + pd.DataFrame( + {"im": list(schema.IM_UNITS), "unit": list(schema.IM_UNITS.values())} + ), + ) + db._insert( + "notes", + pd.DataFrame( + {"topic": list(schema.NOTES), "note": list(schema.NOTES.values())} + ), + ) + + meta: dict[str, str] = { + "schema_version": schema.SCHEMA_VERSION, + "dataset_id": db_path.stem, + "dataset_description": "", + "components": ",".join(components), + "n_periods": str(len(periods)), + "n_frequencies": str(len(frequencies)), + "sort_order": "event", + "created_at": datetime.now(UTC).isoformat(timespec="seconds"), + "creator": getpass.getuser(), + "source": "", + "imdb_version": _imdb_version(), + **{key: "" for key in schema.METADATA_TABLES.values()}, + } + meta.update({key: str(value) for key, value in (db_meta or {}).items()}) + db.set_db_meta(meta) + return db + + def set_db_meta(self, values: Mapping[str, object]) -> None: + """Insert or replace ``db_meta`` entries. + + Parameters + ---------- + values : mapping + Keys and values to write. Values are stored as strings. + """ + self._require_write() + frame = pd.DataFrame( + { + "key": list(values), + "value": [str(value) for value in values.values()], + } + ) + with self._temp_frames({"_meta": frame}): + self.conn.execute( + "INSERT OR REPLACE INTO db_meta SELECT key, value FROM _meta" + ) + self._invalidate() + + def _insert(self, table: str, df: pd.DataFrame) -> None: + """Insert a DataFrame whose columns match the table's, in order. + + Parameters + ---------- + table : str + Target table. + df : pandas.DataFrame + Rows to insert. + """ + if df.empty: + return + with self._temp_frames({"_rows": df}): + self.conn.execute(f"INSERT INTO {table} SELECT * FROM _rows") + + # ------------------------------------------------------------------ + # writing + # ------------------------------------------------------------------ + + def _prepare( + self, df: pd.DataFrame, table: str, required: Sequence[str] + ) -> pd.DataFrame: + """Validate and order a DataFrame against a table's columns. + + Parameters + ---------- + df : pandas.DataFrame + Rows to write. + table : str + Target table. + required : sequence of str + Columns that must be present. + + Returns + ------- + pandas.DataFrame + The rows, with missing columns added as NULL and ordered to match + the table. + """ + columns = self._table_columns(table) + missing = [name for name in required if name not in df.columns] + if missing: + raise ValueError(f"{table} rows are missing columns {missing}") + unknown = [name for name in df.columns if name not in columns] + if unknown: + raise ValueError( + f"columns {unknown} are not in {table}; extra fields belong in metadata" + ) + df = df.copy() + if "metadata" in columns: + df["metadata"] = self._pack_metadata(df.get("metadata"), table) + for name in columns: + if name not in df.columns: + df[name] = None + return df[columns] + + def _pack_metadata(self, values: pd.Series | None, table: str) -> pd.Series | None: + """Serialise a metadata column to JSON and reconcile its declared keys. + + The first write of a table's metadata declares the permitted keys in + ``db_meta``; later writes are validated against that declaration. + + Parameters + ---------- + values : pandas.Series or None + Metadata as dicts or JSON strings. + table : str + Table being written, used to find the ``db_meta`` key. + + Returns + ------- + pandas.Series or None + JSON strings, or ``None`` if there was nothing to pack. + """ + if values is None: + return None + packed, seen = [], set() + for value in values: + if value is None or (isinstance(value, float) and np.isnan(value)): + packed.append(None) + continue + if isinstance(value, str): + value = json.loads(value) + if not isinstance(value, dict): + raise TypeError( + f"{table}.metadata must hold dicts or JSON objects, got {type(value)}" + ) + seen.update(value) + packed.append(json.dumps(value, sort_keys=True, default=str)) + + meta_key = schema.METADATA_TABLES[table] + declared = {k for k in self.db_meta.get(meta_key, "").split(",") if k} + if declared: + undeclared = seen - declared + if undeclared: + raise ValueError( + f"{table}.metadata keys {sorted(undeclared)} are not declared in " + f"db_meta.{meta_key} ({sorted(declared)})" + ) + elif seen: + self.set_db_meta({meta_key: ",".join(sorted(seen))}) + return pd.Series(packed, index=values.index, dtype=object) + + def _next_int_ids(self, table: str, column: str, count: int) -> np.ndarray: + """Allocate contiguous integer surrogates for a dimension table. + + Parameters + ---------- + table : str + Table to extend. + column : str + Surrogate key column. + count : int + How many ids to allocate. + + Returns + ------- + numpy.ndarray + The new ids. + """ + current = self._scalar(f"SELECT max({column}) FROM {table}") + start = 1 if current is None else int(current) + 1 + return np.arange(start, start + count, dtype=np.int64) + + def add_events(self, df: pd.DataFrame) -> None: + """Insert rows into ``events``. + + Parameters + ---------- + df : pandas.DataFrame + Requires ``event_id``. Other columns must be ``events`` columns; + extra fields go in a ``metadata`` dict column. ``event_int_id`` is + assigned here and must not be supplied. + """ + self._require_write() + df = self._prepare(df, "events", ["event_id"]).drop(columns="event_int_id") + df["event_id"] = df["event_id"].astype(str) + self._reject_existing(df["event_id"], self.event_ids, "event_id") + df.insert( + 0, "event_int_id", self._next_int_ids("events", "event_int_id", len(df)) + ) + self._insert("events", df) + self._invalidate() + logger.info("inserted %d events", len(df)) + + def add_realisations(self, df: pd.DataFrame) -> None: + """Insert rows into ``realisations``. + + Parameters + ---------- + df : pandas.DataFrame + Requires ``rel_id`` and ``event_id``. ``rel_int_id`` is assigned + here and must not be supplied. + """ + self._require_write() + df = df.copy() + if "event_id" not in df.columns: + raise ValueError("realisation rows are missing column ['event_id']") + event_int_id = self._resolve(df.pop("event_id"), self.event_ids, "event_id") + df["event_int_id"] = event_int_id + df = self._prepare(df, "realisations", ["rel_id"]).drop(columns="rel_int_id") + df["rel_id"] = df["rel_id"].astype(str) + self._reject_existing(df["rel_id"], self.rel_ids, "rel_id") + df.insert( + 0, "rel_int_id", self._next_int_ids("realisations", "rel_int_id", len(df)) + ) + self._insert("realisations", df) + self._invalidate() + logger.info("inserted %d realisations", len(df)) + + def add_sites(self, df: pd.DataFrame) -> None: + """Insert rows into ``sites``. + + Parameters + ---------- + df : pandas.DataFrame + Requires ``site_id``, ``lat`` and ``lon``. ``site_int_id`` is + assigned here and must not be supplied. + """ + self._require_write() + df = self._prepare(df, "sites", ["site_id", "lat", "lon"]).drop( + columns="site_int_id" + ) + df["site_id"] = df["site_id"].astype(str) + self._reject_existing(df["site_id"], self.site_ids, "site_id") + df.insert(0, "site_int_id", self._next_int_ids("sites", "site_int_id", len(df))) + self._insert("sites", df) + self._invalidate() + logger.info("inserted %d sites", len(df)) + + def add_site_event(self, df: pd.DataFrame) -> None: + """Insert rows into ``site_event``. + + Parameters + ---------- + df : pandas.DataFrame + Requires ``site_id`` and ``event_id``, which are resolved to their + integer surrogates here. + """ + self._require_write() + df = df.copy() + for column in ("site_id", "event_id"): + if column not in df.columns: + raise ValueError(f"site_event rows are missing column ['{column}']") + df["site_int_id"] = self._resolve(df.pop("site_id"), self.site_ids, "site_id") + df["event_int_id"] = self._resolve( + df.pop("event_id"), self.event_ids, "event_id" + ) + df = self._prepare(df, "site_event", ["site_int_id", "event_int_id"]) + self._insert("site_event", df) + logger.info("inserted %d site-event pairs", len(df)) + + def add_records(self, df: pd.DataFrame) -> np.ndarray: + """Insert records and their IM values. + + Writes ``records`` plus whichever of ``psa_ims``, ``fas_ims`` and + ``scalars_ims`` the input covers. ``CAV``, ``AI``, ``Ds575`` and + ``Ds595`` are set to NULL on ``rotd*`` rows. + + Parameters + ---------- + df : pandas.DataFrame + Requires ``rel_id``, ``site_id`` and ``component``. May carry a + ``pSA`` column of arrays, a ``FAS`` column of arrays, and any of the + seven scalar IM columns. + + Returns + ------- + numpy.ndarray + The ``record_id`` assigned to each row, in input order. + """ + self._require_write() + df = df.copy() + for column in ("rel_id", "site_id", "component"): + if column not in df.columns: + raise ValueError(f"record rows are missing column ['{column}']") + + component = df["component"].astype(str) + unknown = set(component.unique()) - set(self.components) + if unknown: + raise ValueError( + f"components {sorted(unknown)} are not declared in db_meta.components " + f"({self.components})" + ) + + rel_int_id = self._resolve(df["rel_id"], self.rel_ids, "rel_id") + site_int_id = self._resolve(df["site_id"], self.site_ids, "site_id") + event_int_id = self.rel_to_event.loc[rel_int_id].to_numpy(dtype=np.int64) + + known = {"rel_id", "site_id", "component", "pSA", "FAS", *schema.SCALAR_IMS} + unknown_columns = [name for name in df.columns if name not in known] + if unknown_columns: + raise ValueError( + f"columns {unknown_columns} are not record or IM columns; expected " + f"rel_id, site_id, component, pSA, FAS or one of {list(schema.SCALAR_IMS)}" + ) + + record_id = ( + self.conn.execute( + "SELECT nextval('record_id_seq') AS record_id FROM range(?)", [len(df)] + ) + .df()["record_id"] + .to_numpy(dtype=np.int64) + ) + + with self._transaction(): + self._write_record_tables( + df, record_id, event_int_id, rel_int_id, site_int_id, component + ) + + logger.info("inserted %d records", len(df)) + return record_id + + def _write_record_tables( + self, + df: pd.DataFrame, + record_id: np.ndarray, + event_int_id: np.ndarray, + rel_int_id: np.ndarray, + site_int_id: np.ndarray, + component: pd.Series, + ) -> None: + """Insert one batch into ``records`` and the IM tables it covers. + + Parameters + ---------- + df : pandas.DataFrame + The validated input rows. + record_id : numpy.ndarray + Allocated record ids. + event_int_id : numpy.ndarray + Event surrogate of each row. + rel_int_id : numpy.ndarray + Realisation surrogate of each row. + site_int_id : numpy.ndarray + Site surrogate of each row. + component : pandas.Series + Component of each row. + """ + self._insert( + "records", + pd.DataFrame( + { + "record_id": record_id, + "event_int_id": event_int_id, + "rel_int_id": rel_int_id, + "site_int_id": site_int_id, + "component": component.to_numpy(dtype=object), + } + ), + ) + + for column, table, grid in ( + ("pSA", "psa_ims", self.periods), + ("FAS", "fas_ims", self.frequencies), + ): + if column not in df.columns: + continue + # a row with no array simply gets no row in this IM table, which is how + # the schema expresses per-record IM coverage + present = df[column].notna().to_numpy() + if not present.any(): + continue + arrays = [ + np.asarray(value, dtype=np.float32) for value in df.loc[present, column] + ] + bad = {array.size for array in arrays} - {len(grid)} + if bad: + raise ValueError( + f"{column} arrays have lengths {sorted(bad)} but this database's " + f"grid has {len(grid)} entries" + ) + self._insert( + table, + pd.DataFrame( + { + "record_id": record_id[present], + column: pd.Series(arrays, dtype=object), + } + ), + ) + + present = [im for im in schema.SCALAR_IMS if im in df.columns] + if present: + scalars = pd.DataFrame({"record_id": record_id}) + is_rotd = component.str.startswith("rotd").to_numpy() + for im in schema.SCALAR_IMS: + values = ( + pd.to_numeric(df[im], errors="raise").astype("float32") + if im in present + else pd.Series(np.nan, index=df.index, dtype="float32") + ) + values = values.to_numpy(dtype=np.float32, copy=True) + if im in schema.ROTD_UNDEFINED: + values[is_rotd] = np.nan + scalars[im] = values + self._insert("scalars_ims", scalars) + + def _reject_existing(self, ids: pd.Series, mapping: pd.Series, what: str) -> None: + """Raise if any of these string ids is already in the database. + + Parameters + ---------- + ids : pandas.Series + String ids about to be written. + mapping : pandas.Series + Existing ids, as the index. + what : str + Name used in the error message. + """ + duplicated = ids[ids.duplicated()].unique() + if len(duplicated): + raise ValueError( + f"duplicate {what} in the input: {sorted(duplicated)[:10]}" + ) + existing = ids[ids.isin(mapping.index)].unique() + if len(existing): + raise ValueError( + f"{what} already in the database: {sorted(existing)[:10]}; " + "use delete_event() to re-ingest" + ) + + def delete_event(self, event_id: str) -> None: + """Delete an event and everything derived from it. + + Removes the event's rows from ``psa_ims``, ``fas_ims``, ``scalars_ims``, + ``records``, ``site_event``, ``realisations`` and ``events``, so a + re-ingest is a delete followed by the same sequence of ``add_*`` calls. + Sites are shared across events and are never deleted. + + Parameters + ---------- + event_id : str + The event to delete. + """ + self._require_write() + event_int_id = int(self._resolve([event_id], self.event_ids, "event_id")[0]) + for table in ("psa_ims", "fas_ims", "scalars_ims"): + self.conn.execute( + f"DELETE FROM {table} WHERE record_id IN " + "(SELECT record_id FROM records WHERE event_int_id = ?)", + [event_int_id], + ) + for table in ("records", "site_event", "realisations", "events"): + self.conn.execute( + f"DELETE FROM {table} WHERE event_int_id = ?", + [event_int_id], + ) + self._invalidate() + logger.info("deleted event %s", event_id) + + # ------------------------------------------------------------------ + # validation + # ------------------------------------------------------------------ + + def validate(self) -> list[str]: + """Check the invariants the large tables do not enforce as constraints. + + Works read-only. Returns problems rather than raising, so an ingest + script can report all of them at once. + + Returns + ------- + list of str + One line per problem found. Empty when the database is consistent. + """ + problems = [] + + def count(query: str) -> int: + return int(self._scalar(query)) + + for column, table, key in ( + ("rel_int_id", "realisations", "rel_int_id"), + ("site_int_id", "sites", "site_int_id"), + ("event_int_id", "events", "event_int_id"), + ): + n = count( + f"SELECT count(*) FROM records r " + f"LEFT JOIN {table} d ON d.{key} = r.{column} WHERE d.{key} IS NULL" + ) + if n: + problems.append(f"records: {n} rows with an orphan {column}") + + for column, table, key in ( + ("site_int_id", "sites", "site_int_id"), + ("event_int_id", "events", "event_int_id"), + ): + n = count( + f"SELECT count(*) FROM site_event se " + f"LEFT JOIN {table} d ON d.{key} = se.{column} WHERE d.{key} IS NULL" + ) + if n: + problems.append(f"site_event: {n} rows with an orphan {column}") + + n = count( + "SELECT count(*) FROM records r JOIN realisations rl USING (rel_int_id) " + "WHERE r.event_int_id != rl.event_int_id" + ) + if n: + problems.append( + f"records: {n} rows whose event_int_id disagrees with their realisation" + ) + + for table in schema.IM_TABLES: + n = count( + f"SELECT count(*) FROM {table} im " + "LEFT JOIN records r USING (record_id) WHERE r.record_id IS NULL" + ) + if n: + problems.append(f"{table}: {n} rows with no matching record") + + for table, column, grid in ( + ("psa_ims", "pSA", "periods"), + ("fas_ims", "FAS", "frequencies"), + ): + n = count( + f"SELECT count(*) FROM {table} " + f"WHERE len({column}) != (SELECT count(*) FROM {grid})" + ) + if n: + problems.append(f"{table}: {n} rows whose {column} length is wrong") + + n = count( + "SELECT count(*) FROM (SELECT 1 FROM records " + "GROUP BY rel_int_id, site_int_id, component HAVING count(*) > 1)" + ) + if n: + problems.append( + f"records: {n} duplicated (rel_int_id, site_int_id, component) keys" + ) + + n = count( + "SELECT count(*) FROM (SELECT 1 FROM site_event " + "GROUP BY site_int_id, event_int_id HAVING count(*) > 1)" + ) + if n: + problems.append( + f"site_event: {n} duplicated (site_int_id, event_int_id) keys" + ) + + for table in schema.IM_TABLES: + n = count( + f"SELECT count(*) FROM (SELECT 1 FROM {table} " + "GROUP BY record_id HAVING count(*) > 1)" + ) + if n: + problems.append(f"{table}: {n} record_ids with more than one row") + + declared = set(self.components) + found = { + row[0] + for row in self.conn.execute( + "SELECT DISTINCT component FROM records" + ).fetchall() + } + if found - declared: + problems.append( + f"records: components {sorted(found - declared)} are not in " + f"db_meta.components ({sorted(declared)})" + ) + + for table, meta_key in schema.METADATA_TABLES.items(): + allowed = {k for k in self.db_meta.get(meta_key, "").split(",") if k} + keys = { + row[0] + for row in self.conn.execute( + f"SELECT DISTINCT unnest(json_keys(metadata)) FROM {table} " + "WHERE metadata IS NOT NULL" + ).fetchall() + } + if keys - allowed: + problems.append( + f"{table}.metadata: keys {sorted(keys - allowed)} are not declared " + f"in db_meta.{meta_key}" + ) + + for key, table in (("n_periods", "periods"), ("n_frequencies", "frequencies")): + declared_n = self.db_meta.get(key) + actual = count(f"SELECT count(*) FROM {table}") + if declared_n is not None and int(declared_n) != actual: + problems.append( + f"db_meta.{key} is {declared_n} but {table} has {actual} rows" + ) + + return problems + + def finalise(self) -> None: + """Refresh the derived ``db_meta`` counts, validate, and checkpoint. + + Call once after the last write. + """ + self._require_write() + self.set_db_meta( + { + "n_periods": len(self.periods), + "n_frequencies": len(self.frequencies), + } + ) + problems = self.validate() + if problems: + raise ValueError("database is inconsistent:\n " + "\n ".join(problems)) + self.conn.execute("CHECKPOINT") + logger.info("finalised %s", self.db_path) + + +def _expand_metadata(df: pd.DataFrame) -> pd.DataFrame: + """Expand a JSON ``metadata`` column into columns. + + Parameters + ---------- + df : pandas.DataFrame + Frame with a ``metadata`` column of JSON strings. + + Returns + ------- + pandas.DataFrame + The frame with ``metadata`` replaced by its fields. + """ + if "metadata" not in df.columns: + return df + parsed = [ + json.loads(value) if isinstance(value, str) else {} for value in df["metadata"] + ] + expanded = pd.json_normalize(parsed) + expanded.index = df.index + return pd.concat([df.drop(columns="metadata"), expanded], axis=1) diff --git a/imdb/schema.py b/imdb/schema.py new file mode 100644 index 0000000..93c7118 --- /dev/null +++ b/imdb/schema.py @@ -0,0 +1,220 @@ +""" +Schema definition for the intensity measure database. + +This module holds facts only: the DDL, the fixed vocabularies and the +documentation text written into every database. The DDL is the single source +of truth for table and column names; nothing else in the package hardcodes a +column list. +""" + +SCHEMA_VERSION = "0" + +COMPONENTS = ("000", "090", "ver", "geom", "rotd0", "rotd50", "rotd100") +"""Ground-motion components, following ``IM_calculation``.""" + +TECT_TYPES = ( + "ACTIVE_SHALLOW", + "VOLCANIC", + "SUBDUCTION_INTERFACE", + "SUBDUCTION_SLAB", +) +"""Tectonic types, following the ``qcore``/``workflow`` ``TectType`` vocabulary.""" + +SCALAR_IMS = ("PGA", "PGV", "PGD", "CAV", "AI", "Ds575", "Ds595") +"""Scalar intensity measures, in ``scalars_ims`` column order.""" + +ROTD_UNDEFINED = frozenset({"CAV", "AI", "Ds575", "Ds595"}) +"""Scalar IMs that are undefined for ``rotd*`` components and stored as NULL.""" + +IM_UNITS = { + "pSA": "g", + "FAS": "g.s", + "PGA": "g", + "PGV": "cm/s", + "PGD": "cm", + "CAV": "m/s", + "AI": "m/s", + "Ds575": "s", + "Ds595": "s", +} +"""Linear physical unit of each intensity measure. Log is a read-time transform.""" + +METADATA_TABLES = { + "events": "event_metadata_keys", + "realisations": "rel_metadata_keys", + "sites": "site_metadata_keys", + "site_event": "site_event_metadata_keys", +} +"""Tables carrying a JSON ``metadata`` column, and the ``db_meta`` key declaring its keys.""" + +IM_TABLES = { + "psa_ims": "pSA", + "fas_ims": "FAS", + "scalars_ims": None, +} +"""IM tables, mapped to their array column where they have one.""" + +NOTES = { + "logical keys": ( + "site_event is keyed on (site_int_id, event_int_id); records on " + "(rel_int_id, site_int_id, component); each IM table on record_id. None of " + "these are declared as constraints. They are enforced by the writer and " + "checked by IMDB.validate()." + ), + "array indexing is 1-based": ( + "periods.period_index and frequencies.freq_index are 1-based, so pSA[period_index] " + "and FAS[freq_index] need no offset. len(pSA) equals the row count of periods and " + "len(FAS) the row count of frequencies, for every row." + ), + "identity and rebuild stability": ( + "event_id, rel_id and site_id are stable. The integer surrogates event_int_id, " + "rel_int_id, site_int_id and record_id are assigned at ingest and change on " + "rebuild. Nothing outside this database may reference them." + ), + "component vocabulary": ( + "000, 090, ver, geom, rotd0, rotd50, rotd100, following IM_calculation. This " + "database holds the subset listed in db_meta.components." + ), + "rotd scalars are undefined": ( + "CAV, AI, Ds575 and Ds595 are undefined for rotd0, rotd50 and rotd100 and are " + "stored as NULL for those components. PGA, PGV and PGD are populated for every " + "component." + ), + "units": ( + "IM units are one row each in im_units, and are linear physical units. Elsewhere: " + "distances km, vs30 m/s, z1p0 and z2p5 km, depths km, angles degrees, coordinates " + "WGS84." + ), + "metadata columns are JSON": ( + "events, realisations, sites and site_event each carry a metadata VARCHAR holding " + "a JSON object. Read with json_extract_string(metadata, '$.key'). Permitted keys " + "are declared in db_meta. A predicate on a metadata field cannot use zone-map " + "pruning." + ), + "distances are event level": ( + "rrup, rjb, rx and ry are measured to the rupture surface and are shared across " + "all realisations of an event. Hypocentral and epicentral distance are not stored; " + "compute them from realisations.hypo_* and sites.lat/lon." + ), + "synthetic realisations": ( + "Every event has at least one realisation. A dataset with no realisation concept " + "gets exactly one per event, with rel_id equal to event_id." + ), + "record_id is file-local": ( + "record_id comes from a sequence and shifts on rebuild. External references must " + "cite (rel_id, site_id, component)." + ), + "physical sort order": ( + "Rows are written in the order recorded by db_meta.sort_order. Filters on the sort " + "key prune row groups; filters on anything else do not." + ), + "provenance": ( + "db_meta records who built this database, when, from what source, and with which " + "version of the imdb library." + ), +} +"""Self-documenting notes written into the ``notes`` table by :meth:`imdb.IMDB.create`.""" + +DDL = """ +-- ---------- documentation ---------- + +CREATE TABLE db_meta (key VARCHAR PRIMARY KEY, value VARCHAR NOT NULL); +CREATE TABLE notes (topic VARCHAR PRIMARY KEY, note VARCHAR NOT NULL); +CREATE TABLE im_units (im VARCHAR PRIMARY KEY, unit VARCHAR NOT NULL); + +-- ---------- IM vocabulary (1-based, matching DuckDB list indexing) ---------- + +CREATE TABLE periods ( + period_index INTEGER PRIMARY KEY, + period DOUBLE NOT NULL UNIQUE -- seconds +); + +CREATE TABLE frequencies ( + freq_index INTEGER PRIMARY KEY, + frequency DOUBLE NOT NULL UNIQUE -- Hz +); + +-- ---------- dimensions ---------- + +CREATE TYPE tect_type_t AS ENUM ( + 'ACTIVE_SHALLOW', 'VOLCANIC', 'SUBDUCTION_INTERFACE', 'SUBDUCTION_SLAB' +); + +CREATE TABLE events ( + event_int_id INTEGER PRIMARY KEY, + event_id VARCHAR NOT NULL UNIQUE, + magnitude FLOAT, + tect_type tect_type_t, + dip FLOAT, + dip_dir FLOAT, + dtop FLOAT, + dbottom FLOAT, + length FLOAT, + source_wkt VARCHAR, -- rupture surface + trace_wkt VARCHAR, -- surface trace + domain_wkt VARCHAR, -- simulation domain + metadata VARCHAR -- JSON: fault_type, sim_type, plane_count, ... +); + +CREATE TABLE realisations ( + rel_int_id INTEGER PRIMARY KEY, + rel_id VARCHAR NOT NULL UNIQUE, + event_int_id INTEGER NOT NULL REFERENCES events(event_int_id), + magnitude FLOAT, + rake FLOAT, + hypo_lat FLOAT, + hypo_lon FLOAT, + hypo_depth FLOAT, + metadata VARCHAR -- JSON: solver, shypo, dhypo, ... +); + +CREATE TABLE sites ( + site_int_id INTEGER PRIMARY KEY, + site_id VARCHAR NOT NULL UNIQUE, + lat FLOAT NOT NULL, + lon FLOAT NOT NULL, + vs30 FLOAT, -- m/s + z1p0 FLOAT, -- km + z2p5 FLOAT, -- km + metadata VARCHAR -- JSON: elevation, basin, grid_level, ... +); + +-- ---------- large tables: no PRIMARY KEY, UNIQUE or FOREIGN KEY ---------- + +CREATE TABLE site_event ( + site_int_id INTEGER NOT NULL, + event_int_id INTEGER NOT NULL, + rrup FLOAT, -- km + rjb FLOAT, + rx FLOAT, + ry FLOAT, + metadata VARCHAR -- JSON +); +-- logical key (site_int_id, event_int_id) + +CREATE SEQUENCE record_id_seq START 1; + +CREATE TABLE records ( + record_id BIGINT DEFAULT nextval('record_id_seq'), + event_int_id INTEGER NOT NULL, -- derived from rel_int_id by the writer + rel_int_id INTEGER NOT NULL, + site_int_id INTEGER NOT NULL, + component VARCHAR NOT NULL +); +-- logical key (rel_int_id, site_int_id, component) + +CREATE TABLE psa_ims (record_id BIGINT NOT NULL, pSA FLOAT[]); -- periods.period_index +CREATE TABLE fas_ims (record_id BIGINT NOT NULL, FAS FLOAT[]); -- frequencies.freq_index + +CREATE TABLE scalars_ims ( + record_id BIGINT NOT NULL, + PGA FLOAT, + PGV FLOAT, + PGD FLOAT, + CAV FLOAT, -- NULL for rotd components + AI FLOAT, -- NULL for rotd components + Ds575 FLOAT, -- NULL for rotd components + Ds595 FLOAT -- NULL for rotd components +); +""" +"""Full schema DDL, executed as one script by :meth:`imdb.IMDB.create`.""" diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..ed67127 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,95 @@ +[build-system] +requires = ["setuptools", "setuptools-scm"] +build-backend = "setuptools.build_meta" + +[project] +name = "imdb" +authors = [{name="ucgmsim"}] +description = "A library for reading and writing intensity measure databases" +readme = "README.md" +requires-python = ">=3.12" +dynamic = ["version"] +dependencies = [ + "duckdb>=1.5", + "numpy>=2", + "pandas>=3", +] + +[dependency-groups] +test = [ + "pytest", + "pytest-cov", + "coverage[toml]", +] +types = [ + "pandas-stubs", + "ty", +] +dev = ["ruff", "deptry", "ty", "numpydoc"] + +[tool.setuptools_scm] + +[tool.setuptools.packages.find] +include = ["imdb*"] + +[tool.ruff.lint] +extend-select = [ + # isort imports + "I", + # Use r'\s+' rather than '\s+' + "W605", + # All the naming errors, like using camel case for function names. + "N", + # Missing docstrings in classes, methods, and functions + "D101", + "D102", + "D103", + "D105", + "D107", + # Use f-string instead of a format call + "UP032", + # Standard library import is deprecated + "UP035", + # Missing function argument type-annotation + "ANN001", + # Using except without specifying an exception type to catch + "BLE001" +] +ignore = ["D104"] + +[tool.ruff.lint.pydocstyle] +convention = "numpy" + +[tool.ruff.lint.isort] +known-first-party = ["imdb", "qcore", "IM", "workflow", "source_modelling"] + +[tool.ruff.lint.per-file-ignores] +# Ignore no docstring in __init__.py +"__init__.py" = ["D104"] +# Ignore docstring errors and fixture annotations in tests folder +"tests/**.py" = ["D", "ANN001"] + +[tool.numpydoc_validation] +checks = [ + "GL05", + "GL08", + "GL10", + "PR01", + "PR02", + "PR03", + "PR04", + "PR05", + "PR06", + "PR07", + "RT01", + "RT02", + "RT03", + "RT04", + "YD01", +] +# remember to use single quotes for regex in TOML +exclude = [ + '\.__repr__$', + '\.__exit__$', + '\.__enter__$', +] diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..d6a4efc --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,117 @@ +import numpy as np +import pandas as pd +import pytest + +from imdb import IMDB + +PERIODS = [0.01, 0.1, 1.0, 3.0, 10.0] +FREQUENCIES = [0.1, 1.0, 10.0] +COMPONENTS = ["geom", "rotd50"] +EVENTS = ["ev1", "ev2"] +SITES = ["stnA", "stnB", "stnC"] +SCALARS = ["PGA", "PGV", "PGD", "CAV", "AI", "Ds575", "Ds595"] + + +def build_frames(): + """Deterministic input frames covering every table.""" + events = pd.DataFrame( + { + "event_id": EVENTS, + "magnitude": [7.1, 6.2], + "tect_type": ["SUBDUCTION_SLAB", "ACTIVE_SHALLOW"], + "dtop": [30.0, 0.5], + "metadata": [ + {"fault_type": "DS_POINT_SOURCE"}, + {"fault_type": "NORMAL_FAULTING"}, + ], + } + ) + rels = pd.DataFrame( + { + "rel_id": [f"{e}_REL{i:02d}" for e in EVENTS for i in (1, 2)], + "event_id": [e for e in EVENTS for _ in (1, 2)], + "magnitude": [7.1, 7.12, 6.2, 6.18], + "rake": [90.0, 88.0, -90.0, -92.0], + "hypo_lat": [-43.5, -43.6, -41.2, -41.3], + "hypo_lon": [172.6, 172.7, 174.8, 174.9], + "hypo_depth": [40.0, 42.0, 8.0, 9.0], + "metadata": [{"solver": "emod3d"}] * 4, + } + ) + sites = pd.DataFrame( + { + "site_id": SITES, + "lat": [-43.5, -43.6, -41.3], + "lon": [172.6, 172.7, 174.8], + "vs30": [300.0, 500.0, 250.0], + "z1p0": [0.3, 0.1, 0.5], + "metadata": [{"elevation": 10.0}, {"elevation": 55.0}, {"elevation": 3.0}], + } + ) + site_event = pd.DataFrame( + { + "site_id": [s for s in SITES for _ in EVENTS], + "event_id": EVENTS * len(SITES), + "rrup": [10.0, 300.0, 25.0, 280.0, 400.0, 5.0], + "rjb": [8.0, 295.0, 22.0, 275.0, 395.0, 3.0], + } + ) + + rows = [] + for rel_id in rels["rel_id"]: + for site_id in SITES: + for component in COMPONENTS: + base = float(len(rows) + 1) + row = { + "rel_id": rel_id, + "site_id": site_id, + "component": component, + "pSA": np.array( + [base + i / 8 for i in range(1, len(PERIODS) + 1)], + dtype=np.float32, + ), + "FAS": np.array( + [base * 2 + i / 8 for i in range(1, len(FREQUENCIES) + 1)], + dtype=np.float32, + ), + } + for k, im in enumerate(SCALARS): + row[im] = base + (k + 1) / 16 + rows.append(row) + records = pd.DataFrame(rows) + return events, rels, sites, site_event, records + + +@pytest.fixture +def db_path(tmp_path): + """Path of a small, fully populated database.""" + path = tmp_path / "test_ims.duckdb" + events, rels, sites, site_event, records = build_frames() + with IMDB.create( + path, + periods=PERIODS, + frequencies=FREQUENCIES, + components=COMPONENTS, + db_meta={"dataset_description": "test fixture", "source": "conftest"}, + ) as db: + db.add_events(events) + db.add_realisations(rels) + db.add_sites(sites) + db.add_site_event(site_event) + db.add_records(records) + db.finalise() + return path + + +@pytest.fixture +def db(db_path): + """The fixture database, open read-only.""" + with IMDB(db_path) as handle: + yield handle + + +@pytest.fixture +def wdb(db_path): + """The fixture database, open for writing.""" + with IMDB(db_path, read_only=False) as handle: + yield handle diff --git a/tests/test_imdb.py b/tests/test_imdb.py new file mode 100644 index 0000000..822ce05 --- /dev/null +++ b/tests/test_imdb.py @@ -0,0 +1,311 @@ +import numpy as np +import pandas as pd +import pytest + +from imdb import IMDB, schema + +from .conftest import ( + COMPONENTS, + EVENTS, + FREQUENCIES, + PERIODS, + SCALARS, + SITES, + build_frames, +) + +N_RECORDS = 4 * len(SITES) * len(COMPONENTS) + + +def test_create_populates_documentation(db): + assert db.db_meta["schema_version"] == schema.SCHEMA_VERSION + assert db.db_meta["dataset_description"] == "test fixture" + assert db.db_meta["n_periods"] == str(len(PERIODS)) + assert db.components == COMPONENTS + assert db.notes == schema.NOTES + assert db.im_units == schema.IM_UNITS + assert db.periods.tolist() == PERIODS + assert db.frequencies.tolist() == FREQUENCIES + assert db.periods.index.tolist() == [1, 2, 3, 4, 5] + + +def test_dimension_reads(db): + assert sorted(db.get_events()["event_id"]) == EVENTS + assert len(db.get_realisations()) == 4 + assert sorted(db.get_sites()["site_id"]) == SITES + assert len(db.get_site_event()) == len(SITES) * len(EVENTS) + assert len(db.get_records()) == N_RECORDS + + +def test_metadata_expansion(db): + events = db.get_events(expand_metadata=True) + assert "metadata" not in events.columns + assert set(events["fault_type"]) == {"DS_POINT_SOURCE", "NORMAL_FAULTING"} + assert db.db_meta["event_metadata_keys"] == "fault_type" + assert db.db_meta["site_metadata_keys"] == "elevation" + + +def test_psa_and_fas_round_trip(db): + _, _, _, _, records = build_frames() + order = db.get_records().sort_index() + + psa = db.get_psa() + assert psa.columns.tolist() == PERIODS + fas = db.get_fas() + assert fas.columns.tolist() == FREQUENCIES + + # rebuild the input keyed the same way the database is + key = ["rel_id", "site_id", "component"] + expected = records.set_index(key) + got = order.reset_index().set_index(key) + + for rid, k in zip(got["record_id"], got.index, strict=True): + np.testing.assert_array_equal( + psa.loc[rid].to_numpy(dtype=np.float32), expected.loc[k, "pSA"] + ) + np.testing.assert_array_equal( + fas.loc[rid].to_numpy(dtype=np.float32), expected.loc[k, "FAS"] + ) + + +def test_scalar_round_trip_and_rotd_nulls(db): + _, _, _, _, records = build_frames() + scalars = db.get_scalars() + assert scalars.columns.tolist() == list(schema.SCALAR_IMS) + + recs = db.get_records() + joined = recs.join(scalars) + expected = records.set_index(["rel_id", "site_id", "component"]) + + for record_id, row in joined.iterrows(): + key = (row["rel_id"], row["site_id"], row["component"]) + for im in SCALARS: + if im in schema.ROTD_UNDEFINED and row["component"].startswith("rotd"): + assert pd.isna(row[im]), (record_id, im) + else: + assert row[im] == pytest.approx(expected.loc[key, im], rel=1e-6) + + +def test_subset_of_periods(db): + full = db.get_psa() + subset = db.get_psa(periods=[1.0, 0.01]) + assert subset.columns.tolist() == [1.0, 0.01] + pd.testing.assert_series_equal(subset[1.0], full[1.0], check_names=False) + + +def test_period_written_with_p(db): + df = db.get_im_df(["pSA_0p1", "pSA_10p0"]) + assert df.columns.tolist() == ["pSA_0p1", "pSA_10p0"] + np.testing.assert_allclose(df["pSA_0p1"], db.get_psa(periods=[0.1])[0.1]) + + +def test_get_im_df_matches_typed_calls(db): + names = ["PGA", "pSA_1.0", "FAS_10.0", "Ds595"] + df = db.get_im_df(names) + assert df.columns.tolist() == names + assert len(df) == N_RECORDS + np.testing.assert_allclose(df["PGA"], db.get_scalars(ims=["PGA"])["PGA"]) + np.testing.assert_allclose(df["pSA_1.0"], db.get_psa(periods=[1.0])[1.0]) + np.testing.assert_allclose(df["FAS_10.0"], db.get_fas(frequencies=[10.0])[10.0]) + + +@pytest.mark.parametrize( + ("filters", "expected"), + [ + ({}, N_RECORDS), + ({"events": ["ev1"]}, 2 * len(SITES) * len(COMPONENTS)), + ({"sites": ["stnA"]}, 4 * len(COMPONENTS)), + ({"rels": ["ev1_REL01"]}, len(SITES) * len(COMPONENTS)), + ({"component": "rotd50"}, N_RECORDS // 2), + ({"component": ["geom", "rotd50"]}, N_RECORDS), + ({"events": ["ev1"], "component": "geom"}, 2 * len(SITES)), + ({"max_rrup": 30.0}, 3 * 2 * len(COMPONENTS)), + ], +) +def test_filters(db, filters, expected): + assert len(db.get_records(**filters)) == expected + assert len(db.get_psa(periods=[1.0], **filters)) == expected + assert len(db.get_scalars(ims=["PGA"], **filters)) == expected + assert len(db.get_im_df(["PGA", "pSA_1.0"], **filters)) == expected + + +def test_record_ids_filter(db): + wanted = db.get_records().index[:5].to_numpy() + got = db.get_psa(periods=[1.0], record_ids=wanted) + assert sorted(got.index) == sorted(wanted) + + +def test_filters_match_raw_sql(db): + got = db.get_records(events=["ev2"], component="geom").index.tolist() + expected = [ + row[0] + for row in db.sql( + """ + SELECT r.record_id FROM records r + JOIN events e ON e.event_int_id = r.event_int_id + WHERE e.event_id = 'ev2' AND r.component = 'geom' + """ + ).fetchall() + ] + assert sorted(got) == sorted(expected) + + +def test_site_event_filters(db): + assert len(db.get_site_event(sites=["stnA"])) == 2 + assert len(db.get_site_event(events=["ev1"])) == 3 + assert len(db.get_site_event(max_rrup=30.0)) == 3 + expanded = db.get_site_event(sites=["stnA"], expand_metadata=True) + assert "metadata" not in expanded.columns + + +def test_bad_requests_raise(db): + with pytest.raises(KeyError, match="not on this database's grid"): + db.get_psa(periods=[2.5]) + with pytest.raises(ValueError, match="cannot parse IM name"): + db.get_im_df(["SA_1.0"]) + with pytest.raises(ValueError, match="not in this database"): + db.get_records(component="000") + with pytest.raises(ValueError, match="unknown scalar IMs"): + db.get_scalars(ims=["MMI"]) + with pytest.raises(KeyError, match="unknown event_id"): + db.get_records(events=["nope"]) + + +def test_read_only_rejects_writes(db): + with pytest.raises(PermissionError): + db.set_db_meta({"source": "nope"}) + + +def test_validate_is_clean(db): + assert db.validate() == [] + + +def test_validate_finds_orphan_record(wdb): + wdb.conn.execute("INSERT INTO records VALUES (9999, 1, 999, 1, 'geom')") + problems = wdb.validate() + assert any("orphan rel_int_id" in p for p in problems) + + +def test_validate_finds_wrong_array_length(wdb): + rid = int(wdb.get_records().index[0]) + wdb.conn.execute("UPDATE psa_ims SET pSA = [1.0, 2.0] WHERE record_id = ?", [rid]) + assert any("pSA length is wrong" in p for p in wdb.validate()) + + +def test_validate_finds_undeclared_metadata_key(wdb): + wdb.conn.execute( + """UPDATE sites SET metadata = '{"basin": "Canterbury"}' WHERE site_int_id = 1""" + ) + assert any( + "not declared in db_meta.site_metadata_keys" in p for p in wdb.validate() + ) + + +def test_validate_finds_duplicate_record_key(wdb): + wdb.conn.execute( + "INSERT INTO records SELECT 9999, event_int_id, rel_int_id, site_int_id, component " + "FROM records LIMIT 1" + ) + assert any("duplicated (rel_int_id" in p for p in wdb.validate()) + + +def test_undeclared_metadata_key_rejected_on_write(wdb): + sites = pd.DataFrame( + { + "site_id": ["stnD"], + "lat": [-42.0], + "lon": [173.0], + "metadata": [{"basin": "Canterbury"}], + } + ) + with pytest.raises(ValueError, match="not declared in db_meta.site_metadata_keys"): + wdb.add_sites(sites) + + +def test_unknown_column_rejected_on_write(wdb): + with pytest.raises(ValueError, match="not in sites"): + wdb.add_sites( + pd.DataFrame( + {"site_id": ["stnD"], "lat": [-42.0], "lon": [173.0], "vs20": [1.0]} + ) + ) + + +def test_duplicate_id_rejected_on_write(wdb): + with pytest.raises(ValueError, match="already in the database"): + wdb.add_sites( + pd.DataFrame({"site_id": ["stnA"], "lat": [-42.0], "lon": [173.0]}) + ) + + +def test_undeclared_component_rejected_on_write(wdb): + _, _, _, _, records = build_frames() + bad = records.head(1).copy() + bad["component"] = "000" + with pytest.raises(ValueError, match="not declared in db_meta.components"): + wdb.add_records(bad) + + +def test_wrong_array_length_rejected_on_write(wdb): + _, _, _, _, records = build_frames() + bad = records.head(1).copy() + bad["rel_id"] = "ev1_REL01" + bad["pSA"] = [np.ones(3, dtype=np.float32)] + with pytest.raises(ValueError, match="grid has 5 entries"): + wdb.add_records(bad) + + +def test_delete_event_then_readd(db_path): + events, rels, _, site_event, records = build_frames() + with IMDB(db_path, read_only=False) as db: + before = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"]).to_numpy() + old_ids = set(db.get_records(events=["ev1"]).index) + + db.delete_event("ev1") + assert db.validate() == [] + assert len(db.get_records()) == N_RECORDS // 2 + assert len(db.get_site_event()) == len(SITES) + + keep = rels["event_id"] == "ev1" + db.add_events(events[events["event_id"] == "ev1"]) + db.add_realisations(rels[keep]) + db.add_site_event(site_event[site_event["event_id"] == "ev1"]) + db.add_records(records[records["rel_id"].isin(rels.loc[keep, "rel_id"])]) + db.finalise() + + after = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"]).to_numpy() + new_ids = set(db.get_records(events=["ev1"]).index) + + np.testing.assert_allclose(np.sort(before, axis=0), np.sort(after, axis=0)) + assert not (old_ids & new_ids) + + +def test_records_without_all_im_tables(tmp_path): + path = tmp_path / "psa_only.duckdb" + events, rels, sites, site_event, records = build_frames() + with IMDB.create(path, periods=PERIODS, components=COMPONENTS) as db: + db.add_events(events) + db.add_realisations(rels) + db.add_sites(sites) + db.add_site_event(site_event) + db.add_records(records[["rel_id", "site_id", "component", "pSA"]]) + db.finalise() + + with IMDB(path) as db: + assert len(db.get_psa()) == N_RECORDS + assert db.get_scalars().empty + assert len(db.frequencies) == 0 + with pytest.raises(KeyError, match="no FAS grid"): + db.get_fas() + + +def test_create_refuses_to_clobber(db_path): + with pytest.raises(FileExistsError): + IMDB.create(db_path, periods=PERIODS) + IMDB.create(db_path, periods=PERIODS, overwrite=True).close() + + +def test_finalise_raises_on_inconsistency(wdb): + wdb.conn.execute("INSERT INTO records VALUES (9999, 1, 999, 1, 'geom')") + with pytest.raises(ValueError, match="database is inconsistent"): + wdb.finalise() diff --git a/uv.lock b/uv.lock new file mode 100644 index 0000000..2744694 --- /dev/null +++ b/uv.lock @@ -0,0 +1,960 @@ +version = 1 +revision = 3 +requires-python = ">=3.12" +resolution-markers = [ + "python_full_version >= '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.14' and sys_platform == 'emscripten'", + "python_full_version >= '3.14' and sys_platform != 'emscripten' and sys_platform != 'win32'", + "python_full_version < '3.14' and sys_platform == 'win32'", + "python_full_version < '3.14' and sys_platform == 'emscripten'", + "python_full_version < '3.14' and sys_platform != 'emscripten' and sys_platform != 'win32'", +] + +[[package]] +name = "alabaster" +version = "1.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a6/f8/d9c74d0daf3f742840fd818d69cfae176fa332022fd44e3469487d5a9420/alabaster-1.0.0.tar.gz", hash = "sha256:c00dca57bca26fa62a6d7d0a9fcce65f3e026e9bfe33e9c538fd3fbb2144fd9e", size = 24210, upload-time = "2024-07-26T18:15:03.762Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7e/b3/6b4067be973ae96ba0d615946e314c5ae35f9f993eca561b356540bb0c2b/alabaster-1.0.0-py3-none-any.whl", hash = "sha256:fc6786402dc3fcb2de3cabd5fe455a2db534b371124f1f21de8731783dec828b", size = 13929, upload-time = "2024-07-26T18:15:02.05Z" }, +] + +[[package]] +name = "babel" +version = "2.18.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/b2/51899539b6ceeeb420d40ed3cd4b7a40519404f9baf3d4ac99dc413a834b/babel-2.18.0.tar.gz", hash = "sha256:b80b99a14bd085fcacfa15c9165f651fbb3406e66cc603abf11c5750937c992d", size = 9959554, upload-time = "2026-02-01T12:30:56.078Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/77/f5/21d2de20e8b8b0408f0681956ca2c69f1320a3848ac50e6e7f39c6159675/babel-2.18.0-py3-none-any.whl", hash = "sha256:e2b422b277c2b9a9630c1d7903c2a00d0830c409c59ac8cae9081c92f1aeba35", size = 10196845, upload-time = "2026-02-01T12:30:53.445Z" }, +] + +[[package]] +name = "certifi" +version = "2026.7.22" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a3/c2/24167ea9858356b47a87a50d39908bfdb72ceeefe0041586e704e5376b3a/certifi-2026.7.22.tar.gz", hash = "sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55", size = 138112, upload-time = "2026-07-22T03:35:12.644Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0b/a7/71ac2cff56fec219ed242bb11b8efb69fcc4bec75db06fb7bfe35de520e6/certifi-2026.7.22-py3-none-any.whl", hash = "sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775", size = 136983, upload-time = "2026-07-22T03:35:11.276Z" }, +] + +[[package]] +name = "charset-normalizer" +version = "3.5.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e5/3f/143b048436775b0f76ac3eec145c019e8173ccc2885c8f20319b996d5e83/charset_normalizer-3.5.1.tar.gz", hash = "sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3", size = 171764, upload-time = "2026-08-15T08:20:44.807Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/30/27/78873dc8b6a56357517b74b6bb9568b80450e7bb4f6ef7e3fa9d22aa0bd7/charset_normalizer-3.5.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f", size = 344456, upload-time = "2026-08-15T08:17:10.072Z" }, + { url = "https://files.pythonhosted.org/packages/9a/4c/be49ada26b1f0232d57aa89bbebf997a5cc2332a5616b6eca26ff680044d/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa", size = 238530, upload-time = "2026-08-15T08:17:11.563Z" }, + { url = "https://files.pythonhosted.org/packages/76/84/6f1290fa07ae6978d3960caa3eb1b8019bf9284ab7c2297b00c099ef4250/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369", size = 230200, upload-time = "2026-08-15T08:17:12.919Z" }, + { url = "https://files.pythonhosted.org/packages/e7/a0/47b18adeed31c8f16ba9700f32c1b18594cfa09f47eb672a488c273c22bf/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893", size = 262222, upload-time = "2026-08-15T08:17:14.571Z" }, + { url = "https://files.pythonhosted.org/packages/38/fe/341861ac118dae06f3ec0eb487488af52128f2ef2faf0b11003944d22259/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0", size = 258951, upload-time = "2026-08-15T08:17:16.158Z" }, + { url = "https://files.pythonhosted.org/packages/6f/89/bb5108dc6c3651dca963f2b0a3ba19bbcb370c94e1b6d3e0e844a58e6dca/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08", size = 248801, upload-time = "2026-08-15T08:17:17.683Z" }, + { url = "https://files.pythonhosted.org/packages/b1/ba/ef83ae3aca816393decfa3530976f38a79812d707b80b580ac33b83f9877/charset_normalizer-3.5.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada", size = 244070, upload-time = "2026-08-15T08:17:19.191Z" }, + { url = "https://files.pythonhosted.org/packages/f6/0b/c5292a2462d69b7378ea89793bbb5b2b6fcf6f7dd6d1667f9619094ad553/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9", size = 240110, upload-time = "2026-08-15T08:17:20.547Z" }, + { url = "https://files.pythonhosted.org/packages/46/22/111e5be3b740d5c2a5bfcedb3d237b6591e5c2e82ae9d6ffcb121fe0909c/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e", size = 232836, upload-time = "2026-08-15T08:17:21.895Z" }, + { url = "https://files.pythonhosted.org/packages/f9/d2/d2aad6fe0dbb44b194bf3becb60f5a0ac48446ade999a47fe7bb41eb09a7/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6", size = 262712, upload-time = "2026-08-15T08:17:23.727Z" }, + { url = "https://files.pythonhosted.org/packages/35/5a/337e4663a5eae6de99db940ee8066d4145caafb61327db62deda15313cce/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf", size = 242977, upload-time = "2026-08-15T08:17:25.157Z" }, + { url = "https://files.pythonhosted.org/packages/ca/85/f82f8a92e31c7519410e2e1afdc630f28ec47490ce2c09a11c1a43cbb459/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71", size = 260207, upload-time = "2026-08-15T08:17:26.602Z" }, + { url = "https://files.pythonhosted.org/packages/b7/52/643d11ffd60e9ac2fd1fb87e167a19285b9eefeff4a40e63c87cbfbeab36/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573", size = 250562, upload-time = "2026-08-15T08:17:27.971Z" }, + { url = "https://files.pythonhosted.org/packages/62/16/46556278c2168d12df9da7fede5dc6fc70e60301b26a82bbeec238c9cfe3/charset_normalizer-3.5.1-cp312-cp312-win32.whl", hash = "sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2", size = 178507, upload-time = "2026-08-15T08:17:29.277Z" }, + { url = "https://files.pythonhosted.org/packages/9d/7a/4c6c298171e6b3e745633180ff59350fc0ca0db1ffd28df1e369e0579f71/charset_normalizer-3.5.1-cp312-cp312-win_amd64.whl", hash = "sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2", size = 200551, upload-time = "2026-08-15T08:17:30.668Z" }, + { url = "https://files.pythonhosted.org/packages/cd/d7/eb95a042f0dd22e304b0b6472b154f3546a1a039a9ee89ccb2a7f61591fc/charset_normalizer-3.5.1-cp312-cp312-win_arm64.whl", hash = "sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a", size = 180700, upload-time = "2026-08-15T08:17:32.028Z" }, + { url = "https://files.pythonhosted.org/packages/bc/61/2cb6ad133dbbb449fa2d37ccae973232f4827e799af258d15e589a3d1e9e/charset_normalizer-3.5.1-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9", size = 211584, upload-time = "2026-08-15T08:17:33.597Z" }, + { url = "https://files.pythonhosted.org/packages/18/57/a305c968be1ca13f3dd1b32f445877e97addf55d80b65c7cb35fac82b777/charset_normalizer-3.5.1-cp313-cp313-android_24_x86_64.whl", hash = "sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491", size = 223359, upload-time = "2026-08-15T08:17:35.022Z" }, + { url = "https://files.pythonhosted.org/packages/09/0a/d3646670292ce8d8f8cc11ac067d44885e697a5591f57a9221128da5e7b3/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7", size = 194464, upload-time = "2026-08-15T08:17:36.452Z" }, + { url = "https://files.pythonhosted.org/packages/de/93/d51ec556e01042fed6f993ea859311bc7917b466684182fbbceb6ca24762/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e", size = 197676, upload-time = "2026-08-15T08:17:37.819Z" }, + { url = "https://files.pythonhosted.org/packages/a4/a0/562247944386f7d4ef94467e84876600cc1e0f1b93239aaa9213d2bc3cbd/charset_normalizer-3.5.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d", size = 340473, upload-time = "2026-08-15T08:17:39.303Z" }, + { url = "https://files.pythonhosted.org/packages/31/e7/1d994be1b93d41e9502b8b0460eaa88a1dd8df335df415db87d6c3e91ab2/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a", size = 240156, upload-time = "2026-08-15T08:17:40.66Z" }, + { url = "https://files.pythonhosted.org/packages/09/53/27923ce5cc6cbccb832037b27dca98882d9c53e9b69e866bbbef4aae7fc8/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe", size = 228246, upload-time = "2026-08-15T08:17:42.003Z" }, + { url = "https://files.pythonhosted.org/packages/ce/48/5a97e84d63af1d55c07439cb80e56d99a8efb4295700eb4e18c0d1615d2c/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac", size = 263660, upload-time = "2026-08-15T08:17:43.627Z" }, + { url = "https://files.pythonhosted.org/packages/7a/c2/071575791dcc88316c0a9a65ce38897a82e4cfe4a325f0f7fe1b1ac47bcf/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e", size = 260354, upload-time = "2026-08-15T08:17:45.094Z" }, + { url = "https://files.pythonhosted.org/packages/fb/af/63240b0c0248c075c2535a1f1bd992821d8251b9f173abc13329661d09e4/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3", size = 250638, upload-time = "2026-08-15T08:17:46.496Z" }, + { url = "https://files.pythonhosted.org/packages/4d/66/70dfad64f15be09c15ccfee81330a7e515895dbe296dd23114e9a231268a/charset_normalizer-3.5.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876", size = 244583, upload-time = "2026-08-15T08:17:47.963Z" }, + { url = "https://files.pythonhosted.org/packages/c0/24/ef36367d38b9ddd4bccbf72888c342e8de1f5ae506fa0b2dcf970e2732a1/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6", size = 242038, upload-time = "2026-08-15T08:17:49.481Z" }, + { url = "https://files.pythonhosted.org/packages/db/ab/55e683ba0fff2e43adafc10daa3001eac90fdaa419a97227d5a7067eedde/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2", size = 233677, upload-time = "2026-08-15T08:17:50.845Z" }, + { url = "https://files.pythonhosted.org/packages/bd/67/0f40eaf8d1b6e7cf15e82382a2965efaca787fc1c2794b7021d37aaf5036/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591", size = 264491, upload-time = "2026-08-15T08:17:52.61Z" }, + { url = "https://files.pythonhosted.org/packages/5c/64/12b4c2a11ee8df4fcc518c78b0d93e3a92bd3d5253d1617ce74ff0e8c7ef/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c", size = 245196, upload-time = "2026-08-15T08:17:54.023Z" }, + { url = "https://files.pythonhosted.org/packages/37/2e/651d910af6d0fba325eee1cda37ec5443462ed25360e666c144166eb6091/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c", size = 261660, upload-time = "2026-08-15T08:17:55.491Z" }, + { url = "https://files.pythonhosted.org/packages/90/c6/b09e05e6db7f64338e0dc067c79577b1138da86c1e38369096851d96be88/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f", size = 252618, upload-time = "2026-08-15T08:17:57.025Z" }, + { url = "https://files.pythonhosted.org/packages/76/4e/362d4f9fdcdf5556fb2aa3ce7d4a58ebce03ed1ff03aa1d9aca8d02f13f3/charset_normalizer-3.5.1-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4", size = 140362, upload-time = "2026-08-15T08:17:58.425Z" }, + { url = "https://files.pythonhosted.org/packages/b4/d4/703be739b26acce318bd29eb3b25b7209e1b1f527f9eae3d1f1f01fdde2b/charset_normalizer-3.5.1-cp313-cp313-win32.whl", hash = "sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3", size = 177755, upload-time = "2026-08-15T08:18:00.037Z" }, + { url = "https://files.pythonhosted.org/packages/8a/33/56d97ade41c8db611e727168c52ae46c9224c362ec28d4b65d7e9869e8da/charset_normalizer-3.5.1-cp313-cp313-win_amd64.whl", hash = "sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6", size = 199295, upload-time = "2026-08-15T08:18:01.506Z" }, + { url = "https://files.pythonhosted.org/packages/5b/75/5b20dd1e6573a01a08158fe104104fa2c8abf941745596954185726cd46c/charset_normalizer-3.5.1-cp313-cp313-win_arm64.whl", hash = "sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0", size = 179856, upload-time = "2026-08-15T08:18:02.929Z" }, + { url = "https://files.pythonhosted.org/packages/29/cd/2b812ce5e888f1ce69a5350281e58aab07ae64a958ecae8912f30865718e/charset_normalizer-3.5.1-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8", size = 212318, upload-time = "2026-08-15T08:18:04.403Z" }, + { url = "https://files.pythonhosted.org/packages/9e/4a/a6ee107430768a5334e6d63f31f148a04a1a491ef161a1ac9415a73f2fa8/charset_normalizer-3.5.1-cp314-cp314-android_24_x86_64.whl", hash = "sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102", size = 224897, upload-time = "2026-08-15T08:18:05.997Z" }, + { url = "https://files.pythonhosted.org/packages/c3/d9/35ae3f64f29d0179c35c3baefe575904df2913dde519129c7f75995a2b1d/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5", size = 194848, upload-time = "2026-08-15T08:18:07.397Z" }, + { url = "https://files.pythonhosted.org/packages/74/76/f2fc7380f056cc273a53af37f50d08ad54b2c59f61078f31432edcf1c2bd/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3", size = 198163, upload-time = "2026-08-15T08:18:08.989Z" }, + { url = "https://files.pythonhosted.org/packages/e9/40/095ce62fa078483cccc1fa2b36e6bc9580b85422a20ee9f925341c50e44f/charset_normalizer-3.5.1-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c", size = 341823, upload-time = "2026-08-15T08:18:10.458Z" }, + { url = "https://files.pythonhosted.org/packages/f1/5a/0e58b1c04a1596e0256f407274a92d5fb2ee21324409d1fab1da48a65b5b/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0", size = 242458, upload-time = "2026-08-15T08:18:11.989Z" }, + { url = "https://files.pythonhosted.org/packages/22/95/b4618ce912e6db0b1aae89ba788e38e8a7eba0f3025cc66e8c0699f977b2/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96", size = 226717, upload-time = "2026-08-15T08:18:13.401Z" }, + { url = "https://files.pythonhosted.org/packages/8a/76/c681192bbda3d55356db5dadd64381d5202b37c6b598fcda5282e88b5d3d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc", size = 266111, upload-time = "2026-08-15T08:18:14.961Z" }, + { url = "https://files.pythonhosted.org/packages/88/be/55127bfca72c0cff6c022488d140d7c5b04c771e3b72e9bdb4836d54979d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f", size = 263128, upload-time = "2026-08-15T08:18:16.515Z" }, + { url = "https://files.pythonhosted.org/packages/e0/91/39c3af510b0aa32bbda03374259200f28430febfd1bf5e511fe765282ce5/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90", size = 251240, upload-time = "2026-08-15T08:18:18.127Z" }, + { url = "https://files.pythonhosted.org/packages/1c/a5/cbe418bbc6ecdfc3e05a0116002897c4b403a5e838d697e64c78e9f0190d/charset_normalizer-3.5.1-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506", size = 245282, upload-time = "2026-08-15T08:18:19.625Z" }, + { url = "https://files.pythonhosted.org/packages/cc/a4/689bb42e8e7cd492f3cb64907c6bc00ad247ec9a3628cd3f8eed126e8ae1/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5", size = 244597, upload-time = "2026-08-15T08:18:21.121Z" }, + { url = "https://files.pythonhosted.org/packages/c1/ce/9962938e179cf9f699d3f1e7b3114b5d7642dee6a893745229f9dd04f274/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e", size = 231376, upload-time = "2026-08-15T08:18:22.57Z" }, + { url = "https://files.pythonhosted.org/packages/85/54/46000450ada53bd9eac5429a2c8c54cd2d9b39c0c255f229aea9af0948a5/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5", size = 266715, upload-time = "2026-08-15T08:18:24.235Z" }, + { url = "https://files.pythonhosted.org/packages/3d/bb/618749d70f792b44252a777bf89bfb86823b9bbc1ea13fe8ce759b07f38a/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3", size = 245848, upload-time = "2026-08-15T08:18:25.726Z" }, + { url = "https://files.pythonhosted.org/packages/7e/3f/ffb64458527c7668031d5eb095d978de561958dc9f5b53f8e488a533e603/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3", size = 264521, upload-time = "2026-08-15T08:18:27.193Z" }, + { url = "https://files.pythonhosted.org/packages/4f/ab/74a55fd803916a35ac461daf002708191aac19b546b80dc8cabfedc63d98/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36", size = 253054, upload-time = "2026-08-15T08:18:28.568Z" }, + { url = "https://files.pythonhosted.org/packages/a0/2a/6a9034b7d3c60b17499afb482df5878bf9fa20b50cc3887d5ef017a833db/charset_normalizer-3.5.1-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7", size = 140580, upload-time = "2026-08-15T08:18:30.214Z" }, + { url = "https://files.pythonhosted.org/packages/f3/46/1d362e1a00d035d66b9869e1281eee115907f7e390a16a07824ab5737360/charset_normalizer-3.5.1-cp314-cp314-win32.whl", hash = "sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b", size = 180325, upload-time = "2026-08-15T08:18:31.877Z" }, + { url = "https://files.pythonhosted.org/packages/7a/7c/4938c329b6a9d446f6a59aa2092ff7118f274209b5ed0e26893d1d30a63c/charset_normalizer-3.5.1-cp314-cp314-win_amd64.whl", hash = "sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b", size = 204175, upload-time = "2026-08-15T08:18:33.466Z" }, + { url = "https://files.pythonhosted.org/packages/ac/33/eeb384dbd8dec570661354592f4f2e1b2fcc92585624d146a000caf53841/charset_normalizer-3.5.1-cp314-cp314-win_arm64.whl", hash = "sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687", size = 184123, upload-time = "2026-08-15T08:18:34.913Z" }, + { url = "https://files.pythonhosted.org/packages/1c/6c/c73fa9d5a85f6ab05395de61c5f6984e0a9ff40bb5ff888d46dff02526c6/charset_normalizer-3.5.1-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348", size = 381682, upload-time = "2026-08-15T08:18:36.349Z" }, + { url = "https://files.pythonhosted.org/packages/30/c7/63565f860921457feba93bae6c86fb7746deb4cffeed2f375cb845318146/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef", size = 240826, upload-time = "2026-08-15T08:18:37.887Z" }, + { url = "https://files.pythonhosted.org/packages/06/ae/7ae8807410dfa33f8e6f1715740adeaafa8a816cc4cb33508f54b1f7c896/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885", size = 227861, upload-time = "2026-08-15T08:18:39.315Z" }, + { url = "https://files.pythonhosted.org/packages/e9/a3/887c1642f0da26000b0e0652d91071113c0e72cea33952e225cf589f49a9/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375", size = 260758, upload-time = "2026-08-15T08:18:40.88Z" }, + { url = "https://files.pythonhosted.org/packages/3e/11/e6f5b9a3d0e55b0ef7505cd3765cdd48f22db89994c947b316f52f801fd8/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1", size = 259950, upload-time = "2026-08-15T08:18:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/1b/ee/e4e10a94d51cd1ee638aa7e00b65399e6b2a4e8376ab6d2eac9f95586671/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65", size = 249329, upload-time = "2026-08-15T08:18:43.914Z" }, + { url = "https://files.pythonhosted.org/packages/c4/25/d5f4198819e6059735a84e8d0bfb72dc33976da67b97adcd3fb5a5e07ec6/charset_normalizer-3.5.1-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5", size = 243137, upload-time = "2026-08-15T08:18:45.368Z" }, + { url = "https://files.pythonhosted.org/packages/a5/e9/e925ca7569cf9fb9701fd82503fee73eea5268fdb856bdd64947092d3daa/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af", size = 242820, upload-time = "2026-08-15T08:18:46.842Z" }, + { url = "https://files.pythonhosted.org/packages/34/17/672c251a888ed2aebcdd2fe830ad0104e25ff83c43f5c4f9c15e9fc6853c/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1", size = 230504, upload-time = "2026-08-15T08:18:48.353Z" }, + { url = "https://files.pythonhosted.org/packages/3f/fc/f6a85abebd42ce4da2f1db0aa56cc6a0df1995e318b3875d14401b8381d1/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9", size = 263087, upload-time = "2026-08-15T08:18:49.859Z" }, + { url = "https://files.pythonhosted.org/packages/98/66/7c42677e739ba66746b297e2046918d793078094dc239e1e72768cffccc6/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a", size = 243269, upload-time = "2026-08-15T08:18:51.601Z" }, + { url = "https://files.pythonhosted.org/packages/de/d8/a50b79237f417af10f8c2a501ce8d1ca87829a22e69117891ca4ba20a69e/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032", size = 258766, upload-time = "2026-08-15T08:18:53.23Z" }, + { url = "https://files.pythonhosted.org/packages/2e/1d/0fc91aeaeb3c83b748f532399ce67cf84604b48297405d740000f7a9e786/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e", size = 250814, upload-time = "2026-08-15T08:18:54.768Z" }, + { url = "https://files.pythonhosted.org/packages/ae/10/3d8c777cf9024615295aa1b808324ad5b4a77855869c00824bad74ffaf8a/charset_normalizer-3.5.1-cp314-cp314t-win32.whl", hash = "sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4", size = 191074, upload-time = "2026-08-15T08:18:56.305Z" }, + { url = "https://files.pythonhosted.org/packages/4d/81/ae557d3c44d1a1d688696d60563413a0866a91b7ebc50f20df838be3d8c8/charset_normalizer-3.5.1-cp314-cp314t-win_amd64.whl", hash = "sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00", size = 216476, upload-time = "2026-08-15T08:18:57.889Z" }, + { url = "https://files.pythonhosted.org/packages/27/e9/61c01fb8b804692569c036b3fc50495814502dcf13a60649c6055390b02c/charset_normalizer-3.5.1-cp314-cp314t-win_arm64.whl", hash = "sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f", size = 194115, upload-time = "2026-08-15T08:18:59.418Z" }, + { url = "https://files.pythonhosted.org/packages/4a/4e/8544831ef59d8f27ce92c80871380fdacc8076a8a56ed62f82e54f991333/charset_normalizer-3.5.1-cp315-cp315-macosx_10_15_universal2.whl", hash = "sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af", size = 342048, upload-time = "2026-08-15T08:19:01.054Z" }, + { url = "https://files.pythonhosted.org/packages/7f/a6/e3b46852424246065355644f4fb6dbccc0239a42a2eee27ecfc8957f0bcd/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8", size = 242997, upload-time = "2026-08-15T08:19:02.492Z" }, + { url = "https://files.pythonhosted.org/packages/03/3b/0cc9a26777334ab2f2e3089b948bbf4e4fe72ea70b897715ef6415043ec8/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90", size = 237014, upload-time = "2026-08-15T08:19:03.943Z" }, + { url = "https://files.pythonhosted.org/packages/8c/c2/027335f0aa337a2a2e121bac1ad88c4f02ba6053ea0926802784f3db11af/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20", size = 266174, upload-time = "2026-08-15T08:19:05.598Z" }, + { url = "https://files.pythonhosted.org/packages/86/d3/e367787febe4e74769dec0f406f2c3c8d1b955fce5aee1fd0f94e8367a45/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449", size = 263361, upload-time = "2026-08-15T08:19:07.251Z" }, + { url = "https://files.pythonhosted.org/packages/af/3d/391b193eb9f3e84b02f9314088c386debdc0debee843535aaea2e2c6715d/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a", size = 252143, upload-time = "2026-08-15T08:19:08.816Z" }, + { url = "https://files.pythonhosted.org/packages/2e/57/de221f1745a90d418199761967e2776bfe2c275a1194220985e8c1d37833/charset_normalizer-3.5.1-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0", size = 252086, upload-time = "2026-08-15T08:19:10.255Z" }, + { url = "https://files.pythonhosted.org/packages/c8/e3/d119f86a01f9331e8186175f24873b1d74a7ee9e2e4b4d68f9947dae5afd/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e", size = 245231, upload-time = "2026-08-15T08:19:11.807Z" }, + { url = "https://files.pythonhosted.org/packages/26/de/d8e48c135ae480879539cdb179c8d3b50c7879497d75dd899b5763b69cee/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_armv7l.whl", hash = "sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2", size = 241546, upload-time = "2026-08-15T08:19:13.416Z" }, + { url = "https://files.pythonhosted.org/packages/67/c4/217755fd1abc50d326c252922cd642002758095a81ff45010337b8b3ef65/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_ppc64le.whl", hash = "sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626", size = 267033, upload-time = "2026-08-15T08:19:14.981Z" }, + { url = "https://files.pythonhosted.org/packages/b8/d7/34d8e404e358d2adcc5a228c2134643af00104c8fb0bf525f3688d756f05/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5", size = 252045, upload-time = "2026-08-15T08:19:16.618Z" }, + { url = "https://files.pythonhosted.org/packages/5e/fa/40414471acf0aa0692ca77305aa00e434fcd8288f0941c93c30e9a5f8f2f/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_s390x.whl", hash = "sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774", size = 264866, upload-time = "2026-08-15T08:19:18.101Z" }, + { url = "https://files.pythonhosted.org/packages/32/90/fcc850bae791abd2e0c041847f13e270aa08692a79f3e00de6d2dce1cb50/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7", size = 253932, upload-time = "2026-08-15T08:19:19.734Z" }, + { url = "https://files.pythonhosted.org/packages/af/af/53afe99068b3c10b4cbae592a52ef72a7c92c0188440e83ee3a078fd8f75/charset_normalizer-3.5.1-cp315-cp315-win32.whl", hash = "sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9", size = 180320, upload-time = "2026-08-15T08:19:21.37Z" }, + { url = "https://files.pythonhosted.org/packages/c9/bc/f46a132041b29e4a8779ed712d3df1bf112e94ca8de58b66d7ec2c0cf8b9/charset_normalizer-3.5.1-cp315-cp315-win_amd64.whl", hash = "sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712", size = 204174, upload-time = "2026-08-15T08:19:23.088Z" }, + { url = "https://files.pythonhosted.org/packages/a1/5d/9ed554480eda8e447b673648628fdc29574d23dbad01fe11837adedd1cae/charset_normalizer-3.5.1-cp315-cp315-win_arm64.whl", hash = "sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7", size = 184126, upload-time = "2026-08-15T08:19:24.471Z" }, + { url = "https://files.pythonhosted.org/packages/3b/32/9b8929bf384061ee1fe5d9c27c6f9776d3d824039ad4e14c88ec00c7808e/charset_normalizer-3.5.1-cp315-cp315t-macosx_10_15_universal2.whl", hash = "sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663", size = 381441, upload-time = "2026-08-15T08:19:26.038Z" }, + { url = "https://files.pythonhosted.org/packages/96/10/e9aa7923d3ddac652c99a1c5f7be494e737e151566a44abe018daf757f2c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11", size = 241742, upload-time = "2026-08-15T08:19:27.532Z" }, + { url = "https://files.pythonhosted.org/packages/28/53/a2d249ebddf47b889a100c0bdcb61a2f9dbb8bc24ef325cc062e4f476877/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc", size = 235298, upload-time = "2026-08-15T08:19:29.274Z" }, + { url = "https://files.pythonhosted.org/packages/7d/07/469f78af590f7d5cd48e20d8dbfa3d66deeff9ba37768c04d886b5afd45c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a", size = 262500, upload-time = "2026-08-15T08:19:30.955Z" }, + { url = "https://files.pythonhosted.org/packages/55/66/3bb56a47f7dcba014055b1a1d33c6f08bbe9c1e74dba154cfa25f90ae885/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4", size = 258888, upload-time = "2026-08-15T08:19:32.458Z" }, + { url = "https://files.pythonhosted.org/packages/ff/c1/2adc2800903fb013210349313b710a5376856578d9e33e6b9a1d8b36714a/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004", size = 250243, upload-time = "2026-08-15T08:19:33.94Z" }, + { url = "https://files.pythonhosted.org/packages/95/b5/a18d0dd1157ab655cc2cb14a545f4a4784bbad70ab3502412e36097502d9/charset_normalizer-3.5.1-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b", size = 249871, upload-time = "2026-08-15T08:19:35.413Z" }, + { url = "https://files.pythonhosted.org/packages/ad/c3/525f508cd1e58d0450ac55ed40ac75bc3a97482c59def5278456a5fbf03c/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263", size = 243580, upload-time = "2026-08-15T08:19:36.886Z" }, + { url = "https://files.pythonhosted.org/packages/7c/c1/49a91fe7e97c8140094ca5c64161ab623a70d9f636bf834eace14048acb5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_armv7l.whl", hash = "sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee", size = 239807, upload-time = "2026-08-15T08:19:38.392Z" }, + { url = "https://files.pythonhosted.org/packages/d3/58/56a48c296601274c4689b864a8e2dfb209b81dfcb39472753ce95eea662b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_ppc64le.whl", hash = "sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c", size = 264083, upload-time = "2026-08-15T08:19:39.856Z" }, + { url = "https://files.pythonhosted.org/packages/10/4c/dc48409274a1817ff349711d26c62aa0c597df865d4d69ef79160c859193/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e", size = 250317, upload-time = "2026-08-15T08:19:41.53Z" }, + { url = "https://files.pythonhosted.org/packages/81/58/d325912115caec62d6bdd77bbab5e0b7da5d234a9f20affdffcbcb530d0b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_s390x.whl", hash = "sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d", size = 258173, upload-time = "2026-08-15T08:19:43.07Z" }, + { url = "https://files.pythonhosted.org/packages/34/f7/b13b1ccae2c8ec63980d13be1890eb73f8aeabbfce02a24aabc0908788f5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61", size = 251960, upload-time = "2026-08-15T08:19:44.587Z" }, + { url = "https://files.pythonhosted.org/packages/1e/25/ed3f9919c5aef8cc818be1f972f565f7610d7b2076b8ebb98839516ffc3c/charset_normalizer-3.5.1-cp315-cp315t-win32.whl", hash = "sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f", size = 191186, upload-time = "2026-08-15T08:19:46.293Z" }, + { url = "https://files.pythonhosted.org/packages/69/d5/43c2b3e9d8267092b913eb8b0603f0f71993c395632886bd37a7223f96cf/charset_normalizer-3.5.1-cp315-cp315t-win_amd64.whl", hash = "sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb", size = 215947, upload-time = "2026-08-15T08:19:47.853Z" }, + { url = "https://files.pythonhosted.org/packages/a8/76/9aad3e9c8865e5e0efa9a7f6f81c37a67635a985145ecd44528a81e088ee/charset_normalizer-3.5.1-cp315-cp315t-win_arm64.whl", hash = "sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a", size = 193909, upload-time = "2026-08-15T08:19:49.383Z" }, + { url = "https://files.pythonhosted.org/packages/5b/97/fb4e82231aba271ffd775a1b4993b0defc4e3059f286ae41d9433409fe85/charset_normalizer-3.5.1-cp37-abi3-macosx_10_9_universal2.whl", hash = "sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2", size = 331467, upload-time = "2026-08-15T08:19:50.959Z" }, + { url = "https://files.pythonhosted.org/packages/9f/2f/fe3f187327aac18e2d54e9d2b08e15d27bf9b642d9e51c219f130fc34d1a/charset_normalizer-3.5.1-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99", size = 253057, upload-time = "2026-08-15T08:19:52.654Z" }, + { url = "https://files.pythonhosted.org/packages/d7/c7/9e48cee5c161fe24da823b61bf381921d77cb994a0a4de148e95018c1984/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2", size = 240930, upload-time = "2026-08-15T08:19:54.163Z" }, + { url = "https://files.pythonhosted.org/packages/49/e0/716601f3cc69be7b198951150c75ead1ece33c3c8036ff6ffa46029659a0/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235", size = 230822, upload-time = "2026-08-15T08:19:55.807Z" }, + { url = "https://files.pythonhosted.org/packages/d3/05/71bfc5caa0abcc45aea1f6a4d50ac68e59605ddc7666fe8494f4cd229665/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598", size = 260037, upload-time = "2026-08-15T08:19:57.312Z" }, + { url = "https://files.pythonhosted.org/packages/c3/92/de7e32ed05341e7a9c4c877c318418197b7f2d66a3b68d561bf2ac57ca3e/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96", size = 255097, upload-time = "2026-08-15T08:19:59.056Z" }, + { url = "https://files.pythonhosted.org/packages/f5/7b/ade0a122600319dfa0b1000ab0f9731c94a817904cf3c5de408c73a4ede7/charset_normalizer-3.5.1-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962", size = 250166, upload-time = "2026-08-15T08:20:00.612Z" }, + { url = "https://files.pythonhosted.org/packages/75/9c/019fbb9f4834491a160951349b1a3714439376f66e5f7cf18b4f18f0c7aa/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3", size = 241821, upload-time = "2026-08-15T08:20:02.321Z" }, + { url = "https://files.pythonhosted.org/packages/2b/b8/11d4840bfc99330cc7fbcc2681ee5a044553a6e77655508d8f9b2bff7b34/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950", size = 232529, upload-time = "2026-08-15T08:20:04.008Z" }, + { url = "https://files.pythonhosted.org/packages/18/96/2b3a21492d9f65171ac75d872f5018260013d00bfa0ff70ec9f179148cbd/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8", size = 260348, upload-time = "2026-08-15T08:20:05.877Z" }, + { url = "https://files.pythonhosted.org/packages/d6/aa/a69a2028e8bd052476c245460ab19d7de595de084dd968f2d75cd50c3e25/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031", size = 247234, upload-time = "2026-08-15T08:20:07.487Z" }, + { url = "https://files.pythonhosted.org/packages/35/8a/3d130aeabcaf3d2466af76b7b141c08d9e89c9016ab4b7cdd0f7dc2d1c62/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_s390x.whl", hash = "sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072", size = 256917, upload-time = "2026-08-15T08:20:09.142Z" }, + { url = "https://files.pythonhosted.org/packages/80/c2/a7379b840292d0c1ab9fbd17d1f3967aa81794dc95bc74be8999d7fedcf7/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d", size = 254846, upload-time = "2026-08-15T08:20:10.727Z" }, + { url = "https://files.pythonhosted.org/packages/01/65/d43b714731bb2f40d4053dfa00ecfc1c5a301f8e3316c5db3a09af59fe94/charset_normalizer-3.5.1-cp37-abi3-win32.whl", hash = "sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc", size = 174216, upload-time = "2026-08-15T08:20:12.334Z" }, + { url = "https://files.pythonhosted.org/packages/35/4f/b911ed898b26a09789eba9c9200c999aff6c61b4bafaf4838e56d1a1e1a3/charset_normalizer-3.5.1-cp37-abi3-win_amd64.whl", hash = "sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959", size = 199764, upload-time = "2026-08-15T08:20:13.908Z" }, + { url = "https://files.pythonhosted.org/packages/f0/a7/920baf467bfd9bf689f3b318340f37aee4572a71f162bd8db51da55ba4fa/charset_normalizer-3.5.1-cp37-abi3-win_arm64.whl", hash = "sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e", size = 287318, upload-time = "2026-08-15T08:20:15.551Z" }, + { url = "https://files.pythonhosted.org/packages/cc/61/d01fc49b8dea277640b55a9e15960dbca9fdc8c9fde18e572d39c59f4019/charset_normalizer-3.5.1-py3-none-any.whl", hash = "sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6", size = 68658, upload-time = "2026-08-15T08:20:43.306Z" }, +] + +[[package]] +name = "click" +version = "8.5.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/c7/0e/7fa0ef50764b67090eca4114772a2abf8b6148198475e54c660b97caeee6/click-8.5.0.tar.gz", hash = "sha256:ba0d2089de75ea0310e2dde03160e6ca10009947fb95a182f9b54021bb272e34", size = 382235, upload-time = "2026-08-26T13:33:14.56Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/58/50/6c0d534c5f134586a8e1ba4e330569e32f057e33372ae556463212fb4cd3/click-8.5.0-py3-none-any.whl", hash = "sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360", size = 125251, upload-time = "2026-08-26T13:33:12.928Z" }, +] + +[[package]] +name = "colorama" +version = "0.4.6" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44", size = 27697, upload-time = "2022-10-25T02:36:22.414Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335, upload-time = "2022-10-25T02:36:20.889Z" }, +] + +[[package]] +name = "coverage" +version = "7.16.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d1/f5/deb1a27aa20746c0278ac998c4179e272004699b2d33959ce020c5ac1615/coverage-7.16.0.tar.gz", hash = "sha256:077f0964087883176ff6ab9b074694cae29f8c708273b13ca62c183c6ed716cd", size = 945620, upload-time = "2026-08-28T21:54:37.74Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/9c/8d2688694f53dc0b0f0e4783c7eb3c4bb1e79beaf1411879f6dabedf4607/coverage-7.16.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:d1c77c3579ac42798f8b7eed6d3dd258debacca32c8753fc8a1f6eaf1db644f5", size = 223194, upload-time = "2026-08-28T21:51:27.767Z" }, + { url = "https://files.pythonhosted.org/packages/ca/11/f002163dd688aa3fa49ac6a424b7c2705c7fcf80fba18ec9f586d77827ca/coverage-7.16.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1f81cb1554c3712e41649ed5dc98656b50b958e4da12f0f5adb681ce3db92831", size = 223553, upload-time = "2026-08-28T21:51:29.46Z" }, + { url = "https://files.pythonhosted.org/packages/81/65/f9d469e97c4554372a710650a109004a2434dfc56f577142e5d6057fa0cc/coverage-7.16.0-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:6e701938ec9081d3e400a0c9a9a8ae0f7ca44214741daeac4454b1c6ef6dbd19", size = 255054, upload-time = "2026-08-28T21:51:31.54Z" }, + { url = "https://files.pythonhosted.org/packages/95/29/dd89fd39af1a3b6e9a9c3eddeaf03f6376ba517d43d6cbf8b519177e2a10/coverage-7.16.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:719a3feb6220dd32ed932d4c3676d17fb8739e2643b29c0e7c3af400ff80ac44", size = 257790, upload-time = "2026-08-28T21:51:33.374Z" }, + { url = "https://files.pythonhosted.org/packages/0a/64/208d26cedc525d6b5db9c492cf9130784c42d9eb08d22badaa7b806005ad/coverage-7.16.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:87771ecf986cff55e87413238cd5e4f54d949c2074bd6fc1657d26a56314ee24", size = 258904, upload-time = "2026-08-28T21:51:35.096Z" }, + { url = "https://files.pythonhosted.org/packages/1f/98/28e2752aa9a8baee5798edade9c95602ca200f4e7eeb503eb64df42e5921/coverage-7.16.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:47d5e1fc0b321c8308a2aacee0497c435b08acaa629b7059798fdf6fc3006352", size = 261165, upload-time = "2026-08-28T21:51:36.744Z" }, + { url = "https://files.pythonhosted.org/packages/eb/77/fa6ae699a0ea2bc12acb38a85d96b786fea0f833c12b5756056350e0e547/coverage-7.16.0-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:01b18b8a6c9cec8d5f45550e2501426ed982cf2c35016b0acd2ba9b5d8b2fb06", size = 255416, upload-time = "2026-08-28T21:51:38.495Z" }, + { url = "https://files.pythonhosted.org/packages/89/c8/5ee46d1de7d34cb00ba08b5c50da1971114dbc09ca9898ccc32975ec74dd/coverage-7.16.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:32c56b5b47c50635081445ac404dd08c2d591b9c837c22570aa9e182c3b42cd4", size = 256825, upload-time = "2026-08-28T21:51:40.27Z" }, + { url = "https://files.pythonhosted.org/packages/15/f6/d59e1c0693ad48855fe20169fbf6ee5befefe5887a7fabf5f0bcb464a2dc/coverage-7.16.0-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:6ad3bbad240ab937512156bc944fdee63ac4dd34a7558a3094548fd4c1150c02", size = 254970, upload-time = "2026-08-28T21:51:43.136Z" }, + { url = "https://files.pythonhosted.org/packages/df/7b/b51bbe05b3a7565927fccfb1be42b8b3c1f4ab15e53d91b303e9923969aa/coverage-7.16.0-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:4c1f16d5555a195295d0dc9c902612270e3dfed6a11f3bf7bc470b7b6a79ed3c", size = 259039, upload-time = "2026-08-28T21:51:44.983Z" }, + { url = "https://files.pythonhosted.org/packages/fa/04/d513f816456a8a43c1859abe88a37d01d7d2515b6c3e24ebb3c9b1dd44ec/coverage-7.16.0-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:f6c9c21a8bf0d19788f3c5f3e020c90317a0a63ef60521b376003801e21250fb", size = 254539, upload-time = "2026-08-28T21:51:46.733Z" }, + { url = "https://files.pythonhosted.org/packages/dc/54/5542190ceb97e0d1333a4ce0c8f95b2ef2efe790f1ad018a4b61766f849e/coverage-7.16.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:06f20145a9eb5bf1fd1dde3c0bc2af2e7c22135ab07ca6284d6ada7cc3904c4e", size = 256410, upload-time = "2026-08-28T21:51:48.363Z" }, + { url = "https://files.pythonhosted.org/packages/ee/28/78643f361ff6bb5b2ade90f8bfc8395fe9ca367a18c101f8991215b4c65b/coverage-7.16.0-cp312-cp312-win32.whl", hash = "sha256:916cf8d25c1ce148f7eceb1d45afc9724841200110adc4e53250391852debd91", size = 225239, upload-time = "2026-08-28T21:51:50.22Z" }, + { url = "https://files.pythonhosted.org/packages/67/61/8e76b36c36b1a033dc933dd2480db96b04ce3be975793ce3fad122e7174d/coverage-7.16.0-cp312-cp312-win_amd64.whl", hash = "sha256:78f8b56261d608be102c62edd3a60b66bcd0b581f3f86fdcabaf8b8d95adc950", size = 225775, upload-time = "2026-08-28T21:51:51.912Z" }, + { url = "https://files.pythonhosted.org/packages/c8/f3/bb4787a4b81c1792ca69b502f5f730dbbb609f73fed552ab074c6b92cb8b/coverage-7.16.0-cp312-cp312-win_arm64.whl", hash = "sha256:577c2ac8c0036f6f8edd3a7783a9e67302b17771d1abf0fd2ed246e3158be51b", size = 225159, upload-time = "2026-08-28T21:51:53.667Z" }, + { url = "https://files.pythonhosted.org/packages/54/c5/e62c87f4799d1e3647d5b2ae16ea1d12205d72fde1ea8529e13fe050f678/coverage-7.16.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:1545c52ce756b8a97007f439a220297f1cd72a2cbbcdffccdf1c1f70e74f9a42", size = 223215, upload-time = "2026-08-28T21:51:55.628Z" }, + { url = "https://files.pythonhosted.org/packages/89/e9/5e62fda9397175fb206f75368b6e85da06d831c181b6d0f67ca073cd2f89/coverage-7.16.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:0598aadae641f30a0796b75b45c0b9c5de8619bd5cfb251bb0cc254e86e6dd13", size = 223585, upload-time = "2026-08-28T21:51:57.355Z" }, + { url = "https://files.pythonhosted.org/packages/b9/40/bede08621b1ba67e88c4d3336c22b52cb7911ff1fa4ef055344b6670e58a/coverage-7.16.0-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:4080ad6bad9f14690e6b2104f5e8d137ccc65a4b5427a36662090637d4bd16d5", size = 254575, upload-time = "2026-08-28T21:51:59.233Z" }, + { url = "https://files.pythonhosted.org/packages/12/d8/ab0bdaa45dfd6b8cbf1a3ec548fdf827684b1997f9724375c5b3e89144fb/coverage-7.16.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:e9883a2f8206ce3af59117dc278e5d043fea06912bca3f199816129e5e2de354", size = 257172, upload-time = "2026-08-28T21:52:01.015Z" }, + { url = "https://files.pythonhosted.org/packages/1d/bb/135de81784bbd7dfedcab2b92b03d71d75b09b0815b42d6dabb052def5a6/coverage-7.16.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:984e5430fc6f858385009e92549955157d79335b1f3e13e1031e0f89d1284261", size = 258410, upload-time = "2026-08-28T21:52:02.76Z" }, + { url = "https://files.pythonhosted.org/packages/ad/72/ce44ecc062fb2e43d9447bb76154d091c2139232f20c125297c4b58f4c6a/coverage-7.16.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:b1374099dd1ad0d31fbb6c95d00a56a3c5e85fb3343dca14fc12f78323a2b42a", size = 260539, upload-time = "2026-08-28T21:52:04.821Z" }, + { url = "https://files.pythonhosted.org/packages/e7/c4/9389c36a41e59406ca2bba493807c2294d2e5186a7e9ebcc2e63a0f2a711/coverage-7.16.0-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:34d8686bce035c8465b318a8c2890e69ba14a00801a27f4eb6bdc97c23944d87", size = 254756, upload-time = "2026-08-28T21:52:06.68Z" }, + { url = "https://files.pythonhosted.org/packages/ad/0f/7762447b15e01fb84263608540123c4d9941f06303265ee74d801ccbec0e/coverage-7.16.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:857fceba6ff4b507ee0ad98798a33d544a8473df0c542bf04251ee4ed5ee6292", size = 256540, upload-time = "2026-08-28T21:52:08.529Z" }, + { url = "https://files.pythonhosted.org/packages/e6/fa/c60dc75a8346c1dbebebc7279b19971c88f70dd575f0bc10bc0cb16f92d5/coverage-7.16.0-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:bbf08d951abaa1ce89e28c998361d56b952413846b459cd017f116ad4c9adbfa", size = 254508, upload-time = "2026-08-28T21:52:10.323Z" }, + { url = "https://files.pythonhosted.org/packages/c3/f0/4e0834f3a1fccaa8bf625a2a1d73bde0fa32577dc3249853c0dd0e7f2b20/coverage-7.16.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:1a03e78f53e4d2ab13adac19958a89322d1829913e5623d642627bf60b35da21", size = 258659, upload-time = "2026-08-28T21:52:12.124Z" }, + { url = "https://files.pythonhosted.org/packages/b4/ec/fe712d3a11fd6e874565a5fa5497c48b8ece561d9611da040b44cdcf8386/coverage-7.16.0-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:dcd3dafcdd78305d27c59a1006b53a4990acb89e68d8fbe0992f4f83503c827f", size = 254326, upload-time = "2026-08-28T21:52:14.181Z" }, + { url = "https://files.pythonhosted.org/packages/e7/78/093e12072e01034c65ff380f76c74b79dd83e44fa92b689a2154389be734/coverage-7.16.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:c1bcfe470a796fbea6234accd81d258a31574dc0b7bf569e16be757572c4de17", size = 256102, upload-time = "2026-08-28T21:52:16.003Z" }, + { url = "https://files.pythonhosted.org/packages/9b/c0/265176117ca5d06e3f65575842884cdda96cf213350a31e9d41c80d65854/coverage-7.16.0-cp313-cp313-win32.whl", hash = "sha256:1420370276f1694b663207b8245c3628aafb9624fe3cebf313a13d860e55ee67", size = 225250, upload-time = "2026-08-28T21:52:17.82Z" }, + { url = "https://files.pythonhosted.org/packages/1f/01/8a87f2c04fde322430b45d16d8f543693e9894c5b2d2ca238a287c00beca/coverage-7.16.0-cp313-cp313-win_amd64.whl", hash = "sha256:496277c8d7beed695e02c7be53516a0152e4caef8738a0feab6a638546cce449", size = 225790, upload-time = "2026-08-28T21:52:19.641Z" }, + { url = "https://files.pythonhosted.org/packages/23/40/c21feacd9edfe7063195bf9cc84d650e9938fc6a23063e4f027199b160e1/coverage-7.16.0-cp313-cp313-win_arm64.whl", hash = "sha256:181c2906b9b3759955c1c33c51fbb91c754fbd0b82ea49e2c81061f5a052082c", size = 225180, upload-time = "2026-08-28T21:52:21.613Z" }, + { url = "https://files.pythonhosted.org/packages/ea/73/850675f262391b322c4c988b6cdc32cdc6629288f0fb158687b587a393a8/coverage-7.16.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:54b7fba6a74d010de34319a0419d5b65af8c00f539ad0b6f39fc6f342ab99697", size = 223258, upload-time = "2026-08-28T21:52:23.558Z" }, + { url = "https://files.pythonhosted.org/packages/61/c1/4f54c6d47c80d1cc58ef8fe6b74e6eb50f9e2c0f6e2de6cf38dbca2937b8/coverage-7.16.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:fa4ff0b3dd52208d2b30903022d5087f82000507b504753dfeee83e4f32d6883", size = 223587, upload-time = "2026-08-28T21:52:25.627Z" }, + { url = "https://files.pythonhosted.org/packages/3c/be/298f2456230fb44e272a4e53a41b3f3c39f0821c242d7b7daa9787b4d6f7/coverage-7.16.0-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:35a9676bf86097f790113ebd9fb67681804ef54d40941d2f10ba68c02239e575", size = 254632, upload-time = "2026-08-28T21:52:27.689Z" }, + { url = "https://files.pythonhosted.org/packages/a3/9c/a1bda6439c19c4783d50df896142b67b9e7d432db36675d339a32778669d/coverage-7.16.0-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:f98d438add63546745e5e847192e3e9ab897ed6f2ca96f8281e2f5a15958ae62", size = 257139, upload-time = "2026-08-28T21:52:29.741Z" }, + { url = "https://files.pythonhosted.org/packages/f8/cd/cd735c9be757f97237c305f36897a5e5b348bdbc12ebed3b2b80060dd8a9/coverage-7.16.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:151855767480be14db595cbc2040f6a4db965cdfeebd354d79b0256742b029e0", size = 258484, upload-time = "2026-08-28T21:52:31.68Z" }, + { url = "https://files.pythonhosted.org/packages/e4/04/84b2e1e8aae9db3f549782f28ce25bba5fd6a9c7bfba3782ffe8b4cd2559/coverage-7.16.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:183613f664718b340589d7f005c7e92b4b601cffd20a8a4117cfda3e983b080f", size = 260798, upload-time = "2026-08-28T21:52:33.642Z" }, + { url = "https://files.pythonhosted.org/packages/8a/4f/e04cf52483619a4dc5dd6367b30c9a8ac52243567fdfacec9b11a441565c/coverage-7.16.0-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:785b114356c99c0dd5b3f57b9696cfd57b7704f4c53847df8dc88c6cc0d9bcb6", size = 254612, upload-time = "2026-08-28T21:52:35.543Z" }, + { url = "https://files.pythonhosted.org/packages/da/33/627c4113f66bfffd43807f54dbf080c4632ecf12e4ef7a3bdd4ec38e46a2/coverage-7.16.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:30f5aee6d1d517abcdfd4f9cad027969ff79a1440a22da263f9514e31b5b66e9", size = 256495, upload-time = "2026-08-28T21:52:37.485Z" }, + { url = "https://files.pythonhosted.org/packages/3c/38/aaca432f4e008a88f2bc4d1459aa7016d8d1bbbe801f7e4fa3cf2746557b/coverage-7.16.0-cp314-cp314-musllinux_1_2_i686.whl", hash = "sha256:190ffa0f5af966254c249fb3aeaca2cef389785e3e287fd577d39e134d20f8a3", size = 254454, upload-time = "2026-08-28T21:52:39.425Z" }, + { url = "https://files.pythonhosted.org/packages/cc/db/8430aa87ef0a508f4c17c1b8fa7e0cf80231988d9081aa36c194036592d6/coverage-7.16.0-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:0ccc37c00e1a5d30840902c54557e104d04aead872cedf6d2281c8725a467e06", size = 258728, upload-time = "2026-08-28T21:52:41.32Z" }, + { url = "https://files.pythonhosted.org/packages/76/88/cd8aa8c82493ffbd291d3ef5554452fffc634c6c6098a04ac848c79c98f3/coverage-7.16.0-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:6c60cde430c0e7e3be612973af39b4cff90ec2e2defe7b2b701daea3a0ffff04", size = 254271, upload-time = "2026-08-28T21:52:43.278Z" }, + { url = "https://files.pythonhosted.org/packages/a8/49/fe16c811ea9314a84b48f34e4bf5a3d9013091093b285a74b2272fc863d7/coverage-7.16.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:c5297028c8df849a61b29129cadfe682f90b5b396f528eb319a57d7678eefdad", size = 255927, upload-time = "2026-08-28T21:52:45.461Z" }, + { url = "https://files.pythonhosted.org/packages/d1/45/d0bd410e78cfbf768acc8099b335e1d5c0d5c26103c796d2bebdee001715/coverage-7.16.0-cp314-cp314-win32.whl", hash = "sha256:136988df5bc5a48795d9c42c75c4bbda5d9a78e750a080c1233010edff93a1af", size = 225424, upload-time = "2026-08-28T21:52:47.658Z" }, + { url = "https://files.pythonhosted.org/packages/17/78/1ce6ce4646822e9308dcdb1942eaf31bfd7da43247b8886338b0d6fe3767/coverage-7.16.0-cp314-cp314-win_amd64.whl", hash = "sha256:ce2ba5e9f1842fe09165825abfb3bc6b527c71a27bc2eb3a10f2284ced64506d", size = 225918, upload-time = "2026-08-28T21:52:49.692Z" }, + { url = "https://files.pythonhosted.org/packages/f9/cd/e1323fe3a7dfcdd709451a43fe708ca1dfd36a7fc07b34eb7bd1dfdfb52d/coverage-7.16.0-cp314-cp314-win_arm64.whl", hash = "sha256:a89d07e48d9baead9a15599923a02f62c6df6c3d85aa84ef34be3c9fd6aeb91f", size = 225344, upload-time = "2026-08-28T21:52:51.665Z" }, + { url = "https://files.pythonhosted.org/packages/39/fb/1c15460d4cf915f09ae3ad3862fef4f901838991c5641b0cec545050d810/coverage-7.16.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:6e2854b62601c89a63814ad5def3b90d99c6724cc4cb977f75b725e5fca4b1e3", size = 223986, upload-time = "2026-08-28T21:52:53.572Z" }, + { url = "https://files.pythonhosted.org/packages/9f/73/347d2d0009ac211f79ee2a2364fd2aa19d6b9628dc22ed13a9b9386097ab/coverage-7.16.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:f093faf23df888518d273be6da65f0ec5a25b5d8b670231e4d87de07361042e7", size = 224254, upload-time = "2026-08-28T21:52:55.59Z" }, + { url = "https://files.pythonhosted.org/packages/5a/2f/51442e6ad9d705369596f08496021647e276d5b57311818fd4312d93509b/coverage-7.16.0-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:b7dbbbf6551eb94618e7bc76ab61cc2740a5b3d13294171bd6adb36e12346c3c", size = 265619, upload-time = "2026-08-28T21:52:57.645Z" }, + { url = "https://files.pythonhosted.org/packages/ea/8e/0f752276f6d13efbd019ab6d90792e20d6272c44cda039dc5c6d27b91e7f/coverage-7.16.0-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:51e7d0e311d2fba3915f971236cbdd4ad821fc7a23988221c0b33c964b0eba22", size = 267734, upload-time = "2026-08-28T21:52:59.611Z" }, + { url = "https://files.pythonhosted.org/packages/fa/02/4df3baef8029881c9d1a380859f2be73f90080d430def567d182e8566a35/coverage-7.16.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0bb04ee77e557d7476471969d35fbbfb5fc8a4152e9409aa5811780c36d9b23e", size = 270156, upload-time = "2026-08-28T21:53:01.658Z" }, + { url = "https://files.pythonhosted.org/packages/9f/30/ce10fdb74055ebbfb5c8a025d8845dc19c76e4b2c42bb5c755b56678990c/coverage-7.16.0-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c72c9b201dc0e8c2c8821d49858fd865010d08181bf877d2320971b6464ebfd5", size = 271279, upload-time = "2026-08-28T21:53:03.698Z" }, + { url = "https://files.pythonhosted.org/packages/71/19/c7e1fc9504d90da848493bad4018dd235c713a80633e48c5f0a41b63d45e/coverage-7.16.0-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0fca700cae4635656668ba6e2b66a85aac9f2622d7b2bcf82e844c409eaa1313", size = 264677, upload-time = "2026-08-28T21:53:05.741Z" }, + { url = "https://files.pythonhosted.org/packages/a4/f3/4021519dd41583ab396c81955387f927779641f6bac26818b6918a45aafc/coverage-7.16.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:584896fb8b650e999e24ef57e9513e482c12f8e15a73ee9d4584e23c99465867", size = 267610, upload-time = "2026-08-28T21:53:07.763Z" }, + { url = "https://files.pythonhosted.org/packages/55/fc/df65aac93938d8f506434c8e96440c1d696f6be0a6a01d3c6bfe5d49403e/coverage-7.16.0-cp314-cp314t-musllinux_1_2_i686.whl", hash = "sha256:949eae7e0f562b1518355aaef4b03523e49a6d3fea12aa3542d9e36c863f8267", size = 265217, upload-time = "2026-08-28T21:53:09.786Z" }, + { url = "https://files.pythonhosted.org/packages/32/2d/dc9a5e62715165fcb4c715f965f411e324917c9daeddde16536e9d36ce3f/coverage-7.16.0-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:64f0611ee05364fc85cc3e5bc371804117a76fd337720e6017332fc7c534257a", size = 268948, upload-time = "2026-08-28T21:53:11.866Z" }, + { url = "https://files.pythonhosted.org/packages/8b/4e/fe73a5560f25fca52acda76fc1554f30de081793ae4de97e920f8ab161d7/coverage-7.16.0-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:050a291b3cfe5e0df5999ef2fa5a7aff6e2db329f069d47eb63f02bde2e7e96b", size = 264061, upload-time = "2026-08-28T21:53:13.996Z" }, + { url = "https://files.pythonhosted.org/packages/b3/f7/bb78cc4b97085ebbd77fa18cbc25abfab462814efa3e2363b4e50885c775/coverage-7.16.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:a336b1e2990a64f5c356a9b8380fb9c029d56c832b801255250c44d603271bfd", size = 266371, upload-time = "2026-08-28T21:53:16.233Z" }, + { url = "https://files.pythonhosted.org/packages/aa/ec/84b4af5cd4ad498477b3bfb2217e47b048da919451053790efda66f7383c/coverage-7.16.0-cp314-cp314t-win32.whl", hash = "sha256:058631257350b31784ed43ceb808298b6f074edf4ebca4c7ce5082e6bf873a61", size = 225736, upload-time = "2026-08-28T21:53:18.632Z" }, + { url = "https://files.pythonhosted.org/packages/7e/43/50fc0e6c675c3ef14895a74bab2d6120cb5d6f4b562a3d3f5046797758dc/coverage-7.16.0-cp314-cp314t-win_amd64.whl", hash = "sha256:ed35097438dfa980c1ec75bc83edf8acbe7a374d7007e571957a257fbd0e2fb3", size = 226570, upload-time = "2026-08-28T21:53:20.754Z" }, + { url = "https://files.pythonhosted.org/packages/fc/24/9effce7bcd3c6eeb4da3561905837509e582dcdde7a7f07d6ef2c8512f76/coverage-7.16.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0466f4a5c0370461b7d8c7eb259d7d1db0b5756f13d66230b04d22a1d380ee11", size = 225879, upload-time = "2026-08-28T21:53:22.747Z" }, + { url = "https://files.pythonhosted.org/packages/4a/2c/318e4379106bc8047ba235e3732ddc87d1b393ac3db9776f5405ff14f322/coverage-7.16.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:80d7d5d744a041f08637df743ac086204ec5acbcd8432a42b00b49e607358024", size = 223257, upload-time = "2026-08-28T21:53:25.376Z" }, + { url = "https://files.pythonhosted.org/packages/81/4d/a5c54d9144e9db6505749758ba50a28be624148873751728a59cbb72d27a/coverage-7.16.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:c5feffce90c3d602e149de1c477578efc34dee5f069f9764cc15808ce01ee15c", size = 223596, upload-time = "2026-08-28T21:53:27.461Z" }, + { url = "https://files.pythonhosted.org/packages/bc/97/38e93a10899c9315964c0a4e729b3e5867f8f46e977808f9c6fbda52525a/coverage-7.16.0-cp315-cp315-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:acadbf2f2a18d7f9c7f119ac798c00c540d7c79c93abd71ed648c87891303633", size = 254699, upload-time = "2026-08-28T21:53:29.715Z" }, + { url = "https://files.pythonhosted.org/packages/fa/7a/acddda030b4630f68167f3daa94b41d22071847822a70d8178d43dcf678e/coverage-7.16.0-cp315-cp315-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:4212cec9b42fd9929e70b462732fefd8b13406371871c82f3c14397499d6550b", size = 257614, upload-time = "2026-08-28T21:53:31.948Z" }, + { url = "https://files.pythonhosted.org/packages/15/7e/225b182497c1ce6d3f0d76a3074a4dbc9f272300e92bb100df53b03de0aa/coverage-7.16.0-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1c5a43cc0ef101637ae920a9eed24cf0549ef815621eae68b3ad577ec5a7ad2f", size = 259236, upload-time = "2026-08-28T21:53:34.291Z" }, + { url = "https://files.pythonhosted.org/packages/2e/19/76641ddc50cb2410ebbd0ed7fe1052614d0e5612e802a2817521adb9febb/coverage-7.16.0-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c76a9b50a344261fe4a9bd20c322b48d3913cc48e8c37f78c21a596008296e68", size = 261433, upload-time = "2026-08-28T21:53:36.401Z" }, + { url = "https://files.pythonhosted.org/packages/12/9e/5f89de8b7c2017f36b68b4e4a25940723a748b21474820bf61e8bce0891c/coverage-7.16.0-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:80cf547379ad6b1878fd03b033b51188beab4b41824c96e7839e014a4cb947be", size = 255182, upload-time = "2026-08-28T21:53:38.496Z" }, + { url = "https://files.pythonhosted.org/packages/1a/c1/ce94b2ec502e79775efb5efa22c741ebb0bd2be10bdd29650825ff57bdcb/coverage-7.16.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:4b1d09cb5d8dc2c7164450f5217e6f0717497de9c588806a0780d352abef904a", size = 257329, upload-time = "2026-08-28T21:53:40.87Z" }, + { url = "https://files.pythonhosted.org/packages/86/8d/3f5374df3a6ca19ee5f98a6bd21dbb05f1e9d399bd9978e9821d260eab5e/coverage-7.16.0-cp315-cp315-musllinux_1_2_i686.whl", hash = "sha256:cd1e85abed2d2499c16664137ac802356316f92b4e2bf3c150bdf0c45f5dd9ae", size = 255210, upload-time = "2026-08-28T21:53:43.393Z" }, + { url = "https://files.pythonhosted.org/packages/8e/b8/1bc5751496d0be6fd9dde8ca547d9a8a9f07847856aba3f3ae5ac594cd81/coverage-7.16.0-cp315-cp315-musllinux_1_2_ppc64le.whl", hash = "sha256:360967a6fd77794c167529eec2d16ff8e38216110619d23acc3fd466a1648bee", size = 259442, upload-time = "2026-08-28T21:53:45.725Z" }, + { url = "https://files.pythonhosted.org/packages/7a/dc/8aca78e47e1e6fcc761cd28a20daf4a84bd847a7369e2701a93ccfc3d1fd/coverage-7.16.0-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:92cbc2bf4f7f67c79f1d3ca4fe8c50faddf48e852a3d07eaaf02dc014889832f", size = 254618, upload-time = "2026-08-28T21:53:48.292Z" }, + { url = "https://files.pythonhosted.org/packages/73/fd/787842cdf6ce16ac5c1bd8a26549bab3b3f27b02500075bc540dc7853bca/coverage-7.16.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:cce4dc8528453128c6fae523b15f3887fbea1d4d7c9eb9639d3d4fdcbe570c73", size = 256541, upload-time = "2026-08-28T21:53:50.805Z" }, + { url = "https://files.pythonhosted.org/packages/ef/79/8df302cbef373dd1f3401044cdb94dfc74517e5af2af27b4d0e721557e0e/coverage-7.16.0-cp315-cp315-win32.whl", hash = "sha256:5205baea687133613dced668a3d0168ea1479349615bfc255849a7944988c889", size = 225429, upload-time = "2026-08-28T21:53:53.177Z" }, + { url = "https://files.pythonhosted.org/packages/85/87/5bad7ac45f76b3728ca211028ee561c2ede3ba44da401129e28bb8737291/coverage-7.16.0-cp315-cp315-win_amd64.whl", hash = "sha256:4fcb5f07a9b7083bfb715115d27ce263ba2b5b89dddeee536b295ba0e3c2c627", size = 225903, upload-time = "2026-08-28T21:53:55.535Z" }, + { url = "https://files.pythonhosted.org/packages/cc/ea/67d84b11caf240f059ec313f616d82212df5004e8bc85802c1edfc50bb3d/coverage-7.16.0-cp315-cp315-win_arm64.whl", hash = "sha256:d568a8adcec0eda42ec23e5e65dfb8c184fc255120f9e99b484f7c869d923fb9", size = 225334, upload-time = "2026-08-28T21:53:57.769Z" }, + { url = "https://files.pythonhosted.org/packages/65/21/a88349cce3ff720729b754916ac47e2e3646a8137552e4fa7cdd5967cc7f/coverage-7.16.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:3e8037e8213adf882e9d7eedd2c5c557933ab0b9632c42d98fe98ec9bcdb4025", size = 223980, upload-time = "2026-08-28T21:54:00.082Z" }, + { url = "https://files.pythonhosted.org/packages/fd/02/4d54abf3e6a4d8b7675921b20e91163b1064a5a9dbefebb71c05065dd136/coverage-7.16.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:289f2ed4d56eebf029b649e7dfc3c1153b111962a75e294cdd8e4a1598a04cc3", size = 224276, upload-time = "2026-08-28T21:54:02.381Z" }, + { url = "https://files.pythonhosted.org/packages/f6/39/10dbc96d95d20b9b041045d293480bd49e536180e93af62dd7662376284d/coverage-7.16.0-cp315-cp315t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:9b83f6ac575530783771c8dcf05284f7c8b5b12f1e7cb226d63445aac4497a3a", size = 265135, upload-time = "2026-08-28T21:54:04.558Z" }, + { url = "https://files.pythonhosted.org/packages/e7/3b/6b326544afd1a8aef3a495bbae109a7ab5baf23e04a2741d8d64e2df2ba2/coverage-7.16.0-cp315-cp315t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:2c3ff6580f2dfc5bec34717b85b2e6cf5ec993b721e7bb58a794babd525a8178", size = 268216, upload-time = "2026-08-28T21:54:06.97Z" }, + { url = "https://files.pythonhosted.org/packages/54/34/1dc8265f3ed990690e24d5f31ff79bc9fb9b25d54f9f89bebad5a6a8b7a1/coverage-7.16.0-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:507596cee23e9968b1934fe86d799b76166541af0a293930918b1b48a5c84bd2", size = 270772, upload-time = "2026-08-28T21:54:09.234Z" }, + { url = "https://files.pythonhosted.org/packages/66/a7/3a8463713a402b44044ec832f4a76e442ce4b3a207804303f4d1dc1a9bb4/coverage-7.16.0-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:edc2be98e6c55ccc5ff7832bb64f023a4b03dba39dfa84b850046cf08a8249b0", size = 271752, upload-time = "2026-08-28T21:54:11.701Z" }, + { url = "https://files.pythonhosted.org/packages/25/3b/dd5e795cfbe1842f69899189089ae289a96d6a68de312960ea668542e33c/coverage-7.16.0-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:9c0690994b84a15a53bdd39e0b2fdb539b22533820623eb86ba75b93760c645b", size = 265589, upload-time = "2026-08-28T21:54:14.12Z" }, + { url = "https://files.pythonhosted.org/packages/1b/b6/fd90636cbd95cb018312f6ca1ca2bbd70fbe8e4ee6f3992fc36a4230364e/coverage-7.16.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:de24c62bf798940a14674a47489a81b79915ec4134f556d5199830e065225dd0", size = 268596, upload-time = "2026-08-28T21:54:16.303Z" }, + { url = "https://files.pythonhosted.org/packages/91/10/ef2d59264f3b3b358cc5885ca375e6cdbda7c195e78304d5aae800a72d9d/coverage-7.16.0-cp315-cp315t-musllinux_1_2_i686.whl", hash = "sha256:69474d81f198774c9d2937599ca5da04c9e1c5de5032da23c607ce4960ce360e", size = 265072, upload-time = "2026-08-28T21:54:18.597Z" }, + { url = "https://files.pythonhosted.org/packages/3f/5b/400891c364c0170408d172501b340b18611800f4c42d8fbb16f9f5497c24/coverage-7.16.0-cp315-cp315t-musllinux_1_2_ppc64le.whl", hash = "sha256:72a0795cc6d34acc2b03dfeabdc82b61b72087f2737018b56ac92c1cf5446c54", size = 269768, upload-time = "2026-08-28T21:54:20.985Z" }, + { url = "https://files.pythonhosted.org/packages/98/93/9792c80271df04d287d21ed5d662fd8fa58b1737888d817679b1ce5d2fab/coverage-7.16.0-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:d9a218d3f9c7d6916684ed5ba94f620661117a730e733cd6ef5e87accc5872eb", size = 265211, upload-time = "2026-08-28T21:54:23.344Z" }, + { url = "https://files.pythonhosted.org/packages/81/67/5b8f827cfa6616e6bd7ba9397acfe7e3c4fd5b9fca4125511d5089f55d5a/coverage-7.16.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:49fa72ead28c8216f8916398a4f3c4669acb30a061822810ee20a727a1be2897", size = 267170, upload-time = "2026-08-28T21:54:25.85Z" }, + { url = "https://files.pythonhosted.org/packages/5c/ee/c135d2d2cb617d744bc3e13c922f2fae66964494176ddef225dc4656bd2c/coverage-7.16.0-cp315-cp315t-win32.whl", hash = "sha256:27461af9f3ed7d2cf2411eb083784f87055ebf42211789ae3a216c48609bc743", size = 225731, upload-time = "2026-08-28T21:54:28.151Z" }, + { url = "https://files.pythonhosted.org/packages/8a/4d/dc3d53eadf155916e183bf5dfacbfc4aa5bfb7f13b7da11c01caa7a05cbc/coverage-7.16.0-cp315-cp315t-win_amd64.whl", hash = "sha256:c5612cc20ca76abc883e50269af47c1494b42958bb63dbb9aa79729a1ab5f7d3", size = 226562, upload-time = "2026-08-28T21:54:30.42Z" }, + { url = "https://files.pythonhosted.org/packages/2f/00/ac9da1a60a4e84c3ad0f7db4723fd327154a8f9add210c0dcd2db3ec5156/coverage-7.16.0-cp315-cp315t-win_arm64.whl", hash = "sha256:2ddaa9e2af4760a329d80008b7a3b4762fbb0dbcb169199360f9a5179c32f2dc", size = 225872, upload-time = "2026-08-28T21:54:32.806Z" }, + { url = "https://files.pythonhosted.org/packages/b1/5a/234e8fadf85c3cc48cb31c247b9e8e0c7f06ece80f5b29f9b8c241f9da4c/coverage-7.16.0-py3-none-any.whl", hash = "sha256:245f7de6d023a5bba375dbec9f2e0869bfa26ac0cc639bbb7b4c814884000b73", size = 214977, upload-time = "2026-08-28T21:54:35.189Z" }, +] + +[[package]] +name = "deptry" +version = "0.25.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "click" }, + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "packaging" }, + { name = "requirements-parser" }, + { name = "tomli", marker = "python_full_version < '3.15'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b8/b2/50ccc99362ae7757342978b7ecb3b98e47fade721fd617d74db1948ec3a1/deptry-0.25.1.tar.gz", hash = "sha256:45c8cd982c85cd4faae573ddff6920de7eec735336db6973f26a765ae7950f7d", size = 509748, upload-time = "2026-03-18T23:22:18.139Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/33/1d/b538dc635e873b25360d761cfe1fa0ccd7d6c69b698047e552f33401e60d/deptry-0.25.1-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:a4dd1148db24a1ddacfa8b840836c6019c2f864fcb7579dd089fd217606338c8", size = 1850319, upload-time = "2026-03-18T23:22:15.65Z" }, + { url = "https://files.pythonhosted.org/packages/fe/a9/511477a8f0ae4f6021d68a80bdca77e7ffb0722008dc24ee5d9ef49f5c88/deptry-0.25.1-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:c67c666d916ef12013c0772e40d78be0f21577a495d8d99ec5fcb18c332d393d", size = 1759259, upload-time = "2026-03-18T23:22:30.853Z" }, + { url = "https://files.pythonhosted.org/packages/4f/4b/c9f0bdda410912a6df79a789cb118fa29acae02a397794ead3c84adcda5c/deptry-0.25.1-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:58d39279828dbf4efc1abb40bf50a71b21499c36759bed5a8d8a3c0e3149b091", size = 1872012, upload-time = "2026-03-18T23:22:19.145Z" }, + { url = "https://files.pythonhosted.org/packages/72/9c/6f6f9125bac74b5d5d2af89536cbdb3fa159b6466aa097b74e7e85e8e030/deptry-0.25.1-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:14bfcc28b4326ed8c6abb30691b19077d4ef8613cfba6c37ef5b1f471775bf6f", size = 1926575, upload-time = "2026-03-18T23:22:11.269Z" }, + { url = "https://files.pythonhosted.org/packages/52/48/2a5e705a7f898295966ade67bd1223e2af96da433e25b39f6b9483ba2c7b/deptry-0.25.1-cp310-abi3-musllinux_1_1_aarch64.whl", hash = "sha256:555f5f9a487899ec9bf301eecba1745e14d212c4b354f4d3a5fd691e907366d3", size = 2050816, upload-time = "2026-03-18T23:22:27.439Z" }, + { url = "https://files.pythonhosted.org/packages/5f/c6/50f189a894e1f3bf21266299112c8a06cb731838976e1b9a9cadd0b4a86e/deptry-0.25.1-cp310-abi3-musllinux_1_1_x86_64.whl", hash = "sha256:18d21b3545ab2bfec53f3f45c6f5f201d55f713323327f8d12674505469ae6b7", size = 2145416, upload-time = "2026-03-18T23:22:24.682Z" }, + { url = "https://files.pythonhosted.org/packages/7a/6a/3f82f7a06217778282bc4456af1b4ffb3bc4b2c8e7891d00e8323f9ad0b8/deptry-0.25.1-cp310-abi3-win_amd64.whl", hash = "sha256:b59a560cb7dffb21832a98bb80d33d614cfb5630ea36ce21833eabf4eae3df99", size = 1718489, upload-time = "2026-03-18T23:22:28.589Z" }, + { url = "https://files.pythonhosted.org/packages/c7/7f/cd6b3ac8cf95f2f1c5c7a74ff6452e9098af89a9b56607381f677880641e/deptry-0.25.1-cp310-abi3-win_arm64.whl", hash = "sha256:6efffd8116fb9d2c45a251382ce4ce1c38dbb17179f581ec9231ed5390f7fc12", size = 1647020, upload-time = "2026-03-18T23:22:23.311Z" }, +] + +[[package]] +name = "docutils" +version = "0.22.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ae/b6/03bb70946330e88ffec97aefd3ea75ba575cb2e762061e0e62a213befee8/docutils-0.22.4.tar.gz", hash = "sha256:4db53b1fde9abecbb74d91230d32ab626d94f6badfc575d6db9194a49df29968", size = 2291750, upload-time = "2025-12-18T19:00:26.443Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/02/10/5da547df7a391dcde17f59520a231527b8571e6f46fc8efb02ccb370ab12/docutils-0.22.4-py3-none-any.whl", hash = "sha256:d0013f540772d1420576855455d050a2180186c91c15779301ac2ccb3eeb68de", size = 633196, upload-time = "2025-12-18T19:00:18.077Z" }, +] + +[[package]] +name = "duckdb" +version = "1.5.5" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/19/e57151753576373c6696a12022648546cca6038e8833fda2908ee2342d9b/duckdb-1.5.5.tar.gz", hash = "sha256:72f33ee57ca7595b23957671a2cc7f7fe2be0ecc2d68f63abedcfcaa3a5c1238", size = 18066741, upload-time = "2026-07-22T10:55:17.819Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d6/40/2e05d324400fdaa5656c9f48d6298da421cb034d85e509fa0e6e325cf04b/duckdb-1.5.5-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:d4dd65f8941a604b947e0b9b4b4f7165988e29a23ec0b69b4038520956d9933e", size = 32753858, upload-time = "2026-07-22T10:54:05.514Z" }, + { url = "https://files.pythonhosted.org/packages/79/15/5ceb58ffb5bb8a62b3fd7abb39c41467cdf94850ece02e6d88664dfc75ce/duckdb-1.5.5-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:33db46679b071f108d57139493dee2d37e1f5efcf5c5c039c2969eed11a6c8a7", size = 17368293, upload-time = "2026-07-22T10:54:09.139Z" }, + { url = "https://files.pythonhosted.org/packages/bf/5c/bf02da0b354fe83cca4f95a4fbf762181af466f7d551ab2a093f7698882a/duckdb-1.5.5-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f0b88535a5d86fdd63dba6ea02ab68c003dfb9e4892b11256ef24c4da208baae", size = 15509131, upload-time = "2026-07-22T10:54:12.228Z" }, + { url = "https://files.pythonhosted.org/packages/ea/a9/5f1f09da421d8e930e0b063d11c1b3f90363f40ede74438cd188afdd13a2/duckdb-1.5.5-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f316eae2323d9a851883fdf2dee91c1f9efe251ab33e14a2272f82a913422ed6", size = 19391959, upload-time = "2026-07-22T10:54:15.551Z" }, + { url = "https://files.pythonhosted.org/packages/4f/98/6549769f158126fa64fd6c1ac2eb59a18282146c939867a3eb31b7c1db07/duckdb-1.5.5-cp312-cp312-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7a6d2d11859d82a936ebdcb30ce3d8a1cbb3e990bff05c12abb9b54c44fa7bd1", size = 21510909, upload-time = "2026-07-22T10:54:19.681Z" }, + { url = "https://files.pythonhosted.org/packages/af/b7/5753b41d3124838f868f9f523362812d9fc45409e9e4dd70dcbb0a25826e/duckdb-1.5.5-cp312-cp312-win_amd64.whl", hash = "sha256:ddfbdb096c11d51ee22492397d342c90a82e62c5d09961477895934d0a25372f", size = 13168544, upload-time = "2026-07-22T10:54:22.789Z" }, + { url = "https://files.pythonhosted.org/packages/5c/28/44b679c7d46245f8398feae7edac959d1b83d4eb143e25b3fce0630b78bd/duckdb-1.5.5-cp312-cp312-win_arm64.whl", hash = "sha256:2725d2b9ace3a4e75d72fc5a239f6a44b502c580edadb8fb2676db772c5f9282", size = 13988684, upload-time = "2026-07-22T10:54:26.003Z" }, + { url = "https://files.pythonhosted.org/packages/47/37/4a38116e7700720fd152c666292214fd3abdf916496991296d8d1f66efbf/duckdb-1.5.5-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:cd98829b67788609017e65c761bd42a5dd0f9129441bed8bda4d6881ccf819f0", size = 32754294, upload-time = "2026-07-22T10:54:29.822Z" }, + { url = "https://files.pythonhosted.org/packages/66/42/7d392f1ba1eee0eaf4ab4c8c7a604bfe3536cd63f979cf5c98798664f807/duckdb-1.5.5-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:feead93c56679b79592d437c62975d39cb67adedffa7592c763baf8160ac7366", size = 17368211, upload-time = "2026-07-22T10:54:33.359Z" }, + { url = "https://files.pythonhosted.org/packages/9f/a5/0a6f4fa60562faa615e55e15bd1953a2f2b17a8edd8105e5cda215e43457/duckdb-1.5.5-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:49c963d9469373d7aba8d750d9ea565ab823e94166efed953f184dd9b169b98c", size = 15509136, upload-time = "2026-07-22T10:54:36.369Z" }, + { url = "https://files.pythonhosted.org/packages/e4/cb/023c89f51978545b9fab318581bba0c457a58e7530d2d933e54ae7d8647c/duckdb-1.5.5-cp313-cp313-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a736217825461732b5442d05a220f3da2e23a0dae114efbf08c9bf171b53098a", size = 19392147, upload-time = "2026-07-22T10:54:39.551Z" }, + { url = "https://files.pythonhosted.org/packages/3e/c5/41bef391fb8b23dbc133c9f2ba016e7a7a8124513d2cc1b430f1897d87e4/duckdb-1.5.5-cp313-cp313-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:078e6a60dd8eedde5832f45422ca5c4a6b8c837aeabd8a56ca0b7d933f588053", size = 21511060, upload-time = "2026-07-22T10:54:42.788Z" }, + { url = "https://files.pythonhosted.org/packages/07/9f/c44dfc1f924ac29b3252dc1b91393c01d009dbfe9f8ed33f10b986151bd1/duckdb-1.5.5-cp313-cp313-win_amd64.whl", hash = "sha256:6826504277dba513c0c5d71d828456c94d729c9d2482f94b2e289f90a9167e28", size = 13168028, upload-time = "2026-07-22T10:54:46.127Z" }, + { url = "https://files.pythonhosted.org/packages/ca/88/591384b2cd59abddd6f5dc175e60374f9abae6064429f0c4402854c10f44/duckdb-1.5.5-cp313-cp313-win_arm64.whl", hash = "sha256:baa9c5702002fabb559ded2a39008f9f421fcbc7237d388b8213eff1e08858de", size = 13989955, upload-time = "2026-07-22T10:54:49.262Z" }, + { url = "https://files.pythonhosted.org/packages/3e/56/12c65bfa2d2605b81981b264788891bcf11ec72227889554cead5d8d13b9/duckdb-1.5.5-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:8e6413dd40facb7b8ab21bd844450cd8f549b29e138635be9cf090ef4d2049e2", size = 32761946, upload-time = "2026-07-22T10:54:53.412Z" }, + { url = "https://files.pythonhosted.org/packages/b9/46/682ce155f17e0d2822d4f13ee3db9ca4b5b7c2da61b841b2629035e1f4bc/duckdb-1.5.5-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:64078acfd16541132ac6e191eb81b2845554444a0305cc1aa581ba107e514aa8", size = 17375069, upload-time = "2026-07-22T10:54:57.269Z" }, + { url = "https://files.pythonhosted.org/packages/39/ce/a24bcbd3289c8f305a430759c5fc12242740b4af3e17f7593f3a34e333d2/duckdb-1.5.5-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:8c11775cc99a447618d5f1840126db17f2652f3eae05529df4f81f40e2df7151", size = 15519791, upload-time = "2026-07-22T10:55:00.681Z" }, + { url = "https://files.pythonhosted.org/packages/d9/76/3a01afbc615c1d418c0de58a6b68ac5ce2a8563232c0464bfbc2ce552398/duckdb-1.5.5-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:77bbc1e6ba12e1e06f9020117bdf848627ecfdf36f907550e62e008e6109dece", size = 19398251, upload-time = "2026-07-22T10:55:04.168Z" }, + { url = "https://files.pythonhosted.org/packages/a1/43/3a5e81d1728f4d234c79bfe385808ee7c04834f7c37a4b5c257459c25614/duckdb-1.5.5-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fbf0f2d48b43c6c304d00463b463c27ead6c4b01c3c1816b750f728decf71afe", size = 21513851, upload-time = "2026-07-22T10:55:07.864Z" }, + { url = "https://files.pythonhosted.org/packages/91/41/fc7c829172c60ca22485251eab285f4f1a0d87b486a024c726f21471d86e/duckdb-1.5.5-cp314-cp314-win_amd64.whl", hash = "sha256:9dc826c4b50e64f6c4e4d07a3a9cb075ef70ba3899dc43ec5493dc3d7b04b353", size = 13691858, upload-time = "2026-07-22T10:55:11.181Z" }, + { url = "https://files.pythonhosted.org/packages/e1/2c/95d9216b79e9273689d7ebce125a54503ed0c9bd7da931f0265888e99779/duckdb-1.5.5-cp314-cp314-win_arm64.whl", hash = "sha256:63e48d4b74b15aeacd688976432a7225163df8c226eddeb8536bba2d4d4ff433", size = 14470180, upload-time = "2026-07-22T10:55:14.445Z" }, +] + +[[package]] +name = "idna" +version = "3.19" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/5f/f7/abb373e5757eaec4b922b92f97ec8d6d7e057cf06778247604fbc4e7c3f3/idna-3.19.tar.gz", hash = "sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15", size = 215237, upload-time = "2026-08-18T05:14:24.27Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/57/b0/0e52c878c53f245edd3a11020f20979b3f490f245af532c7cae3027754b5/idna-3.19-py3-none-any.whl", hash = "sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4", size = 68550, upload-time = "2026-08-18T05:14:22.343Z" }, +] + +[[package]] +name = "imagesize" +version = "2.0.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/fb/5e/513ff06670c84e7b9887c1fdf61b2d42b4f574a831f2f1d2222023049d8a/imagesize-2.0.1.tar.gz", hash = "sha256:b2ba6a4dea487a7ebcd53248d3476aca449d30db12a2dde5e0c5ca9624fd77e5", size = 1883774, upload-time = "2026-08-24T12:35:19.13Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/01/f9/575c8d760eae1fc99651b7cc5efd96ad5379ca4d6b53750b0fb4fe983f34/imagesize-2.0.1-py3-none-any.whl", hash = "sha256:ea0c9a0384df69ed86a943a15cde37d0360b82491b3910dc2215e202e62b5b02", size = 14794, upload-time = "2026-08-24T12:35:12.548Z" }, +] + +[[package]] +name = "imdb" +source = { editable = "." } +dependencies = [ + { name = "duckdb" }, + { name = "numpy" }, + { name = "pandas" }, +] + +[package.dev-dependencies] +dev = [ + { name = "deptry" }, + { name = "numpydoc" }, + { name = "ruff" }, + { name = "ty" }, +] +test = [ + { name = "coverage" }, + { name = "pytest" }, + { name = "pytest-cov" }, +] +types = [ + { name = "pandas-stubs" }, + { name = "ty" }, +] + +[package.metadata] +requires-dist = [ + { name = "duckdb", specifier = ">=1.5" }, + { name = "numpy", specifier = ">=2" }, + { name = "pandas", specifier = ">=3" }, +] + +[package.metadata.requires-dev] +dev = [ + { name = "deptry" }, + { name = "numpydoc" }, + { name = "ruff" }, + { name = "ty" }, +] +test = [ + { name = "coverage", extras = ["toml"] }, + { name = "pytest" }, + { name = "pytest-cov" }, +] +types = [ + { name = "pandas-stubs" }, + { name = "ty" }, +] + +[[package]] +name = "iniconfig" +version = "2.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730", size = 20503, upload-time = "2025-10-18T21:55:43.219Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12", size = 7484, upload-time = "2025-10-18T21:55:41.639Z" }, +] + +[[package]] +name = "jinja2" +version = "3.1.6" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markupsafe" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/df/bf/f7da0350254c0ed7c72f3e33cef02e048281fec7ecec5f032d4aac52226b/jinja2-3.1.6.tar.gz", hash = "sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d", size = 245115, upload-time = "2025-03-05T20:05:02.478Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" }, +] + +[[package]] +name = "markupsafe" +version = "3.0.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7e/99/7690b6d4034fffd95959cbe0c02de8deb3098cc577c67bb6a24fe5d7caa7/markupsafe-3.0.3.tar.gz", hash = "sha256:722695808f4b6457b320fdc131280796bdceb04ab50fe1795cd540799ebe1698", size = 80313, upload-time = "2025-09-27T18:37:40.426Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5a/72/147da192e38635ada20e0a2e1a51cf8823d2119ce8883f7053879c2199b5/markupsafe-3.0.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:d53197da72cc091b024dd97249dfc7794d6a56530370992a5e1a08983ad9230e", size = 11615, upload-time = "2025-09-27T18:36:30.854Z" }, + { url = "https://files.pythonhosted.org/packages/9a/81/7e4e08678a1f98521201c3079f77db69fb552acd56067661f8c2f534a718/markupsafe-3.0.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1872df69a4de6aead3491198eaf13810b565bdbeec3ae2dc8780f14458ec73ce", size = 12020, upload-time = "2025-09-27T18:36:31.971Z" }, + { url = "https://files.pythonhosted.org/packages/1e/2c/799f4742efc39633a1b54a92eec4082e4f815314869865d876824c257c1e/markupsafe-3.0.3-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3a7e8ae81ae39e62a41ec302f972ba6ae23a5c5396c8e60113e9066ef893da0d", size = 24332, upload-time = "2025-09-27T18:36:32.813Z" }, + { url = "https://files.pythonhosted.org/packages/3c/2e/8d0c2ab90a8c1d9a24f0399058ab8519a3279d1bd4289511d74e909f060e/markupsafe-3.0.3-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d6dd0be5b5b189d31db7cda48b91d7e0a9795f31430b7f271219ab30f1d3ac9d", size = 22947, upload-time = "2025-09-27T18:36:33.86Z" }, + { url = "https://files.pythonhosted.org/packages/2c/54/887f3092a85238093a0b2154bd629c89444f395618842e8b0c41783898ea/markupsafe-3.0.3-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:94c6f0bb423f739146aec64595853541634bde58b2135f27f61c1ffd1cd4d16a", size = 21962, upload-time = "2025-09-27T18:36:35.099Z" }, + { url = "https://files.pythonhosted.org/packages/c9/2f/336b8c7b6f4a4d95e91119dc8521402461b74a485558d8f238a68312f11c/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:be8813b57049a7dc738189df53d69395eba14fb99345e0a5994914a3864c8a4b", size = 23760, upload-time = "2025-09-27T18:36:36.001Z" }, + { url = "https://files.pythonhosted.org/packages/32/43/67935f2b7e4982ffb50a4d169b724d74b62a3964bc1a9a527f5ac4f1ee2b/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:83891d0e9fb81a825d9a6d61e3f07550ca70a076484292a70fde82c4b807286f", size = 21529, upload-time = "2025-09-27T18:36:36.906Z" }, + { url = "https://files.pythonhosted.org/packages/89/e0/4486f11e51bbba8b0c041098859e869e304d1c261e59244baa3d295d47b7/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:77f0643abe7495da77fb436f50f8dab76dbc6e5fd25d39589a0f1fe6548bfa2b", size = 23015, upload-time = "2025-09-27T18:36:37.868Z" }, + { url = "https://files.pythonhosted.org/packages/2f/e1/78ee7a023dac597a5825441ebd17170785a9dab23de95d2c7508ade94e0e/markupsafe-3.0.3-cp312-cp312-win32.whl", hash = "sha256:d88b440e37a16e651bda4c7c2b930eb586fd15ca7406cb39e211fcff3bf3017d", size = 14540, upload-time = "2025-09-27T18:36:38.761Z" }, + { url = "https://files.pythonhosted.org/packages/aa/5b/bec5aa9bbbb2c946ca2733ef9c4ca91c91b6a24580193e891b5f7dbe8e1e/markupsafe-3.0.3-cp312-cp312-win_amd64.whl", hash = "sha256:26a5784ded40c9e318cfc2bdb30fe164bdb8665ded9cd64d500a34fb42067b1c", size = 15105, upload-time = "2025-09-27T18:36:39.701Z" }, + { url = "https://files.pythonhosted.org/packages/e5/f1/216fc1bbfd74011693a4fd837e7026152e89c4bcf3e77b6692fba9923123/markupsafe-3.0.3-cp312-cp312-win_arm64.whl", hash = "sha256:35add3b638a5d900e807944a078b51922212fb3dedb01633a8defc4b01a3c85f", size = 13906, upload-time = "2025-09-27T18:36:40.689Z" }, + { url = "https://files.pythonhosted.org/packages/38/2f/907b9c7bbba283e68f20259574b13d005c121a0fa4c175f9bed27c4597ff/markupsafe-3.0.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:e1cf1972137e83c5d4c136c43ced9ac51d0e124706ee1c8aa8532c1287fa8795", size = 11622, upload-time = "2025-09-27T18:36:41.777Z" }, + { url = "https://files.pythonhosted.org/packages/9c/d9/5f7756922cdd676869eca1c4e3c0cd0df60ed30199ffd775e319089cb3ed/markupsafe-3.0.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:116bb52f642a37c115f517494ea5feb03889e04df47eeff5b130b1808ce7c219", size = 12029, upload-time = "2025-09-27T18:36:43.257Z" }, + { url = "https://files.pythonhosted.org/packages/00/07/575a68c754943058c78f30db02ee03a64b3c638586fba6a6dd56830b30a3/markupsafe-3.0.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:133a43e73a802c5562be9bbcd03d090aa5a1fe899db609c29e8c8d815c5f6de6", size = 24374, upload-time = "2025-09-27T18:36:44.508Z" }, + { url = "https://files.pythonhosted.org/packages/a9/21/9b05698b46f218fc0e118e1f8168395c65c8a2c750ae2bab54fc4bd4e0e8/markupsafe-3.0.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ccfcd093f13f0f0b7fdd0f198b90053bf7b2f02a3927a30e63f3ccc9df56b676", size = 22980, upload-time = "2025-09-27T18:36:45.385Z" }, + { url = "https://files.pythonhosted.org/packages/7f/71/544260864f893f18b6827315b988c146b559391e6e7e8f7252839b1b846a/markupsafe-3.0.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:509fa21c6deb7a7a273d629cf5ec029bc209d1a51178615ddf718f5918992ab9", size = 21990, upload-time = "2025-09-27T18:36:46.916Z" }, + { url = "https://files.pythonhosted.org/packages/c2/28/b50fc2f74d1ad761af2f5dcce7492648b983d00a65b8c0e0cb457c82ebbe/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a4afe79fb3de0b7097d81da19090f4df4f8d3a2b3adaa8764138aac2e44f3af1", size = 23784, upload-time = "2025-09-27T18:36:47.884Z" }, + { url = "https://files.pythonhosted.org/packages/ed/76/104b2aa106a208da8b17a2fb72e033a5a9d7073c68f7e508b94916ed47a9/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:795e7751525cae078558e679d646ae45574b47ed6e7771863fcc079a6171a0fc", size = 21588, upload-time = "2025-09-27T18:36:48.82Z" }, + { url = "https://files.pythonhosted.org/packages/b5/99/16a5eb2d140087ebd97180d95249b00a03aa87e29cc224056274f2e45fd6/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:8485f406a96febb5140bfeca44a73e3ce5116b2501ac54fe953e488fb1d03b12", size = 23041, upload-time = "2025-09-27T18:36:49.797Z" }, + { url = "https://files.pythonhosted.org/packages/19/bc/e7140ed90c5d61d77cea142eed9f9c303f4c4806f60a1044c13e3f1471d0/markupsafe-3.0.3-cp313-cp313-win32.whl", hash = "sha256:bdd37121970bfd8be76c5fb069c7751683bdf373db1ed6c010162b2a130248ed", size = 14543, upload-time = "2025-09-27T18:36:51.584Z" }, + { url = "https://files.pythonhosted.org/packages/05/73/c4abe620b841b6b791f2edc248f556900667a5a1cf023a6646967ae98335/markupsafe-3.0.3-cp313-cp313-win_amd64.whl", hash = "sha256:9a1abfdc021a164803f4d485104931fb8f8c1efd55bc6b748d2f5774e78b62c5", size = 15113, upload-time = "2025-09-27T18:36:52.537Z" }, + { url = "https://files.pythonhosted.org/packages/f0/3a/fa34a0f7cfef23cf9500d68cb7c32dd64ffd58a12b09225fb03dd37d5b80/markupsafe-3.0.3-cp313-cp313-win_arm64.whl", hash = "sha256:7e68f88e5b8799aa49c85cd116c932a1ac15caaa3f5db09087854d218359e485", size = 13911, upload-time = "2025-09-27T18:36:53.513Z" }, + { url = "https://files.pythonhosted.org/packages/e4/d7/e05cd7efe43a88a17a37b3ae96e79a19e846f3f456fe79c57ca61356ef01/markupsafe-3.0.3-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:218551f6df4868a8d527e3062d0fb968682fe92054e89978594c28e642c43a73", size = 11658, upload-time = "2025-09-27T18:36:54.819Z" }, + { url = "https://files.pythonhosted.org/packages/99/9e/e412117548182ce2148bdeacdda3bb494260c0b0184360fe0d56389b523b/markupsafe-3.0.3-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:3524b778fe5cfb3452a09d31e7b5adefeea8c5be1d43c4f810ba09f2ceb29d37", size = 12066, upload-time = "2025-09-27T18:36:55.714Z" }, + { url = "https://files.pythonhosted.org/packages/bc/e6/fa0ffcda717ef64a5108eaa7b4f5ed28d56122c9a6d70ab8b72f9f715c80/markupsafe-3.0.3-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4e885a3d1efa2eadc93c894a21770e4bc67899e3543680313b09f139e149ab19", size = 25639, upload-time = "2025-09-27T18:36:56.908Z" }, + { url = "https://files.pythonhosted.org/packages/96/ec/2102e881fe9d25fc16cb4b25d5f5cde50970967ffa5dddafdb771237062d/markupsafe-3.0.3-cp313-cp313t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8709b08f4a89aa7586de0aadc8da56180242ee0ada3999749b183aa23df95025", size = 23569, upload-time = "2025-09-27T18:36:57.913Z" }, + { url = "https://files.pythonhosted.org/packages/4b/30/6f2fce1f1f205fc9323255b216ca8a235b15860c34b6798f810f05828e32/markupsafe-3.0.3-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:b8512a91625c9b3da6f127803b166b629725e68af71f8184ae7e7d54686a56d6", size = 23284, upload-time = "2025-09-27T18:36:58.833Z" }, + { url = "https://files.pythonhosted.org/packages/58/47/4a0ccea4ab9f5dcb6f79c0236d954acb382202721e704223a8aafa38b5c8/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:9b79b7a16f7fedff2495d684f2b59b0457c3b493778c9eed31111be64d58279f", size = 24801, upload-time = "2025-09-27T18:36:59.739Z" }, + { url = "https://files.pythonhosted.org/packages/6a/70/3780e9b72180b6fecb83a4814d84c3bf4b4ae4bf0b19c27196104149734c/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_riscv64.whl", hash = "sha256:12c63dfb4a98206f045aa9563db46507995f7ef6d83b2f68eda65c307c6829eb", size = 22769, upload-time = "2025-09-27T18:37:00.719Z" }, + { url = "https://files.pythonhosted.org/packages/98/c5/c03c7f4125180fc215220c035beac6b9cb684bc7a067c84fc69414d315f5/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:8f71bc33915be5186016f675cd83a1e08523649b0e33efdb898db577ef5bb009", size = 23642, upload-time = "2025-09-27T18:37:01.673Z" }, + { url = "https://files.pythonhosted.org/packages/80/d6/2d1b89f6ca4bff1036499b1e29a1d02d282259f3681540e16563f27ebc23/markupsafe-3.0.3-cp313-cp313t-win32.whl", hash = "sha256:69c0b73548bc525c8cb9a251cddf1931d1db4d2258e9599c28c07ef3580ef354", size = 14612, upload-time = "2025-09-27T18:37:02.639Z" }, + { url = "https://files.pythonhosted.org/packages/2b/98/e48a4bfba0a0ffcf9925fe2d69240bfaa19c6f7507b8cd09c70684a53c1e/markupsafe-3.0.3-cp313-cp313t-win_amd64.whl", hash = "sha256:1b4b79e8ebf6b55351f0d91fe80f893b4743f104bff22e90697db1590e47a218", size = 15200, upload-time = "2025-09-27T18:37:03.582Z" }, + { url = "https://files.pythonhosted.org/packages/0e/72/e3cc540f351f316e9ed0f092757459afbc595824ca724cbc5a5d4263713f/markupsafe-3.0.3-cp313-cp313t-win_arm64.whl", hash = "sha256:ad2cf8aa28b8c020ab2fc8287b0f823d0a7d8630784c31e9ee5edea20f406287", size = 13973, upload-time = "2025-09-27T18:37:04.929Z" }, + { url = "https://files.pythonhosted.org/packages/33/8a/8e42d4838cd89b7dde187011e97fe6c3af66d8c044997d2183fbd6d31352/markupsafe-3.0.3-cp314-cp314-macosx_10_13_x86_64.whl", hash = "sha256:eaa9599de571d72e2daf60164784109f19978b327a3910d3e9de8c97b5b70cfe", size = 11619, upload-time = "2025-09-27T18:37:06.342Z" }, + { url = "https://files.pythonhosted.org/packages/b5/64/7660f8a4a8e53c924d0fa05dc3a55c9cee10bbd82b11c5afb27d44b096ce/markupsafe-3.0.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:c47a551199eb8eb2121d4f0f15ae0f923d31350ab9280078d1e5f12b249e0026", size = 12029, upload-time = "2025-09-27T18:37:07.213Z" }, + { url = "https://files.pythonhosted.org/packages/da/ef/e648bfd021127bef5fa12e1720ffed0c6cbb8310c8d9bea7266337ff06de/markupsafe-3.0.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f34c41761022dd093b4b6896d4810782ffbabe30f2d443ff5f083e0cbbb8c737", size = 24408, upload-time = "2025-09-27T18:37:09.572Z" }, + { url = "https://files.pythonhosted.org/packages/41/3c/a36c2450754618e62008bf7435ccb0f88053e07592e6028a34776213d877/markupsafe-3.0.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:457a69a9577064c05a97c41f4e65148652db078a3a509039e64d3467b9e7ef97", size = 23005, upload-time = "2025-09-27T18:37:10.58Z" }, + { url = "https://files.pythonhosted.org/packages/bc/20/b7fdf89a8456b099837cd1dc21974632a02a999ec9bf7ca3e490aacd98e7/markupsafe-3.0.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:e8afc3f2ccfa24215f8cb28dcf43f0113ac3c37c2f0f0806d8c70e4228c5cf4d", size = 22048, upload-time = "2025-09-27T18:37:11.547Z" }, + { url = "https://files.pythonhosted.org/packages/9a/a7/591f592afdc734f47db08a75793a55d7fbcc6902a723ae4cfbab61010cc5/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ec15a59cf5af7be74194f7ab02d0f59a62bdcf1a537677ce67a2537c9b87fcda", size = 23821, upload-time = "2025-09-27T18:37:12.48Z" }, + { url = "https://files.pythonhosted.org/packages/7d/33/45b24e4f44195b26521bc6f1a82197118f74df348556594bd2262bda1038/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:0eb9ff8191e8498cca014656ae6b8d61f39da5f95b488805da4bb029cccbfbaf", size = 21606, upload-time = "2025-09-27T18:37:13.485Z" }, + { url = "https://files.pythonhosted.org/packages/ff/0e/53dfaca23a69fbfbbf17a4b64072090e70717344c52eaaaa9c5ddff1e5f0/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:2713baf880df847f2bece4230d4d094280f4e67b1e813eec43b4c0e144a34ffe", size = 23043, upload-time = "2025-09-27T18:37:14.408Z" }, + { url = "https://files.pythonhosted.org/packages/46/11/f333a06fc16236d5238bfe74daccbca41459dcd8d1fa952e8fbd5dccfb70/markupsafe-3.0.3-cp314-cp314-win32.whl", hash = "sha256:729586769a26dbceff69f7a7dbbf59ab6572b99d94576a5592625d5b411576b9", size = 14747, upload-time = "2025-09-27T18:37:15.36Z" }, + { url = "https://files.pythonhosted.org/packages/28/52/182836104b33b444e400b14f797212f720cbc9ed6ba34c800639d154e821/markupsafe-3.0.3-cp314-cp314-win_amd64.whl", hash = "sha256:bdc919ead48f234740ad807933cdf545180bfbe9342c2bb451556db2ed958581", size = 15341, upload-time = "2025-09-27T18:37:16.496Z" }, + { url = "https://files.pythonhosted.org/packages/6f/18/acf23e91bd94fd7b3031558b1f013adfa21a8e407a3fdb32745538730382/markupsafe-3.0.3-cp314-cp314-win_arm64.whl", hash = "sha256:5a7d5dc5140555cf21a6fefbdbf8723f06fcd2f63ef108f2854de715e4422cb4", size = 14073, upload-time = "2025-09-27T18:37:17.476Z" }, + { url = "https://files.pythonhosted.org/packages/3c/f0/57689aa4076e1b43b15fdfa646b04653969d50cf30c32a102762be2485da/markupsafe-3.0.3-cp314-cp314t-macosx_10_13_x86_64.whl", hash = "sha256:1353ef0c1b138e1907ae78e2f6c63ff67501122006b0f9abad68fda5f4ffc6ab", size = 11661, upload-time = "2025-09-27T18:37:18.453Z" }, + { url = "https://files.pythonhosted.org/packages/89/c3/2e67a7ca217c6912985ec766c6393b636fb0c2344443ff9d91404dc4c79f/markupsafe-3.0.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1085e7fbddd3be5f89cc898938f42c0b3c711fdcb37d75221de2666af647c175", size = 12069, upload-time = "2025-09-27T18:37:19.332Z" }, + { url = "https://files.pythonhosted.org/packages/f0/00/be561dce4e6ca66b15276e184ce4b8aec61fe83662cce2f7d72bd3249d28/markupsafe-3.0.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1b52b4fb9df4eb9ae465f8d0c228a00624de2334f216f178a995ccdcf82c4634", size = 25670, upload-time = "2025-09-27T18:37:20.245Z" }, + { url = "https://files.pythonhosted.org/packages/50/09/c419f6f5a92e5fadde27efd190eca90f05e1261b10dbd8cbcb39cd8ea1dc/markupsafe-3.0.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50", size = 23598, upload-time = "2025-09-27T18:37:21.177Z" }, + { url = "https://files.pythonhosted.org/packages/22/44/a0681611106e0b2921b3033fc19bc53323e0b50bc70cffdd19f7d679bb66/markupsafe-3.0.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f190daf01f13c72eac4efd5c430a8de82489d9cff23c364c3ea822545032993e", size = 23261, upload-time = "2025-09-27T18:37:22.167Z" }, + { url = "https://files.pythonhosted.org/packages/5f/57/1b0b3f100259dc9fffe780cfb60d4be71375510e435efec3d116b6436d43/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:e56b7d45a839a697b5eb268c82a71bd8c7f6c94d6fd50c3d577fa39a9f1409f5", size = 24835, upload-time = "2025-09-27T18:37:23.296Z" }, + { url = "https://files.pythonhosted.org/packages/26/6a/4bf6d0c97c4920f1597cc14dd720705eca0bf7c787aebc6bb4d1bead5388/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:f3e98bb3798ead92273dc0e5fd0f31ade220f59a266ffd8a4f6065e0a3ce0523", size = 22733, upload-time = "2025-09-27T18:37:24.237Z" }, + { url = "https://files.pythonhosted.org/packages/14/c7/ca723101509b518797fedc2fdf79ba57f886b4aca8a7d31857ba3ee8281f/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:5678211cb9333a6468fb8d8be0305520aa073f50d17f089b5b4b477ea6e67fdc", size = 23672, upload-time = "2025-09-27T18:37:25.271Z" }, + { url = "https://files.pythonhosted.org/packages/fb/df/5bd7a48c256faecd1d36edc13133e51397e41b73bb77e1a69deab746ebac/markupsafe-3.0.3-cp314-cp314t-win32.whl", hash = "sha256:915c04ba3851909ce68ccc2b8e2cd691618c4dc4c4232fb7982bca3f41fd8c3d", size = 14819, upload-time = "2025-09-27T18:37:26.285Z" }, + { url = "https://files.pythonhosted.org/packages/1a/8a/0402ba61a2f16038b48b39bccca271134be00c5c9f0f623208399333c448/markupsafe-3.0.3-cp314-cp314t-win_amd64.whl", hash = "sha256:4faffd047e07c38848ce017e8725090413cd80cbc23d86e55c587bf979e579c9", size = 15426, upload-time = "2025-09-27T18:37:27.316Z" }, + { url = "https://files.pythonhosted.org/packages/70/bc/6f1c2f612465f5fa89b95bead1f44dcb607670fd42891d8fdcd5d039f4f4/markupsafe-3.0.3-cp314-cp314t-win_arm64.whl", hash = "sha256:32001d6a8fc98c8cb5c947787c5d08b0a50663d139f1305bac5885d98d9b40fa", size = 14146, upload-time = "2025-09-27T18:37:28.327Z" }, +] + +[[package]] +name = "numpy" +version = "2.5.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/13/01/11703282db468b85f6f7b8c7f22d058de5970d5c7e60a3a8aaa313c3de36/numpy-2.5.3.tar.gz", hash = "sha256:df2d5874ff183595a4ba404edd04f6bd9b5505c1d7708573f6a6c17489a67563", size = 20791231, upload-time = "2026-09-06T16:27:47.073Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d6/50/8fdbb16af64895706a45f06a4068e29db732ec180f3c1375f14123359138/numpy-2.5.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:cb189f09db39283b26bfd061ec16189e14f71c6755207f72a0f7540867afe5b9", size = 16994982, upload-time = "2026-09-06T16:24:29.244Z" }, + { url = "https://files.pythonhosted.org/packages/60/39/789131c1188c078dcb3a1692e72e1e050c68b88ffe72c9ccaac9bcd7a9cd/numpy-2.5.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f59a878c33d6b88122d80d239bb3b845d58708750b0cb06a09aebb9b18ec696c", size = 12009327, upload-time = "2026-09-06T16:24:32.491Z" }, + { url = "https://files.pythonhosted.org/packages/9c/59/a312e95696e5f601914dd8b6dd844692ba61670807417e24b68e337b5c70/numpy-2.5.3-cp312-cp312-macosx_14_0_arm64.whl", hash = "sha256:a72f874bc9e10e4b8f80426fb49716d5141f64442a0c8418065093ec8017fbb0", size = 5445405, upload-time = "2026-09-06T16:24:35.071Z" }, + { url = "https://files.pythonhosted.org/packages/30/d0/5623a1707ed4fe16e3909fe3cf5ee3da004ae677ad23d83bbf3adf1a6faf/numpy-2.5.3-cp312-cp312-macosx_14_0_x86_64.whl", hash = "sha256:fc36dc566135b5eceec4cf89758fcb719266a019ef07dae1754ae7c9f617ef3e", size = 6783213, upload-time = "2026-09-06T16:24:37.253Z" }, + { url = "https://files.pythonhosted.org/packages/f1/32/84146fc020ad3c25f805f70ab60da46fe3c540a21369754a7e4369754b6f/numpy-2.5.3-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:76c2c1e6bfa5c84adc6434dfbf013aa92096a7985221762c8f11fedfd20fff58", size = 15687872, upload-time = "2026-09-06T16:24:39.751Z" }, + { url = "https://files.pythonhosted.org/packages/65/af/aa78d1a88805456e212b65461354cd943197fb9acecc4c90fd12295123a3/numpy-2.5.3-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b7e18c623bb5c95acb3b3328861272816ba199fb531921c5d6d0b675f1fde9e3", size = 16717410, upload-time = "2026-09-06T16:24:42.745Z" }, + { url = "https://files.pythonhosted.org/packages/3b/24/faa79d865e69a97ba17473b23a1b74094b2259c03e820c70297293b9ea49/numpy-2.5.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:4f8929ee6c96bfbd7b4ed2032e0c03af86fe1826740ab61ddabf9072d06e57ff", size = 17040975, upload-time = "2026-09-06T16:24:45.961Z" }, + { url = "https://files.pythonhosted.org/packages/62/4a/8877e629445a7176297dffcaf9c485faa96a95d81728a62521ad55bd4c0f/numpy-2.5.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:b5d93cf48f687479941d12b69c873ad2cc76bbd487f0091c2200636497f34034", size = 18476479, upload-time = "2026-09-06T16:24:49.35Z" }, + { url = "https://files.pythonhosted.org/packages/c8/db/35e1c2d38b04cbd5b731f9d71495e055e813197669d22b612f11748d2ff9/numpy-2.5.3-cp312-cp312-win32.whl", hash = "sha256:bf63afbe037eb5d2fe87fbcc7778e61da53ebaf21d938a4515aa73b62532a5d4", size = 6133378, upload-time = "2026-09-06T16:24:51.915Z" }, + { url = "https://files.pythonhosted.org/packages/3c/a1/accf6d4f0c80c5d9ba9735d6b1550e444180599f34dec69ca01360f717ad/numpy-2.5.3-cp312-cp312-win_amd64.whl", hash = "sha256:0a59a421a32580a009e8a1751345bf829631b990dc1794b80514ab722b435def", size = 12567828, upload-time = "2026-09-06T16:24:54.255Z" }, + { url = "https://files.pythonhosted.org/packages/22/43/1764aff32e4652526ae2f71fa8b3efd8d25c8a3d6926914454e47138ed1e/numpy-2.5.3-cp312-cp312-win_arm64.whl", hash = "sha256:ccb32e0525d29e8b0572eb84c9a57af0e7a4e615726927506f55063c62414034", size = 10485432, upload-time = "2026-09-06T16:24:57.278Z" }, + { url = "https://files.pythonhosted.org/packages/79/e5/8fb89cd46d14e35699d13bf943a5f5f441ecee8667120a1f6105ab89e349/numpy-2.5.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:66a78fe4556c60aceda5916f9eacd638b18e9e681016ec302dcb4682d6d4d034", size = 16991061, upload-time = "2026-09-06T16:25:00.411Z" }, + { url = "https://files.pythonhosted.org/packages/2f/06/9dc9e48b5e5e941c8b10350c5ff2d721da42a20517d911d15544246775ff/numpy-2.5.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:92f30e89b8ee0ecf363033576c422b2f58fed6a80bed0aa48dff6d14c654663e", size = 12003676, upload-time = "2026-09-06T16:25:03.475Z" }, + { url = "https://files.pythonhosted.org/packages/ab/2a/98282aa5b8f58b1157d440bb6282eed47e3632a5de53a714fbab17e659fe/numpy-2.5.3-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:f9a2353b37a1a9e78fd82b27ad7e2a32a2d036604d18f02b05e3136c62ca3b09", size = 5439695, upload-time = "2026-09-06T16:25:05.978Z" }, + { url = "https://files.pythonhosted.org/packages/a1/f9/b6533d777be9d6ffd29dc1be0867e563e6e8cc9a220ff1b716adc317f060/numpy-2.5.3-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:ccbc4665079665c3cf3bab4db9f6b095370cd6437d66be549b6c2a1fd19e1958", size = 6779395, upload-time = "2026-09-06T16:25:08.599Z" }, + { url = "https://files.pythonhosted.org/packages/73/85/735720d04ec197c5dcfacdfc9922667c7f1f5f496a279b7ba4d7c74c4cc7/numpy-2.5.3-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c76d5dde9f445058f83d0c02af00557a4db91de9a9a57c0df87d1535001d654b", size = 15681750, upload-time = "2026-09-06T16:25:11.173Z" }, + { url = "https://files.pythonhosted.org/packages/3a/1b/3b16a9bc514a440a7a0883684111dcb1ef1aee960af2ca95da8fc775f124/numpy-2.5.3-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:a5fa86b80fd24bcd1aff83ad23be44ea323de3f787be8f8b15d4a65621e25321", size = 16708577, upload-time = "2026-09-06T16:25:14.171Z" }, + { url = "https://files.pythonhosted.org/packages/69/c4/386f397831b07328b639c96c5b62719346cf4baf07c68d927239752b1534/numpy-2.5.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:bd4cb9ad3c7889b9b3fe0a9a9fb5d2ed26f9879bff2608d9f01aed147a20d231", size = 17042047, upload-time = "2026-09-06T16:25:17.582Z" }, + { url = "https://files.pythonhosted.org/packages/5f/3e/a700ecbf36e85ae8328fd3b0e12eeddc22ed6358a64cb2bd913e0d195d65/numpy-2.5.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:1302b90c0e52281681b2975adfe8a860cb7b12216a27b4b0b4207c44bf7bccf0", size = 18465724, upload-time = "2026-09-06T16:25:20.949Z" }, + { url = "https://files.pythonhosted.org/packages/41/ee/38e785e88a4045f6ad1d1f2808dcdfafdca48c760260c0587bf171e29fc9/numpy-2.5.3-cp313-cp313-win32.whl", hash = "sha256:1c80eabb4035ecf4ca9cd49cde8a9fdd69a729e63e6474887d1523ade7aa277f", size = 6129003, upload-time = "2026-09-06T16:25:23.664Z" }, + { url = "https://files.pythonhosted.org/packages/f3/ec/100f2b1794ede74a9b3d7ec6b9736927f56713414c1dfe19ab6c383494bf/numpy-2.5.3-cp313-cp313-win_amd64.whl", hash = "sha256:71cad2b2a7451ab79d8f5e71b453485b6775963d5cf794179144a7463fe6e8ec", size = 12560965, upload-time = "2026-09-06T16:25:26.602Z" }, + { url = "https://files.pythonhosted.org/packages/80/b1/7dc825ca94c12acebbce4c37caa5e198695eb31424bc579679f32b1bb49d/numpy-2.5.3-cp313-cp313-win_arm64.whl", hash = "sha256:8e4dd766076855b5ff7ea52fa5f07ce26286726e0f8bff446b7739d02e6ea204", size = 10482343, upload-time = "2026-09-06T16:25:29.772Z" }, + { url = "https://files.pythonhosted.org/packages/70/78/cf416f15dc29375a229d9dfebf8db6e313f291580b39fa1a568b6052bb07/numpy-2.5.3-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:350ba9783ce969cf9f7ce6e6a9a58e1a6e2a19ca025b7ee448c4db727706212a", size = 16998686, upload-time = "2026-09-06T16:25:33.171Z" }, + { url = "https://files.pythonhosted.org/packages/9e/59/abcc2d8def4fd60eec7d87f92d27c13448ffd9ab14339bcc63a0d7a2fdea/numpy-2.5.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:012e66aca395d795496446e52aeeb5866312a5d4d3f27da270e5a0b43f70dc5c", size = 12013862, upload-time = "2026-09-06T16:25:36.748Z" }, + { url = "https://files.pythonhosted.org/packages/94/75/4640d2d6e4b64a049e48425a82728a41ef4adb61332d2cba68055774878b/numpy-2.5.3-cp314-cp314-macosx_14_0_arm64.whl", hash = "sha256:adc1ada2662f8a5f960b8a10d9986897e7499ef07e06d4cfe7197f8cce923c07", size = 5449793, upload-time = "2026-09-06T16:25:39.476Z" }, + { url = "https://files.pythonhosted.org/packages/96/cd/625b57ae33d4ca560f32cc0b47b4a5922146d9beb998ddf773900d440a73/numpy-2.5.3-cp314-cp314-macosx_14_0_x86_64.whl", hash = "sha256:54a115e5a73b8fc44f0cebef486365a1894b5c9760685d4558b72b7c3eb846e0", size = 6785176, upload-time = "2026-09-06T16:25:42.069Z" }, + { url = "https://files.pythonhosted.org/packages/9c/72/12918652e7912ef9751e8694c88820fcd1908e0618cb23f5f3caa6004b7b/numpy-2.5.3-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:be5a8381859b6da607c84f4f7d6847725f1cf1853ef8a2c9e115b7d58bef47dc", size = 15703377, upload-time = "2026-09-06T16:25:45.135Z" }, + { url = "https://files.pythonhosted.org/packages/45/8f/9beacf79ca7c650688ad0baa80931adb988fe6e6e5d5903c23cc3dbd70eb/numpy-2.5.3-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b0521d0f4aebb6e06189451025fa17a913287b13c03d5fe05c017333b654ea5b", size = 16711928, upload-time = "2026-09-06T16:25:48.461Z" }, + { url = "https://files.pythonhosted.org/packages/09/8d/41d0a56e1ac4c87495c897a211b1368691b7237aadabec8b3b8f3a74d48f/numpy-2.5.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:9deb49575e5b0b94ed72c8a64ec4d033381adc27e9060ae842971f697ba96104", size = 17059507, upload-time = "2026-09-06T16:25:51.873Z" }, + { url = "https://files.pythonhosted.org/packages/08/1e/0dfbc5cc251d54e2af790f254d24ec38637fa97ec7d5d11de7ffed787098/numpy-2.5.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:b00eefbcf0f292945c4b4dec2ae845389ef5bcdcd596e6e4328051db5b5ba694", size = 18471002, upload-time = "2026-09-06T16:25:55.233Z" }, + { url = "https://files.pythonhosted.org/packages/b5/2c/dfa40f6991f8185c8c30ffd023dfcbb11888e823cfab9557b920f3bb7bed/numpy-2.5.3-cp314-cp314-win32.whl", hash = "sha256:c2381f82999704f818e2c987a865050e285ec3621262c66d40f5a96c8f899f8e", size = 6180485, upload-time = "2026-09-06T16:25:58.157Z" }, + { url = "https://files.pythonhosted.org/packages/a4/73/d2c08231e4fde7e415501fd02c715d96e98599b2d8384445933944152984/numpy-2.5.3-cp314-cp314-win_amd64.whl", hash = "sha256:2c25dfa72943e4336ddb6b0ee4277b47a0c85bede0807530ec68103bf58e2c10", size = 12698179, upload-time = "2026-09-06T16:26:00.789Z" }, + { url = "https://files.pythonhosted.org/packages/5c/e9/dcdcc9b95cf5f49815055573aee1b11cfbf5299f38a180e437ded050810f/numpy-2.5.3-cp314-cp314-win_arm64.whl", hash = "sha256:15aa985ac73a8db02db7663381aa109510449d3819d37206caed27b33a65a8a6", size = 10769383, upload-time = "2026-09-06T16:26:04.011Z" }, + { url = "https://files.pythonhosted.org/packages/49/c4/af8bc08a7ef4e1529a7c0cf24969accce316b783999802089a581ec99272/numpy-2.5.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:ac7bb1c52d445bd4f8f7f97fefe6abc3a084dc4d63df50d79b17fa2b78e89297", size = 12132668, upload-time = "2026-09-06T16:26:07.138Z" }, + { url = "https://files.pythonhosted.org/packages/c5/ae/0f15eb56d4ec5e13c1f7ff04ff407f997d1acbadb45d3e1f2e2645a8f43c/numpy-2.5.3-cp314-cp314t-macosx_14_0_arm64.whl", hash = "sha256:e6ab667ba76450084eb64013762c438ea76d9d29cc676dcd6c2e9892ba37f841", size = 5568580, upload-time = "2026-09-06T16:26:09.828Z" }, + { url = "https://files.pythonhosted.org/packages/23/fb/c72a8f25d4b6e96c354e7ab45ace3b27dc11e5d6a13b6c7d0cd6b08bf112/numpy-2.5.3-cp314-cp314t-macosx_14_0_x86_64.whl", hash = "sha256:f7fabeb6cea87d65f3b926de33d03fb016cfdc29314c90974383b5582ae72891", size = 6882634, upload-time = "2026-09-06T16:26:12.524Z" }, + { url = "https://files.pythonhosted.org/packages/07/a9/968c90ed2ab15060c338e8137f1215b5a60756ae07328e0a60d1c6734df4/numpy-2.5.3-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1fb6f8fb9ff0b3a69f52c66ce397b0246583e9f28616231b0e32ca49259a5fa6", size = 15748923, upload-time = "2026-09-06T16:26:15.092Z" }, + { url = "https://files.pythonhosted.org/packages/59/08/9df04103947b95e3b6b1f2ed1a70521f325647a31b82da6a2aae3a485508/numpy-2.5.3-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:93e1f5447e2b1e479d7bd74701e84746b86450cff1fc368b132d195e2b8f8211", size = 16746748, upload-time = "2026-09-06T16:26:18.43Z" }, + { url = "https://files.pythonhosted.org/packages/41/a0/14c8d5fe5b53a334aabb653deb391c0fef49558f491880ea300ed6785224/numpy-2.5.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:c00abe94c1a69d75d827dcf1c025b25c8a45d230b3bcd77a9020883a1b047653", size = 17111561, upload-time = "2026-09-06T16:26:22.113Z" }, + { url = "https://files.pythonhosted.org/packages/c4/a6/d7e96e42f01522e154c32489640f16dfc4f6181d165d05fc3bec8c2c4999/numpy-2.5.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:536f963710a4e63934d80ac0dc4f478804a83e9a84b6828018f25d09953ada33", size = 18513945, upload-time = "2026-09-06T16:26:25.401Z" }, + { url = "https://files.pythonhosted.org/packages/25/39/3453afb7119d0449ef11c886874120ff180e2c337760e0e2d88f70f1a945/numpy-2.5.3-cp314-cp314t-win32.whl", hash = "sha256:4c8a6d2ebce6305fd82fbefca827775437147052a976ee7c94b36a0c1b52ac6c", size = 6335421, upload-time = "2026-09-06T16:26:28.175Z" }, + { url = "https://files.pythonhosted.org/packages/99/01/22815d2b19a1a746b1d45205cffebb3fe511a18acb75fba6c88491fc9894/numpy-2.5.3-cp314-cp314t-win_amd64.whl", hash = "sha256:9a37475425b431b4d060f23b4f52cd2f3aef6bc7c654bd760adf0040eec9d435", size = 12896420, upload-time = "2026-09-06T16:26:31.265Z" }, + { url = "https://files.pythonhosted.org/packages/fa/ee/a7cbba67eeaff038dc29ca8b98a88396c8b0cc9c89d4924f4a27a5c9150b/numpy-2.5.3-cp314-cp314t-win_arm64.whl", hash = "sha256:2d8240cb4c16fd831074aa2b2cf9fc54664d826341d61c372245b96a74a49a9a", size = 10857177, upload-time = "2026-09-06T16:26:34.167Z" }, + { url = "https://files.pythonhosted.org/packages/45/56/78194492883ff5eec90423fe56a3a44b154da047d88a6307f629713c584f/numpy-2.5.3-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:a6391fafaba97500887132cd582abc6e19452b1ac775a47caa7b24490e152058", size = 16996531, upload-time = "2026-09-06T16:26:37.287Z" }, + { url = "https://files.pythonhosted.org/packages/11/39/dd55c0af90bbab564b09ae3b0aa60ec5c02b900fa4f1ba23440525c8b32d/numpy-2.5.3-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:09d5a423c71ad5feb5625844ad58050e35df43871004b52ac9c0ad44a56775be", size = 12012569, upload-time = "2026-09-06T16:26:40.707Z" }, + { url = "https://files.pythonhosted.org/packages/b6/51/04f67d32e4862b281b1cb84ceeaed3421189a84fb6fb51a391cd6d5009f7/numpy-2.5.3-cp315-cp315-macosx_14_0_arm64.whl", hash = "sha256:f9579f383d1bf9df80081e72760e84960a7fd4f88cf0c9e535a8597c9bb646f5", size = 5448498, upload-time = "2026-09-06T16:26:43.435Z" }, + { url = "https://files.pythonhosted.org/packages/a3/c9/25b4dc0dd1344ec26c7319e84fd4e9809d2b5628f4e12decd618036e5178/numpy-2.5.3-cp315-cp315-macosx_14_0_x86_64.whl", hash = "sha256:86bff898a431c0fb71f7610b75726e75a54d47b37edc9d537f48de63bb3c0b90", size = 6783026, upload-time = "2026-09-06T16:26:46.374Z" }, + { url = "https://files.pythonhosted.org/packages/fc/c7/29285be1e5232a6e7ee3268a33c85843f5a8ee93350c6465cddd66ebbf76/numpy-2.5.3-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1f3ed25271581281f2fccb1adcedfcde4c07362eec69189b50baf6f90e3ae159", size = 15697322, upload-time = "2026-09-06T16:26:49.415Z" }, + { url = "https://files.pythonhosted.org/packages/55/49/bbad5335fb4996a16881f853ff3e0ba582f01720e55c89b1c06b8fc42a90/numpy-2.5.3-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ffdc76bfcae6b255dff75202c5e7feaf95b40246bc0a17944facc1fecf9f79ab", size = 16708995, upload-time = "2026-09-06T16:26:53.127Z" }, + { url = "https://files.pythonhosted.org/packages/ef/e9/1df35483760b04a65ea44669f89dc64f30e5aca098b48ceb8b1310b0e0fe/numpy-2.5.3-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:116f96cadd935c6122e9228d676fe7ede19e741f5c8bb1c3cddbe0c51ccebea2", size = 17052508, upload-time = "2026-09-06T16:26:56.464Z" }, + { url = "https://files.pythonhosted.org/packages/b8/99/66e54da8265cc8be8a7382bf96edce17aaa2837d6f484432025932a3caa5/numpy-2.5.3-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:09ffa5d903faeaa5c4dd05009cf81c8bab9f2cb37c548b8d39b65b4cfa7c97f7", size = 18468224, upload-time = "2026-09-06T16:26:59.966Z" }, + { url = "https://files.pythonhosted.org/packages/01/bc/b5e90a91c115168d793dfd2ad9c69c438c2fe7a13a437e770bc5b078e732/numpy-2.5.3-cp315-cp315-win32.whl", hash = "sha256:e01c918ac3d48e18a927cf7b14a26a3e29ff2bdf2eacb976da0aecd6a43ed034", size = 6179919, upload-time = "2026-09-06T16:27:03.166Z" }, + { url = "https://files.pythonhosted.org/packages/37/ea/780748fd3985109075514ef8fc64cd25f943e40dde13a6d59141eb268fc8/numpy-2.5.3-cp315-cp315-win_amd64.whl", hash = "sha256:e931e4f499e0dc7ef29d269a8e5b35dd722e5d14be07df6240166ea7c6532fae", size = 12697656, upload-time = "2026-09-06T16:27:06.153Z" }, + { url = "https://files.pythonhosted.org/packages/b3/16/407be69a2a87c8cab64d95975a8977a426a29e138f07e276ec258f0fe4e5/numpy-2.5.3-cp315-cp315-win_arm64.whl", hash = "sha256:26e15e4aecd8617dfbaecb37d223e365d7b39411fba20454be2670a96aa74cb5", size = 10767601, upload-time = "2026-09-06T16:27:09.297Z" }, + { url = "https://files.pythonhosted.org/packages/44/bf/a97ffb01e41d50a32a9177aef942a4d0e389a3daf451d04e5f38ef6afb87/numpy-2.5.3-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:6cef4bb1706dfec49243c05d921eefb4e190d41e2528b30d8035ea1f36b4c24a", size = 17090092, upload-time = "2026-09-06T16:27:12.907Z" }, + { url = "https://files.pythonhosted.org/packages/d1/24/136c02f2c2af9a067a84d0c3aa10c99012c0476fa5066732fa4a4202557d/numpy-2.5.3-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:d1c89973648c85069c5046ad460f7b8a00218b29a2e42359ac8cc63e9ab94832", size = 12129429, upload-time = "2026-09-06T16:27:16.089Z" }, + { url = "https://files.pythonhosted.org/packages/fe/6c/b47582d6597789bf946d5efbeb6b9e56fd8bcbd5efc6fbf51dbe1ea31eb3/numpy-2.5.3-cp315-cp315t-macosx_14_0_arm64.whl", hash = "sha256:214045a5bf00113a146ab9ee9730c44501af6723cdf1f6830932f7b5ef2e7af0", size = 5565452, upload-time = "2026-09-06T16:27:19.868Z" }, + { url = "https://files.pythonhosted.org/packages/be/b4/ef3cc6da73774202d4deae16bb321fd8298a4e0561e3539f8c4be237d916/numpy-2.5.3-cp315-cp315t-macosx_14_0_x86_64.whl", hash = "sha256:8617bbfae4486cf99c9f899966699428d19da931d06ca94ad3da986c76e15997", size = 6876736, upload-time = "2026-09-06T16:27:22.232Z" }, + { url = "https://files.pythonhosted.org/packages/9e/24/e3813329498596cb842703dcacac1741612ed9fb9c4e6a3e0c7e2ebbc597/numpy-2.5.3-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:595d020938c84e320bcf40ad71089e108eac0d377cd018e14a8c094f39e98d85", size = 15745777, upload-time = "2026-09-06T16:27:25.181Z" }, + { url = "https://files.pythonhosted.org/packages/4a/9e/4e7a07fd0776dc2210cdacf2010be8665194d094defc10c419d7dea794cc/numpy-2.5.3-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6f24021b9f22bc6301c37b196974a92c1c18dccedb6fef3dd252e95f2d6adbe4", size = 16746949, upload-time = "2026-09-06T16:27:28.576Z" }, + { url = "https://files.pythonhosted.org/packages/91/db/01674c0e20335057813a00c2ebd546ed25bff9ed7914f9bced00f8c55d94/numpy-2.5.3-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:71b39d9f935b6ec0f8753e3e2afb51e3efba6f2e05b68b32a40754d24bcd4a3c", size = 17108994, upload-time = "2026-09-06T16:27:31.946Z" }, + { url = "https://files.pythonhosted.org/packages/45/7a/584c5e71f8d378e57cac0b033891ed65c683ef90573ba4854e8c28203db0/numpy-2.5.3-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:6b05c171afb3aa07adbd20abc00aea86fe375beb0fdb9ef780ec5b7f63bab1c0", size = 18512266, upload-time = "2026-09-06T16:27:35.196Z" }, + { url = "https://files.pythonhosted.org/packages/a1/d2/4e1014173aa3c55e6a756e0e567290743a6ab33a288460374d7ef6bcd239/numpy-2.5.3-cp315-cp315t-win32.whl", hash = "sha256:f54660b0eb6b0b9f36e7fe1cdfdff472028dd0d14acd9b9b65098efbad059469", size = 6330292, upload-time = "2026-09-06T16:27:38.149Z" }, + { url = "https://files.pythonhosted.org/packages/6c/b0/ff5658a58199b7bcaad87bf260eef6713d9d42cca4e028f935b4fc5fbac6/numpy-2.5.3-cp315-cp315t-win_amd64.whl", hash = "sha256:1aad64d99730d013cfc6debafed22783b4fc5a7f4b8bc744d2d8cf7dcc880551", size = 12884918, upload-time = "2026-09-06T16:27:40.965Z" }, + { url = "https://files.pythonhosted.org/packages/fb/0b/b12a2df5d1b774bd9007a6fdff9381145b6223d37f11afc9c37ab0efd9a1/numpy-2.5.3-cp315-cp315t-win_arm64.whl", hash = "sha256:befa1ae5bd6030b3f512b43ff3fa5290bbed6b84411a44244b14adf835f5b89d", size = 10850807, upload-time = "2026-09-06T16:27:43.868Z" }, +] + +[[package]] +name = "numpydoc" +version = "1.10.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sphinx" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/e9/3c/dfccc9e7dee357fb2aa13c3890d952a370dd0ed071e0f7ed62ed0df567c1/numpydoc-1.10.0.tar.gz", hash = "sha256:3f7970f6eee30912260a6b31ac72bba2432830cd6722569ec17ee8d3ef5ffa01", size = 94027, upload-time = "2025-12-02T16:39:12.937Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/62/5e/3a6a3e90f35cea3853c45e5d5fb9b7192ce4384616f932cf7591298ab6e1/numpydoc-1.10.0-py3-none-any.whl", hash = "sha256:3149da9874af890bcc2a82ef7aae5484e5aa81cb2778f08e3c307ba6d963721b", size = 69255, upload-time = "2025-12-02T16:39:11.561Z" }, +] + +[[package]] +name = "packaging" +version = "26.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" }, +] + +[[package]] +name = "pandas" +version = "3.0.5" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy" }, + { name = "python-dateutil" }, + { name = "tzdata", marker = "sys_platform == 'emscripten' or sys_platform == 'win32'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/be/4f/5f3422a2afec5ffc46308b79e53291365a93748b498ac2e58bead0197916/pandas-3.0.5.tar.gz", hash = "sha256:dca3734d6ab7c906e6730f0788b0a1dbb9f2467731f9711f77995c8e9d62d712", size = 4658219, upload-time = "2026-07-22T22:19:28.819Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1c/54/1dc810ea558d1320b597aa140a514f2fdf1d2ea09c38cf556f13ea712ec9/pandas-3.0.5-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:fa290c16964d4963fbfbc358928239cf3bd755b20e988ce944877def2f44471d", size = 10411717, upload-time = "2026-07-22T22:18:08.307Z" }, + { url = "https://files.pythonhosted.org/packages/68/56/fbe81c09195924d8b7b8d4461a20458fe80a6a5ed6b24f0314da684277e1/pandas-3.0.5-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:c2e26bb46934b8a2ca0c3de1d3d606fc5f6746584791b2db264d58cf370e08dc", size = 9957095, upload-time = "2026-07-22T22:18:10.6Z" }, + { url = "https://files.pythonhosted.org/packages/e0/51/fac252f4a913ed5eabf3c11b880a9e8d5a6c10f0b2129d0462212d238b4d/pandas-3.0.5-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:73fa87b08a7ef706f8aafda39ddaccf2a99047bea62d8c88a0361bcafb2237bc", size = 10485458, upload-time = "2026-07-22T22:18:12.834Z" }, + { url = "https://files.pythonhosted.org/packages/12/98/e976540c1addf70442be7842a18cf70884a964abbf69442504f4d2939989/pandas-3.0.5-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d373ce03ffd84010ed9839fa73672a9c8256990532e158440c0085db7d914b34", size = 10998091, upload-time = "2026-07-22T22:18:15.209Z" }, + { url = "https://files.pythonhosted.org/packages/a4/8c/1f29b5be8d3fc47dd7567eb167fabba2085879b31e0287ce7cba6d3d2ff4/pandas-3.0.5-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:2a29c53d85ea98c5e792c59ef82ee9fbe6ca902c0d0adb6b23f45ef894cd7bf6", size = 11499501, upload-time = "2026-07-22T22:18:17.689Z" }, + { url = "https://files.pythonhosted.org/packages/9d/e2/bd9c98ad2df7b38bde002adde4cdf353519da51881634323b126c55997f9/pandas-3.0.5-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:a5ad3b02ed6bc7d7ae9b70804b2c6aa31827489d150f8e623ce82491b82085d7", size = 12060559, upload-time = "2026-07-22T22:18:20.147Z" }, + { url = "https://files.pythonhosted.org/packages/f3/9a/ffbd852d58bd74a617fe2f8ee6a58a96982271ce41cf981eab22190b4a4b/pandas-3.0.5-cp312-cp312-pyemscripten_2024_0_wasm32.whl", hash = "sha256:b2acb4650527eec6822c3dadb2b771277b65e7dae7a267d4bccf65fd1bb3fbce", size = 7197652, upload-time = "2026-07-22T22:18:22.502Z" }, + { url = "https://files.pythonhosted.org/packages/70/b5/d2d3e9ae73362ba4229651b0ee1455cf78073a1ce585f6ff693782ce263e/pandas-3.0.5-cp312-cp312-win_amd64.whl", hash = "sha256:80a611068e8a3ac23f7398c6c14eb46dc974e5cc9997f653e2dcfd1da74edd41", size = 9831691, upload-time = "2026-07-22T22:18:24.534Z" }, + { url = "https://files.pythonhosted.org/packages/52/51/dea1e89d6a6796b9c43f85a09b484ee03edb8a4c4842e73e200a8c11301c/pandas-3.0.5-cp312-cp312-win_arm64.whl", hash = "sha256:25ff585b972a18ef1fe9ffa3ac6544d9950508aa76832e5147640b6022821e49", size = 9105796, upload-time = "2026-07-22T22:18:27.064Z" }, + { url = "https://files.pythonhosted.org/packages/bf/09/7b95c4a0025227d6f118c4039b423412ac6a982db02864166185d812fbc7/pandas-3.0.5-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:c1c05a767fe8e5b4fe9e1c29806829c582052eaedb9120a3da83ba3f69e24a5b", size = 10385742, upload-time = "2026-07-22T22:18:29.346Z" }, + { url = "https://files.pythonhosted.org/packages/8d/0c/dc78fd8c4da477b4b5e8ad37295af352190d21ef63a9ee1bc071753074cc/pandas-3.0.5-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:b86765f268b56f7e665b93bce9d5df69dee7f99e595cf8fb839483ab315942a3", size = 9932067, upload-time = "2026-07-22T22:18:31.833Z" }, + { url = "https://files.pythonhosted.org/packages/3e/71/3592c055cf44df9808550f9368ceda80ff2b224d355ef73fe251dcda1802/pandas-3.0.5-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c597ecf5616b5c420372c1d4d4c00dbbfba7398bea857dcc984347e1ea48417b", size = 10466756, upload-time = "2026-07-22T22:18:34.195Z" }, + { url = "https://files.pythonhosted.org/packages/e3/70/4363150359f95b4cb4bcbb34ca23572bb5495749a621a8f3d5a1ddfd293c/pandas-3.0.5-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4b11c36e218331d0387cbe3a0a5f75162357a1d92d57b2b08a336ff94b19b2be", size = 10938525, upload-time = "2026-07-22T22:18:36.81Z" }, + { url = "https://files.pythonhosted.org/packages/f7/d0/317e7a0c67c0e69fa905a0161409397a7dc2d46ff611f6ca4803352c042b/pandas-3.0.5-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:cf52e1f61d229496da17dc7ab54acdee627357e7008fd4fecba3d0ba2937fa58", size = 11489303, upload-time = "2026-07-22T22:18:39.287Z" }, + { url = "https://files.pythonhosted.org/packages/f1/8d/36dade89b49e4f9d5cbdbe863772581f98c0c6d78fc39ad4c557f6f2e17e/pandas-3.0.5-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:db172144bb56422bd157812f3b021eacc255451470b31e2c633c349490a1cfee", size = 11989004, upload-time = "2026-07-22T22:18:42.208Z" }, + { url = "https://files.pythonhosted.org/packages/9c/ba/18c4ec8a746e177da05a9e7a7963781d8ea195780724f854601b6ebd6b78/pandas-3.0.5-cp313-cp313-win_amd64.whl", hash = "sha256:0d298e951f23016ce4699951d044ae6418dbc91bf68cefca0f77666fcbb4e5c6", size = 9826896, upload-time = "2026-07-22T22:18:44.539Z" }, + { url = "https://files.pythonhosted.org/packages/de/ec/28a57266b753799a87b8bc79e7887ac6fd981b8c6d2978a0b7e7b6bd708c/pandas-3.0.5-cp313-cp313-win_arm64.whl", hash = "sha256:66266d3442a5e8b3c90274c2b8b230bee42dd1c286bc822cc2f9f2c7e12b883e", size = 9094790, upload-time = "2026-07-22T22:18:47.468Z" }, + { url = "https://files.pythonhosted.org/packages/51/2f/cf6aae281264f4463f0875bcbb15fd2bb6d291cc535187dad1732475e4a9/pandas-3.0.5-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:2f264fc46911cc8131a7322a16199bbf8e353d27c10bb211f5bd0c814324dc36", size = 10390034, upload-time = "2026-07-22T22:18:49.818Z" }, + { url = "https://files.pythonhosted.org/packages/06/ec/5189518c7a7659c4bdcc6b1eb32c46c6f3c86b0661ffd84143d1112c7732/pandas-3.0.5-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:53730687fcd161883b24e10411c06d6a4c0f2275d2faf3bb2bc25deb4ba8007c", size = 9980065, upload-time = "2026-07-22T22:18:52.249Z" }, + { url = "https://files.pythonhosted.org/packages/ea/f1/598503ce8d7e3c35601e0747ba288c7864baae66380725bc12f13f884dfe/pandas-3.0.5-cp314-cp314-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:960d3ebcf249f75206899fcd2c6de53f736b7265759ced0d3e559df0b8b709b0", size = 10545532, upload-time = "2026-07-22T22:18:54.813Z" }, + { url = "https://files.pythonhosted.org/packages/fa/de/ceae2adf7034e07e9910299fe412e1819c4f0dd520700a888bcb03625448/pandas-3.0.5-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:9e94c2c5ca43bd3ca32bf64d32308887b65e5f9bfd8023ea52755107a999f93b", size = 10963120, upload-time = "2026-07-22T22:18:57.42Z" }, + { url = "https://files.pythonhosted.org/packages/66/25/86e0f4451874eb79e688deeebe3c451fec4557f8952005818d800ee8ac7e/pandas-3.0.5-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:e819dd5f62966b481a8cb649d3299ebd886a1ea91ed5a99bf7ce77c98d18ab94", size = 11563178, upload-time = "2026-07-22T22:18:59.729Z" }, + { url = "https://files.pythonhosted.org/packages/f3/45/8643daa3b4147e433adfcccefdd0380d3aad79d86b15d8999730fe1944d5/pandas-3.0.5-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:3c5ed2e7c06e91d340dfd091d7934f9bc82e4a36b95f647f090b9d1c9ac649da", size = 12028708, upload-time = "2026-07-22T22:19:02.164Z" }, + { url = "https://files.pythonhosted.org/packages/96/58/ad979ae617615576e8aafd569c9d4b62f1191d896e38f51d66ba06f3b89a/pandas-3.0.5-cp314-cp314-win_amd64.whl", hash = "sha256:cd8f7c6dc98527058ee6264219343f5392240a6f1bfa654fc5d79023020d0c92", size = 9951806, upload-time = "2026-07-22T22:19:04.596Z" }, + { url = "https://files.pythonhosted.org/packages/69/32/7ac03886b304049a9d2625ee88f59af760d8a93bd30ed9239bce7b9869a8/pandas-3.0.5-cp314-cp314-win_arm64.whl", hash = "sha256:5183427f5a8156d480f30333777bc978be93650a49a7c01db26adffe95b31e85", size = 9238297, upload-time = "2026-07-22T22:19:06.836Z" }, + { url = "https://files.pythonhosted.org/packages/be/ed/1d1f2ee5547d5167face2376d11c8b2a4c7bfff5a416ee7a9046891fab1e/pandas-3.0.5-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:303da736987d481074ca720ada325f8bd80c64ebc2d45ed79b29df3aaa4a26ca", size = 10849690, upload-time = "2026-07-22T22:19:09.391Z" }, + { url = "https://files.pythonhosted.org/packages/57/55/17e17152e98fbb0c4b1e562bc65387a2f20a80db0f4a86bf8d3a0e4248d4/pandas-3.0.5-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:3b2801bbb049d0136f6c213eae02b5fca969384fc2064dd728d8620552aa49da", size = 10509945, upload-time = "2026-07-22T22:19:11.773Z" }, + { url = "https://files.pythonhosted.org/packages/88/90/817d44dbf83facf9556f33576d9af0a241981e7bb5c00606c0bcb5df8dda/pandas-3.0.5-cp314-cp314t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cce3a9d11d2b1f82c69a27ec1f4948a170e2c403c4bbfa8cca62e3fdebe2ef3a", size = 10392197, upload-time = "2026-07-22T22:19:14.024Z" }, + { url = "https://files.pythonhosted.org/packages/f1/da/889f00c0a6f5aa1545add70abbf01502dff87ab577adb855bd631c54d2f2/pandas-3.0.5-cp314-cp314t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ef01af4d8dc6cd2c8d6c7736f149574ef93fe043811eeb5e445f2647154b5040", size = 10862726, upload-time = "2026-07-22T22:19:16.351Z" }, + { url = "https://files.pythonhosted.org/packages/bc/98/f1e934fb3c98fce859c6147c6785816c7b5b9ab7821115c5d8c4de9842b9/pandas-3.0.5-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:e2759e890db96dfcffdbd9b86c3c2cb6afaf58def482820317e06163ec1066cd", size = 11414864, upload-time = "2026-07-22T22:19:18.981Z" }, + { url = "https://files.pythonhosted.org/packages/fe/be/d448af7d657d82e1888dd8551f79c6d6fb161080b5b9752d84d910ec2319/pandas-3.0.5-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:b58b1b39d46a5862e3fb18f50d1a201398619d16a0f9f73f57eea5583cf0e63c", size = 11925105, upload-time = "2026-07-22T22:19:21.515Z" }, + { url = "https://files.pythonhosted.org/packages/29/c1/ccb4238212c8c4f496c584f3044d94e0c030ed8e1d68999db46c91c2242f/pandas-3.0.5-cp314-cp314t-win_amd64.whl", hash = "sha256:1c10461f6eeb35d8f05b6184c65c8b9991663b66c46b1d559b682cb34ae7c6ea", size = 10387612, upload-time = "2026-07-22T22:19:24.257Z" }, + { url = "https://files.pythonhosted.org/packages/d2/cf/6a51b2c38980e04c279fd2fa908a1b0982064e860444acfca4ec2e2c8359/pandas-3.0.5-cp314-cp314t-win_arm64.whl", hash = "sha256:3c5015fd1730fbf883647e88068176c839c102cea883ba1769a6f4593bfc1f8c", size = 9509776, upload-time = "2026-07-22T22:19:26.694Z" }, +] + +[[package]] +name = "pandas-stubs" +version = "3.0.5.260730" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c2/d2/dea4a3a56b7b5f69c5fbca9f14625fcf28e1a39a657e9833d4a10bcac593/pandas_stubs-3.0.5.260730.tar.gz", hash = "sha256:f70a232c57d93a5a2c81f8a53953e10891a5374bc92652277deb325e2e4d0ff3", size = 114631, upload-time = "2026-07-30T14:31:42.271Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/60/c2/959caec5c46f484b5f8bb6def4b0cf7a45ba6acda26f12b75142d98cc5ae/pandas_stubs-3.0.5.260730-py3-none-any.whl", hash = "sha256:60e90e3e1eda6937e337e243cbe6217e151c11137cd7eddf832af537c7310bfd", size = 174807, upload-time = "2026-07-30T14:31:41.17Z" }, +] + +[[package]] +name = "pluggy" +version = "1.6.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3", size = 69412, upload-time = "2025-05-15T12:30:07.975Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, +] + +[[package]] +name = "pygments" +version = "2.21.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c", size = 5005329, upload-time = "2026-08-17T08:02:48.824Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9", size = 1250147, upload-time = "2026-08-17T08:02:44.912Z" }, +] + +[[package]] +name = "pytest" +version = "9.1.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "iniconfig" }, + { name = "packaging" }, + { name = "pluggy" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", hash = "sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313", size = 1636369, upload-time = "2026-06-19T10:58:32.857Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c", size = 386536, upload-time = "2026-06-19T10:58:31.347Z" }, +] + +[[package]] +name = "pytest-cov" +version = "7.1.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "coverage" }, + { name = "pluggy" }, + { name = "pytest" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b1/51/a849f96e117386044471c8ec2bd6cfebacda285da9525c9106aeb28da671/pytest_cov-7.1.0.tar.gz", hash = "sha256:30674f2b5f6351aa09702a9c8c364f6a01c27aae0c1366ae8016160d1efc56b2", size = 55592, upload-time = "2026-03-21T20:11:16.284Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", hash = "sha256:a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678", size = 22876, upload-time = "2026-03-21T20:11:14.438Z" }, +] + +[[package]] +name = "python-dateutil" +version = "2.9.0.post0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "six" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/66/c0/0c8b6ad9f17a802ee498c46e004a0eb49bc148f2fd230864601a86dcf6db/python-dateutil-2.9.0.post0.tar.gz", hash = "sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3", size = 342432, upload-time = "2024-03-01T18:36:20.211Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ec/57/56b9bcc3c9c6a792fcbaf139543cee77261f3651ca9da0c93f5c1221264b/python_dateutil-2.9.0.post0-py2.py3-none-any.whl", hash = "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427", size = 229892, upload-time = "2024-03-01T18:36:18.57Z" }, +] + +[[package]] +name = "requests" +version = "2.34.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "certifi" }, + { name = "charset-normalizer" }, + { name = "idna" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ac/c3/e2a2b89f2d3e2179abd6d00ebd70bff6273f37fb3e0cc209f48b39d00cbf/requests-2.34.2.tar.gz", hash = "sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed", size = 142856, upload-time = "2026-05-14T19:25:27.735Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", size = 73075, upload-time = "2026-05-14T19:25:26.443Z" }, +] + +[[package]] +name = "requirements-parser" +version = "0.13.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "packaging" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/89/1a/5f3c22d38bf1d87d1f4a961489d9eba35c4370a21395562d94410cdd0e73/requirements_parser-0.13.1.tar.gz", hash = "sha256:78811383b2089b6c5197a1431bc2c12ff950245edca39a23eea3460782038dd3", size = 22783, upload-time = "2026-06-18T07:52:25.291Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bb/f9/15b44d5e4401b0013bbcefe3c09d7bfddcce28cc3d41b1d3077bcedf5b1f/requirements_parser-0.13.1-py3-none-any.whl", hash = "sha256:6e385663eb32589d16e5b22bb6e5251a57908e73803ffff438b53cd6ea2056e0", size = 14926, upload-time = "2026-06-18T07:52:24.171Z" }, +] + +[[package]] +name = "roman-numerals" +version = "4.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ae/f9/41dc953bbeb056c17d5f7a519f50fdf010bd0553be2d630bc69d1e022703/roman_numerals-4.1.0.tar.gz", hash = "sha256:1af8b147eb1405d5839e78aeb93131690495fe9da5c91856cb33ad55a7f1e5b2", size = 9077, upload-time = "2025-12-17T18:25:34.381Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/04/54/6f679c435d28e0a568d8e8a7c0a93a09010818634c3c3907fc98d8983770/roman_numerals-4.1.0-py3-none-any.whl", hash = "sha256:647ba99caddc2cc1e55a51e4360689115551bf4476d90e8162cf8c345fe233c7", size = 7676, upload-time = "2025-12-17T18:25:33.098Z" }, +] + +[[package]] +name = "ruff" +version = "0.16.6" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a4/7c/6adb35d70e7c027e308274557901c7e00fb3407750faf3620c184ae058cb/ruff-0.16.6.tar.gz", hash = "sha256:dcf8a73d2ff77e99dde91244b4da16feba7f14e6beeb4015dee7c5a909e99050", size = 4921251, upload-time = "2026-09-03T16:57:29.037Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a4/28/9cc1b79639e284ec103f43c88c644db4eb58cbd0ea1ca11f1193435369ac/ruff-0.16.6-py3-none-linux_armv6l.whl", hash = "sha256:61c368c26bf8e973e5ab14a2772de587bc068ea3f9a277f673380749b4898fb8", size = 10015638, upload-time = "2026-09-03T16:56:40.986Z" }, + { url = "https://files.pythonhosted.org/packages/71/11/627d342ef727ea7794edf74fe23d60a074b02c3acc2e9436684e782286ca/ruff-0.16.6-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:ecf4f068e2e123e43a26e9db4e19524cc56563912404e83bbfca375757e45a32", size = 10220762, upload-time = "2026-09-03T16:56:44.681Z" }, + { url = "https://files.pythonhosted.org/packages/43/d9/b75668ce41e4c8d073d18d6d08672ba6906ce45d5c06ea4fdb2e84ce3853/ruff-0.16.6-py3-none-macosx_11_0_arm64.whl", hash = "sha256:99b62ea33baf130f50368798d841f0d95527b6d817bf31817b65dd058f1d314c", size = 9835082, upload-time = "2026-09-03T16:56:47.142Z" }, + { url = "https://files.pythonhosted.org/packages/99/97/123ab10b05cde889c107c20f5a9774955104b5552796a2a8584b089ae8eb/ruff-0.16.6-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:7fbf89013f2bb3f6835a6038ff658dc8a1b38c98dc8e724b964168ad4e881876", size = 9949304, upload-time = "2026-09-03T16:56:49.813Z" }, + { url = "https://files.pythonhosted.org/packages/3e/58/a4a2c59dd2e5b85929c912d9cac3056eb9ee8c7e75e9b9fe3e109174966b/ruff-0.16.6-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:56a67065e22efa6bc4d498299d3bb06c0c90aace8fac2068b5a12f9dc4d8d51d", size = 9840612, upload-time = "2026-09-03T16:56:52.368Z" }, + { url = "https://files.pythonhosted.org/packages/61/6a/ff8c8626a786c4f49d48ced4a752dadbca65f5263005f9c2416578194694/ruff-0.16.6-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:e25cc89174874b176a157e4428d66761c2c0c006654419bf384f967f361ff1b1", size = 10543465, upload-time = "2026-09-03T16:56:55.089Z" }, + { url = "https://files.pythonhosted.org/packages/ad/bb/c47535923365f337b82e28192e4e9eef2176511007cfd99a62fc22df5dad/ruff-0.16.6-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:0700580ed5303723cb3c11c2f1d2a8913ce77b7ea86646dddb887f5417a9ba70", size = 11267576, upload-time = "2026-09-03T16:56:57.791Z" }, + { url = "https://files.pythonhosted.org/packages/ba/50/e5119a5212b5cd63b51e1f4b25e7bd636a6668fc069a3160b108ad7e3c16/ruff-0.16.6-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:15f1d0b6e165a6e56567befb6629f8209271311d990bae0f37e6d065035ef5f3", size = 10781993, upload-time = "2026-09-03T16:57:00.666Z" }, + { url = "https://files.pythonhosted.org/packages/8b/98/083d8b4ef3c51a0d19db84367791cbe9f44e4b53343d19dfa83556e1cd9a/ruff-0.16.6-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:d72c591a96986ee4268860e2b7235082129ca5e4cb9cbba653a4b57c11893757", size = 10317748, upload-time = "2026-09-03T16:57:03.428Z" }, + { url = "https://files.pythonhosted.org/packages/9a/29/68f7ff2c5ad95f19f00627ac2de95644e25fe47371ea60b2db1fd952315e/ruff-0.16.6-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:65a006baa18f33324325814c864daef03541d51564b98c517610ea756ab7003e", size = 10540096, upload-time = "2026-09-03T16:57:06.182Z" }, + { url = "https://files.pythonhosted.org/packages/c4/f9/79a8f6de85968641d68a7863aeec577551924ef066a990a48ff93167beab/ruff-0.16.6-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:cd02a7bf1a21a8735228a3e8c95a9dc5cf86bd2a52194f4aaae2a5755b4de0f4", size = 10100494, upload-time = "2026-09-03T16:57:09.194Z" }, + { url = "https://files.pythonhosted.org/packages/d9/e8/b81a22d9b90c00b892ccf2fa2ac36fa95de4c13ab85aea3e73795cfe4651/ruff-0.16.6-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:31b36f1e5ad85e0737f09d2be4e512e2e283583c14015da3b9dc07359ac0fc88", size = 9843663, upload-time = "2026-09-03T16:57:12.168Z" }, + { url = "https://files.pythonhosted.org/packages/39/aa/54f516ec5e5a11c4afdceb1c454ebb054ffb96e4f4a1705580b4346abd35/ruff-0.16.6-py3-none-musllinux_1_2_i686.whl", hash = "sha256:61029b4ab4aa723fd3064fab96b1d814492596bf0c792679fffcbde1e1679953", size = 10282461, upload-time = "2026-09-03T16:57:15.077Z" }, + { url = "https://files.pythonhosted.org/packages/52/0b/38d0aa8aa32372b96dc44f97b22e576c4147808271aab7b2cb1e353d4445/ruff-0.16.6-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:9ac8998457832c2061709d900856b7ad271dace0cb41f346588d540162bfa718", size = 10728808, upload-time = "2026-09-03T16:57:17.797Z" }, + { url = "https://files.pythonhosted.org/packages/5e/e5/9e274e24eeb027640ffc7442f21239f16d17f47acec15ae34f32e03a5c79/ruff-0.16.6-py3-none-win32.whl", hash = "sha256:0b87d9d16fcb63e8018423ca1d50b7260f15cb2da33e30db4baad4183a948c25", size = 10049212, upload-time = "2026-09-03T16:57:20.55Z" }, + { url = "https://files.pythonhosted.org/packages/22/31/72472449414223ed1a2da236b992adbb1a2ae59e34794574810f60ce068e/ruff-0.16.6-py3-none-win_amd64.whl", hash = "sha256:10d21c51c3495d8eaea7b703a16592117ea6eb1d649e36335aa965ff1173eb39", size = 10556402, upload-time = "2026-09-03T16:57:23.501Z" }, + { url = "https://files.pythonhosted.org/packages/fc/07/d781f8f8e1ac24bef9f3269cf62ffb1407ca24c3a8f12e5e22874f90528c/ruff-0.16.6-py3-none-win_arm64.whl", hash = "sha256:7a976c79b958f94e50a022a19f0f8c87387448020935ec14fc74331bd0a7f2c5", size = 10412850, upload-time = "2026-09-03T16:57:26.416Z" }, +] + +[[package]] +name = "six" +version = "1.17.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/94/e7/b2c673351809dca68a0e064b6af791aa332cf192da575fd474ed7d6f16a2/six-1.17.0.tar.gz", hash = "sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81", size = 34031, upload-time = "2024-12-04T17:35:28.174Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b7/ce/149a00dd41f10bc29e5921b496af8b574d8413afcd5e30dfa0ed46c2cc5e/six-1.17.0-py2.py3-none-any.whl", hash = "sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274", size = 11050, upload-time = "2024-12-04T17:35:26.475Z" }, +] + +[[package]] +name = "snowballstemmer" +version = "3.1.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/43/f8/0a71edf031f03c40db17503cb8ca78a69a171254e568e7db241b0ab57ea1/snowballstemmer-3.1.1.tar.gz", hash = "sha256:e07bbc54a0d798fe6010a12398422e62a8bfbba95c394fd0956ef58cb4d3e260", size = 123314, upload-time = "2026-06-03T00:56:40.194Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/4c/07/2ebca9b11fb9be7340a818d8d6f63feaebb146be2c4afbd6061701d6df6e/snowballstemmer-3.1.1-py3-none-any.whl", hash = "sha256:7e207fa178741da09cdee59d3ecec3827ad5f92b1fc5c9ff3755b639f71f5752", size = 104164, upload-time = "2026-06-03T00:56:38.614Z" }, +] + +[[package]] +name = "sphinx" +version = "9.1.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "alabaster" }, + { name = "babel" }, + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "docutils" }, + { name = "imagesize" }, + { name = "jinja2" }, + { name = "packaging" }, + { name = "pygments" }, + { name = "requests" }, + { name = "roman-numerals" }, + { name = "snowballstemmer" }, + { name = "sphinxcontrib-applehelp" }, + { name = "sphinxcontrib-devhelp" }, + { name = "sphinxcontrib-htmlhelp" }, + { name = "sphinxcontrib-jsmath" }, + { name = "sphinxcontrib-qthelp" }, + { name = "sphinxcontrib-serializinghtml" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/cd/bd/f08eb0f4eed5c83f1ba2a3bd18f7745a2b1525fad70660a1c00224ec468a/sphinx-9.1.0.tar.gz", hash = "sha256:7741722357dd75f8190766926071fed3bdc211c74dd2d7d4df5404da95930ddb", size = 8718324, upload-time = "2025-12-31T15:09:27.646Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/73/f7/b1884cb3188ab181fc81fa00c266699dab600f927a964df02ec3d5d1916a/sphinx-9.1.0-py3-none-any.whl", hash = "sha256:c84fdd4e782504495fe4f2c0b3413d6c2bf388589bb352d439b2a3bb99991978", size = 3921742, upload-time = "2025-12-31T15:09:25.561Z" }, +] + +[[package]] +name = "sphinxcontrib-applehelp" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ba/6e/b837e84a1a704953c62ef8776d45c3e8d759876b4a84fe14eba2859106fe/sphinxcontrib_applehelp-2.0.0.tar.gz", hash = "sha256:2f29ef331735ce958efa4734873f084941970894c6090408b079c61b2e1c06d1", size = 20053, upload-time = "2024-07-29T01:09:00.465Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5d/85/9ebeae2f76e9e77b952f4b274c27238156eae7979c5421fba91a28f4970d/sphinxcontrib_applehelp-2.0.0-py3-none-any.whl", hash = "sha256:4cd3f0ec4ac5dd9c17ec65e9ab272c9b867ea77425228e68ecf08d6b28ddbdb5", size = 119300, upload-time = "2024-07-29T01:08:58.99Z" }, +] + +[[package]] +name = "sphinxcontrib-devhelp" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f6/d2/5beee64d3e4e747f316bae86b55943f51e82bb86ecd325883ef65741e7da/sphinxcontrib_devhelp-2.0.0.tar.gz", hash = "sha256:411f5d96d445d1d73bb5d52133377b4248ec79db5c793ce7dbe59e074b4dd1ad", size = 12967, upload-time = "2024-07-29T01:09:23.417Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/35/7a/987e583882f985fe4d7323774889ec58049171828b58c2217e7f79cdf44e/sphinxcontrib_devhelp-2.0.0-py3-none-any.whl", hash = "sha256:aefb8b83854e4b0998877524d1029fd3e6879210422ee3780459e28a1f03a8a2", size = 82530, upload-time = "2024-07-29T01:09:21.945Z" }, +] + +[[package]] +name = "sphinxcontrib-htmlhelp" +version = "2.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/43/93/983afd9aa001e5201eab16b5a444ed5b9b0a7a010541e0ddfbbfd0b2470c/sphinxcontrib_htmlhelp-2.1.0.tar.gz", hash = "sha256:c9e2916ace8aad64cc13a0d233ee22317f2b9025b9cf3295249fa985cc7082e9", size = 22617, upload-time = "2024-07-29T01:09:37.889Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0a/7b/18a8c0bcec9182c05a0b3ec2a776bba4ead82750a55ff798e8d406dae604/sphinxcontrib_htmlhelp-2.1.0-py3-none-any.whl", hash = "sha256:166759820b47002d22914d64a075ce08f4c46818e17cfc9470a9786b759b19f8", size = 98705, upload-time = "2024-07-29T01:09:36.407Z" }, +] + +[[package]] +name = "sphinxcontrib-jsmath" +version = "1.0.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b2/e8/9ed3830aeed71f17c026a07a5097edcf44b692850ef215b161b8ad875729/sphinxcontrib-jsmath-1.0.1.tar.gz", hash = "sha256:a9925e4a4587247ed2191a22df5f6970656cb8ca2bd6284309578f2153e0c4b8", size = 5787, upload-time = "2019-01-21T16:10:16.347Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c2/42/4c8646762ee83602e3fb3fbe774c2fac12f317deb0b5dbeeedd2d3ba4b77/sphinxcontrib_jsmath-1.0.1-py2.py3-none-any.whl", hash = "sha256:2ec2eaebfb78f3f2078e73666b1415417a116cc848b72e5172e596c871103178", size = 5071, upload-time = "2019-01-21T16:10:14.333Z" }, +] + +[[package]] +name = "sphinxcontrib-qthelp" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/68/bc/9104308fc285eb3e0b31b67688235db556cd5b0ef31d96f30e45f2e51cae/sphinxcontrib_qthelp-2.0.0.tar.gz", hash = "sha256:4fe7d0ac8fc171045be623aba3e2a8f613f8682731f9153bb2e40ece16b9bbab", size = 17165, upload-time = "2024-07-29T01:09:56.435Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/27/83/859ecdd180cacc13b1f7e857abf8582a64552ea7a061057a6c716e790fce/sphinxcontrib_qthelp-2.0.0-py3-none-any.whl", hash = "sha256:b18a828cdba941ccd6ee8445dbe72ffa3ef8cbe7505d8cd1fa0d42d3f2d5f3eb", size = 88743, upload-time = "2024-07-29T01:09:54.885Z" }, +] + +[[package]] +name = "sphinxcontrib-serializinghtml" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3b/44/6716b257b0aa6bfd51a1b31665d1c205fb12cb5ad56de752dfa15657de2f/sphinxcontrib_serializinghtml-2.0.0.tar.gz", hash = "sha256:e9d912827f872c029017a53f0ef2180b327c3f7fd23c87229f7a8e8b70031d4d", size = 16080, upload-time = "2024-07-29T01:10:09.332Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/52/a7/d2782e4e3f77c8450f727ba74a8f12756d5ba823d81b941f1b04da9d033a/sphinxcontrib_serializinghtml-2.0.0-py3-none-any.whl", hash = "sha256:6e2cb0eef194e10c27ec0023bfeb25badbbb5868244cf5bc5bdc04e4464bf331", size = 92072, upload-time = "2024-07-29T01:10:08.203Z" }, +] + +[[package]] +name = "tomli" +version = "2.4.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/22/de/48c59722572767841493b26183a0d1cc411d54fd759c5607c4590b6563a6/tomli-2.4.1.tar.gz", hash = "sha256:7c7e1a961a0b2f2472c1ac5b69affa0ae1132c39adcb67aba98568702b9cc23f", size = 17543, upload-time = "2026-03-25T20:22:03.828Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c1/ba/42f134a3fe2b370f555f44b1d72feebb94debcab01676bf918d0cb70e9aa/tomli-2.4.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:c742f741d58a28940ce01d58f0ab2ea3ced8b12402f162f4d534dfe18ba1cd6a", size = 155924, upload-time = "2026-03-25T20:21:21.626Z" }, + { url = "https://files.pythonhosted.org/packages/dc/c7/62d7a17c26487ade21c5422b646110f2162f1fcc95980ef7f63e73c68f14/tomli-2.4.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:7f86fd587c4ed9dd76f318225e7d9b29cfc5a9d43de44e5754db8d1128487085", size = 150018, upload-time = "2026-03-25T20:21:23.002Z" }, + { url = "https://files.pythonhosted.org/packages/5c/05/79d13d7c15f13bdef410bdd49a6485b1c37d28968314eabee452c22a7fda/tomli-2.4.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ff18e6a727ee0ab0388507b89d1bc6a22b138d1e2fa56d1ad494586d61d2eae9", size = 244948, upload-time = "2026-03-25T20:21:24.04Z" }, + { url = "https://files.pythonhosted.org/packages/10/90/d62ce007a1c80d0b2c93e02cab211224756240884751b94ca72df8a875ca/tomli-2.4.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:136443dbd7e1dee43c68ac2694fde36b2849865fa258d39bf822c10e8068eac5", size = 253341, upload-time = "2026-03-25T20:21:25.177Z" }, + { url = "https://files.pythonhosted.org/packages/1a/7e/caf6496d60152ad4ed09282c1885cca4eea150bfd007da84aea07bcc0a3e/tomli-2.4.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:5e262d41726bc187e69af7825504c933b6794dc3fbd5945e41a79bb14c31f585", size = 248159, upload-time = "2026-03-25T20:21:26.364Z" }, + { url = "https://files.pythonhosted.org/packages/99/e7/c6f69c3120de34bbd882c6fba7975f3d7a746e9218e56ab46a1bc4b42552/tomli-2.4.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:5cb41aa38891e073ee49d55fbc7839cfdb2bc0e600add13874d048c94aadddd1", size = 253290, upload-time = "2026-03-25T20:21:27.46Z" }, + { url = "https://files.pythonhosted.org/packages/d6/2f/4a3c322f22c5c66c4b836ec58211641a4067364f5dcdd7b974b4c5da300c/tomli-2.4.1-cp312-cp312-win32.whl", hash = "sha256:da25dc3563bff5965356133435b757a795a17b17d01dbc0f42fb32447ddfd917", size = 98141, upload-time = "2026-03-25T20:21:28.492Z" }, + { url = "https://files.pythonhosted.org/packages/24/22/4daacd05391b92c55759d55eaee21e1dfaea86ce5c571f10083360adf534/tomli-2.4.1-cp312-cp312-win_amd64.whl", hash = "sha256:52c8ef851d9a240f11a88c003eacb03c31fc1c9c4ec64a99a0f922b93874fda9", size = 108847, upload-time = "2026-03-25T20:21:29.386Z" }, + { url = "https://files.pythonhosted.org/packages/68/fd/70e768887666ddd9e9f5d85129e84910f2db2796f9096aa02b721a53098d/tomli-2.4.1-cp312-cp312-win_arm64.whl", hash = "sha256:f758f1b9299d059cc3f6546ae2af89670cb1c4d48ea29c3cacc4fe7de3058257", size = 95088, upload-time = "2026-03-25T20:21:30.677Z" }, + { url = "https://files.pythonhosted.org/packages/07/06/b823a7e818c756d9a7123ba2cda7d07bc2dd32835648d1a7b7b7a05d848d/tomli-2.4.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:36d2bd2ad5fb9eaddba5226aa02c8ec3fa4f192631e347b3ed28186d43be6b54", size = 155866, upload-time = "2026-03-25T20:21:31.65Z" }, + { url = "https://files.pythonhosted.org/packages/14/6f/12645cf7f08e1a20c7eb8c297c6f11d31c1b50f316a7e7e1e1de6e2e7b7e/tomli-2.4.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:eb0dc4e38e6a1fd579e5d50369aa2e10acfc9cace504579b2faabb478e76941a", size = 149887, upload-time = "2026-03-25T20:21:33.028Z" }, + { url = "https://files.pythonhosted.org/packages/5c/e0/90637574e5e7212c09099c67ad349b04ec4d6020324539297b634a0192b0/tomli-2.4.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c7f2c7f2b9ca6bdeef8f0fa897f8e05085923eb091721675170254cbc5b02897", size = 243704, upload-time = "2026-03-25T20:21:34.51Z" }, + { url = "https://files.pythonhosted.org/packages/10/8f/d3ddb16c5a4befdf31a23307f72828686ab2096f068eaf56631e136c1fdd/tomli-2.4.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f3c6818a1a86dd6dca7ddcaaf76947d5ba31aecc28cb1b67009a5877c9a64f3f", size = 251628, upload-time = "2026-03-25T20:21:36.012Z" }, + { url = "https://files.pythonhosted.org/packages/e3/f1/dbeeb9116715abee2485bf0a12d07a8f31af94d71608c171c45f64c0469d/tomli-2.4.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:d312ef37c91508b0ab2cee7da26ec0b3ed2f03ce12bd87a588d771ae15dcf82d", size = 247180, upload-time = "2026-03-25T20:21:37.136Z" }, + { url = "https://files.pythonhosted.org/packages/d3/74/16336ffd19ed4da28a70959f92f506233bd7cfc2332b20bdb01591e8b1d1/tomli-2.4.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:51529d40e3ca50046d7606fa99ce3956a617f9b36380da3b7f0dd3dd28e68cb5", size = 251674, upload-time = "2026-03-25T20:21:38.298Z" }, + { url = "https://files.pythonhosted.org/packages/16/f9/229fa3434c590ddf6c0aa9af64d3af4b752540686cace29e6281e3458469/tomli-2.4.1-cp313-cp313-win32.whl", hash = "sha256:2190f2e9dd7508d2a90ded5ed369255980a1bcdd58e52f7fe24b8162bf9fedbd", size = 97976, upload-time = "2026-03-25T20:21:39.316Z" }, + { url = "https://files.pythonhosted.org/packages/6a/1e/71dfd96bcc1c775420cb8befe7a9d35f2e5b1309798f009dca17b7708c1e/tomli-2.4.1-cp313-cp313-win_amd64.whl", hash = "sha256:8d65a2fbf9d2f8352685bc1364177ee3923d6baf5e7f43ea4959d7d8bc326a36", size = 108755, upload-time = "2026-03-25T20:21:40.248Z" }, + { url = "https://files.pythonhosted.org/packages/83/7a/d34f422a021d62420b78f5c538e5b102f62bea616d1d75a13f0a88acb04a/tomli-2.4.1-cp313-cp313-win_arm64.whl", hash = "sha256:4b605484e43cdc43f0954ddae319fb75f04cc10dd80d830540060ee7cd0243cd", size = 95265, upload-time = "2026-03-25T20:21:41.219Z" }, + { url = "https://files.pythonhosted.org/packages/3c/fb/9a5c8d27dbab540869f7c1f8eb0abb3244189ce780ba9cd73f3770662072/tomli-2.4.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:fd0409a3653af6c147209d267a0e4243f0ae46b011aa978b1080359fddc9b6cf", size = 155726, upload-time = "2026-03-25T20:21:42.23Z" }, + { url = "https://files.pythonhosted.org/packages/62/05/d2f816630cc771ad836af54f5001f47a6f611d2d39535364f148b6a92d6b/tomli-2.4.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:a120733b01c45e9a0c34aeef92bf0cf1d56cfe81ed9d47d562f9ed591a9828ac", size = 149859, upload-time = "2026-03-25T20:21:43.386Z" }, + { url = "https://files.pythonhosted.org/packages/ce/48/66341bdb858ad9bd0ceab5a86f90eddab127cf8b046418009f2125630ecb/tomli-2.4.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:559db847dc486944896521f68d8190be1c9e719fced785720d2216fe7022b662", size = 244713, upload-time = "2026-03-25T20:21:44.474Z" }, + { url = "https://files.pythonhosted.org/packages/df/6d/c5fad00d82b3c7a3ab6189bd4b10e60466f22cfe8a08a9394185c8a8111c/tomli-2.4.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:01f520d4f53ef97964a240a035ec2a869fe1a37dde002b57ebc4417a27ccd853", size = 252084, upload-time = "2026-03-25T20:21:45.62Z" }, + { url = "https://files.pythonhosted.org/packages/00/71/3a69e86f3eafe8c7a59d008d245888051005bd657760e96d5fbfb0b740c2/tomli-2.4.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:7f94b27a62cfad8496c8d2513e1a222dd446f095fca8987fceef261225538a15", size = 247973, upload-time = "2026-03-25T20:21:46.937Z" }, + { url = "https://files.pythonhosted.org/packages/67/50/361e986652847fec4bd5e4a0208752fbe64689c603c7ae5ea7cb16b1c0ca/tomli-2.4.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:ede3e6487c5ef5d28634ba3f31f989030ad6af71edfb0055cbbd14189ff240ba", size = 256223, upload-time = "2026-03-25T20:21:48.467Z" }, + { url = "https://files.pythonhosted.org/packages/8c/9a/b4173689a9203472e5467217e0154b00e260621caa227b6fa01feab16998/tomli-2.4.1-cp314-cp314-win32.whl", hash = "sha256:3d48a93ee1c9b79c04bb38772ee1b64dcf18ff43085896ea460ca8dec96f35f6", size = 98973, upload-time = "2026-03-25T20:21:49.526Z" }, + { url = "https://files.pythonhosted.org/packages/14/58/640ac93bf230cd27d002462c9af0d837779f8773bc03dee06b5835208214/tomli-2.4.1-cp314-cp314-win_amd64.whl", hash = "sha256:88dceee75c2c63af144e456745e10101eb67361050196b0b6af5d717254dddf7", size = 109082, upload-time = "2026-03-25T20:21:50.506Z" }, + { url = "https://files.pythonhosted.org/packages/d5/2f/702d5e05b227401c1068f0d386d79a589bb12bf64c3d2c72ce0631e3bc49/tomli-2.4.1-cp314-cp314-win_arm64.whl", hash = "sha256:b8c198f8c1805dc42708689ed6864951fd2494f924149d3e4bce7710f8eb5232", size = 96490, upload-time = "2026-03-25T20:21:51.474Z" }, + { url = "https://files.pythonhosted.org/packages/45/4b/b877b05c8ba62927d9865dd980e34a755de541eb65fffba52b4cc495d4d2/tomli-2.4.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:d4d8fe59808a54658fcc0160ecfb1b30f9089906c50b23bcb4c69eddc19ec2b4", size = 164263, upload-time = "2026-03-25T20:21:52.543Z" }, + { url = "https://files.pythonhosted.org/packages/24/79/6ab420d37a270b89f7195dec5448f79400d9e9c1826df982f3f8e97b24fd/tomli-2.4.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:7008df2e7655c495dd12d2a4ad038ff878d4ca4b81fccaf82b714e07eae4402c", size = 160736, upload-time = "2026-03-25T20:21:53.674Z" }, + { url = "https://files.pythonhosted.org/packages/02/e0/3630057d8eb170310785723ed5adcdfb7d50cb7e6455f85ba8a3deed642b/tomli-2.4.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1d8591993e228b0c930c4bb0db464bdad97b3289fb981255d6c9a41aedc84b2d", size = 270717, upload-time = "2026-03-25T20:21:55.129Z" }, + { url = "https://files.pythonhosted.org/packages/7a/b4/1613716072e544d1a7891f548d8f9ec6ce2faf42ca65acae01d76ea06bb0/tomli-2.4.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:734e20b57ba95624ecf1841e72b53f6e186355e216e5412de414e3c51e5e3c41", size = 278461, upload-time = "2026-03-25T20:21:56.228Z" }, + { url = "https://files.pythonhosted.org/packages/05/38/30f541baf6a3f6df77b3df16b01ba319221389e2da59427e221ef417ac0c/tomli-2.4.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:8a650c2dbafa08d42e51ba0b62740dae4ecb9338eefa093aa5c78ceb546fcd5c", size = 274855, upload-time = "2026-03-25T20:21:57.653Z" }, + { url = "https://files.pythonhosted.org/packages/77/a3/ec9dd4fd2c38e98de34223b995a3b34813e6bdadf86c75314c928350ed14/tomli-2.4.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:504aa796fe0569bb43171066009ead363de03675276d2d121ac1a4572397870f", size = 283144, upload-time = "2026-03-25T20:21:59.089Z" }, + { url = "https://files.pythonhosted.org/packages/ef/be/605a6261cac79fba2ec0c9827e986e00323a1945700969b8ee0b30d85453/tomli-2.4.1-cp314-cp314t-win32.whl", hash = "sha256:b1d22e6e9387bf4739fbe23bfa80e93f6b0373a7f1b96c6227c32bef95a4d7a8", size = 108683, upload-time = "2026-03-25T20:22:00.214Z" }, + { url = "https://files.pythonhosted.org/packages/12/64/da524626d3b9cc40c168a13da8335fe1c51be12c0a63685cc6db7308daae/tomli-2.4.1-cp314-cp314t-win_amd64.whl", hash = "sha256:2c1c351919aca02858f740c6d33adea0c5deea37f9ecca1cc1ef9e884a619d26", size = 121196, upload-time = "2026-03-25T20:22:01.169Z" }, + { url = "https://files.pythonhosted.org/packages/5a/cd/e80b62269fc78fc36c9af5a6b89c835baa8af28ff5ad28c7028d60860320/tomli-2.4.1-cp314-cp314t-win_arm64.whl", hash = "sha256:eab21f45c7f66c13f2a9e0e1535309cee140182a9cdae1e041d02e47291e8396", size = 100393, upload-time = "2026-03-25T20:22:02.137Z" }, + { url = "https://files.pythonhosted.org/packages/7b/61/cceae43728b7de99d9b847560c262873a1f6c98202171fd5ed62640b494b/tomli-2.4.1-py3-none-any.whl", hash = "sha256:0d85819802132122da43cb86656f8d1f8c6587d54ae7dcaf30e90533028b49fe", size = 14583, upload-time = "2026-03-25T20:22:03.012Z" }, +] + +[[package]] +name = "ty" +version = "0.0.78" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d8/c7/2ba0861384c5b5097ac354383abb98188112cb330208c21d5197e98a29e5/ty-0.0.78.tar.gz", hash = "sha256:770b45854f85fa11595208f08c0f28df80943164d10a2832d86be6ac29f135b2", size = 7050609, upload-time = "2026-09-02T22:41:33.92Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ea/ed/f34cfc06a9ba72979df219e48dae1a0b6a6c74a340763ccd30a547edf913/ty-0.0.78-py3-none-linux_armv6l.whl", hash = "sha256:122700b98f9d45785c1ce91a9418154562e7f64f4bf34679f6f7b38c5b97453f", size = 13304624, upload-time = "2026-09-02T22:40:56.649Z" }, + { url = "https://files.pythonhosted.org/packages/1c/28/5576e2a08b57676d9b2a736d528f077a6c6e32c3d1de2d75dbbe28346966/ty-0.0.78-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:cbe7e3709ccf29ef3d9f58f5914cc30ec36fb647008a5a4ff8483abe401bfe8d", size = 12928383, upload-time = "2026-09-02T22:40:59.162Z" }, + { url = "https://files.pythonhosted.org/packages/8e/e0/027d49c3da5235e634da3d3fc52c4874ec89f50830388c9c259f8fb71e0e/ty-0.0.78-py3-none-macosx_11_0_arm64.whl", hash = "sha256:c3528897b3ab9d3589561bc2b5e61a4d5d686de527a4400bb09a439b922a6c70", size = 12730058, upload-time = "2026-09-02T22:41:01.235Z" }, + { url = "https://files.pythonhosted.org/packages/bb/bc/6bf8ffefd8063730a70ae889cf85def72bc155fd1453135ebf8a09c2f1c2/ty-0.0.78-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:8bfba4c44a06484093f527b8ef43a317abcf91395b49396170266629eedf74aa", size = 12813321, upload-time = "2026-09-02T22:41:03.251Z" }, + { url = "https://files.pythonhosted.org/packages/6d/c0/b3cdf26f82108908d92fdb7ddb741019c446ceb666cf204d4dbf611bde33/ty-0.0.78-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:5f903c06fdee17baf8373173039ab28f1bce28e66b1c734929a5f50b37eb54d9", size = 13072219, upload-time = "2026-09-02T22:41:05.225Z" }, + { url = "https://files.pythonhosted.org/packages/4d/36/16bfb0abdd178dae11dbc4b57b9a86c2b62642a18c1103fd2c2930aa5879/ty-0.0.78-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:dd83e5fe3f07291d1bd4e591f8d2607c204f3eab53bcb1b8060c40e876e1aa61", size = 13903958, upload-time = "2026-09-02T22:41:07.333Z" }, + { url = "https://files.pythonhosted.org/packages/0f/0c/74d3b1f0344b13156c719dd68a7edf7e48c2051718c9f326ba3cbf45ea53/ty-0.0.78-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:325285377319cf7168a2b8b771ae5027f411530e3c7eab1faf4583cdd72fd7c9", size = 14355581, upload-time = "2026-09-02T22:41:09.56Z" }, + { url = "https://files.pythonhosted.org/packages/8d/d7/7ca0359e1e61b15b8b02328c8d120db3c507876b9b7c6bd9f26935353e0e/ty-0.0.78-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:54b7846a404da6492697b524a769755ee9bee157d67674063ecd9c5f69dc52ff", size = 14044749, upload-time = "2026-09-02T22:41:11.678Z" }, + { url = "https://files.pythonhosted.org/packages/3b/6a/731f16ff42c5fc96e742c2f4e0a6915356bae114ae02ae4af6f33502175f/ty-0.0.78-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:169d3b9d134c0b48fe1af8a844142d374655552054a9d6c15a4b6e51bb0382ea", size = 13391647, upload-time = "2026-09-02T22:41:13.928Z" }, + { url = "https://files.pythonhosted.org/packages/0c/af/250cc29daf310e837188509ea4d78460c495d8a74c4fb84b4152d9ef2d4f/ty-0.0.78-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6108cb3b2d28dac5981d4e25008a0a5547d8c877a67f6da9e8c38e7e946be44f", size = 13945020, upload-time = "2026-09-02T22:41:15.968Z" }, + { url = "https://files.pythonhosted.org/packages/b8/f9/96aab1dee4535e66554e8dc7657c69f6c61c181d8e70b9c3527908e93ef4/ty-0.0.78-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:589ad03608d9d2975ef4b23c41c4f847e325bcc7686f51d4c66066b8dbc18a2a", size = 12851149, upload-time = "2026-09-02T22:41:18.06Z" }, + { url = "https://files.pythonhosted.org/packages/85/b0/53d8fe9a847534ef5fe2165c7d5292cb853b29dfd7011cbfb1cdb90a8012/ty-0.0.78-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b3fa0786edc1af06030f872d83499c0cea7c261dbbb651e0aee8306bf2c8a869", size = 13091464, upload-time = "2026-09-02T22:41:20.227Z" }, + { url = "https://files.pythonhosted.org/packages/21/65/94a4e5a02de559f6c6c0ee7a14b88660b4eee058ec6fec5c0ff6ecfbf5cc/ty-0.0.78-py3-none-musllinux_1_2_i686.whl", hash = "sha256:92c5639befc577578c8abd4e5a7fccf98d828db98b09e8d4dd81605dc0e54aef", size = 13389389, upload-time = "2026-09-02T22:41:22.27Z" }, + { url = "https://files.pythonhosted.org/packages/2e/b6/d0c7fe6be64ca5a15100c4b1f0e39c19660d53bc544f609fa117b346e661/ty-0.0.78-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:0dce70bc51652b2775debd1d1e422fb0179f5207c73deb4d5d0aca37e5f61695", size = 13691928, upload-time = "2026-09-02T22:41:24.292Z" }, + { url = "https://files.pythonhosted.org/packages/0a/5d/36081205611ba3fa802efc4ad63e4b1e949bda2f7bb37b34ac9bb6c00db0/ty-0.0.78-py3-none-win32.whl", hash = "sha256:1c80976ca9185d7a9d1baab1fb57240331b8dac58d3b11e46200284086a6de8d", size = 12647346, upload-time = "2026-09-02T22:41:26.699Z" }, + { url = "https://files.pythonhosted.org/packages/f1/89/b925fe1ea1bc56fc7f11d2496e29072fdd2ceb7b389884fc7b2d07cf0c41/ty-0.0.78-py3-none-win_amd64.whl", hash = "sha256:32e82b704471eab34f67b51c151660ca8a00815977b28278905d76fba54f7415", size = 13241810, upload-time = "2026-09-02T22:41:28.857Z" }, + { url = "https://files.pythonhosted.org/packages/91/f1/090ef7b52355bcedfbfbff6ce70fa81ba5bd5f6ed3f8d99ae05e6f2fe75b/ty-0.0.78-py3-none-win_arm64.whl", hash = "sha256:3a14d641a3c04fa9a80f2a46be1531d915f60d4fb79d4b894627bbe46bb35d64", size = 13077433, upload-time = "2026-09-02T22:41:31.525Z" }, +] + +[[package]] +name = "tzdata" +version = "2026.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/92/ff/5a28bdfd8c3ebec42564ac7d0e54ca3db65044a9314a97f9564fa7a1e926/tzdata-2026.3.tar.gz", hash = "sha256:4a1518b8993086a7982523e071643f3c0e5f213e75b21318e78bcabfff9d1415", size = 198674, upload-time = "2026-07-10T08:50:37.887Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e5/6d/b53b99a9f2766d095985947a5782f1702cabb129a34f7a802d7197af832f/tzdata-2026.3-py2.py3-none-any.whl", hash = "sha256:dc096730c87af6cab1b171c9d532be840741ff5d459015e7f6947bd7d7e54931", size = 348168, upload-time = "2026-07-10T08:50:36.46Z" }, +] + +[[package]] +name = "urllib3" +version = "2.7.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/53/0c/06f8b233b8fd13b9e5ee11424ef85419ba0d8ba0b3138bf360be2ff56953/urllib3-2.7.0.tar.gz", hash = "sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c", size = 433602, upload-time = "2026-05-07T16:13:18.596Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7f/3e/5db95bcf282c52709639744ca2a8b149baccf648e39c8cc87553df9eae0c/urllib3-2.7.0-py3-none-any.whl", hash = "sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897", size = 131087, upload-time = "2026-05-07T16:13:17.151Z" }, +] From 102a5e77a5f27434f8924391320a71d058fa4afe Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 9 Sep 2026 12:22:09 +1200 Subject: [PATCH 02/25] Saving --- README.md | 103 +-- imdb/__init__.py | 2 +- imdb/imdb.py | 1875 ++++++++++---------------------------------- imdb/schema.py | 254 +++--- pyproject.toml | 2 +- tests/conftest.py | 168 ++-- tests/test_imdb.py | 320 +------- uv.lock | 199 ++++- 8 files changed, 797 insertions(+), 2126 deletions(-) diff --git a/README.md b/README.md index 2e806de..fa30faa 100644 --- a/README.md +++ b/README.md @@ -1,92 +1,37 @@ # imdb -Reading and writing intensity measure databases (IMDBs): DuckDB files holding -intensity measures from physics-based ground-motion simulation. +A library for reading and writing intensity measure databases (IMDBs) — DuckDB +databases of simulated ground-motion intensity measures. Schema is documented in +`imdb/schema.py`. -One file holds one simulation run set. Every file uses the same schema and is -self-contained; combine several with `ATTACH` and `UNION ALL`. - -## Schema - -Thirteen tables: four dimensions (`events`, `realisations`, `sites`, -`site_event`), one identity table (`records`), three IM tables (`psa_ims`, -`fas_ims`, `scalars_ims`), two vocabulary tables (`periods`, `frequencies`) and -three documentation tables (`db_meta`, `im_units`, `notes`). - -A ground motion is identified by `(rel_id, site_id, component)`. pSA and FAS are -stored as one `FLOAT[]` per record, indexed by `periods.period_index` and -`frequencies.freq_index` (both 1-based). Scalar IMs are named columns. - -`imdb/schema.py` holds the DDL and is the single source of truth. Every database -also documents itself: read the `notes` and `db_meta` tables. - -## Reading +## Usage ```python from imdb import IMDB -with IMDB("cs200.duckdb") as db: - df = db.get_im_df( - ["PGA", "pSA_0.1", "pSA_1.0", "FAS_5.0"], - events=["AlpineF2K"], component="rotd50", max_rrup=200, - ) +# read +with IMDB("run_set.duckdb") as db: + records = db.get_records(event_ids=["event1"], component="rotd50") + psa = db.get_psa(periods=[0.1, 1.0], event_ids=["event1"]) + scalars = db.get_scalars(ims=["PGA", "PGV"]) + +# write +db = IMDB.create("new.duckdb", periods=[0.1, 0.2, 1.0]) +db.add_events(events_df) +db.add_realisations(realisations_df) +db.add_sites(sites_df) +db.add_site_event(site_event_df) +db.add_records(records_df) # rel_id, site_id, component, pSA, FAS, scalar IM columns +db.validate() +db.close() ``` -`get_im_df` returns one DataFrame indexed by `record_id`, columns named as -requested. Underneath it are `get_psa(periods=...)`, `get_fas(frequencies=...)` -and `get_scalars(ims=...)`, which label their columns with the period in -seconds, the frequency in Hz, and the IM name respectively. - -Every read takes the same keyword filters: `events`, `rels`, `sites`, -`component`, `max_rrup`, `record_ids`. Anything more specific is a raw query -through `db.sql(...)` or `db.conn`. - -Dimension tables come back whole: `get_events`, `get_realisations`, `get_sites`, -`get_site_event`. Pass `expand_metadata=True` to unpack their JSON `metadata` -column into columns. - -## Writing - -```python -with IMDB.create("new.duckdb", periods=[0.01, 0.1, 1.0], frequencies=[1.0, 10.0], - components=["rotd50"], db_meta={"source": "CyberShake v24p1"}) as db: - db.add_events(event_df) # event_id, magnitude, tect_type, ... - db.add_realisations(rel_df) # rel_id, event_id, rake, hypo_*, ... - db.add_sites(site_df) # site_id, lat, lon, vs30, ... - db.add_site_event(site_event_df) # site_id, event_id, rrup, rjb, ... - db.add_records(record_df) # rel_id, site_id, component, pSA, FAS, PGA, ... - db.finalise() -``` - -Integer surrogates (`event_int_id`, `rel_int_id`, `site_int_id`, `record_id`) are -assigned here and are file-local: they change on rebuild, so nothing outside the -file may reference them. Callers work in string IDs throughout. - -`add_records` writes `records` plus whichever IM tables the input covers; a row -with no `pSA` array simply gets no `psa_ims` row. `CAV`, `AI`, `Ds575` and -`Ds595` are undefined for `rotd*` components and are stored as NULL there. - -Extra source-specific fields go in a `metadata` column, passed as a dict. The -first write of a table's metadata declares its permitted keys in `db_meta`; -later writes are validated against that declaration. - -Statements autocommit, so writes are not rolled back on error. To re-ingest an -event, `delete_event(event_id)` first: it removes the event and everything -derived from it, leaving shared sites alone. - -## Validation - -The large tables carry no `PRIMARY KEY`, `UNIQUE` or `FOREIGN KEY`, because at -these row counts each one is an ART index loaded into memory on open and buys no -lookup speed. `validate()` checks what they would have enforced (orphan keys, -duplicated logical keys, array lengths, undeclared components and metadata keys) -and returns the problems as a list. `finalise()` runs it and raises. - ## Development -```bash +``` uv sync --all-groups -uv run pytest -uv run ruff check imdb tests && uv run ruff format --check imdb tests -uv run ty check imdb +uv run pytest -q +uv run ruff check +uv run ruff format +uv run ty check ``` diff --git a/imdb/__init__.py b/imdb/__init__.py index 3331754..ca128c4 100644 --- a/imdb/__init__.py +++ b/imdb/__init__.py @@ -1,4 +1,4 @@ -"""Reading and writing intensity measure databases.""" +"""A library for reading and writing intensity measure databases.""" from imdb.imdb import IMDB diff --git a/imdb/imdb.py b/imdb/imdb.py index d8a392f..a01a340 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -1,1627 +1,544 @@ -""" -Read and write intensity measure databases. +"""Read and write intensity measure databases (IMDBs).""" -An intensity measure database (IMDB) is a DuckDB file holding intensity -measures (IMs) from physics-based ground-motion simulation, laid out according -to :mod:`imdb.schema`. One file holds one simulation run set. - -Reading: - ->>> with IMDB("path/to/ims.duckdb") as db: -... df = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"], component="rotd50") - -Writing: - ->>> with IMDB.create("new.duckdb", periods=[0.1, 1.0], frequencies=[1.0]) as db: -... db.add_events(event_df) -... db.add_realisations(rel_df) -... db.add_sites(site_df) -... db.add_site_event(site_event_df) -... db.add_records(record_df) -... db.finalise() -""" - -import contextlib -import getpass -import json +import datetime import logging -from collections.abc import Iterable, Mapping, Sequence -from datetime import UTC, datetime -from functools import cached_property -from importlib.metadata import PackageNotFoundError, version +from importlib.metadata import version from pathlib import Path -from types import TracebackType from typing import Any, Self -import duckdb +import ibis import numpy as np import pandas as pd +from ibis import _ +from ibis.backends.duckdb import Backend as DuckDBBackend from imdb import schema logger = logging.getLogger(__name__) -_CACHED = ( - "db_meta", - "notes", - "im_units", - "components", - "periods", - "frequencies", - "event_ids", - "rel_ids", - "site_ids", - "rel_to_event", -) - -_DIM_JOINS = { - "e": "JOIN events e ON e.event_int_id = r.event_int_id", - "rl": "JOIN realisations rl ON rl.rel_int_id = r.rel_int_id", - "s": "JOIN sites s ON s.site_int_id = r.site_int_id", - "se": ( - "JOIN site_event se ON se.site_int_id = r.site_int_id " - "AND se.event_int_id = r.event_int_id" - ), -} - -_MATCH_RTOL = 1e-6 -"""Relative tolerance when matching a requested period or frequency to the grid.""" - - -def _imdb_version() -> str: - """Return the installed version of this library, or ``unknown``. - - Returns - ------- - str - Version string. - """ - try: - return version("imdb") - except PackageNotFoundError: - return "unknown" - -def _parse_number(text: str) -> float: - """Parse a period or frequency written with either ``.`` or ``p`` as the point. +class IMDB: + """A DuckDB-backed intensity measure database. Parameters ---------- - text : str - Number to parse, for example ``1.0`` or ``0p1``. - - Returns - ------- - float - The parsed value. + path : Path + Path to the database file. + read_only : bool + Open the database read-only. """ - try: - return float(text) - except ValueError: - return float(text.replace("p", ".", 1)) - -class IMDB(contextlib.AbstractContextManager): - """An intensity measure database. - - Parameters - ---------- - db_path : Path or str - Path to the DuckDB file. - read_only : bool, optional - Open read-only. Write methods raise when true. - memory_limit : str, optional - DuckDB memory limit applied on open. - """ - - def __init__( - self, - db_path: Path | str, - read_only: bool = True, - memory_limit: str = "8GB", - ) -> None: - """Create a handle. The connection is opened lazily, or by :meth:`open`.""" - self.db_path = Path(db_path) + def __init__(self, path: Path, read_only: bool = True) -> None: + """Set up the database path; does not open a connection.""" + self.path = Path(path) self.read_only = read_only - self.memory_limit = memory_limit - self._conn: duckdb.DuckDBPyConnection | None = None + self._con: DuckDBBackend | None = None - def open(self) -> Self: - """Open the database connection. + @property + def con(self) -> DuckDBBackend: + """The underlying ibis connection. Raises if the database is not open.""" + if self._con is None: + raise RuntimeError( + "database is not open; call .open() or use as a context manager" + ) + return self._con - Returns - ------- - IMDB - This database, for chaining. - """ - if self._conn is not None: - return self - if self.read_only and not self.db_path.exists(): - raise FileNotFoundError(self.db_path) - self._conn = duckdb.connect(str(self.db_path), read_only=self.read_only) - self._conn.execute(f"SET memory_limit='{self.memory_limit}'") - self._conn.execute("SET enable_progress_bar = false") + def open(self) -> Self: + """Open the database connection, if not already open.""" + if self._con is None: + self._con = ibis.duckdb.connect(self.path, read_only=self.read_only) return self def close(self) -> None: """Close the database connection.""" - if self._conn is not None: - self._conn.close() - self._conn = None - self._invalidate() - - @property - def conn(self) -> duckdb.DuckDBPyConnection: - """The open DuckDB connection. - - Returns - ------- - duckdb.DuckDBPyConnection - The connection. - """ - if self._conn is None: - self.open() - assert self._conn is not None - return self._conn - - def sql( - self, query: str, params: Sequence | None = None - ) -> duckdb.DuckDBPyRelation: - """Run an arbitrary query against the database. - - The escape hatch for reads the keyword filters cannot express. - - Parameters - ---------- - query : str - SQL to execute. - params : sequence, optional - Prepared-statement parameters. - - Returns - ------- - duckdb.DuckDBPyRelation - The query result. - """ - return self.conn.sql(query, params=params) + if self._con is not None: + self._con.disconnect() + self._con = None def __enter__(self) -> Self: - """Open the connection and return this database.""" + """Open the database connection.""" return self.open() - def __exit__( - self, - exc_type: type[BaseException] | None, - exc_val: BaseException | None, - exc_tb: TracebackType | None, - ) -> None: - """Close the connection. - - Statements autocommit, so a failed write is not rolled back here. Repair - a partial ingest with :meth:`delete_event`, which is what makes a - re-ingest idempotent. - """ + def __exit__(self, *exc: object) -> None: + """Close the database connection.""" self.close() - def _invalidate(self) -> None: - """Drop every cached lookup, after a write or a close.""" - for name in _CACHED: - self.__dict__.pop(name, None) - - def _require_write(self) -> None: - """Raise if the database was opened read-only.""" - if self.read_only: - raise PermissionError(f"{self.db_path} is open read-only") - - @contextlib.contextmanager - def _temp_frames(self, frames: Mapping[str, pd.DataFrame]): - """Register DataFrames as views for the duration of a query. - - Parameters - ---------- - frames : mapping of str to pandas.DataFrame - View name to DataFrame. - - Yields - ------ - None - """ - for name, df in frames.items(): - self.conn.register(name, df) - try: - yield - finally: - for name in frames: - self.conn.unregister(name) - - @contextlib.contextmanager - def _transaction(self): - """Group several statements into one atomic write. - - Yields - ------ - None - """ - self.conn.execute("BEGIN TRANSACTION") - try: - yield - except Exception: - self.conn.execute("ROLLBACK") - raise - else: - self.conn.execute("COMMIT") - - # ------------------------------------------------------------------ - # cached vocabulary - # ------------------------------------------------------------------ - - @cached_property - def db_meta(self) -> dict[str, str]: - """Contents of the ``db_meta`` table. - - Returns - ------- - dict of str to str - Key to value. - """ - return dict(self.conn.execute("SELECT key, value FROM db_meta").fetchall()) - - @cached_property - def notes(self) -> dict[str, str]: - """Contents of the ``notes`` table. - - Returns - ------- - dict of str to str - Topic to note. - """ - return dict(self.conn.execute("SELECT topic, note FROM notes").fetchall()) - - @cached_property - def im_units(self) -> dict[str, str]: - """Contents of the ``im_units`` table. - - Returns - ------- - dict of str to str - IM name to unit. - """ - return dict(self.conn.execute("SELECT im, unit FROM im_units").fetchall()) - - @cached_property - def components(self) -> list[str]: - """Components held by this database, from ``db_meta``. - - Returns - ------- - list of str - Component names. - """ - value = self.db_meta.get("components", "") - return [c for c in (part.strip() for part in value.split(",")) if c] - - @cached_property - def periods(self) -> pd.Series: - """The pSA period grid. - - Returns - ------- - pandas.Series - Period in seconds, indexed by 1-based ``period_index``. - """ - return ( - self.conn.execute("SELECT period_index, period FROM periods ORDER BY 1") - .df() - .set_index("period_index")["period"] - ) - - @cached_property - def frequencies(self) -> pd.Series: - """The FAS frequency grid. - - Returns - ------- - pandas.Series - Frequency in Hz, indexed by 1-based ``freq_index``. - """ - return ( - self.conn.execute( - "SELECT freq_index, frequency FROM frequencies ORDER BY 1" - ) - .df() - .set_index("freq_index")["frequency"] - ) - - @cached_property - def event_ids(self) -> pd.Series: - """Mapping of ``event_id`` to ``event_int_id``. - - Returns - ------- - pandas.Series - Integer id indexed by string id. - """ - return self._id_map("events", "event_id", "event_int_id") - - @cached_property - def rel_ids(self) -> pd.Series: - """Mapping of ``rel_id`` to ``rel_int_id``. - - Returns - ------- - pandas.Series - Integer id indexed by string id. - """ - return self._id_map("realisations", "rel_id", "rel_int_id") - - @cached_property - def site_ids(self) -> pd.Series: - """Mapping of ``site_id`` to ``site_int_id``. - - Returns - ------- - pandas.Series - Integer id indexed by string id. - """ - return self._id_map("sites", "site_id", "site_int_id") - - @cached_property - def rel_to_event(self) -> pd.Series: - """Mapping of ``rel_int_id`` to ``event_int_id``. - - Returns - ------- - pandas.Series - Event integer id indexed by realisation integer id. - """ - return self._id_map("realisations", "rel_int_id", "event_int_id") - - def _id_map(self, table: str, key: str, value: str) -> pd.Series: - """Read a two-column mapping out of a dimension table. - - Parameters - ---------- - table : str - Table to read. - key : str - Column to index by. - value : str - Column to map to. - - Returns - ------- - pandas.Series - ``value`` indexed by ``key``. - """ - return ( - self.conn.execute(f"SELECT {key}, {value} FROM {table}") - .df() - .set_index(key)[value] - ) - - def _resolve(self, ids: Iterable[str], mapping: pd.Series, what: str) -> np.ndarray: - """Resolve string ids to integer surrogates. - - Parameters - ---------- - ids : iterable of str - String ids to resolve. - mapping : pandas.Series - Mapping to resolve against. - what : str - Name used in the error message. - - Returns - ------- - numpy.ndarray - Integer ids, in the order given. - """ - # the id columns are VARCHAR, so a numeric id from the source reaches the - # database as its string form and must be looked up that way - ids = np.asarray([str(value) for value in ids], dtype=object) - unknown = pd.unique(ids[~pd.Index(ids).isin(mapping.index)]) - if len(unknown): - raise KeyError(f"unknown {what}: {sorted(unknown)[:10]}") - return mapping.loc[ids].to_numpy(dtype=np.int64) - - def _scalar(self, query: str, params: Sequence | None = None) -> Any: - """Run a query returning exactly one row and one column. - - Parameters - ---------- - query : str - SQL to execute. - params : sequence, optional - Prepared-statement parameters. - - Returns - ------- - Any - The single value. - """ - row = self.conn.execute(query, params).fetchone() - if row is None: - raise RuntimeError(f"query returned no rows: {query}") - return row[0] - - def _table_columns(self, table: str) -> list[str]: - """Column names of a table, in declaration order. - - Parameters - ---------- - table : str - Table name. - - Returns - ------- - list of str - Column names. - """ - rows = self.conn.execute( - "SELECT column_name FROM information_schema.columns " - "WHERE table_name = ? ORDER BY ordinal_position", - [table], - ).fetchall() - if not rows: - raise KeyError(f"no such table: {table}") - return [row[0] for row in rows] - - def _grid_indices( - self, values: Iterable[float] | None, grid: pd.Series, what: str - ) -> tuple[list[int], list[float]]: - """Resolve requested grid values to 1-based array indices. - - Parameters - ---------- - values : iterable of float, optional - Values to resolve. ``None`` returns the whole grid. - grid : pandas.Series - Grid to resolve against, indexed by array index. - what : str - Name used in the error message. - - Returns - ------- - tuple of (list of int, list of float) - Array indices and the matched grid values, in the order requested. - """ - if grid.empty: - raise KeyError(f"this database has no {what} grid") - if values is None: - return list(grid.index.astype(int)), list(grid.to_numpy(dtype=float)) - - available = grid.to_numpy(dtype=float) - indices, matched = [], [] - for value in values: - value = float(value) - close = np.flatnonzero( - np.isclose(available, value, rtol=_MATCH_RTOL, atol=0.0) - ) - if close.size == 0: - raise KeyError( - f"{what} {value} is not on this database's grid; " - f"available: {np.array2string(available, threshold=20)}" - ) - indices.append(int(grid.index[close[0]])) - matched.append(float(available[close[0]])) - return indices, matched - - # ------------------------------------------------------------------ - # record filtering - # ------------------------------------------------------------------ - - def _record_filter( - self, - need: Iterable[str] = (), - events: Iterable[str] | None = None, - rels: Iterable[str] | None = None, - sites: Iterable[str] | None = None, - component: str | Iterable[str] | None = None, - max_rrup: float | None = None, - record_ids: Iterable[int] | None = None, - ) -> tuple[str, str, dict[str, pd.DataFrame]]: - """Build the FROM and WHERE clauses shared by every record query. - - Parameters - ---------- - need : iterable of str, optional - Dimension aliases the caller's SELECT needs, from ``e``, ``rl``, ``s``, ``se``. - events : iterable of str, optional - Keep only these ``event_id`` values. - rels : iterable of str, optional - Keep only these ``rel_id`` values. - sites : iterable of str, optional - Keep only these ``site_id`` values. - component : str or iterable of str, optional - Keep only these components. - max_rrup : float, optional - Keep only records whose ``site_event.rrup`` is at most this. - record_ids : iterable of int, optional - Keep only these ``record_id`` values. - - Returns - ------- - tuple of (str, str, dict of str to pandas.DataFrame) - FROM clause, WHERE clause and the frames to register while querying. - """ - dims, joins, wheres, frames = set(need), [], [], {} - - for alias, column, table, values, mapping in ( - ("e", "event_id", "events", events, self.event_ids), - ("rl", "rel_id", "realisations", rels, self.rel_ids), - ("s", "site_id", "sites", sites, self.site_ids), - ): - if values is None: - continue - dims.add(alias) - values = [str(value) for value in values] - self._resolve(values, mapping, column) - view = f"_f_{table}" - frames[view] = pd.DataFrame({"_key": np.asarray(values, dtype=object)}) - joins.append(f"JOIN {view} ON {view}._key = {alias}.{column}") - - if record_ids is not None: - frames["_f_records"] = pd.DataFrame( - {"_key": np.asarray(list(record_ids), dtype=np.int64)} - ) - joins.append("JOIN _f_records ON _f_records._key = r.record_id") - - if max_rrup is not None: - dims.add("se") - wheres.append(f"se.rrup <= {float(max_rrup)}") - - if component is not None: - wanted = [component] if isinstance(component, str) else list(component) - unknown = set(wanted) - set(self.components) - if unknown: - raise ValueError( - f"components {sorted(unknown)} are not in this database; " - f"it holds {self.components}" - ) - wheres.append( - "r.component IN (" + ", ".join(f"'{c}'" for c in wanted) + ")" - ) - - from_sql = "\n".join( - ["FROM records r"] - + [_DIM_JOINS[alias] for alias in ("e", "rl", "s", "se") if alias in dims] - + joins - ) - where_sql = ("WHERE " + " AND ".join(wheres)) if wheres else "" - return from_sql, where_sql, frames - - def _record_query( - self, - select: str, - need: Iterable[str] = (), - joins: Iterable[str] = (), - **filters, - ) -> pd.DataFrame: - """Run a query over ``records``, indexed by ``record_id``. - - Parameters - ---------- - select : str - SELECT list, without the leading ``record_id``. - need : iterable of str, optional - Dimension aliases the SELECT needs. - joins : iterable of str, optional - Extra join clauses, for example onto an IM table. - **filters - Passed to :meth:`_record_filter`. - - Returns - ------- - pandas.DataFrame - Query result indexed by ``record_id``. - """ - from_sql, where_sql, frames = self._record_filter(need=need, **filters) - from_sql = "\n".join([from_sql, *joins]) - query = f"SELECT r.record_id, {select}\n{from_sql}\n{where_sql}" - logger.debug("query: %s", query) - with self._temp_frames(frames): - return self.conn.execute(query).df().set_index("record_id") - - # ------------------------------------------------------------------ - # dimension reads - # ------------------------------------------------------------------ - - def get_events(self, expand_metadata: bool = False) -> pd.DataFrame: - """Read the ``events`` table. - - Parameters - ---------- - expand_metadata : bool, optional - Expand the JSON ``metadata`` column into columns. - - Returns - ------- - pandas.DataFrame - Events indexed by ``event_int_id``. - """ - df = self.conn.execute("SELECT * FROM events").df().set_index("event_int_id") - return _expand_metadata(df) if expand_metadata else df - - def get_realisations(self, expand_metadata: bool = False) -> pd.DataFrame: - """Read the ``realisations`` table. - - Parameters - ---------- - expand_metadata : bool, optional - Expand the JSON ``metadata`` column into columns. - - Returns - ------- - pandas.DataFrame - Realisations indexed by ``rel_int_id``. - """ - df = ( - self.conn.execute("SELECT * FROM realisations").df().set_index("rel_int_id") - ) - return _expand_metadata(df) if expand_metadata else df - - def get_sites(self, expand_metadata: bool = False) -> pd.DataFrame: - """Read the ``sites`` table. - - Parameters - ---------- - expand_metadata : bool, optional - Expand the JSON ``metadata`` column into columns. - - Returns - ------- - pandas.DataFrame - Sites indexed by ``site_int_id``. - """ - df = self.conn.execute("SELECT * FROM sites").df().set_index("site_int_id") - return _expand_metadata(df) if expand_metadata else df - - def get_site_event( - self, - sites: Iterable[str] | None = None, - events: Iterable[str] | None = None, - max_rrup: float | None = None, - expand_metadata: bool = False, - ) -> pd.DataFrame: - """Read the ``site_event`` table. - - Parameters - ---------- - sites : iterable of str, optional - Keep only these ``site_id`` values. - events : iterable of str, optional - Keep only these ``event_id`` values. - max_rrup : float, optional - Keep only pairs whose ``rrup`` is at most this. - expand_metadata : bool, optional - Expand the JSON ``metadata`` column into columns. - - Returns - ------- - pandas.DataFrame - Site-event pairs, with ``site_id`` and ``event_id`` added. - """ - wheres, frames = [], {} - if sites is not None: - sites = [str(value) for value in sites] - self._resolve(sites, self.site_ids, "site_id") - frames["_f_sites"] = pd.DataFrame({"_key": np.asarray(sites, dtype=object)}) - wheres.append("s.site_id IN (SELECT _key FROM _f_sites)") - if events is not None: - events = [str(value) for value in events] - self._resolve(events, self.event_ids, "event_id") - frames["_f_events"] = pd.DataFrame( - {"_key": np.asarray(events, dtype=object)} - ) - wheres.append("e.event_id IN (SELECT _key FROM _f_events)") - if max_rrup is not None: - wheres.append(f"se.rrup <= {float(max_rrup)}") - where_sql = ("WHERE " + " AND ".join(wheres)) if wheres else "" - - query = f""" - SELECT se.*, s.site_id, e.event_id - FROM site_event se - JOIN sites s ON s.site_int_id = se.site_int_id - JOIN events e ON e.event_int_id = se.event_int_id - {where_sql} - """ - with self._temp_frames(frames): - df = self.conn.execute(query).df() - return _expand_metadata(df) if expand_metadata else df - - # ------------------------------------------------------------------ - # record and IM reads - # ------------------------------------------------------------------ - - def get_records(self, **filters) -> pd.DataFrame: - """Read record identities. - - Parameters - ---------- - **filters - Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, - ``record_ids``. - - Returns - ------- - pandas.DataFrame - ``event_id``, ``rel_id``, ``site_id`` and ``component``, indexed by - ``record_id``. - """ - return self._record_query( - "e.event_id, rl.rel_id, s.site_id, r.component", - need=("e", "rl", "s"), - **filters, - ) - - def get_psa( - self, periods: Iterable[float] | None = None, **filters - ) -> pd.DataFrame: - """Read pSA values. - - Parameters - ---------- - periods : iterable of float, optional - Periods in seconds. ``None`` reads the whole grid. - **filters - Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, - ``record_ids``. - - Returns - ------- - pandas.DataFrame - pSA in g, indexed by ``record_id``, one column per period, labelled - with the period in seconds. Records with no ``psa_ims`` row are absent. - """ - return self._spectral_read("psa_ims", "pSA", self.periods, periods, **filters) - - def get_fas( - self, frequencies: Iterable[float] | None = None, **filters - ) -> pd.DataFrame: - """Read FAS values. - - Parameters - ---------- - frequencies : iterable of float, optional - Frequencies in Hz. ``None`` reads the whole grid. - **filters - Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, - ``record_ids``. - - Returns - ------- - pandas.DataFrame - FAS in g.s, indexed by ``record_id``, one column per frequency, - labelled with the frequency in Hz. Records with no ``fas_ims`` row - are absent. - """ - return self._spectral_read( - "fas_ims", "FAS", self.frequencies, frequencies, **filters - ) - - def _spectral_read( - self, - table: str, - column: str, - grid: pd.Series, - values: Iterable[float] | None, - **filters, - ) -> pd.DataFrame: - """Project selected array elements out of a spectral IM table. - - Parameters - ---------- - table : str - IM table to read. - column : str - Array column in that table. - grid : pandas.Series - The database's grid for that column. - values : iterable of float, optional - Grid values to read. ``None`` reads all of them. - **filters - Passed to :meth:`_record_filter`. - - Returns - ------- - pandas.DataFrame - One column per requested grid value, indexed by ``record_id``. - """ - indices, matched = self._grid_indices(values, grid, column) - select = ", ".join(f'im.{column}[{i}] AS "c{n}"' for n, i in enumerate(indices)) - df = self._record_query( - select, joins=[f"JOIN {table} im USING (record_id)"], **filters - ) - df.columns = pd.Index(matched, name=column) - return df - - def get_scalars(self, ims: Iterable[str] | None = None, **filters) -> pd.DataFrame: - """Read scalar IM values. - - Parameters - ---------- - ims : iterable of str, optional - Scalar IM names. ``None`` reads all of them. - **filters - Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, - ``record_ids``. - - Returns - ------- - pandas.DataFrame - One column per requested IM, indexed by ``record_id``. Records with - no ``scalars_ims`` row are absent. ``CAV``, ``AI``, ``Ds575`` and - ``Ds595`` are NULL for ``rotd*`` components. - """ - wanted = list(schema.SCALAR_IMS) if ims is None else list(ims) - unknown = set(wanted) - set(schema.SCALAR_IMS) - if unknown: - raise ValueError( - f"unknown scalar IMs {sorted(unknown)}; known: {list(schema.SCALAR_IMS)}" - ) - select = ", ".join(f"im.{im}" for im in wanted) - return self._record_query( - select, joins=["JOIN scalars_ims im USING (record_id)"], **filters - ) - - def get_im_df(self, ims: Iterable[str], **filters) -> pd.DataFrame: - """Read named intensity measures into one DataFrame. - - Names are ``PGA``, ``PGV``, ``PGD``, ``CAV``, ``AI``, ``Ds575``, - ``Ds595`` for scalars, ``pSA_`` for response spectra and - ``FAS_`` for Fourier spectra. The point may be written as - ``.`` or ``p``, so ``pSA_0.1`` and ``pSA_0p1`` are the same. - - Parameters - ---------- - ims : iterable of str - IM names to read. - **filters - Any of ``events``, ``rels``, ``sites``, ``component``, ``max_rrup``, - ``record_ids``. - - Returns - ------- - pandas.DataFrame - One column per requested name, in the order requested, indexed by - ``record_id``. IM tables are joined outer, so a record missing from - one of them reads as NaN in its columns. - """ - ims = list(ims) - scalars, psa, fas = [], [], [] - for name in ims: - if name in schema.SCALAR_IMS: - scalars.append(name) - elif name.startswith("pSA_"): - psa.append((name, _parse_number(name[4:]))) - elif name.startswith("FAS_"): - fas.append((name, _parse_number(name[4:]))) - else: - raise ValueError( - f"cannot parse IM name {name!r}; expected one of " - f"{list(schema.SCALAR_IMS)}, pSA_ or FAS_" - ) - - parts = [] - if scalars: - parts.append(self.get_scalars(ims=scalars, **filters)) - for getter, requested in ((self.get_psa, psa), (self.get_fas, fas)): - if not requested: - continue - part = getter([value for _, value in requested], **filters) - part.columns = pd.Index([name for name, _ in requested]) - parts.append(part) - - df = parts[0] if len(parts) == 1 else pd.concat(parts, axis=1, join="outer") - return df[ims] - - # ------------------------------------------------------------------ - # creation - # ------------------------------------------------------------------ + # ---- create ------------------------------------------------------- @classmethod def create( cls, - db_path: Path | str, - periods: Iterable[float], - frequencies: Iterable[float] = (), - components: Iterable[str] = schema.COMPONENTS, - db_meta: Mapping[str, object] | None = None, - overwrite: bool = False, - **kwargs, - ) -> Self: - """Create an empty database and return it open for writing. - - Parameters - ---------- - db_path : Path or str - Path of the file to create. - periods : iterable of float - pSA period grid in seconds. Sorted ascending on write. - frequencies : iterable of float, optional - FAS frequency grid in Hz. Sorted ascending on write. - components : iterable of str, optional + path: Path, + periods: list[float], + frequencies: list[float] | None = None, + components: tuple[str, ...] = schema.COMPONENTS, + db_meta: dict[str, str] | None = None, + ) -> "IMDB": + """Create a new, empty IMDB and return it open for writing. + + Parameters + ---------- + path : Path + Path to the database file to create. Must not already exist. + periods : list of float + Response spectral periods, in seconds, that `pSA` arrays are indexed by. + frequencies : list of float, optional + Frequencies, in Hz, that `FAS` arrays are indexed by. + components : tuple of str Components this database will hold. - db_meta : mapping, optional - Values merged over the defaults, for example ``dataset_description``, - ``source`` and the ``*_metadata_keys`` declarations. - overwrite : bool, optional - Replace an existing file. - **kwargs - Passed to the constructor. + db_meta : dict of str to str, optional + Extra `db_meta` entries, merged over the defaults. Returns ------- IMDB - The new database, open for writing. + The newly created database, open for writing. """ - db_path = Path(db_path) - if db_path.exists(): - if not overwrite: - raise FileExistsError(db_path) - db_path.unlink() + frequencies = frequencies or [] + db = cls(path, read_only=False).open() + con = db.con + for statement in schema.DDL.strip().split(";"): + if statement.strip(): + con.raw_sql(statement) - unknown = set(components) - set(schema.COMPONENTS) - if unknown: - raise ValueError( - f"unknown components {sorted(unknown)}; known: {list(schema.COMPONENTS)}" - ) - - db = cls(db_path, read_only=False, **kwargs).open() - db.conn.execute(schema.DDL) - - periods = np.unique(np.asarray(list(periods), dtype=float)) - frequencies = np.unique(np.asarray(list(frequencies), dtype=float)) - db._insert( + con.insert( "periods", pd.DataFrame( - {"period_index": np.arange(1, len(periods) + 1), "period": periods} + {"period_index": range(1, len(periods) + 1), "period": periods} ), ) - db._insert( + con.insert( "frequencies", pd.DataFrame( - { - "freq_index": np.arange(1, len(frequencies) + 1), - "frequency": frequencies, - } + {"freq_index": range(1, len(frequencies) + 1), "frequency": frequencies} ), ) - db._insert( + con.insert( "im_units", pd.DataFrame( - {"im": list(schema.IM_UNITS), "unit": list(schema.IM_UNITS.values())} + {"im": schema.IM_UNITS.keys(), "unit": schema.IM_UNITS.values()} ), ) - db._insert( + con.insert( "notes", - pd.DataFrame( - {"topic": list(schema.NOTES), "note": list(schema.NOTES.values())} - ), + pd.DataFrame({"topic": schema.NOTES.keys(), "note": schema.NOTES.values()}), ) - meta: dict[str, str] = { + meta = { "schema_version": schema.SCHEMA_VERSION, - "dataset_id": db_path.stem, - "dataset_description": "", "components": ",".join(components), "n_periods": str(len(periods)), "n_frequencies": str(len(frequencies)), - "sort_order": "event", - "created_at": datetime.now(UTC).isoformat(timespec="seconds"), - "creator": getpass.getuser(), - "source": "", - "imdb_version": _imdb_version(), - **{key: "" for key in schema.METADATA_TABLES.values()}, + "created_at": datetime.datetime.now(datetime.UTC).isoformat(), + "imdb_version": version("imdb"), + **(db_meta or {}), } - meta.update({key: str(value) for key, value in (db_meta or {}).items()}) - db.set_db_meta(meta) - return db - - def set_db_meta(self, values: Mapping[str, object]) -> None: - """Insert or replace ``db_meta`` entries. - - Parameters - ---------- - values : mapping - Keys and values to write. Values are stored as strings. - """ - self._require_write() - frame = pd.DataFrame( - { - "key": list(values), - "value": [str(value) for value in values.values()], - } + con.insert( + "db_meta", pd.DataFrame({"key": meta.keys(), "value": meta.values()}) ) - with self._temp_frames({"_meta": frame}): - self.conn.execute( - "INSERT OR REPLACE INTO db_meta SELECT key, value FROM _meta" - ) - self._invalidate() - - def _insert(self, table: str, df: pd.DataFrame) -> None: - """Insert a DataFrame whose columns match the table's, in order. - - Parameters - ---------- - table : str - Target table. - df : pandas.DataFrame - Rows to insert. - """ - if df.empty: - return - with self._temp_frames({"_rows": df}): - self.conn.execute(f"INSERT INTO {table} SELECT * FROM _rows") - - # ------------------------------------------------------------------ - # writing - # ------------------------------------------------------------------ - - def _prepare( - self, df: pd.DataFrame, table: str, required: Sequence[str] - ) -> pd.DataFrame: - """Validate and order a DataFrame against a table's columns. + return db - Parameters - ---------- - df : pandas.DataFrame - Rows to write. - table : str - Target table. - required : sequence of str - Columns that must be present. + # ---- write helpers -------------------------------------------------- - Returns - ------- - pandas.DataFrame - The rows, with missing columns added as NULL and ordered to match - the table. - """ - columns = self._table_columns(table) - missing = [name for name in required if name not in df.columns] - if missing: - raise ValueError(f"{table} rows are missing columns {missing}") - unknown = [name for name in df.columns if name not in columns] - if unknown: - raise ValueError( - f"columns {unknown} are not in {table}; extra fields belong in metadata" - ) - df = df.copy() - if "metadata" in columns: - df["metadata"] = self._pack_metadata(df.get("metadata"), table) - for name in columns: - if name not in df.columns: - df[name] = None - return df[columns] - - def _pack_metadata(self, values: pd.Series | None, table: str) -> pd.Series | None: - """Serialise a metadata column to JSON and reconcile its declared keys. + def _next_ids(self, table: str, int_col: str, n: int) -> np.ndarray: + """Return `n` new contiguous integer ids for `table`, starting after the current max.""" + current = self.con.table(table)[int_col].max().to_pandas() + start = 0 if pd.isna(current) else int(current) + 1 # ty: ignore[invalid-argument-type] + return np.arange(start, start + n) - The first write of a table's metadata declares the permitted keys in - ``db_meta``; later writes are validated against that declaration. + def _id_map(self, table: str, id_col: str, int_col: str) -> pd.Series: + """Return a `pd.Series` mapping string id to int id for `table`.""" + df = self.con.table(table).select(id_col, int_col).to_pandas() + return df.set_index(id_col)[int_col] - Parameters - ---------- - values : pandas.Series or None - Metadata as dicts or JSON strings. - table : str - Table being written, used to find the ``db_meta`` key. - - Returns - ------- - pandas.Series or None - JSON strings, or ``None`` if there was nothing to pack. - """ - if values is None: - return None - packed, seen = [], set() - for value in values: - if value is None or (isinstance(value, float) and np.isnan(value)): - packed.append(None) - continue - if isinstance(value, str): - value = json.loads(value) - if not isinstance(value, dict): - raise TypeError( - f"{table}.metadata must hold dicts or JSON objects, got {type(value)}" - ) - seen.update(value) - packed.append(json.dumps(value, sort_keys=True, default=str)) - - meta_key = schema.METADATA_TABLES[table] - declared = {k for k in self.db_meta.get(meta_key, "").split(",") if k} - if declared: - undeclared = seen - declared - if undeclared: - raise ValueError( - f"{table}.metadata keys {sorted(undeclared)} are not declared in " - f"db_meta.{meta_key} ({sorted(declared)})" - ) - elif seen: - self.set_db_meta({meta_key: ",".join(sorted(seen))}) - return pd.Series(packed, index=values.index, dtype=object) - - def _next_int_ids(self, table: str, column: str, count: int) -> np.ndarray: - """Allocate contiguous integer surrogates for a dimension table. - - Parameters - ---------- - table : str - Table to extend. - column : str - Surrogate key column. - count : int - How many ids to allocate. - - Returns - ------- - numpy.ndarray - The new ids. - """ - current = self._scalar(f"SELECT max({column}) FROM {table}") - start = 1 if current is None else int(current) + 1 - return np.arange(start, start + count, dtype=np.int64) + # ---- write ---------------------------------------------------------- def add_events(self, df: pd.DataFrame) -> None: - """Insert rows into ``events``. + """Insert new events. Parameters ---------- - df : pandas.DataFrame - Requires ``event_id``. Other columns must be ``events`` columns; - extra fields go in a ``metadata`` dict column. ``event_int_id`` is - assigned here and must not be supplied. + df : pd.DataFrame + Must have an `event_id` column; other columns match `events`. """ - self._require_write() - df = self._prepare(df, "events", ["event_id"]).drop(columns="event_int_id") - df["event_id"] = df["event_id"].astype(str) - self._reject_existing(df["event_id"], self.event_ids, "event_id") - df.insert( - 0, "event_int_id", self._next_int_ids("events", "event_int_id", len(df)) - ) - self._insert("events", df) - self._invalidate() - logger.info("inserted %d events", len(df)) + df = df.copy() + df["event_int_id"] = self._next_ids("events", "event_int_id", len(df)) + self.con.insert("events", df) def add_realisations(self, df: pd.DataFrame) -> None: - """Insert rows into ``realisations``. + """Insert new realisations. Parameters ---------- - df : pandas.DataFrame - Requires ``rel_id`` and ``event_id``. ``rel_int_id`` is assigned - here and must not be supplied. + df : pd.DataFrame + Must have `rel_id` and `event_id` columns; other columns match + `realisations`. """ - self._require_write() df = df.copy() - if "event_id" not in df.columns: - raise ValueError("realisation rows are missing column ['event_id']") - event_int_id = self._resolve(df.pop("event_id"), self.event_ids, "event_id") - df["event_int_id"] = event_int_id - df = self._prepare(df, "realisations", ["rel_id"]).drop(columns="rel_int_id") - df["rel_id"] = df["rel_id"].astype(str) - self._reject_existing(df["rel_id"], self.rel_ids, "rel_id") - df.insert( - 0, "rel_int_id", self._next_int_ids("realisations", "rel_int_id", len(df)) - ) - self._insert("realisations", df) - self._invalidate() - logger.info("inserted %d realisations", len(df)) + event_int_id = self._id_map("events", "event_id", "event_int_id") + df["event_int_id"] = event_int_id.loc[df["event_id"]].to_numpy() + df = df.drop(columns="event_id") + df["rel_int_id"] = self._next_ids("realisations", "rel_int_id", len(df)) + self.con.insert("realisations", df) def add_sites(self, df: pd.DataFrame) -> None: - """Insert rows into ``sites``. + """Insert new sites. Parameters ---------- - df : pandas.DataFrame - Requires ``site_id``, ``lat`` and ``lon``. ``site_int_id`` is - assigned here and must not be supplied. + df : pd.DataFrame + Must have `site_id`, `lat` and `lon` columns; other columns match `sites`. """ - self._require_write() - df = self._prepare(df, "sites", ["site_id", "lat", "lon"]).drop( - columns="site_int_id" - ) - df["site_id"] = df["site_id"].astype(str) - self._reject_existing(df["site_id"], self.site_ids, "site_id") - df.insert(0, "site_int_id", self._next_int_ids("sites", "site_int_id", len(df))) - self._insert("sites", df) - self._invalidate() - logger.info("inserted %d sites", len(df)) + df = df.copy() + df["site_int_id"] = self._next_ids("sites", "site_int_id", len(df)) + self.con.insert("sites", df) def add_site_event(self, df: pd.DataFrame) -> None: - """Insert rows into ``site_event``. + """Insert new site-event distance rows. Parameters ---------- - df : pandas.DataFrame - Requires ``site_id`` and ``event_id``, which are resolved to their - integer surrogates here. + df : pd.DataFrame + Must have `site_id` and `event_id` columns; other columns match + `site_event`. """ - self._require_write() df = df.copy() - for column in ("site_id", "event_id"): - if column not in df.columns: - raise ValueError(f"site_event rows are missing column ['{column}']") - df["site_int_id"] = self._resolve(df.pop("site_id"), self.site_ids, "site_id") - df["event_int_id"] = self._resolve( - df.pop("event_id"), self.event_ids, "event_id" - ) - df = self._prepare(df, "site_event", ["site_int_id", "event_int_id"]) - self._insert("site_event", df) - logger.info("inserted %d site-event pairs", len(df)) - + site_int_id = self._id_map("sites", "site_id", "site_int_id") + event_int_id = self._id_map("events", "event_id", "event_int_id") + df["site_int_id"] = site_int_id.loc[df["site_id"]].to_numpy() + df["event_int_id"] = event_int_id.loc[df["event_id"]].to_numpy() + df = df.drop(columns=["site_id", "event_id"]) + self.con.insert("site_event", df) + + ## TODO: CHange this to take a record_df (containing rel_id, site_id, component) + scalar IM columns, + # plus optional pSA and FAS numpy arrays. Update logic accordingly def add_records(self, df: pd.DataFrame) -> np.ndarray: - """Insert records and their IM values. + """Insert new records, and whichever IM tables the input covers. - Writes ``records`` plus whichever of ``psa_ims``, ``fas_ims`` and - ``scalars_ims`` the input covers. ``CAV``, ``AI``, ``Ds575`` and - ``Ds595`` are set to NULL on ``rotd*`` rows. + A row with a missing (`None`) `pSA` or `FAS` array is not written to that IM + table at all, matching the schema's row-presence-means-coverage convention. Parameters ---------- - df : pandas.DataFrame - Requires ``rel_id``, ``site_id`` and ``component``. May carry a - ``pSA`` column of arrays, a ``FAS`` column of arrays, and any of the - seven scalar IM columns. + df : pd.DataFrame + Must have `rel_id`, `site_id` and `component` columns. May also have a + `pSA` column (list of float, one per period), a `FAS` column (list of + float, one per frequency), and any of the scalar IM columns + (`schema.SCALAR_IMS`). Returns ------- - numpy.ndarray - The ``record_id`` assigned to each row, in input order. + np.ndarray + The `record_id` assigned to each input row, in input order. """ - self._require_write() df = df.copy() - for column in ("rel_id", "site_id", "component"): - if column not in df.columns: - raise ValueError(f"record rows are missing column ['{column}']") - component = df["component"].astype(str) - unknown = set(component.unique()) - set(self.components) + ## TODO: I think this should be a property? + components = set( + self.con.table("db_meta") + .filter(_.key == "components") + .to_pandas()["value"] + .iloc[0] + .split(",") + ) + unknown = set(df["component"]) - components if unknown: - raise ValueError( - f"components {sorted(unknown)} are not declared in db_meta.components " - f"({self.components})" - ) - - rel_int_id = self._resolve(df["rel_id"], self.rel_ids, "rel_id") - site_int_id = self._resolve(df["site_id"], self.site_ids, "site_id") - event_int_id = self.rel_to_event.loc[rel_int_id].to_numpy(dtype=np.int64) + raise ValueError(f"components not in this database: {sorted(unknown)}") - known = {"rel_id", "site_id", "component", "pSA", "FAS", *schema.SCALAR_IMS} - unknown_columns = [name for name in df.columns if name not in known] - if unknown_columns: - raise ValueError( - f"columns {unknown_columns} are not record or IM columns; expected " - f"rel_id, site_id, component, pSA, FAS or one of {list(schema.SCALAR_IMS)}" - ) + rel_int_id = self._id_map("realisations", "rel_id", "rel_int_id") + rel_event_int_id = self._id_map("realisations", "rel_int_id", "event_int_id") + site_int_id = self._id_map("sites", "site_id", "site_int_id") + df["rel_int_id"] = rel_int_id.loc[df["rel_id"]].to_numpy() + df["site_int_id"] = site_int_id.loc[df["site_id"]].to_numpy() + df["event_int_id"] = rel_event_int_id.loc[df["rel_int_id"]].to_numpy() record_id = ( - self.conn.execute( - "SELECT nextval('record_id_seq') AS record_id FROM range(?)", [len(df)] - ) - .df()["record_id"] - .to_numpy(dtype=np.int64) + self.con.raw_sql(f"SELECT nextval('record_id_seq') FROM range({len(df)})") + .df()["nextval('record_id_seq')"] + .to_numpy() ) + df["record_id"] = record_id - with self._transaction(): - self._write_record_tables( - df, record_id, event_int_id, rel_int_id, site_int_id, component + self.con.raw_sql("BEGIN TRANSACTION") + try: + self.con.insert( + "records", + df[ + [ + "record_id", + "event_int_id", + "rel_int_id", + "site_int_id", + "component", + ] + ], ) - + if "pSA" in df: + # Why not just + mask = df["pSA"].apply(lambda x: x is not None) + self.con.insert("psa_ims", df.loc[mask, ["record_id", "pSA"]]) + if "FAS" in df: + mask = df["FAS"].apply(lambda x: x is not None) + self.con.insert("fas_ims", df.loc[mask, ["record_id", "FAS"]]) + scalar_cols = [c for c in schema.SCALAR_IMS if c in df] + if scalar_cols: + scalars = df[["record_id", *scalar_cols]].copy() + rotd = df["component"].str.startswith("rotd") + for col in schema.ROTD_UNDEFINED & set(scalar_cols): + scalars.loc[rotd, col] = None + self.con.insert("scalars_ims", scalars) + except Exception: + self.con.raw_sql("ROLLBACK") + raise + self.con.raw_sql("COMMIT") logger.info("inserted %d records", len(df)) return record_id - def _write_record_tables( - self, - df: pd.DataFrame, - record_id: np.ndarray, - event_int_id: np.ndarray, - rel_int_id: np.ndarray, - site_int_id: np.ndarray, - component: pd.Series, - ) -> None: - """Insert one batch into ``records`` and the IM tables it covers. - - Parameters - ---------- - df : pandas.DataFrame - The validated input rows. - record_id : numpy.ndarray - Allocated record ids. - event_int_id : numpy.ndarray - Event surrogate of each row. - rel_int_id : numpy.ndarray - Realisation surrogate of each row. - site_int_id : numpy.ndarray - Site surrogate of each row. - component : pandas.Series - Component of each row. - """ - self._insert( - "records", - pd.DataFrame( - { - "record_id": record_id, - "event_int_id": event_int_id, - "rel_int_id": rel_int_id, - "site_int_id": site_int_id, - "component": component.to_numpy(dtype=object), - } - ), - ) - - for column, table, grid in ( - ("pSA", "psa_ims", self.periods), - ("FAS", "fas_ims", self.frequencies), - ): - if column not in df.columns: - continue - # a row with no array simply gets no row in this IM table, which is how - # the schema expresses per-record IM coverage - present = df[column].notna().to_numpy() - if not present.any(): - continue - arrays = [ - np.asarray(value, dtype=np.float32) for value in df.loc[present, column] - ] - bad = {array.size for array in arrays} - {len(grid)} - if bad: - raise ValueError( - f"{column} arrays have lengths {sorted(bad)} but this database's " - f"grid has {len(grid)} entries" - ) - self._insert( - table, - pd.DataFrame( - { - "record_id": record_id[present], - column: pd.Series(arrays, dtype=object), - } - ), - ) - - present = [im for im in schema.SCALAR_IMS if im in df.columns] - if present: - scalars = pd.DataFrame({"record_id": record_id}) - is_rotd = component.str.startswith("rotd").to_numpy() - for im in schema.SCALAR_IMS: - values = ( - pd.to_numeric(df[im], errors="raise").astype("float32") - if im in present - else pd.Series(np.nan, index=df.index, dtype="float32") - ) - values = values.to_numpy(dtype=np.float32, copy=True) - if im in schema.ROTD_UNDEFINED: - values[is_rotd] = np.nan - scalars[im] = values - self._insert("scalars_ims", scalars) - - def _reject_existing(self, ids: pd.Series, mapping: pd.Series, what: str) -> None: - """Raise if any of these string ids is already in the database. - - Parameters - ---------- - ids : pandas.Series - String ids about to be written. - mapping : pandas.Series - Existing ids, as the index. - what : str - Name used in the error message. - """ - duplicated = ids[ids.duplicated()].unique() - if len(duplicated): - raise ValueError( - f"duplicate {what} in the input: {sorted(duplicated)[:10]}" - ) - existing = ids[ids.isin(mapping.index)].unique() - if len(existing): - raise ValueError( - f"{what} already in the database: {sorted(existing)[:10]}; " - "use delete_event() to re-ingest" - ) - def delete_event(self, event_id: str) -> None: - """Delete an event and everything derived from it. - - Removes the event's rows from ``psa_ims``, ``fas_ims``, ``scalars_ims``, - ``records``, ``site_event``, ``realisations`` and ``events``, so a - re-ingest is a delete followed by the same sequence of ``add_*`` calls. - Sites are shared across events and are never deleted. + """Delete an event and everything derived from it, for a clean re-ingest. Parameters ---------- event_id : str The event to delete. """ - self._require_write() - event_int_id = int(self._resolve([event_id], self.event_ids, "event_id")[0]) - for table in ("psa_ims", "fas_ims", "scalars_ims"): - self.conn.execute( - f"DELETE FROM {table} WHERE record_id IN " - "(SELECT record_id FROM records WHERE event_int_id = ?)", - [event_int_id], - ) - for table in ("records", "site_event", "realisations", "events"): - self.conn.execute( - f"DELETE FROM {table} WHERE event_int_id = ?", - [event_int_id], - ) - self._invalidate() - logger.info("deleted event %s", event_id) - - # ------------------------------------------------------------------ - # validation - # ------------------------------------------------------------------ - + event_int_id_subquery = "(SELECT event_int_id FROM events WHERE event_id = ?)" + record_subquery = f"(SELECT record_id FROM records WHERE event_int_id = {event_int_id_subquery})" + statements = [ + f"DELETE FROM psa_ims WHERE record_id IN {record_subquery}", + f"DELETE FROM fas_ims WHERE record_id IN {record_subquery}", + f"DELETE FROM scalars_ims WHERE record_id IN {record_subquery}", + f"DELETE FROM records WHERE event_int_id = {event_int_id_subquery}", + f"DELETE FROM site_event WHERE event_int_id = {event_int_id_subquery}", + f"DELETE FROM realisations WHERE event_int_id = {event_int_id_subquery}", + ] + for statement in statements: + self.con.raw_sql(statement, parameters=[event_id]) + + ## TODO: Simplify, it should just check for basic stuff, e.g. that all integer ids are unique, fks are valid. Keep it simple. def validate(self) -> list[str]: - """Check the invariants the large tables do not enforce as constraints. - - Works read-only. Returns problems rather than raising, so an ingest - script can report all of them at once. + """Check the invariants the schema itself cannot enforce. Returns ------- list of str - One line per problem found. Empty when the database is consistent. + One entry per problem found; empty if the database is consistent. """ problems = [] - - def count(query: str) -> int: - return int(self._scalar(query)) - - for column, table, key in ( - ("rel_int_id", "realisations", "rel_int_id"), - ("site_int_id", "sites", "site_int_id"), - ("event_int_id", "events", "event_int_id"), - ): - n = count( - f"SELECT count(*) FROM records r " - f"LEFT JOIN {table} d ON d.{key} = r.{column} WHERE d.{key} IS NULL" - ) - if n: - problems.append(f"records: {n} rows with an orphan {column}") - - for column, table, key in ( - ("site_int_id", "sites", "site_int_id"), - ("event_int_id", "events", "event_int_id"), - ): - n = count( - f"SELECT count(*) FROM site_event se " - f"LEFT JOIN {table} d ON d.{key} = se.{column} WHERE d.{key} IS NULL" - ) + records = self.con.table("records") + n_periods = len(self.con.table("periods").to_pandas()) + n_frequencies = len(self.con.table("frequencies").to_pandas()) + + orphans = { + "rel_int_id": self.con.table("realisations").rel_int_id, + "site_int_id": self.con.table("sites").site_int_id, + "event_int_id": self.con.table("events").event_int_id, + } + for col, valid in orphans.items(): + n = records.filter(~records[col].isin(valid)).count().to_pandas() if n: - problems.append(f"site_event: {n} rows with an orphan {column}") + problems.append(f"records: {n} rows with an unknown {col}") - n = count( - "SELECT count(*) FROM records r JOIN realisations rl USING (rel_int_id) " - "WHERE r.event_int_id != rl.event_int_id" + n = ( + self.con.table("psa_ims") + .filter(_.pSA.length() != n_periods) + .count() + .to_pandas() ) if n: - problems.append( - f"records: {n} rows whose event_int_id disagrees with their realisation" - ) + problems.append(f"psa_ims: {n} rows with pSA length != {n_periods}") - for table in schema.IM_TABLES: - n = count( - f"SELECT count(*) FROM {table} im " - "LEFT JOIN records r USING (record_id) WHERE r.record_id IS NULL" - ) - if n: - problems.append(f"{table}: {n} rows with no matching record") - - for table, column, grid in ( - ("psa_ims", "pSA", "periods"), - ("fas_ims", "FAS", "frequencies"), - ): - n = count( - f"SELECT count(*) FROM {table} " - f"WHERE len({column}) != (SELECT count(*) FROM {grid})" - ) - if n: - problems.append(f"{table}: {n} rows whose {column} length is wrong") - - n = count( - "SELECT count(*) FROM (SELECT 1 FROM records " - "GROUP BY rel_int_id, site_int_id, component HAVING count(*) > 1)" + n = ( + self.con.table("fas_ims") + .filter(_.FAS.length() != n_frequencies) + .count() + .to_pandas() ) if n: - problems.append( - f"records: {n} duplicated (rel_int_id, site_int_id, component) keys" - ) + problems.append(f"fas_ims: {n} rows with FAS length != {n_frequencies}") + + return problems + + # ---- read ------------------------------------------------------------- + + def get_events(self) -> pd.DataFrame: + """Return all events, indexed by `event_id`.""" + return self.con.table("events").to_pandas().set_index("event_id") + + def get_realisations(self) -> pd.DataFrame: + """Return all realisations, indexed by `rel_id`.""" + return self.con.table("realisations").to_pandas().set_index("rel_id") - n = count( - "SELECT count(*) FROM (SELECT 1 FROM site_event " - "GROUP BY site_int_id, event_int_id HAVING count(*) > 1)" + def get_sites(self) -> pd.DataFrame: + """Return all sites, indexed by `site_id`.""" + return self.con.table("sites").to_pandas().set_index("site_id") + + def get_site_event( + self, + event_ids: list[str] | None = None, + site_ids: list[str] | None = None, + max_rrup: float | None = None, + ) -> pd.DataFrame: + """Return site-event distance rows. + + Parameters + ---------- + event_ids : list of str, optional + Only these events. + site_ids : list of str, optional + Only these sites. + max_rrup : float, optional + Only rows with `rrup` at most this value. + + Returns + ------- + pd.DataFrame + One row per (site, event), with `site_id` and `event_id` columns. + """ + sites = self.con.table("sites").select("site_int_id", "site_id") + events = self.con.table("events").select("event_int_id", "event_id") + t = ( + self.con.table("site_event") + .join(sites, "site_int_id") + .join(events, "event_int_id") ) - if n: - problems.append( - f"site_event: {n} duplicated (site_int_id, event_int_id) keys" - ) + if event_ids is not None: + t = t.filter(t.event_id.isin(event_ids)) + if site_ids is not None: + t = t.filter(t.site_id.isin(site_ids)) + if max_rrup is not None: + t = t.filter(t.rrup <= max_rrup) + return t.drop("site_int_id", "event_int_id").to_pandas() - for table in schema.IM_TABLES: - n = count( - f"SELECT count(*) FROM (SELECT 1 FROM {table} " - "GROUP BY record_id HAVING count(*) > 1)" - ) - if n: - problems.append(f"{table}: {n} record_ids with more than one row") - - declared = set(self.components) - found = { - row[0] - for row in self.conn.execute( - "SELECT DISTINCT component FROM records" - ).fetchall() - } - if found - declared: - problems.append( - f"records: components {sorted(found - declared)} are not in " - f"db_meta.components ({sorted(declared)})" - ) + def get_records( + self, + event_ids: list[str] | None = None, + rel_ids: list[str] | None = None, + site_ids: list[str] | None = None, + component: str | None = None, + record_ids: list[int] | None = None, + ) -> pd.DataFrame: + """Return record identity rows. - for table, meta_key in schema.METADATA_TABLES.items(): - allowed = {k for k in self.db_meta.get(meta_key, "").split(",") if k} - keys = { - row[0] - for row in self.conn.execute( - f"SELECT DISTINCT unnest(json_keys(metadata)) FROM {table} " - "WHERE metadata IS NOT NULL" - ).fetchall() - } - if keys - allowed: - problems.append( - f"{table}.metadata: keys {sorted(keys - allowed)} are not declared " - f"in db_meta.{meta_key}" - ) - - for key, table in (("n_periods", "periods"), ("n_frequencies", "frequencies")): - declared_n = self.db_meta.get(key) - actual = count(f"SELECT count(*) FROM {table}") - if declared_n is not None and int(declared_n) != actual: - problems.append( - f"db_meta.{key} is {declared_n} but {table} has {actual} rows" - ) + Parameters + ---------- + event_ids : list of str, optional + Only records for these events. + rel_ids : list of str, optional + Only records for these realisations. + site_ids : list of str, optional + Only records for these sites. + component : str, optional + Only records with this component. + record_ids : list of int, optional + Only these `record_id` values. - return problems + Returns + ------- + pd.DataFrame + Indexed by `record_id`, with `event_id`, `rel_id`, `site_id` and + `component` columns. + """ + events = self.con.table("events").select("event_int_id", "event_id") + realisations = self.con.table("realisations").select("rel_int_id", "rel_id") + sites = self.con.table("sites").select("site_int_id", "site_id") + t = ( + self.con.table("records") + .join(events, "event_int_id") + .join(realisations, "rel_int_id") + .join(sites, "site_int_id") + ) + if event_ids is not None: + t = t.filter(t.event_id.isin(event_ids)) + if rel_ids is not None: + t = t.filter(t.rel_id.isin(rel_ids)) + if site_ids is not None: + t = t.filter(t.site_id.isin(site_ids)) + if component is not None: + t = t.filter(t.component == component) + if record_ids is not None: + ## QUESTION: How performant is this for a large number of record_ids? I previously had issues with ISIN (SQL) queries being slow. + t = t.filter(t.record_id.isin(record_ids)) + return ( + t.select("record_id", "event_id", "rel_id", "site_id", "component") + .to_pandas() + .set_index("record_id") + ) - def finalise(self) -> None: - """Refresh the derived ``db_meta`` counts, validate, and checkpoint. + def get_psa( + self, periods: list[float] | None = None, **filters: Any + ) -> pd.DataFrame: + """Return response spectral acceleration. - Call once after the last write. + Parameters + ---------- + periods : list of float, optional + Periods to return, in seconds. Defaults to every period on the grid. + **filters + Passed to `get_records` to select which records to return. + + Returns + ------- + pd.DataFrame + Indexed by `record_id`, one column per requested period. """ - self._require_write() - self.set_db_meta( - { - "n_periods": len(self.periods), - "n_frequencies": len(self.frequencies), - } + grid = self.con.table("periods").to_pandas().set_index("period")["period_index"] + if periods is None: + periods = grid.index.tolist() + record_ids = self.get_records(**filters).index.tolist() + t = self.con.table("psa_ims").filter(_.record_id.isin(record_ids)) + cols = {str(p): t.pSA[int(grid.loc[p]) - 1] for p in periods} + return ( + t.select(record_id=t.record_id, **cols).to_pandas().set_index("record_id") ) - problems = self.validate() - if problems: - raise ValueError("database is inconsistent:\n " + "\n ".join(problems)) - self.conn.execute("CHECKPOINT") - logger.info("finalised %s", self.db_path) + def get_fas( + self, frequencies: list[float] | None = None, **filters: Any + ) -> pd.DataFrame: + """Return Fourier amplitude spectra. -def _expand_metadata(df: pd.DataFrame) -> pd.DataFrame: - """Expand a JSON ``metadata`` column into columns. + Parameters + ---------- + frequencies : list of float, optional + Frequencies to return, in Hz. Defaults to every frequency on the grid. + **filters + Passed to `get_records` to select which records to return. - Parameters - ---------- - df : pandas.DataFrame - Frame with a ``metadata`` column of JSON strings. + Returns + ------- + pd.DataFrame + Indexed by `record_id`, one column per requested frequency. + """ + grid = ( + self.con.table("frequencies") + .to_pandas() + .set_index("frequency")["freq_index"] + ) + if frequencies is None: + frequencies = grid.index.tolist() + record_ids = self.get_records(**filters).index.tolist() + t = self.con.table("fas_ims").filter(_.record_id.isin(record_ids)) + cols = {str(f): t.FAS[int(grid.loc[f]) - 1] for f in frequencies} + return ( + t.select(record_id=t.record_id, **cols).to_pandas().set_index("record_id") + ) - Returns - ------- - pandas.DataFrame - The frame with ``metadata`` replaced by its fields. - """ - if "metadata" not in df.columns: - return df - parsed = [ - json.loads(value) if isinstance(value, str) else {} for value in df["metadata"] - ] - expanded = pd.json_normalize(parsed) - expanded.index = df.index - return pd.concat([df.drop(columns="metadata"), expanded], axis=1) + def get_scalars(self, ims: list[str] | None = None, **filters: Any) -> pd.DataFrame: + """Return scalar intensity measures. + + Parameters + ---------- + ims : list of str, optional + Which scalar IMs to return. Defaults to all of `schema.SCALAR_IMS`. + **filters + Passed to `get_records` to select which records to return. + + Returns + ------- + pd.DataFrame + Indexed by `record_id`, one column per requested IM. + """ + ims = ims or list(schema.SCALAR_IMS) + record_ids = self.get_records(**filters).index.tolist() + t = self.con.table("scalars_ims").filter(_.record_id.isin(record_ids)) + return t.select("record_id", *ims).to_pandas().set_index("record_id") diff --git a/imdb/schema.py b/imdb/schema.py index 93c7118..3e149dd 100644 --- a/imdb/schema.py +++ b/imdb/schema.py @@ -1,141 +1,26 @@ -""" -Schema definition for the intensity measure database. +"""DDL and fixed vocabulary for the IMDB schema, version 0. -This module holds facts only: the DDL, the fixed vocabularies and the -documentation text written into every database. The DDL is the single source -of truth for table and column names; nothing else in the package hardcodes a -column list. +Facts only, no logic. `DDL` is the single source of truth for the schema; nothing +else in this library composes column lists by hand. """ SCHEMA_VERSION = "0" -COMPONENTS = ("000", "090", "ver", "geom", "rotd0", "rotd50", "rotd100") -"""Ground-motion components, following ``IM_calculation``.""" - -TECT_TYPES = ( - "ACTIVE_SHALLOW", - "VOLCANIC", - "SUBDUCTION_INTERFACE", - "SUBDUCTION_SLAB", -) -"""Tectonic types, following the ``qcore``/``workflow`` ``TectType`` vocabulary.""" - -SCALAR_IMS = ("PGA", "PGV", "PGD", "CAV", "AI", "Ds575", "Ds595") -"""Scalar intensity measures, in ``scalars_ims`` column order.""" - -ROTD_UNDEFINED = frozenset({"CAV", "AI", "Ds575", "Ds595"}) -"""Scalar IMs that are undefined for ``rotd*`` components and stored as NULL.""" - -IM_UNITS = { - "pSA": "g", - "FAS": "g.s", - "PGA": "g", - "PGV": "cm/s", - "PGD": "cm", - "CAV": "m/s", - "AI": "m/s", - "Ds575": "s", - "Ds595": "s", -} -"""Linear physical unit of each intensity measure. Log is a read-time transform.""" - -METADATA_TABLES = { - "events": "event_metadata_keys", - "realisations": "rel_metadata_keys", - "sites": "site_metadata_keys", - "site_event": "site_event_metadata_keys", -} -"""Tables carrying a JSON ``metadata`` column, and the ``db_meta`` key declaring its keys.""" - -IM_TABLES = { - "psa_ims": "pSA", - "fas_ims": "FAS", - "scalars_ims": None, -} -"""IM tables, mapped to their array column where they have one.""" - -NOTES = { - "logical keys": ( - "site_event is keyed on (site_int_id, event_int_id); records on " - "(rel_int_id, site_int_id, component); each IM table on record_id. None of " - "these are declared as constraints. They are enforced by the writer and " - "checked by IMDB.validate()." - ), - "array indexing is 1-based": ( - "periods.period_index and frequencies.freq_index are 1-based, so pSA[period_index] " - "and FAS[freq_index] need no offset. len(pSA) equals the row count of periods and " - "len(FAS) the row count of frequencies, for every row." - ), - "identity and rebuild stability": ( - "event_id, rel_id and site_id are stable. The integer surrogates event_int_id, " - "rel_int_id, site_int_id and record_id are assigned at ingest and change on " - "rebuild. Nothing outside this database may reference them." - ), - "component vocabulary": ( - "000, 090, ver, geom, rotd0, rotd50, rotd100, following IM_calculation. This " - "database holds the subset listed in db_meta.components." - ), - "rotd scalars are undefined": ( - "CAV, AI, Ds575 and Ds595 are undefined for rotd0, rotd50 and rotd100 and are " - "stored as NULL for those components. PGA, PGV and PGD are populated for every " - "component." - ), - "units": ( - "IM units are one row each in im_units, and are linear physical units. Elsewhere: " - "distances km, vs30 m/s, z1p0 and z2p5 km, depths km, angles degrees, coordinates " - "WGS84." - ), - "metadata columns are JSON": ( - "events, realisations, sites and site_event each carry a metadata VARCHAR holding " - "a JSON object. Read with json_extract_string(metadata, '$.key'). Permitted keys " - "are declared in db_meta. A predicate on a metadata field cannot use zone-map " - "pruning." - ), - "distances are event level": ( - "rrup, rjb, rx and ry are measured to the rupture surface and are shared across " - "all realisations of an event. Hypocentral and epicentral distance are not stored; " - "compute them from realisations.hypo_* and sites.lat/lon." - ), - "synthetic realisations": ( - "Every event has at least one realisation. A dataset with no realisation concept " - "gets exactly one per event, with rel_id equal to event_id." - ), - "record_id is file-local": ( - "record_id comes from a sequence and shifts on rebuild. External references must " - "cite (rel_id, site_id, component)." - ), - "physical sort order": ( - "Rows are written in the order recorded by db_meta.sort_order. Filters on the sort " - "key prune row groups; filters on anything else do not." - ), - "provenance": ( - "db_meta records who built this database, when, from what source, and with which " - "version of the imdb library." - ), -} -"""Self-documenting notes written into the ``notes`` table by :meth:`imdb.IMDB.create`.""" - DDL = """ --- ---------- documentation ---------- - CREATE TABLE db_meta (key VARCHAR PRIMARY KEY, value VARCHAR NOT NULL); CREATE TABLE notes (topic VARCHAR PRIMARY KEY, note VARCHAR NOT NULL); CREATE TABLE im_units (im VARCHAR PRIMARY KEY, unit VARCHAR NOT NULL); --- ---------- IM vocabulary (1-based, matching DuckDB list indexing) ---------- - CREATE TABLE periods ( period_index INTEGER PRIMARY KEY, - period DOUBLE NOT NULL UNIQUE -- seconds + period DOUBLE NOT NULL UNIQUE ); CREATE TABLE frequencies ( freq_index INTEGER PRIMARY KEY, - frequency DOUBLE NOT NULL UNIQUE -- Hz + frequency DOUBLE NOT NULL UNIQUE ); --- ---------- dimensions ---------- - CREATE TYPE tect_type_t AS ENUM ( 'ACTIVE_SHALLOW', 'VOLCANIC', 'SUBDUCTION_INTERFACE', 'SUBDUCTION_SLAB' ); @@ -150,10 +35,10 @@ dtop FLOAT, dbottom FLOAT, length FLOAT, - source_wkt VARCHAR, -- rupture surface - trace_wkt VARCHAR, -- surface trace - domain_wkt VARCHAR, -- simulation domain - metadata VARCHAR -- JSON: fault_type, sim_type, plane_count, ... + source_wkt VARCHAR, + trace_wkt VARCHAR, + domain_wkt VARCHAR, + metadata VARCHAR ); CREATE TABLE realisations ( @@ -165,7 +50,7 @@ hypo_lat FLOAT, hypo_lon FLOAT, hypo_depth FLOAT, - metadata VARCHAR -- JSON: solver, shypo, dhypo, ... + metadata VARCHAR ); CREATE TABLE sites ( @@ -173,48 +58,131 @@ site_id VARCHAR NOT NULL UNIQUE, lat FLOAT NOT NULL, lon FLOAT NOT NULL, - vs30 FLOAT, -- m/s - z1p0 FLOAT, -- km - z2p5 FLOAT, -- km - metadata VARCHAR -- JSON: elevation, basin, grid_level, ... + vs30 FLOAT, + z1p0 FLOAT, + z2p5 FLOAT, + metadata VARCHAR ); --- ---------- large tables: no PRIMARY KEY, UNIQUE or FOREIGN KEY ---------- - CREATE TABLE site_event ( site_int_id INTEGER NOT NULL, event_int_id INTEGER NOT NULL, - rrup FLOAT, -- km + rrup FLOAT, rjb FLOAT, rx FLOAT, ry FLOAT, - metadata VARCHAR -- JSON + metadata VARCHAR ); --- logical key (site_int_id, event_int_id) CREATE SEQUENCE record_id_seq START 1; CREATE TABLE records ( record_id BIGINT DEFAULT nextval('record_id_seq'), - event_int_id INTEGER NOT NULL, -- derived from rel_int_id by the writer + event_int_id INTEGER NOT NULL, rel_int_id INTEGER NOT NULL, site_int_id INTEGER NOT NULL, component VARCHAR NOT NULL ); --- logical key (rel_int_id, site_int_id, component) -CREATE TABLE psa_ims (record_id BIGINT NOT NULL, pSA FLOAT[]); -- periods.period_index -CREATE TABLE fas_ims (record_id BIGINT NOT NULL, FAS FLOAT[]); -- frequencies.freq_index +CREATE TABLE psa_ims (record_id BIGINT NOT NULL, pSA FLOAT[]); +CREATE TABLE fas_ims (record_id BIGINT NOT NULL, FAS FLOAT[]); CREATE TABLE scalars_ims ( record_id BIGINT NOT NULL, PGA FLOAT, PGV FLOAT, PGD FLOAT, - CAV FLOAT, -- NULL for rotd components - AI FLOAT, -- NULL for rotd components - Ds575 FLOAT, -- NULL for rotd components - Ds595 FLOAT -- NULL for rotd components + CAV FLOAT, + AI FLOAT, + Ds575 FLOAT, + Ds595 FLOAT ); """ -"""Full schema DDL, executed as one script by :meth:`imdb.IMDB.create`.""" + +COMPONENTS = ("000", "090", "ver", "geom", "rotd0", "rotd50", "rotd100") + +SCALAR_IMS = ("PGA", "PGV", "PGD", "CAV", "AI", "Ds575", "Ds595") + +ROTD_UNDEFINED = frozenset({"CAV", "AI", "Ds575", "Ds595"}) + +TECT_TYPES = ("ACTIVE_SHALLOW", "VOLCANIC", "SUBDUCTION_INTERFACE", "SUBDUCTION_SLAB") + +IM_UNITS = { + "pSA": "g", + "FAS": "g.s", + "PGA": "g", + "PGV": "cm/s", + "PGD": "cm", + "CAV": "m/s", + "AI": "m/s", + "Ds575": "s", + "Ds595": "s", +} + +METADATA_TABLES = { + "events": "event_metadata_keys", + "realisations": "rel_metadata_keys", + "sites": "site_metadata_keys", + "site_event": "site_event_metadata_keys", +} + +NOTES = { + "logical keys": ( + "site_event's logical key is (site_int_id, event_int_id); records' logical key is " + "(rel_int_id, site_int_id, component). Neither is enforced by a constraint; the " + "writer is responsible for not creating duplicates." + ), + "array indexing is 1-based": ( + "periods.period_index and frequencies.freq_index are 1-based, matching DuckDB list " + "indexing, so pSA[period_index] and FAS[freq_index] need no offset. len(pSA) equals " + "the row count of periods; len(FAS) equals the row count of frequencies." + ), + "identity and rebuild stability": ( + "event_id, rel_id and site_id are stable. The integer surrogates event_int_id, " + "rel_int_id, site_int_id and record_id are assigned at ingest and change on rebuild; " + "nothing outside the database may reference them. External references use " + "(rel_id, site_id, component)." + ), + "component vocabulary": ( + "000, 090, ver, geom, rotd0, rotd50, rotd100, following IM_calculation. A database " + "may hold any subset; db_meta.components lists which. The writer validates against " + "that list." + ), + "rotd scalars are undefined": ( + "scalars_ims.CAV, AI, Ds575 and Ds595 are NULL for rotd* components. PGA, PGV and " + "PGD are populated for every component." + ), + "units": ( + "Linear, physical units; log is a read-time transform. Every IM unit is a row in " + "im_units. Outside the IM tables: distances km, vs30 m/s, z1p0 and z2p5 km, depths " + "km, angles degrees, coordinates WGS84." + ), + "metadata columns are JSON": ( + "events, realisations, sites and site_event each carry a metadata VARCHAR holding a " + "JSON object. Read with json_extract_string(metadata, '$.key'). Permitted keys are " + "declared in db_meta and validated by the writer." + ), + "distances are event level": ( + "rrup, rjb, rx and ry are measured to the rupture surface and are defined at the " + "event level, shared across all realisations of an event. Hypocentral and " + "epicentral distance are not stored; compute them from realisations.hypo_* and " + "sites.lat/lon." + ), + "synthetic realisations": ( + "Every event has at least one realisation. A dataset with no realisation concept " + "gets exactly one per event, with rel_id equal to event_id." + ), + "record_id is file-local": ( + "record_id is a surrogate scoped to this database file only; it is not stable " + "across rebuilds or between databases." + ), + "physical sort order": ( + "Rows are written in event order, so event_int_id and record_id are monotonic and " + "event filters prune row groups. Site filters do not prune. db_meta.sort_order " + "records the ordering used." + ), + "provenance": ( + "db_meta.source, creator, created_at and imdb_version record where the data in this " + "database came from and what built it." + ), +} diff --git a/pyproject.toml b/pyproject.toml index ed67127..348efb1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,7 @@ readme = "README.md" requires-python = ">=3.12" dynamic = ["version"] dependencies = [ - "duckdb>=1.5", + "ibis-framework[duckdb]>=9", "numpy>=2", "pandas>=3", ] diff --git a/tests/conftest.py b/tests/conftest.py index d6a4efc..e83c73b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,117 +1,79 @@ +"""Shared fixtures: one small synthetic IMDB.""" + import numpy as np import pandas as pd import pytest from imdb import IMDB -PERIODS = [0.01, 0.1, 1.0, 3.0, 10.0] -FREQUENCIES = [0.1, 1.0, 10.0] -COMPONENTS = ["geom", "rotd50"] -EVENTS = ["ev1", "ev2"] -SITES = ["stnA", "stnB", "stnC"] -SCALARS = ["PGA", "PGV", "PGD", "CAV", "AI", "Ds575", "Ds595"] +PERIODS = [0.1, 0.2, 0.5, 1.0, 2.0] +FREQUENCIES = [1.0, 5.0, 10.0] +COMPONENTS = ["000", "090"] +EVENTS = ["eventA", "eventB"] +SITES = ["siteA", "siteB", "siteC"] -def build_frames(): - """Deterministic input frames covering every table.""" - events = pd.DataFrame( - { - "event_id": EVENTS, - "magnitude": [7.1, 6.2], - "tect_type": ["SUBDUCTION_SLAB", "ACTIVE_SHALLOW"], - "dtop": [30.0, 0.5], - "metadata": [ - {"fault_type": "DS_POINT_SOURCE"}, - {"fault_type": "NORMAL_FAULTING"}, - ], - } - ) - rels = pd.DataFrame( - { - "rel_id": [f"{e}_REL{i:02d}" for e in EVENTS for i in (1, 2)], - "event_id": [e for e in EVENTS for _ in (1, 2)], - "magnitude": [7.1, 7.12, 6.2, 6.18], - "rake": [90.0, 88.0, -90.0, -92.0], - "hypo_lat": [-43.5, -43.6, -41.2, -41.3], - "hypo_lon": [172.6, 172.7, 174.8, 174.9], - "hypo_depth": [40.0, 42.0, 8.0, 9.0], - "metadata": [{"solver": "emod3d"}] * 4, - } - ) - sites = pd.DataFrame( - { - "site_id": SITES, - "lat": [-43.5, -43.6, -41.3], - "lon": [172.6, 172.7, 174.8], - "vs30": [300.0, 500.0, 250.0], - "z1p0": [0.3, 0.1, 0.5], - "metadata": [{"elevation": 10.0}, {"elevation": 55.0}, {"elevation": 3.0}], - } - ) - site_event = pd.DataFrame( - { - "site_id": [s for s in SITES for _ in EVENTS], - "event_id": EVENTS * len(SITES), - "rrup": [10.0, 300.0, 25.0, 280.0, 400.0, 5.0], - "rjb": [8.0, 295.0, 22.0, 275.0, 395.0, 3.0], - } +@pytest.fixture +def db(tmp_path): + """A small, fully populated IMDB: 2 events, 2 realisations each, 3 sites, 2 components.""" + path = tmp_path / "test.duckdb" + db = IMDB.create( + path, periods=PERIODS, frequencies=FREQUENCIES, components=COMPONENTS ) - rows = [] - for rel_id in rels["rel_id"]: - for site_id in SITES: - for component in COMPONENTS: - base = float(len(rows) + 1) - row = { - "rel_id": rel_id, - "site_id": site_id, - "component": component, - "pSA": np.array( - [base + i / 8 for i in range(1, len(PERIODS) + 1)], - dtype=np.float32, - ), - "FAS": np.array( - [base * 2 + i / 8 for i in range(1, len(FREQUENCIES) + 1)], - dtype=np.float32, - ), - } - for k, im in enumerate(SCALARS): - row[im] = base + (k + 1) / 16 - rows.append(row) - records = pd.DataFrame(rows) - return events, rels, sites, site_event, records - - -@pytest.fixture -def db_path(tmp_path): - """Path of a small, fully populated database.""" - path = tmp_path / "test_ims.duckdb" - events, rels, sites, site_event, records = build_frames() - with IMDB.create( - path, - periods=PERIODS, - frequencies=FREQUENCIES, - components=COMPONENTS, - db_meta={"dataset_description": "test fixture", "source": "conftest"}, - ) as db: - db.add_events(events) - db.add_realisations(rels) - db.add_sites(sites) - db.add_site_event(site_event) - db.add_records(records) - db.finalise() - return path + db.add_events(pd.DataFrame({"event_id": EVENTS, "magnitude": [6.0, 7.0]})) + rel_ids = [f"{e}_rel{i}" for e in EVENTS for i in range(2)] + db.add_realisations( + pd.DataFrame( + { + "rel_id": rel_ids, + "event_id": [e for e in EVENTS for _ in range(2)], + } + ) + ) -@pytest.fixture -def db(db_path): - """The fixture database, open read-only.""" - with IMDB(db_path) as handle: - yield handle + db.add_sites( + pd.DataFrame( + { + "site_id": SITES, + "lat": [-43.5, -43.6, -43.7], + "lon": [172.6, 172.7, 172.8], + } + ) + ) + db.add_site_event( + pd.DataFrame( + { + "site_id": [s for _ in EVENTS for s in SITES], + "event_id": [e for e in EVENTS for _ in SITES], + "rrup": np.arange(len(EVENTS) * len(SITES), dtype=float), + } + ) + ) -@pytest.fixture -def wdb(db_path): - """The fixture database, open for writing.""" - with IMDB(db_path, read_only=False) as handle: - yield handle + rng = np.random.default_rng(0) + rows = [ + (rel_id, site_id, component) + for rel_id in rel_ids + for site_id in SITES + for component in COMPONENTS + ] + n = len(rows) + db.add_records( + pd.DataFrame( + { + "rel_id": [r[0] for r in rows], + "site_id": [r[1] for r in rows], + "component": [r[2] for r in rows], + "pSA": [rng.uniform(size=len(PERIODS)) for _ in range(n)], + "FAS": [rng.uniform(size=len(FREQUENCIES)) for _ in range(n)], + "PGA": rng.uniform(size=n), + "PGV": rng.uniform(size=n), + "PGD": rng.uniform(size=n), + } + ) + ) + yield db + db.close() diff --git a/tests/test_imdb.py b/tests/test_imdb.py index 822ce05..fc1b99e 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -1,311 +1,37 @@ -import numpy as np -import pandas as pd -import pytest +"""Basic tests: the library works for its intended, correct usage.""" -from imdb import IMDB, schema +from tests.conftest import COMPONENTS, EVENTS, FREQUENCIES, PERIODS, SITES -from .conftest import ( - COMPONENTS, - EVENTS, - FREQUENCIES, - PERIODS, - SCALARS, - SITES, - build_frames, -) -N_RECORDS = 4 * len(SITES) * len(COMPONENTS) - - -def test_create_populates_documentation(db): - assert db.db_meta["schema_version"] == schema.SCHEMA_VERSION - assert db.db_meta["dataset_description"] == "test fixture" - assert db.db_meta["n_periods"] == str(len(PERIODS)) - assert db.components == COMPONENTS - assert db.notes == schema.NOTES - assert db.im_units == schema.IM_UNITS - assert db.periods.tolist() == PERIODS - assert db.frequencies.tolist() == FREQUENCIES - assert db.periods.index.tolist() == [1, 2, 3, 4, 5] - - -def test_dimension_reads(db): - assert sorted(db.get_events()["event_id"]) == EVENTS - assert len(db.get_realisations()) == 4 - assert sorted(db.get_sites()["site_id"]) == SITES - assert len(db.get_site_event()) == len(SITES) * len(EVENTS) - assert len(db.get_records()) == N_RECORDS - - -def test_metadata_expansion(db): - events = db.get_events(expand_metadata=True) - assert "metadata" not in events.columns - assert set(events["fault_type"]) == {"DS_POINT_SOURCE", "NORMAL_FAULTING"} - assert db.db_meta["event_metadata_keys"] == "fault_type" - assert db.db_meta["site_metadata_keys"] == "elevation" - - -def test_psa_and_fas_round_trip(db): - _, _, _, _, records = build_frames() - order = db.get_records().sort_index() +def test_round_trip(db): + records = db.get_records() + n = len(EVENTS) * 2 * len(SITES) * len(COMPONENTS) + assert len(records) == n psa = db.get_psa() - assert psa.columns.tolist() == PERIODS - fas = db.get_fas() - assert fas.columns.tolist() == FREQUENCIES - - # rebuild the input keyed the same way the database is - key = ["rel_id", "site_id", "component"] - expected = records.set_index(key) - got = order.reset_index().set_index(key) - - for rid, k in zip(got["record_id"], got.index, strict=True): - np.testing.assert_array_equal( - psa.loc[rid].to_numpy(dtype=np.float32), expected.loc[k, "pSA"] - ) - np.testing.assert_array_equal( - fas.loc[rid].to_numpy(dtype=np.float32), expected.loc[k, "FAS"] - ) - - -def test_scalar_round_trip_and_rotd_nulls(db): - _, _, _, _, records = build_frames() - scalars = db.get_scalars() - assert scalars.columns.tolist() == list(schema.SCALAR_IMS) - - recs = db.get_records() - joined = recs.join(scalars) - expected = records.set_index(["rel_id", "site_id", "component"]) - - for record_id, row in joined.iterrows(): - key = (row["rel_id"], row["site_id"], row["component"]) - for im in SCALARS: - if im in schema.ROTD_UNDEFINED and row["component"].startswith("rotd"): - assert pd.isna(row[im]), (record_id, im) - else: - assert row[im] == pytest.approx(expected.loc[key, im], rel=1e-6) - - -def test_subset_of_periods(db): - full = db.get_psa() - subset = db.get_psa(periods=[1.0, 0.01]) - assert subset.columns.tolist() == [1.0, 0.01] - pd.testing.assert_series_equal(subset[1.0], full[1.0], check_names=False) - - -def test_period_written_with_p(db): - df = db.get_im_df(["pSA_0p1", "pSA_10p0"]) - assert df.columns.tolist() == ["pSA_0p1", "pSA_10p0"] - np.testing.assert_allclose(df["pSA_0p1"], db.get_psa(periods=[0.1])[0.1]) - - -def test_get_im_df_matches_typed_calls(db): - names = ["PGA", "pSA_1.0", "FAS_10.0", "Ds595"] - df = db.get_im_df(names) - assert df.columns.tolist() == names - assert len(df) == N_RECORDS - np.testing.assert_allclose(df["PGA"], db.get_scalars(ims=["PGA"])["PGA"]) - np.testing.assert_allclose(df["pSA_1.0"], db.get_psa(periods=[1.0])[1.0]) - np.testing.assert_allclose(df["FAS_10.0"], db.get_fas(frequencies=[10.0])[10.0]) - - -@pytest.mark.parametrize( - ("filters", "expected"), - [ - ({}, N_RECORDS), - ({"events": ["ev1"]}, 2 * len(SITES) * len(COMPONENTS)), - ({"sites": ["stnA"]}, 4 * len(COMPONENTS)), - ({"rels": ["ev1_REL01"]}, len(SITES) * len(COMPONENTS)), - ({"component": "rotd50"}, N_RECORDS // 2), - ({"component": ["geom", "rotd50"]}, N_RECORDS), - ({"events": ["ev1"], "component": "geom"}, 2 * len(SITES)), - ({"max_rrup": 30.0}, 3 * 2 * len(COMPONENTS)), - ], -) -def test_filters(db, filters, expected): - assert len(db.get_records(**filters)) == expected - assert len(db.get_psa(periods=[1.0], **filters)) == expected - assert len(db.get_scalars(ims=["PGA"], **filters)) == expected - assert len(db.get_im_df(["PGA", "pSA_1.0"], **filters)) == expected - - -def test_record_ids_filter(db): - wanted = db.get_records().index[:5].to_numpy() - got = db.get_psa(periods=[1.0], record_ids=wanted) - assert sorted(got.index) == sorted(wanted) - - -def test_filters_match_raw_sql(db): - got = db.get_records(events=["ev2"], component="geom").index.tolist() - expected = [ - row[0] - for row in db.sql( - """ - SELECT r.record_id FROM records r - JOIN events e ON e.event_int_id = r.event_int_id - WHERE e.event_id = 'ev2' AND r.component = 'geom' - """ - ).fetchall() - ] - assert sorted(got) == sorted(expected) - - -def test_site_event_filters(db): - assert len(db.get_site_event(sites=["stnA"])) == 2 - assert len(db.get_site_event(events=["ev1"])) == 3 - assert len(db.get_site_event(max_rrup=30.0)) == 3 - expanded = db.get_site_event(sites=["stnA"], expand_metadata=True) - assert "metadata" not in expanded.columns + assert list(psa.columns) == [str(p) for p in PERIODS] + assert (psa.index == records.index).all() + fas = db.get_fas() + assert list(fas.columns) == [str(f) for f in FREQUENCIES] -def test_bad_requests_raise(db): - with pytest.raises(KeyError, match="not on this database's grid"): - db.get_psa(periods=[2.5]) - with pytest.raises(ValueError, match="cannot parse IM name"): - db.get_im_df(["SA_1.0"]) - with pytest.raises(ValueError, match="not in this database"): - db.get_records(component="000") - with pytest.raises(ValueError, match="unknown scalar IMs"): - db.get_scalars(ims=["MMI"]) - with pytest.raises(KeyError, match="unknown event_id"): - db.get_records(events=["nope"]) + scalars = db.get_scalars(ims=["PGA", "PGV", "PGD"]) + assert set(scalars.columns) == {"PGA", "PGV", "PGD"} + assert not scalars.isna().any().any() -def test_read_only_rejects_writes(db): - with pytest.raises(PermissionError): - db.set_db_meta({"source": "nope"}) +def test_filter_by_event(db): + records = db.get_records(event_ids=["eventA"]) + assert (records["event_id"] == "eventA").all() + assert len(records) == 2 * len(SITES) * len(COMPONENTS) -def test_validate_is_clean(db): +def test_validate_clean(db): assert db.validate() == [] -def test_validate_finds_orphan_record(wdb): - wdb.conn.execute("INSERT INTO records VALUES (9999, 1, 999, 1, 'geom')") - problems = wdb.validate() - assert any("orphan rel_int_id" in p for p in problems) - - -def test_validate_finds_wrong_array_length(wdb): - rid = int(wdb.get_records().index[0]) - wdb.conn.execute("UPDATE psa_ims SET pSA = [1.0, 2.0] WHERE record_id = ?", [rid]) - assert any("pSA length is wrong" in p for p in wdb.validate()) - - -def test_validate_finds_undeclared_metadata_key(wdb): - wdb.conn.execute( - """UPDATE sites SET metadata = '{"basin": "Canterbury"}' WHERE site_int_id = 1""" - ) - assert any( - "not declared in db_meta.site_metadata_keys" in p for p in wdb.validate() - ) - - -def test_validate_finds_duplicate_record_key(wdb): - wdb.conn.execute( - "INSERT INTO records SELECT 9999, event_int_id, rel_int_id, site_int_id, component " - "FROM records LIMIT 1" - ) - assert any("duplicated (rel_int_id" in p for p in wdb.validate()) - - -def test_undeclared_metadata_key_rejected_on_write(wdb): - sites = pd.DataFrame( - { - "site_id": ["stnD"], - "lat": [-42.0], - "lon": [173.0], - "metadata": [{"basin": "Canterbury"}], - } - ) - with pytest.raises(ValueError, match="not declared in db_meta.site_metadata_keys"): - wdb.add_sites(sites) - - -def test_unknown_column_rejected_on_write(wdb): - with pytest.raises(ValueError, match="not in sites"): - wdb.add_sites( - pd.DataFrame( - {"site_id": ["stnD"], "lat": [-42.0], "lon": [173.0], "vs20": [1.0]} - ) - ) - - -def test_duplicate_id_rejected_on_write(wdb): - with pytest.raises(ValueError, match="already in the database"): - wdb.add_sites( - pd.DataFrame({"site_id": ["stnA"], "lat": [-42.0], "lon": [173.0]}) - ) - - -def test_undeclared_component_rejected_on_write(wdb): - _, _, _, _, records = build_frames() - bad = records.head(1).copy() - bad["component"] = "000" - with pytest.raises(ValueError, match="not declared in db_meta.components"): - wdb.add_records(bad) - - -def test_wrong_array_length_rejected_on_write(wdb): - _, _, _, _, records = build_frames() - bad = records.head(1).copy() - bad["rel_id"] = "ev1_REL01" - bad["pSA"] = [np.ones(3, dtype=np.float32)] - with pytest.raises(ValueError, match="grid has 5 entries"): - wdb.add_records(bad) - - -def test_delete_event_then_readd(db_path): - events, rels, _, site_event, records = build_frames() - with IMDB(db_path, read_only=False) as db: - before = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"]).to_numpy() - old_ids = set(db.get_records(events=["ev1"]).index) - - db.delete_event("ev1") - assert db.validate() == [] - assert len(db.get_records()) == N_RECORDS // 2 - assert len(db.get_site_event()) == len(SITES) - - keep = rels["event_id"] == "ev1" - db.add_events(events[events["event_id"] == "ev1"]) - db.add_realisations(rels[keep]) - db.add_site_event(site_event[site_event["event_id"] == "ev1"]) - db.add_records(records[records["rel_id"].isin(rels.loc[keep, "rel_id"])]) - db.finalise() - - after = db.get_im_df(["PGA", "pSA_1.0"], events=["ev1"]).to_numpy() - new_ids = set(db.get_records(events=["ev1"]).index) - - np.testing.assert_allclose(np.sort(before, axis=0), np.sort(after, axis=0)) - assert not (old_ids & new_ids) - - -def test_records_without_all_im_tables(tmp_path): - path = tmp_path / "psa_only.duckdb" - events, rels, sites, site_event, records = build_frames() - with IMDB.create(path, periods=PERIODS, components=COMPONENTS) as db: - db.add_events(events) - db.add_realisations(rels) - db.add_sites(sites) - db.add_site_event(site_event) - db.add_records(records[["rel_id", "site_id", "component", "pSA"]]) - db.finalise() - - with IMDB(path) as db: - assert len(db.get_psa()) == N_RECORDS - assert db.get_scalars().empty - assert len(db.frequencies) == 0 - with pytest.raises(KeyError, match="no FAS grid"): - db.get_fas() - - -def test_create_refuses_to_clobber(db_path): - with pytest.raises(FileExistsError): - IMDB.create(db_path, periods=PERIODS) - IMDB.create(db_path, periods=PERIODS, overwrite=True).close() - - -def test_finalise_raises_on_inconsistency(wdb): - wdb.conn.execute("INSERT INTO records VALUES (9999, 1, 999, 1, 'geom')") - with pytest.raises(ValueError, match="database is inconsistent"): - wdb.finalise() +def test_delete_and_readd(db): + db.delete_event("eventA") + assert len(db.get_records(event_ids=["eventA"])) == 0 + assert len(db.get_realisations()) == 2 + assert db.validate() == [] diff --git a/uv.lock b/uv.lock index 2744694..53e4d1b 100644 --- a/uv.lock +++ b/uv.lock @@ -19,6 +19,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7e/b3/6b4067be973ae96ba0d615946e314c5ae35f9f993eca561b356540bb0c2b/alabaster-1.0.0-py3-none-any.whl", hash = "sha256:fc6786402dc3fcb2de3cabd5fe455a2db534b371124f1f21de8731783dec828b", size = 13929, upload-time = "2024-07-26T18:15:02.05Z" }, ] +[[package]] +name = "atpublic" +version = "7.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a9/05/e2e131a0debaf0f01b8a1b586f5f11713f6affc3e711b406f15f11eafc92/atpublic-7.0.0.tar.gz", hash = "sha256:466ef10d0c8bbd14fd02a5fbd5a8b6af6a846373d91106d3a07c16d72d96b63e", size = 17801, upload-time = "2025-11-29T05:56:45.45Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/96/c0/271f3e1e3502a8decb8ee5c680dbed2d8dc2cd504f5e20f7ed491d5f37e1/atpublic-7.0.0-py3-none-any.whl", hash = "sha256:6702bd9e7245eb4e8220a3e222afcef7f87412154732271ee7deee4433b72b4b", size = 6421, upload-time = "2025-11-29T05:56:44.604Z" }, +] + [[package]] name = "babel" version = "2.18.0" @@ -346,6 +355,35 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e1/2c/95d9216b79e9273689d7ebce125a54503ed0c9bd7da931f0265888e99779/duckdb-1.5.5-cp314-cp314-win_arm64.whl", hash = "sha256:63e48d4b74b15aeacd688976432a7225163df8c226eddeb8536bba2d4d4ff433", size = 14470180, upload-time = "2026-07-22T10:55:14.445Z" }, ] +[[package]] +name = "ibis-framework" +version = "12.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "atpublic" }, + { name = "parsy" }, + { name = "python-dateutil" }, + { name = "sqlglot" }, + { name = "toolz" }, + { name = "typing-extensions" }, + { name = "tzdata" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/f2/8e/2e7ad9bdeaf45350da7beeb67a0d4317d400dac882825eb7c3bd4d3c6ae1/ibis_framework-12.0.0.tar.gz", hash = "sha256:238624f2c14fdab8382ca2f4f667c3cdb81e29844cd5f8db8a325d0743767c61", size = 1351369, upload-time = "2026-02-07T14:31:13.171Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9d/b3/11d406849715b47c9d69bb22f50874f80caee96bd1cbe7b61abbebbf5a05/ibis_framework-12.0.0-py3-none-any.whl", hash = "sha256:0bbd790f268da9cb87926d5eaad2b827a573927113c4ed3be5095efa89b9e512", size = 2079219, upload-time = "2026-02-07T14:31:10.646Z" }, +] + +[package.optional-dependencies] +duckdb = [ + { name = "duckdb" }, + { name = "numpy" }, + { name = "packaging" }, + { name = "pandas" }, + { name = "pyarrow" }, + { name = "pyarrow-hotfix" }, + { name = "rich" }, +] + [[package]] name = "idna" version = "3.19" @@ -368,7 +406,7 @@ wheels = [ name = "imdb" source = { editable = "." } dependencies = [ - { name = "duckdb" }, + { name = "ibis-framework", extra = ["duckdb"] }, { name = "numpy" }, { name = "pandas" }, ] @@ -392,7 +430,7 @@ types = [ [package.metadata] requires-dist = [ - { name = "duckdb", specifier = ">=1.5" }, + { name = "ibis-framework", extras = ["duckdb"], specifier = ">=9" }, { name = "numpy", specifier = ">=2" }, { name = "pandas", specifier = ">=3" }, ] @@ -435,6 +473,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" }, ] +[[package]] +name = "markdown-it-py" +version = "4.2.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "mdurl" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/06/ff/7841249c247aa650a76b9ee4bbaeae59370dc8bfd2f6c01f3630c35eb134/markdown_it_py-4.2.0.tar.gz", hash = "sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49", size = 82454, upload-time = "2026-05-07T12:08:28.36Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b3/81/4da04ced5a082363ecfa159c010d200ecbd959ae410c10c0264a38cac0f5/markdown_it_py-4.2.0-py3-none-any.whl", hash = "sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a", size = 91687, upload-time = "2026-05-07T12:08:27.182Z" }, +] + [[package]] name = "markupsafe" version = "3.0.3" @@ -498,6 +548,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/70/bc/6f1c2f612465f5fa89b95bead1f44dcb607670fd42891d8fdcd5d039f4f4/markupsafe-3.0.3-cp314-cp314t-win_arm64.whl", hash = "sha256:32001d6a8fc98c8cb5c947787c5d08b0a50663d139f1305bac5885d98d9b40fa", size = 14146, upload-time = "2025-09-27T18:37:28.327Z" }, ] +[[package]] +name = "mdurl" +version = "0.1.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d6/54/cfe61301667036ec958cb99bd3efefba235e65cdeb9c84d24a8293ba1d90/mdurl-0.1.2.tar.gz", hash = "sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba", size = 8729, upload-time = "2022-08-14T12:40:10.846Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b3/38/89ba8ad64ae25be8de66a6d463314cf1eb366222074cfda9ee839c56a4b4/mdurl-0.1.2-py3-none-any.whl", hash = "sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8", size = 9979, upload-time = "2022-08-14T12:40:09.779Z" }, +] + [[package]] name = "numpy" version = "2.5.3" @@ -650,6 +709,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/60/c2/959caec5c46f484b5f8bb6def4b0cf7a45ba6acda26f12b75142d98cc5ae/pandas_stubs-3.0.5.260730-py3-none-any.whl", hash = "sha256:60e90e3e1eda6937e337e243cbe6217e151c11137cd7eddf832af537c7310bfd", size = 174807, upload-time = "2026-07-30T14:31:41.17Z" }, ] +[[package]] +name = "parsy" +version = "2.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/cc/58/1e3f382eef9e50a2a115486b0c178d22bb97d2fbb85421ccbe5d3a783530/parsy-2.2.tar.gz", hash = "sha256:e943147644a8cf0d82d1bcb5c5867dd517495254cea3e3eb058b1e421cb7561f", size = 47296, upload-time = "2025-09-12T11:39:26.783Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/77/fc/8cb9073bb1bee54eb49a1ae501a36402d01763812962ac811cdc1c81a9d7/parsy-2.2-py3-none-any.whl", hash = "sha256:5e981613d9d2d8b68012d1dd0afe928967bea2e4eefdb76c2f545af0dd02a9e7", size = 9538, upload-time = "2025-09-12T11:39:25.749Z" }, +] + [[package]] name = "pluggy" version = "1.6.0" @@ -659,6 +727,51 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, ] +[[package]] +name = "pyarrow" +version = "25.0.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3d/e3/27f57f80141379d60defe6703eb50a707325706f07fedfd1312c7a751995/pyarrow-25.0.1.tar.gz", hash = "sha256:9150a83248bfed9813ea3c3af74c3856c1984d444aa28e58bf7733b9750ddf6a", size = 1201653, upload-time = "2026-08-10T12:40:53.904Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a6/e2/9ab15b88cbfac28e16419ce5439ec29234c5172cb8259301b4ba639bdec0/pyarrow-25.0.1-cp312-cp312-macosx_12_0_arm64.whl", hash = "sha256:df961f2e7ae9cf496459259d798652c70625f6c080650d6952f8c04053c58ee9", size = 35861559, upload-time = "2026-08-10T12:38:02.567Z" }, + { url = "https://files.pythonhosted.org/packages/58/79/a0036dbe1eabe1f73127427342f1d99982584c4a2cde2651d6c93499c6f6/pyarrow-25.0.1-cp312-cp312-macosx_12_0_x86_64.whl", hash = "sha256:cc4aa407fde9fc660be3939e49ea31f50f3e9fec17c0ec63159f7711edd3efc9", size = 37628383, upload-time = "2026-08-10T12:38:09.083Z" }, + { url = "https://files.pythonhosted.org/packages/13/49/d93a57d375f4bf0cf82913dd6bb54acafde83dd993be2282c81ac5616cad/pyarrow-25.0.1-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:4340f0ba6c1d2e13f21658de1d7c662ca2545018568d0030a1e9afca159d87e3", size = 46820190, upload-time = "2026-08-10T12:38:15.458Z" }, + { url = "https://files.pythonhosted.org/packages/60/c9/711ca85d79f1ec98f29a5eae2b051e25b4ecec5de3e3c0e2d5c5dcb15664/pyarrow-25.0.1-cp312-cp312-manylinux_2_28_x86_64.whl", hash = "sha256:5389cdf79447ed1515c9e31620e6e1e2302249564d603f2ad727d4f6d313e4c3", size = 50102437, upload-time = "2026-08-10T12:38:22.487Z" }, + { url = "https://files.pythonhosted.org/packages/80/53/8fb8359ff17cfb6263a1cf3ebf7caec9fe197de118719e84fcb1d0618026/pyarrow-25.0.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:d51592cb7561e87877c506113e7adbf1342ab579e6c21f0ef44b8ba41cb74c80", size = 49942424, upload-time = "2026-08-10T12:38:28.755Z" }, + { url = "https://files.pythonhosted.org/packages/e8/83/4e5ae02a9341571b18a6fca380ac7a58ce6ddae7ab3c060208c0a1e79f02/pyarrow-25.0.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:6109c94d8b9f3b17a041daca16cacb2f651ad8f1ef70a4232c2c0f37a23da2a8", size = 53144206, upload-time = "2026-08-10T12:38:34.862Z" }, + { url = "https://files.pythonhosted.org/packages/65/ee/197cbf47e49f83e6ebeb946a5259a48a638dea27ac774db42fe78022179d/pyarrow-25.0.1-cp312-cp312-win_amd64.whl", hash = "sha256:8858d7bfc22e3f51529aeaa4077225029724623e4595dc9eff8c793935c34140", size = 27953934, upload-time = "2026-08-10T12:38:39.808Z" }, + { url = "https://files.pythonhosted.org/packages/cc/8d/8f271a7a034c834910ec925d56fa4b29733b1380f5289419f5aaa3b02777/pyarrow-25.0.1-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:c7c534ec03c358a76ea3e505e74c1b6aef290af90c444dfd092dbfe23e755b85", size = 35855328, upload-time = "2026-08-10T12:38:45.489Z" }, + { url = "https://files.pythonhosted.org/packages/d2/cd/5bac242f4e841b9971d5eb94fdfe2577e2b70be983e27401e72055786037/pyarrow-25.0.1-cp313-cp313-macosx_12_0_x86_64.whl", hash = "sha256:dda9470024204d7bbf2042b47c6e8a0e47a3eeb8e34405882dfaea6577e0c153", size = 37622415, upload-time = "2026-08-10T12:38:51.107Z" }, + { url = "https://files.pythonhosted.org/packages/63/1f/96d03b4e1506524f7087adb0fd6b2f69f0c9c7aaff1ec36d8030082e15a5/pyarrow-25.0.1-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:44a9120ce5bd81936b8ab9a88076e3fd47c2c6838e0e43630fed83626aca81d9", size = 46813813, upload-time = "2026-08-10T12:38:57.773Z" }, + { url = "https://files.pythonhosted.org/packages/98/d6/33a411115b61dbfc16ad6ad73e71730f6fea654ee3667673bc53ab0e2fe7/pyarrow-25.0.1-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:0befcf816e45a1af33ac775a9970b749e4868a230c7372f0ae5e932bee27039f", size = 50104452, upload-time = "2026-08-10T12:39:04.579Z" }, + { url = "https://files.pythonhosted.org/packages/33/ae/b1b97c9ca87f9f9ddbb5230c798df94eccce61bd79b9b45458c69a478588/pyarrow-25.0.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:3f89685964f46e4216103c75483aac0c0692a5f72212d7ca835adba5ede56ce3", size = 49951343, upload-time = "2026-08-10T12:39:11.8Z" }, + { url = "https://files.pythonhosted.org/packages/98/9e/a112df5cfd5a68cb1d9fc31cfe38c28d5aec9f10865ce37ecef2e4450873/pyarrow-25.0.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6943e2fe7954d29d84de45d29d34c8dc36ce96570e67d89aa9976e650a4a9138", size = 53144784, upload-time = "2026-08-10T12:39:20.503Z" }, + { url = "https://files.pythonhosted.org/packages/31/24/97e8bd98f1e3b07e2ba08bcdff690674fbe16d69a7d2712cc3884665e615/pyarrow-25.0.1-cp313-cp313-win_amd64.whl", hash = "sha256:31e49a7888fcdf3a835da33ae777f6bb9a866334e5a789282fc26dcf426f7f15", size = 27870159, upload-time = "2026-08-10T12:39:26.161Z" }, + { url = "https://files.pythonhosted.org/packages/36/4c/b525824ad3094076919273cd97db61fb3d78252dee76fa3b8dc8f76774aa/pyarrow-25.0.1-cp314-cp314-macosx_12_0_arm64.whl", hash = "sha256:bf0b672390cdcb640d7288f96b826d71ff4e9abb254a86c89890baf51a29cee6", size = 35885255, upload-time = "2026-08-10T12:39:32.366Z" }, + { url = "https://files.pythonhosted.org/packages/08/62/448bb0e940de41aec31d1a956e63ad9c54afdf122a103cc3ab20c2a3ce33/pyarrow-25.0.1-cp314-cp314-macosx_12_0_x86_64.whl", hash = "sha256:38a9a4b4b9613380e200641891495a56c3d5a98a092db4a870af9975e220471d", size = 37644461, upload-time = "2026-08-10T12:39:38.142Z" }, + { url = "https://files.pythonhosted.org/packages/6e/9a/13587e38bd4806fd218f50fd13b8903fab60588a699ff0c406372e5b4043/pyarrow-25.0.1-cp314-cp314-manylinux_2_28_aarch64.whl", hash = "sha256:0b726ad7e7b669be982b0c71c07fe4b037d654354130da79a7902a669e93a66b", size = 46877146, upload-time = "2026-08-10T12:39:43.722Z" }, + { url = "https://files.pythonhosted.org/packages/8d/61/1c5d1229fa21da4cff5365e41e57177aaac57c563c727f35419b8513d1c1/pyarrow-25.0.1-cp314-cp314-manylinux_2_28_x86_64.whl", hash = "sha256:9171748cdf796972d85a4b60157c279913e242992e350c90c7450182a9838b2a", size = 50131616, upload-time = "2026-08-10T12:39:49.304Z" }, + { url = "https://files.pythonhosted.org/packages/43/20/291e1d65cc0b09aa19f03cf25cf51a2f5fa94b5db315178f2d254ed5cad4/pyarrow-25.0.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:b7a296aac7a71fa0886c08e155ddb6c636a50013f801f6178daafa0f9e726188", size = 50008879, upload-time = "2026-08-10T12:39:56.891Z" }, + { url = "https://files.pythonhosted.org/packages/8b/7c/1b7c9ec28e76576337e4f97b31141c9a181b89b6d1d6221e9d8205621a58/pyarrow-25.0.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:0fe7c8b6c03969b49c8c66182e4a18e3819ab92d07cfab5d8370c531b9369ef0", size = 53170864, upload-time = "2026-08-10T12:40:04.918Z" }, + { url = "https://files.pythonhosted.org/packages/b7/75/f3d789dc06011a765d14d86bda799cf72ac1d715b6a6edecaa0d73d95062/pyarrow-25.0.1-cp314-cp314-win_amd64.whl", hash = "sha256:f729cfdbd36fd99d543b67a914d2de044c84ebe45be8b34902b299b608c15c8f", size = 28620729, upload-time = "2026-08-10T12:40:51.41Z" }, + { url = "https://files.pythonhosted.org/packages/fc/05/647a8ee6f7c2662feb6921315617bc04dcd6034763fb61b1199720bf6162/pyarrow-25.0.1-cp314-cp314t-macosx_12_0_arm64.whl", hash = "sha256:59a2de54c0cbd954da861eee4d1d330f8e909c45b53455baef696380f2c55033", size = 36130288, upload-time = "2026-08-10T12:40:11.014Z" }, + { url = "https://files.pythonhosted.org/packages/93/f8/c9ee997554d7bea94520667dd1933f109ac1da3ee3556d2b49381e023484/pyarrow-25.0.1-cp314-cp314t-macosx_12_0_x86_64.whl", hash = "sha256:35935cd5de130aa5cf4dea052a63e6bf2e17006c35c3a468194242b9b2bf5956", size = 37762187, upload-time = "2026-08-10T12:40:16.592Z" }, + { url = "https://files.pythonhosted.org/packages/a2/08/a28c01c7fe9e96e8233ce2d13df1d402f4f999f848f51d2daacd6bb4c036/pyarrow-25.0.1-cp314-cp314t-manylinux_2_28_aarch64.whl", hash = "sha256:f3831aaa25c67a99f99dc8b05873cb9d64560390372e2aa197ce9dd4a3f06a44", size = 46888003, upload-time = "2026-08-10T12:40:23.242Z" }, + { url = "https://files.pythonhosted.org/packages/1b/b9/58612e977d28dc58c878448866838369ee8da2f1e7cc8ed2c84b952aafee/pyarrow-25.0.1-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:6a1fdfc6659b6b19022f2e50627fb5cf7156a66c46bf4299379955cbe742382a", size = 50079036, upload-time = "2026-08-10T12:40:29.169Z" }, + { url = "https://files.pythonhosted.org/packages/72/13/66e1402dcc860e1dc2760b1e0292c9a569b62b3bccab69def1b3e907d006/pyarrow-25.0.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:169d3429d5be7c752125890620f75a60776d38b0035eddae939651640822332e", size = 50040226, upload-time = "2026-08-10T12:40:35.186Z" }, + { url = "https://files.pythonhosted.org/packages/78/10/3f1a5497a7ef732ab0f03ecca3e66d89d9c0f57fdc61b4794c456b781f01/pyarrow-25.0.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:119297a6dc197e45d9c6d4415f7814a67ffa36c180d26f68c154c58067ae782d", size = 53149035, upload-time = "2026-08-10T12:40:41.454Z" }, + { url = "https://files.pythonhosted.org/packages/93/c0/37d4a7e8e2f7a6076283673d5298018ca26478b934c6ee369e10505ab32c/pyarrow-25.0.1-cp314-cp314t-win_amd64.whl", hash = "sha256:4288f27577352d608ca08553b0865e4a9b3aa14820c5d95b53337218d609835b", size = 28753071, upload-time = "2026-08-10T12:40:46.623Z" }, +] + +[[package]] +name = "pyarrow-hotfix" +version = "0.7" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d2/ed/c3e8677f7abf3981838c2af7b5ac03e3589b3ef94fcb31d575426abae904/pyarrow_hotfix-0.7.tar.gz", hash = "sha256:59399cd58bdd978b2e42816a4183a55c6472d4e33d183351b6069f11ed42661d", size = 9910, upload-time = "2025-04-25T10:17:06.247Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2e/c3/94ade4906a2f88bc935772f59c934013b4205e773bcb4239db114a6da136/pyarrow_hotfix-0.7-py3-none-any.whl", hash = "sha256:3236f3b5f1260f0e2ac070a55c1a7b339c4bb7267839bd2015e283234e758100", size = 7923, upload-time = "2025-04-25T10:17:05.224Z" }, +] + [[package]] name = "pygments" version = "2.21.0" @@ -737,6 +850,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/bb/f9/15b44d5e4401b0013bbcefe3c09d7bfddcce28cc3d41b1d3077bcedf5b1f/requirements_parser-0.13.1-py3-none-any.whl", hash = "sha256:6e385663eb32589d16e5b22bb6e5251a57908e73803ffff438b53cd6ea2056e0", size = 14926, upload-time = "2026-06-18T07:52:24.171Z" }, ] +[[package]] +name = "rich" +version = "15.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markdown-it-py" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c0/8f/0722ca900cc807c13a6a0c696dacf35430f72e0ec571c4275d2371fca3e9/rich-15.0.0.tar.gz", hash = "sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36", size = 230680, upload-time = "2026-04-12T08:24:00.75Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/82/3b/64d4899d73f91ba49a8c18a8ff3f0ea8f1c1d75481760df8c68ef5235bf5/rich-15.0.0-py3-none-any.whl", hash = "sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb", size = 310654, upload-time = "2026-04-12T08:24:02.83Z" }, +] + [[package]] name = "roman-numerals" version = "4.1.0" @@ -871,6 +997,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/52/a7/d2782e4e3f77c8450f727ba74a8f12756d5ba823d81b941f1b04da9d033a/sphinxcontrib_serializinghtml-2.0.0-py3-none-any.whl", hash = "sha256:6e2cb0eef194e10c27ec0023bfeb25badbbb5868244cf5bc5bdc04e4464bf331", size = 92072, upload-time = "2024-07-29T01:10:08.203Z" }, ] +[[package]] +name = "sqlglot" +version = "30.18.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e4/73/5b5ce3e23b3ded3ea1986c2c2991217460d5fd5161b3e00a8425f5dcb8a3/sqlglot-30.18.0.tar.gz", hash = "sha256:e57e1b205e341979d1df5b1212c1435c598a0437e4619e3f428b15d5bc3a5cc6", size = 6018496, upload-time = "2026-09-03T13:12:58.206Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/08/9f/3dd2ec8dd84a33e0092f1c815267bc055073f9e833b93e389c59c10f62cc/sqlglot-30.18.0-py3-none-any.whl", hash = "sha256:ee0f9a9f3e2193e763c326e52dfb377b96fdb4292f6305c3ff2d9a220e71c601", size = 748788, upload-time = "2026-09-03T13:12:56.467Z" }, +] + [[package]] name = "tomli" version = "2.4.1" @@ -916,29 +1051,47 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7b/61/cceae43728b7de99d9b847560c262873a1f6c98202171fd5ed62640b494b/tomli-2.4.1-py3-none-any.whl", hash = "sha256:0d85819802132122da43cb86656f8d1f8c6587d54ae7dcaf30e90533028b49fe", size = 14583, upload-time = "2026-03-25T20:22:03.012Z" }, ] +[[package]] +name = "toolz" +version = "1.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/11/d6/114b492226588d6ff54579d95847662fc69196bdeec318eb45393b24c192/toolz-1.1.0.tar.gz", hash = "sha256:27a5c770d068c110d9ed9323f24f1543e83b2f300a687b7891c1a6d56b697b5b", size = 52613, upload-time = "2025-10-17T04:03:21.661Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/12/5911ae3eeec47800503a238d971e51722ccea5feb8569b735184d5fcdbc0/toolz-1.1.0-py3-none-any.whl", hash = "sha256:15ccc861ac51c53696de0a5d6d4607f99c210739caf987b5d2054f3efed429d8", size = 58093, upload-time = "2025-10-17T04:03:20.435Z" }, +] + [[package]] name = "ty" -version = "0.0.78" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/d8/c7/2ba0861384c5b5097ac354383abb98188112cb330208c21d5197e98a29e5/ty-0.0.78.tar.gz", hash = "sha256:770b45854f85fa11595208f08c0f28df80943164d10a2832d86be6ac29f135b2", size = 7050609, upload-time = "2026-09-02T22:41:33.92Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/ea/ed/f34cfc06a9ba72979df219e48dae1a0b6a6c74a340763ccd30a547edf913/ty-0.0.78-py3-none-linux_armv6l.whl", hash = "sha256:122700b98f9d45785c1ce91a9418154562e7f64f4bf34679f6f7b38c5b97453f", size = 13304624, upload-time = "2026-09-02T22:40:56.649Z" }, - { url = "https://files.pythonhosted.org/packages/1c/28/5576e2a08b57676d9b2a736d528f077a6c6e32c3d1de2d75dbbe28346966/ty-0.0.78-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:cbe7e3709ccf29ef3d9f58f5914cc30ec36fb647008a5a4ff8483abe401bfe8d", size = 12928383, upload-time = "2026-09-02T22:40:59.162Z" }, - { url = "https://files.pythonhosted.org/packages/8e/e0/027d49c3da5235e634da3d3fc52c4874ec89f50830388c9c259f8fb71e0e/ty-0.0.78-py3-none-macosx_11_0_arm64.whl", hash = "sha256:c3528897b3ab9d3589561bc2b5e61a4d5d686de527a4400bb09a439b922a6c70", size = 12730058, upload-time = "2026-09-02T22:41:01.235Z" }, - { url = "https://files.pythonhosted.org/packages/bb/bc/6bf8ffefd8063730a70ae889cf85def72bc155fd1453135ebf8a09c2f1c2/ty-0.0.78-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:8bfba4c44a06484093f527b8ef43a317abcf91395b49396170266629eedf74aa", size = 12813321, upload-time = "2026-09-02T22:41:03.251Z" }, - { url = "https://files.pythonhosted.org/packages/6d/c0/b3cdf26f82108908d92fdb7ddb741019c446ceb666cf204d4dbf611bde33/ty-0.0.78-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:5f903c06fdee17baf8373173039ab28f1bce28e66b1c734929a5f50b37eb54d9", size = 13072219, upload-time = "2026-09-02T22:41:05.225Z" }, - { url = "https://files.pythonhosted.org/packages/4d/36/16bfb0abdd178dae11dbc4b57b9a86c2b62642a18c1103fd2c2930aa5879/ty-0.0.78-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:dd83e5fe3f07291d1bd4e591f8d2607c204f3eab53bcb1b8060c40e876e1aa61", size = 13903958, upload-time = "2026-09-02T22:41:07.333Z" }, - { url = "https://files.pythonhosted.org/packages/0f/0c/74d3b1f0344b13156c719dd68a7edf7e48c2051718c9f326ba3cbf45ea53/ty-0.0.78-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:325285377319cf7168a2b8b771ae5027f411530e3c7eab1faf4583cdd72fd7c9", size = 14355581, upload-time = "2026-09-02T22:41:09.56Z" }, - { url = "https://files.pythonhosted.org/packages/8d/d7/7ca0359e1e61b15b8b02328c8d120db3c507876b9b7c6bd9f26935353e0e/ty-0.0.78-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:54b7846a404da6492697b524a769755ee9bee157d67674063ecd9c5f69dc52ff", size = 14044749, upload-time = "2026-09-02T22:41:11.678Z" }, - { url = "https://files.pythonhosted.org/packages/3b/6a/731f16ff42c5fc96e742c2f4e0a6915356bae114ae02ae4af6f33502175f/ty-0.0.78-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:169d3b9d134c0b48fe1af8a844142d374655552054a9d6c15a4b6e51bb0382ea", size = 13391647, upload-time = "2026-09-02T22:41:13.928Z" }, - { url = "https://files.pythonhosted.org/packages/0c/af/250cc29daf310e837188509ea4d78460c495d8a74c4fb84b4152d9ef2d4f/ty-0.0.78-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6108cb3b2d28dac5981d4e25008a0a5547d8c877a67f6da9e8c38e7e946be44f", size = 13945020, upload-time = "2026-09-02T22:41:15.968Z" }, - { url = "https://files.pythonhosted.org/packages/b8/f9/96aab1dee4535e66554e8dc7657c69f6c61c181d8e70b9c3527908e93ef4/ty-0.0.78-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:589ad03608d9d2975ef4b23c41c4f847e325bcc7686f51d4c66066b8dbc18a2a", size = 12851149, upload-time = "2026-09-02T22:41:18.06Z" }, - { url = "https://files.pythonhosted.org/packages/85/b0/53d8fe9a847534ef5fe2165c7d5292cb853b29dfd7011cbfb1cdb90a8012/ty-0.0.78-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b3fa0786edc1af06030f872d83499c0cea7c261dbbb651e0aee8306bf2c8a869", size = 13091464, upload-time = "2026-09-02T22:41:20.227Z" }, - { url = "https://files.pythonhosted.org/packages/21/65/94a4e5a02de559f6c6c0ee7a14b88660b4eee058ec6fec5c0ff6ecfbf5cc/ty-0.0.78-py3-none-musllinux_1_2_i686.whl", hash = "sha256:92c5639befc577578c8abd4e5a7fccf98d828db98b09e8d4dd81605dc0e54aef", size = 13389389, upload-time = "2026-09-02T22:41:22.27Z" }, - { url = "https://files.pythonhosted.org/packages/2e/b6/d0c7fe6be64ca5a15100c4b1f0e39c19660d53bc544f609fa117b346e661/ty-0.0.78-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:0dce70bc51652b2775debd1d1e422fb0179f5207c73deb4d5d0aca37e5f61695", size = 13691928, upload-time = "2026-09-02T22:41:24.292Z" }, - { url = "https://files.pythonhosted.org/packages/0a/5d/36081205611ba3fa802efc4ad63e4b1e949bda2f7bb37b34ac9bb6c00db0/ty-0.0.78-py3-none-win32.whl", hash = "sha256:1c80976ca9185d7a9d1baab1fb57240331b8dac58d3b11e46200284086a6de8d", size = 12647346, upload-time = "2026-09-02T22:41:26.699Z" }, - { url = "https://files.pythonhosted.org/packages/f1/89/b925fe1ea1bc56fc7f11d2496e29072fdd2ceb7b389884fc7b2d07cf0c41/ty-0.0.78-py3-none-win_amd64.whl", hash = "sha256:32e82b704471eab34f67b51c151660ca8a00815977b28278905d76fba54f7415", size = 13241810, upload-time = "2026-09-02T22:41:28.857Z" }, - { url = "https://files.pythonhosted.org/packages/91/f1/090ef7b52355bcedfbfbff6ce70fa81ba5bd5f6ed3f8d99ae05e6f2fe75b/ty-0.0.78-py3-none-win_arm64.whl", hash = "sha256:3a14d641a3c04fa9a80f2a46be1531d915f60d4fb79d4b894627bbe46bb35d64", size = 13077433, upload-time = "2026-09-02T22:41:31.525Z" }, +version = "0.0.79" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b5/d3/4fff47468a976c7a5ded9fe350734ca09b9e8460750327f2245ea3288d5d/ty-0.0.79.tar.gz", hash = "sha256:159a1aca70edebae32be08bfba2e5d543ffd8f9f380af160e0e713e85313b733", size = 7162150, upload-time = "2026-09-07T21:51:58.065Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/90/a3/2fa5b5495fe37351d77571eb0985476461be02c822bb39ad669b9fdb5bb8/ty-0.0.79-py3-none-linux_armv6l.whl", hash = "sha256:f60d968bbc6d52b663d3df4cae647d3cc68df4a7135bd9098e8d570bf9a5e0da", size = 13557478, upload-time = "2026-09-07T21:51:14.089Z" }, + { url = "https://files.pythonhosted.org/packages/f5/f0/1659a926d2b19351839ca64787e1531c97fbe8bf35c3c47d20954f338428/ty-0.0.79-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:4c95ae0ca67483c4f1231c4b2fd332b76f212e8d93154c385d77511d90dcbd8e", size = 13163780, upload-time = "2026-09-07T21:51:17.066Z" }, + { url = "https://files.pythonhosted.org/packages/e2/b2/e0a8a8cf58b39f0d1f509a22550aed1e2cc12bb4a868170b43aede9fcd93/ty-0.0.79-py3-none-macosx_11_0_arm64.whl", hash = "sha256:685888adb29b6e732ee325c6b6db92a422f72f3e4031c12450728c846e5e93fe", size = 12964040, upload-time = "2026-09-07T21:51:19.818Z" }, + { url = "https://files.pythonhosted.org/packages/88/a7/7d8fb6c958d88a63ad932421eff8848267bb9229136ef65d9f372aa2be7a/ty-0.0.79-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b0535640db5dc02f9e14bf419ad2cf3acdd49681e70f233c96e96d16cfcefc00", size = 13010025, upload-time = "2026-09-07T21:51:22.468Z" }, + { url = "https://files.pythonhosted.org/packages/e3/4c/94aee26446f058e67b2c8ef0b25bdc1cf555e40eb692112d7a609f30c238/ty-0.0.79-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:48cb24e21488f9d11cffa81d5a5f71f314c87383b7a5d28dabe6feaa3f3d7a34", size = 13332191, upload-time = "2026-09-07T21:51:25.031Z" }, + { url = "https://files.pythonhosted.org/packages/6c/21/9e7c0415fa6275500f10ce13f2bad62e03211ba5f3ea61a702d819328407/ty-0.0.79-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:1a35b3b116a591a68da5958c082247f0c3a50f134d0a60ff45d060eb3d2ad612", size = 14153955, upload-time = "2026-09-07T21:51:27.421Z" }, + { url = "https://files.pythonhosted.org/packages/0d/5f/b27b510f973a314200cfda4f0500de3c65f6caa45a590355f6acc5e96289/ty-0.0.79-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:fcb760f059d660dfd5c9c6c33609478a267c3185c7448d72e6f7ff4572796bae", size = 14594578, upload-time = "2026-09-07T21:51:30.345Z" }, + { url = "https://files.pythonhosted.org/packages/a6/fa/0ccc2e510c1c24682ae21c836c09be6dcf0b1aacab071c51a297d301601f/ty-0.0.79-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:fcdc1ae0c740f9b6d6536ffecf485616a8909bed5de88c6e870a65546a6a2d91", size = 14266409, upload-time = "2026-09-07T21:51:32.868Z" }, + { url = "https://files.pythonhosted.org/packages/e8/c3/3a404d44e9768578daf7dfda06cba714eda6feac53272eda295947a8d4c7/ty-0.0.79-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0c1e77d58e192c81b958783328b303ceda8706328a307c852ceb257505765458", size = 13658441, upload-time = "2026-09-07T21:51:35.283Z" }, + { url = "https://files.pythonhosted.org/packages/f6/23/b90a512055fdf6356e95c0f745cb9b443b9a4ac9d3a79f07ae6ed53ae9ca/ty-0.0.79-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:a87f52183b976d9eda5fad28405b37621a265c3de5e9e279660556c715ed1805", size = 14195669, upload-time = "2026-09-07T21:51:37.848Z" }, + { url = "https://files.pythonhosted.org/packages/b0/05/d051248dcff852e8a3165a2feb164dc5b857ef237fa82e7ea2b3b7bac375/ty-0.0.79-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:fb429a44bc9e90649e69f739d70db4a305102a36586a810caee00c984c379997", size = 13127929, upload-time = "2026-09-07T21:51:40.431Z" }, + { url = "https://files.pythonhosted.org/packages/5a/d1/60b5c980df55b1d5010a41b10d7c8cf6111b5d02ec0807f6f782bb3d88aa/ty-0.0.79-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:1d857fa34d60b0981cf12ca19180ae5de70dd46306a9277bee772bb09e82ae5c", size = 13334425, upload-time = "2026-09-07T21:51:42.956Z" }, + { url = "https://files.pythonhosted.org/packages/9b/3c/fcbfa731128f06b1c9be5074f51ce47da06842dc8f33c81e7c1c483b6e91/ty-0.0.79-py3-none-musllinux_1_2_i686.whl", hash = "sha256:a95cffa2f30289b0b949db7611ae96ede0352f0f1d7ba07ea7cabb27616fccb9", size = 13629671, upload-time = "2026-09-07T21:51:45.467Z" }, + { url = "https://files.pythonhosted.org/packages/0e/00/3004a1d375c2a835c7dbf01c480205a190f24b74c8b5c849fd2dbe078d69/ty-0.0.79-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:d33df0f1bdee62dc366551d35b9830b1fa2e9ee2936a437abd435f71b3fce73e", size = 13939769, upload-time = "2026-09-07T21:51:48.033Z" }, + { url = "https://files.pythonhosted.org/packages/10/7d/f357f5768872ffa3a026707477a880aba7582af5f6c7c7eab74f61167f3d/ty-0.0.79-py3-none-win32.whl", hash = "sha256:ca266de079a187ed6f0f5b802f6fc3b26164753c7347a04daed21d86c938d33d", size = 12833644, upload-time = "2026-09-07T21:51:50.766Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f5/2a2d967be6286ee8fb626161e610fd48624ba5e6712cea932adb2d6b40b6/ty-0.0.79-py3-none-win_amd64.whl", hash = "sha256:88cb357d36ad79181015581365769fff4b1ec580cd12de83addfa98663214fd3", size = 13494258, upload-time = "2026-09-07T21:51:53.105Z" }, + { url = "https://files.pythonhosted.org/packages/8b/74/49b88f104f88d9757497758bf57ba53e3612442b49500d1092d4d6f1daa3/ty-0.0.79-py3-none-win_arm64.whl", hash = "sha256:d29da73f2840ae2631bc62ef8f8d509a94edea596e5b3072851f67e4729d744c", size = 13327517, upload-time = "2026-09-07T21:51:55.602Z" }, +] + +[[package]] +name = "typing-extensions" +version = "4.16.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f6/cc/6253133b5bb138fc3306cebfbda2c520f545d36b5be2c7255cc528bb45d6/typing_extensions-4.16.0.tar.gz", hash = "sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5", size = 113555, upload-time = "2026-07-02T08:40:05.92Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/49/d3/b8441a820a491ddfc024b0b0cf0393375b75ea13866d9c66727e54c2fc80/typing_extensions-4.16.0-py3-none-any.whl", hash = "sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8", size = 45571, upload-time = "2026-07-02T08:40:04.659Z" }, ] [[package]] From 02d5933bd19e68068c4f7ebfb12ba2f87eaafb15 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 9 Sep 2026 12:32:11 +1200 Subject: [PATCH 03/25] Saving --- imdb/imdb.py | 127 ++++++++++++++++++++++------------------------ tests/conftest.py | 6 +-- 2 files changed, 65 insertions(+), 68 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index a01a340..0250049 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -49,6 +49,12 @@ def open(self) -> Self: self._con = ibis.duckdb.connect(self.path, read_only=self.read_only) return self + @property + def db_meta(self) -> dict[str, str]: + """The `db_meta` table, as a dict.""" + df = self.con.table("db_meta").to_pandas() + return dict(zip(df["key"], df["value"], strict=True)) + def close(self) -> None: """Close the database connection.""" if self._con is not None: @@ -63,8 +69,6 @@ def __exit__(self, *exc: object) -> None: """Close the database connection.""" self.close() - # ---- create ------------------------------------------------------- - @classmethod def create( cls, @@ -138,8 +142,6 @@ def create( ) return db - # ---- write helpers -------------------------------------------------- - def _next_ids(self, table: str, int_col: str, n: int) -> np.ndarray: """Return `n` new contiguous integer ids for `table`, starting after the current max.""" current = self.con.table(table)[int_col].max().to_pandas() @@ -151,8 +153,6 @@ def _id_map(self, table: str, id_col: str, int_col: str) -> pd.Series: df = self.con.table(table).select(id_col, int_col).to_pandas() return df.set_index(id_col)[int_col] - # ---- write ---------------------------------------------------------- - def add_events(self, df: pd.DataFrame) -> None: """Insert new events. @@ -210,48 +210,49 @@ def add_site_event(self, df: pd.DataFrame) -> None: df = df.drop(columns=["site_id", "event_id"]) self.con.insert("site_event", df) - ## TODO: CHange this to take a record_df (containing rel_id, site_id, component) + scalar IM columns, - # plus optional pSA and FAS numpy arrays. Update logic accordingly - def add_records(self, df: pd.DataFrame) -> np.ndarray: + def add_records( + self, + df: pd.DataFrame, + pSA: np.ndarray | None = None, # noqa: N803 + FAS: np.ndarray | None = None, # noqa: N803 + ) -> np.ndarray: """Insert new records, and whichever IM tables the input covers. - A row with a missing (`None`) `pSA` or `FAS` array is not written to that IM + A record whose `pSA`/`FAS` row is entirely NaN is not written to that IM table at all, matching the schema's row-presence-means-coverage convention. Parameters ---------- df : pd.DataFrame - Must have `rel_id`, `site_id` and `component` columns. May also have a - `pSA` column (list of float, one per period), a `FAS` column (list of - float, one per frequency), and any of the scalar IM columns - (`schema.SCALAR_IMS`). + Must have `rel_id`, `site_id` and `component` columns, one row per + record. May also have any of the scalar IM columns (`schema.SCALAR_IMS`). + pSA : np.ndarray, optional + Shape `(len(df), n_periods)`. A row of all NaN means that record has + no pSA. + FAS : np.ndarray, optional + Shape `(len(df), n_frequencies)`. A row of all NaN means that record + has no FAS. Returns ------- np.ndarray - The `record_id` assigned to each input row, in input order. + The `record_id` assigned to each row of `df`, in input order. """ df = df.copy() - - ## TODO: I think this should be a property? - components = set( - self.con.table("db_meta") - .filter(_.key == "components") - .to_pandas()["value"] - .iloc[0] - .split(",") - ) - unknown = set(df["component"]) - components + unknown = set(df["component"]) - set(self.db_meta["components"].split(",")) if unknown: raise ValueError(f"components not in this database: {sorted(unknown)}") - rel_int_id = self._id_map("realisations", "rel_id", "rel_int_id") - rel_event_int_id = self._id_map("realisations", "rel_int_id", "event_int_id") - site_int_id = self._id_map("sites", "site_id", "site_int_id") + assert pSA is None or pSA.shape[0] == len(df), "pSA must have one row per record" + assert FAS is None or FAS.shape[0] == len(df), "FAS must have one row per record" - df["rel_int_id"] = rel_int_id.loc[df["rel_id"]].to_numpy() - df["site_int_id"] = site_int_id.loc[df["site_id"]].to_numpy() - df["event_int_id"] = rel_event_int_id.loc[df["rel_int_id"]].to_numpy() + rel_int_id_mapping = self._id_map("realisations", "rel_id", "rel_int_id") + rel_event_int_id_mapping = self._id_map("realisations", "rel_int_id", "event_int_id") + site_int_id_mapping = self._id_map("sites", "site_id", "site_int_id") + + df["rel_int_id"] = rel_int_id_mapping.loc[df["rel_id"]].to_numpy() + df["site_int_id"] = site_int_id_mapping.loc[df["site_id"]].to_numpy() + df["event_int_id"] = rel_event_int_id_mapping.loc[df["rel_int_id"]].to_numpy() record_id = ( self.con.raw_sql(f"SELECT nextval('record_id_seq') FROM range({len(df)})") .df()["nextval('record_id_seq')"] @@ -273,13 +274,22 @@ def add_records(self, df: pd.DataFrame) -> np.ndarray: ] ], ) - if "pSA" in df: - # Why not just - mask = df["pSA"].apply(lambda x: x is not None) - self.con.insert("psa_ims", df.loc[mask, ["record_id", "pSA"]]) - if "FAS" in df: - mask = df["FAS"].apply(lambda x: x is not None) - self.con.insert("fas_ims", df.loc[mask, ["record_id", "FAS"]]) + if pSA is not None: + mask = ~np.isnan(pSA).all(axis=1) + self.con.insert( + "psa_ims", + pd.DataFrame( + {"record_id": record_id[mask], "pSA": list(pSA[mask])} + ), + ) + if FAS is not None: + mask = ~np.isnan(FAS).all(axis=1) + self.con.insert( + "fas_ims", + pd.DataFrame( + {"record_id": record_id[mask], "FAS": list(FAS[mask])} + ), + ) scalar_cols = [c for c in schema.SCALAR_IMS if c in df] if scalar_cols: scalars = df[["record_id", *scalar_cols]].copy() @@ -315,9 +325,8 @@ def delete_event(self, event_id: str) -> None: for statement in statements: self.con.raw_sql(statement, parameters=[event_id]) - ## TODO: Simplify, it should just check for basic stuff, e.g. that all integer ids are unique, fks are valid. Keep it simple. def validate(self) -> list[str]: - """Check the invariants the schema itself cannot enforce. + """Check the invariants the schema itself cannot enforce: unique ids, valid FKs. Returns ------- @@ -325,38 +334,26 @@ def validate(self) -> list[str]: One entry per problem found; empty if the database is consistent. """ problems = [] - records = self.con.table("records") - n_periods = len(self.con.table("periods").to_pandas()) - n_frequencies = len(self.con.table("frequencies").to_pandas()) + for table in ("records", "psa_ims", "fas_ims", "scalars_ims"): + t = self.con.table(table) + n_rows = t.count().to_pandas() + n_unique = t.record_id.nunique().to_pandas() + if n_rows != n_unique: + problems.append( + f"{table}: record_id is not unique ({n_rows} rows, {n_unique} unique)" + ) - orphans = { + records = self.con.table("records") + fks = { "rel_int_id": self.con.table("realisations").rel_int_id, "site_int_id": self.con.table("sites").site_int_id, "event_int_id": self.con.table("events").event_int_id, } - for col, valid in orphans.items(): + for col, valid in fks.items(): n = records.filter(~records[col].isin(valid)).count().to_pandas() if n: problems.append(f"records: {n} rows with an unknown {col}") - n = ( - self.con.table("psa_ims") - .filter(_.pSA.length() != n_periods) - .count() - .to_pandas() - ) - if n: - problems.append(f"psa_ims: {n} rows with pSA length != {n_periods}") - - n = ( - self.con.table("fas_ims") - .filter(_.FAS.length() != n_frequencies) - .count() - .to_pandas() - ) - if n: - problems.append(f"fas_ims: {n} rows with FAS length != {n_frequencies}") - return problems # ---- read ------------------------------------------------------------- @@ -457,8 +454,8 @@ def get_records( if component is not None: t = t.filter(t.component == component) if record_ids is not None: - ## QUESTION: How performant is this for a large number of record_ids? I previously had issues with ISIN (SQL) queries being slow. - t = t.filter(t.record_id.isin(record_ids)) + ids = ibis.memtable({"record_id": list(record_ids)}) + t = t.semi_join(ids, "record_id") return ( t.select("record_id", "event_id", "rel_id", "site_id", "component") .to_pandas() diff --git a/tests/conftest.py b/tests/conftest.py index e83c73b..4722d07 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -67,13 +67,13 @@ def db(tmp_path): "rel_id": [r[0] for r in rows], "site_id": [r[1] for r in rows], "component": [r[2] for r in rows], - "pSA": [rng.uniform(size=len(PERIODS)) for _ in range(n)], - "FAS": [rng.uniform(size=len(FREQUENCIES)) for _ in range(n)], "PGA": rng.uniform(size=n), "PGV": rng.uniform(size=n), "PGD": rng.uniform(size=n), } - ) + ), + pSA=rng.uniform(size=(n, len(PERIODS))), + FAS=rng.uniform(size=(n, len(FREQUENCIES))), ) yield db db.close() From 7e30e8ea0a3823aae44dd901ce8bacd0d2f54167 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 9 Sep 2026 12:36:12 +1200 Subject: [PATCH 04/25] Variable renaming --- imdb/imdb.py | 92 ++++++++++++++++++++++++++++---------------------- imdb/schema.py | 26 +++++++------- 2 files changed, 65 insertions(+), 53 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index 0250049..8cf819d 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -236,29 +236,37 @@ def add_records( Returns ------- np.ndarray - The `record_id` assigned to each row of `df`, in input order. + The `record_int_id` assigned to each row of `df`, in input order. """ df = df.copy() unknown = set(df["component"]) - set(self.db_meta["components"].split(",")) if unknown: raise ValueError(f"components not in this database: {sorted(unknown)}") - assert pSA is None or pSA.shape[0] == len(df), "pSA must have one row per record" - assert FAS is None or FAS.shape[0] == len(df), "FAS must have one row per record" + assert pSA is None or pSA.shape[0] == len(df), ( + "pSA must have one row per record" + ) + assert FAS is None or FAS.shape[0] == len(df), ( + "FAS must have one row per record" + ) rel_int_id_mapping = self._id_map("realisations", "rel_id", "rel_int_id") - rel_event_int_id_mapping = self._id_map("realisations", "rel_int_id", "event_int_id") + rel_event_int_id_mapping = self._id_map( + "realisations", "rel_int_id", "event_int_id" + ) site_int_id_mapping = self._id_map("sites", "site_id", "site_int_id") df["rel_int_id"] = rel_int_id_mapping.loc[df["rel_id"]].to_numpy() df["site_int_id"] = site_int_id_mapping.loc[df["site_id"]].to_numpy() df["event_int_id"] = rel_event_int_id_mapping.loc[df["rel_int_id"]].to_numpy() - record_id = ( - self.con.raw_sql(f"SELECT nextval('record_id_seq') FROM range({len(df)})") - .df()["nextval('record_id_seq')"] + record_int_id = ( + self.con.raw_sql( + f"SELECT nextval('record_int_id_seq') FROM range({len(df)})" + ) + .df()["nextval('record_int_id_seq')"] .to_numpy() ) - df["record_id"] = record_id + df["record_int_id"] = record_int_id self.con.raw_sql("BEGIN TRANSACTION") try: @@ -266,7 +274,7 @@ def add_records( "records", df[ [ - "record_id", + "record_int_id", "event_int_id", "rel_int_id", "site_int_id", @@ -279,7 +287,7 @@ def add_records( self.con.insert( "psa_ims", pd.DataFrame( - {"record_id": record_id[mask], "pSA": list(pSA[mask])} + {"record_int_id": record_int_id[mask], "pSA": list(pSA[mask])} ), ) if FAS is not None: @@ -287,12 +295,12 @@ def add_records( self.con.insert( "fas_ims", pd.DataFrame( - {"record_id": record_id[mask], "FAS": list(FAS[mask])} + {"record_int_id": record_int_id[mask], "FAS": list(FAS[mask])} ), ) scalar_cols = [c for c in schema.SCALAR_IMS if c in df] if scalar_cols: - scalars = df[["record_id", *scalar_cols]].copy() + scalars = df[["record_int_id", *scalar_cols]].copy() rotd = df["component"].str.startswith("rotd") for col in schema.ROTD_UNDEFINED & set(scalar_cols): scalars.loc[rotd, col] = None @@ -302,7 +310,7 @@ def add_records( raise self.con.raw_sql("COMMIT") logger.info("inserted %d records", len(df)) - return record_id + return record_int_id def delete_event(self, event_id: str) -> None: """Delete an event and everything derived from it, for a clean re-ingest. @@ -313,11 +321,11 @@ def delete_event(self, event_id: str) -> None: The event to delete. """ event_int_id_subquery = "(SELECT event_int_id FROM events WHERE event_id = ?)" - record_subquery = f"(SELECT record_id FROM records WHERE event_int_id = {event_int_id_subquery})" + record_subquery = f"(SELECT record_int_id FROM records WHERE event_int_id = {event_int_id_subquery})" statements = [ - f"DELETE FROM psa_ims WHERE record_id IN {record_subquery}", - f"DELETE FROM fas_ims WHERE record_id IN {record_subquery}", - f"DELETE FROM scalars_ims WHERE record_id IN {record_subquery}", + f"DELETE FROM psa_ims WHERE record_int_id IN {record_subquery}", + f"DELETE FROM fas_ims WHERE record_int_id IN {record_subquery}", + f"DELETE FROM scalars_ims WHERE record_int_id IN {record_subquery}", f"DELETE FROM records WHERE event_int_id = {event_int_id_subquery}", f"DELETE FROM site_event WHERE event_int_id = {event_int_id_subquery}", f"DELETE FROM realisations WHERE event_int_id = {event_int_id_subquery}", @@ -337,10 +345,10 @@ def validate(self) -> list[str]: for table in ("records", "psa_ims", "fas_ims", "scalars_ims"): t = self.con.table(table) n_rows = t.count().to_pandas() - n_unique = t.record_id.nunique().to_pandas() + n_unique = t.record_int_id.nunique().to_pandas() if n_rows != n_unique: problems.append( - f"{table}: record_id is not unique ({n_rows} rows, {n_unique} unique)" + f"{table}: record_int_id is not unique ({n_rows} rows, {n_unique} unique)" ) records = self.con.table("records") @@ -413,7 +421,7 @@ def get_records( rel_ids: list[str] | None = None, site_ids: list[str] | None = None, component: str | None = None, - record_ids: list[int] | None = None, + record_int_ids: list[int] | None = None, ) -> pd.DataFrame: """Return record identity rows. @@ -427,13 +435,13 @@ def get_records( Only records for these sites. component : str, optional Only records with this component. - record_ids : list of int, optional - Only these `record_id` values. + record_int_ids : list of int, optional + Only these `record_int_id` values. Returns ------- pd.DataFrame - Indexed by `record_id`, with `event_id`, `rel_id`, `site_id` and + Indexed by `record_int_id`, with `event_id`, `rel_id`, `site_id` and `component` columns. """ events = self.con.table("events").select("event_int_id", "event_id") @@ -453,13 +461,13 @@ def get_records( t = t.filter(t.site_id.isin(site_ids)) if component is not None: t = t.filter(t.component == component) - if record_ids is not None: - ids = ibis.memtable({"record_id": list(record_ids)}) - t = t.semi_join(ids, "record_id") + if record_int_ids is not None: + ids = ibis.memtable({"record_int_id": list(record_int_ids)}) + t = t.semi_join(ids, "record_int_id") return ( - t.select("record_id", "event_id", "rel_id", "site_id", "component") + t.select("record_int_id", "event_id", "rel_id", "site_id", "component") .to_pandas() - .set_index("record_id") + .set_index("record_int_id") ) def get_psa( @@ -477,16 +485,18 @@ def get_psa( Returns ------- pd.DataFrame - Indexed by `record_id`, one column per requested period. + Indexed by `record_int_id`, one column per requested period. """ grid = self.con.table("periods").to_pandas().set_index("period")["period_index"] if periods is None: periods = grid.index.tolist() - record_ids = self.get_records(**filters).index.tolist() - t = self.con.table("psa_ims").filter(_.record_id.isin(record_ids)) + record_int_ids = self.get_records(**filters).index.tolist() + t = self.con.table("psa_ims").filter(_.record_int_id.isin(record_int_ids)) cols = {str(p): t.pSA[int(grid.loc[p]) - 1] for p in periods} return ( - t.select(record_id=t.record_id, **cols).to_pandas().set_index("record_id") + t.select(record_int_id=t.record_int_id, **cols) + .to_pandas() + .set_index("record_int_id") ) def get_fas( @@ -504,7 +514,7 @@ def get_fas( Returns ------- pd.DataFrame - Indexed by `record_id`, one column per requested frequency. + Indexed by `record_int_id`, one column per requested frequency. """ grid = ( self.con.table("frequencies") @@ -513,11 +523,13 @@ def get_fas( ) if frequencies is None: frequencies = grid.index.tolist() - record_ids = self.get_records(**filters).index.tolist() - t = self.con.table("fas_ims").filter(_.record_id.isin(record_ids)) + record_int_ids = self.get_records(**filters).index.tolist() + t = self.con.table("fas_ims").filter(_.record_int_id.isin(record_int_ids)) cols = {str(f): t.FAS[int(grid.loc[f]) - 1] for f in frequencies} return ( - t.select(record_id=t.record_id, **cols).to_pandas().set_index("record_id") + t.select(record_int_id=t.record_int_id, **cols) + .to_pandas() + .set_index("record_int_id") ) def get_scalars(self, ims: list[str] | None = None, **filters: Any) -> pd.DataFrame: @@ -533,9 +545,9 @@ def get_scalars(self, ims: list[str] | None = None, **filters: Any) -> pd.DataFr Returns ------- pd.DataFrame - Indexed by `record_id`, one column per requested IM. + Indexed by `record_int_id`, one column per requested IM. """ ims = ims or list(schema.SCALAR_IMS) - record_ids = self.get_records(**filters).index.tolist() - t = self.con.table("scalars_ims").filter(_.record_id.isin(record_ids)) - return t.select("record_id", *ims).to_pandas().set_index("record_id") + record_int_ids = self.get_records(**filters).index.tolist() + t = self.con.table("scalars_ims").filter(_.record_int_id.isin(record_int_ids)) + return t.select("record_int_id", *ims).to_pandas().set_index("record_int_id") diff --git a/imdb/schema.py b/imdb/schema.py index 3e149dd..e5f4aa1 100644 --- a/imdb/schema.py +++ b/imdb/schema.py @@ -74,21 +74,21 @@ metadata VARCHAR ); -CREATE SEQUENCE record_id_seq START 1; +CREATE SEQUENCE record_int_id_seq START 1; CREATE TABLE records ( - record_id BIGINT DEFAULT nextval('record_id_seq'), - event_int_id INTEGER NOT NULL, - rel_int_id INTEGER NOT NULL, - site_int_id INTEGER NOT NULL, - component VARCHAR NOT NULL + record_int_id BIGINT DEFAULT nextval('record_int_id_seq'), + event_int_id INTEGER NOT NULL, + rel_int_id INTEGER NOT NULL, + site_int_id INTEGER NOT NULL, + component VARCHAR NOT NULL ); -CREATE TABLE psa_ims (record_id BIGINT NOT NULL, pSA FLOAT[]); -CREATE TABLE fas_ims (record_id BIGINT NOT NULL, FAS FLOAT[]); +CREATE TABLE psa_ims (record_int_id BIGINT NOT NULL, pSA FLOAT[]); +CREATE TABLE fas_ims (record_int_id BIGINT NOT NULL, FAS FLOAT[]); CREATE TABLE scalars_ims ( - record_id BIGINT NOT NULL, + record_int_id BIGINT NOT NULL, PGA FLOAT, PGV FLOAT, PGD FLOAT, @@ -139,7 +139,7 @@ ), "identity and rebuild stability": ( "event_id, rel_id and site_id are stable. The integer surrogates event_int_id, " - "rel_int_id, site_int_id and record_id are assigned at ingest and change on rebuild; " + "rel_int_id, site_int_id and record_int_id are assigned at ingest and change on rebuild; " "nothing outside the database may reference them. External references use " "(rel_id, site_id, component)." ), @@ -172,12 +172,12 @@ "Every event has at least one realisation. A dataset with no realisation concept " "gets exactly one per event, with rel_id equal to event_id." ), - "record_id is file-local": ( - "record_id is a surrogate scoped to this database file only; it is not stable " + "record_int_id is file-local": ( + "record_int_id is a surrogate scoped to this database file only; it is not stable " "across rebuilds or between databases." ), "physical sort order": ( - "Rows are written in event order, so event_int_id and record_id are monotonic and " + "Rows are written in event order, so event_int_id and record_int_id are monotonic and " "event filters prune row groups. Site filters do not prune. db_meta.sort_order " "records the ordering used." ), From f3bf806b1ce7948c03fe5b39b38afde4e6310c52 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 9 Sep 2026 12:51:12 +1200 Subject: [PATCH 05/25] Saving --- imdb/imdb.py | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index 8cf819d..901c078 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -487,12 +487,14 @@ def get_psa( pd.DataFrame Indexed by `record_int_id`, one column per requested period. """ - grid = self.con.table("periods").to_pandas().set_index("period")["period_index"] + period_index = ( + self.con.table("periods").to_pandas().set_index("period")["period_index"] + ) if periods is None: - periods = grid.index.tolist() + periods = period_index.index.tolist() record_int_ids = self.get_records(**filters).index.tolist() t = self.con.table("psa_ims").filter(_.record_int_id.isin(record_int_ids)) - cols = {str(p): t.pSA[int(grid.loc[p]) - 1] for p in periods} + cols = {str(p): t.pSA[int(period_index.loc[p]) - 1] for p in periods} return ( t.select(record_int_id=t.record_int_id, **cols) .to_pandas() @@ -516,16 +518,16 @@ def get_fas( pd.DataFrame Indexed by `record_int_id`, one column per requested frequency. """ - grid = ( + freq_index = ( self.con.table("frequencies") .to_pandas() .set_index("frequency")["freq_index"] ) if frequencies is None: - frequencies = grid.index.tolist() + frequencies = freq_index.index.tolist() record_int_ids = self.get_records(**filters).index.tolist() t = self.con.table("fas_ims").filter(_.record_int_id.isin(record_int_ids)) - cols = {str(f): t.FAS[int(grid.loc[f]) - 1] for f in frequencies} + cols = {str(f): t.FAS[int(freq_index.loc[f]) - 1] for f in frequencies} return ( t.select(record_int_id=t.record_int_id, **cols) .to_pandas() From 867678f921b88edf8b704aff262262d5b3640218 Mon Sep 17 00:00:00 2001 From: Claudio Date: Thu, 10 Sep 2026 08:09:58 +1200 Subject: [PATCH 06/25] GMM support --- imdb/imdb.py | 131 ++++++++++++++++++++++++++++++++++++++------- imdb/schema.py | 50 ++++++++++++----- tests/test_imdb.py | 31 +++++++++++ 3 files changed, 179 insertions(+), 33 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index 901c078..5512bc4 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -215,6 +215,8 @@ def add_records( df: pd.DataFrame, pSA: np.ndarray | None = None, # noqa: N803 FAS: np.ndarray | None = None, # noqa: N803 + pSA_sigma: np.ndarray | None = None, # noqa: N803 + FAS_sigma: np.ndarray | None = None, # noqa: N803 ) -> np.ndarray: """Insert new records, and whichever IM tables the input covers. @@ -225,13 +227,22 @@ def add_records( ---------- df : pd.DataFrame Must have `rel_id`, `site_id` and `component` columns, one row per - record. May also have any of the scalar IM columns (`schema.SCALAR_IMS`). + record. May also have a `kind` column (default `"simulated"`), a + `gmm_key` column (default `NULL`, only meaningful for `kind="gmm"`), + any of the scalar IM columns (`schema.SCALAR_IMS`), and their paired + `_sigma` columns. pSA : np.ndarray, optional Shape `(len(df), n_periods)`. A row of all NaN means that record has no pSA. FAS : np.ndarray, optional Shape `(len(df), n_frequencies)`. A row of all NaN means that record has no FAS. + pSA_sigma : np.ndarray, optional + Shape `(len(df), n_periods)`, ln-space total sigma paired with `pSA`. + Only valid together with `pSA`. + FAS_sigma : np.ndarray, optional + Shape `(len(df), n_frequencies)`, ln-space total sigma paired with + `FAS`. Only valid together with `FAS`. Returns ------- @@ -239,6 +250,13 @@ def add_records( The `record_int_id` assigned to each row of `df`, in input order. """ df = df.copy() + + ## TODO: Make kind a required function argument. + if "kind" not in df: + df["kind"] = "simulated" + if "gmm_key" not in df: + df["gmm_key"] = None + unknown = set(df["component"]) - set(self.db_meta["components"].split(",")) if unknown: raise ValueError(f"components not in this database: {sorted(unknown)}") @@ -249,6 +267,12 @@ def add_records( assert FAS is None or FAS.shape[0] == len(df), ( "FAS must have one row per record" ) + assert pSA_sigma is None or (pSA is not None and pSA_sigma.shape == pSA.shape), ( + "pSA_sigma must match pSA's shape, and only be given together with pSA" + ) + assert FAS_sigma is None or (FAS is not None and FAS_sigma.shape == FAS.shape), ( + "FAS_sigma must match FAS's shape, and only be given together with FAS" + ) rel_int_id_mapping = self._id_map("realisations", "rel_id", "rel_int_id") rel_event_int_id_mapping = self._id_map( @@ -279,31 +303,38 @@ def add_records( "rel_int_id", "site_int_id", "component", + "kind", + "gmm_key", ] ], ) if pSA is not None: mask = ~np.isnan(pSA).all(axis=1) - self.con.insert( - "psa_ims", - pd.DataFrame( - {"record_int_id": record_int_id[mask], "pSA": list(pSA[mask])} - ), + psa_df = pd.DataFrame( + {"record_int_id": record_int_id[mask], "pSA": list(pSA[mask])} + ) + psa_df["pSA_sigma"] = ( + list(pSA_sigma[mask]) if pSA_sigma is not None else None ) + self.con.insert("psa_ims", psa_df) if FAS is not None: mask = ~np.isnan(FAS).all(axis=1) - self.con.insert( - "fas_ims", - pd.DataFrame( - {"record_int_id": record_int_id[mask], "FAS": list(FAS[mask])} - ), + fas_df = pd.DataFrame( + {"record_int_id": record_int_id[mask], "FAS": list(FAS[mask])} ) + fas_df["FAS_sigma"] = ( + list(FAS_sigma[mask]) if FAS_sigma is not None else None + ) + self.con.insert("fas_ims", fas_df) scalar_cols = [c for c in schema.SCALAR_IMS if c in df] + sigma_cols = [f"{c}_sigma" for c in scalar_cols if f"{c}_sigma" in df] if scalar_cols: - scalars = df[["record_int_id", *scalar_cols]].copy() + scalars = df[["record_int_id", *scalar_cols, *sigma_cols]].copy() rotd = df["component"].str.startswith("rotd") for col in schema.ROTD_UNDEFINED & set(scalar_cols): scalars.loc[rotd, col] = None + if f"{col}_sigma" in scalars: + scalars.loc[rotd, f"{col}_sigma"] = None self.con.insert("scalars_ims", scalars) except Exception: self.con.raw_sql("ROLLBACK") @@ -362,6 +393,19 @@ def validate(self) -> list[str]: if n: problems.append(f"records: {n} rows with an unknown {col}") + bad_gmm_key = ( + records.filter( + ((records.kind == "gmm") & records.gmm_key.isnull()) + | ((records.kind != "gmm") & records.gmm_key.notnull()) + ) + .count() + .to_pandas() + ) + if bad_gmm_key: + problems.append( + f"records: {bad_gmm_key} rows with gmm_key inconsistent with kind" + ) + return problems # ---- read ------------------------------------------------------------- @@ -421,6 +465,8 @@ def get_records( rel_ids: list[str] | None = None, site_ids: list[str] | None = None, component: str | None = None, + kind: str | None = None, + gmm_key: str | None = None, record_int_ids: list[int] | None = None, ) -> pd.DataFrame: """Return record identity rows. @@ -435,14 +481,18 @@ def get_records( Only records for these sites. component : str, optional Only records with this component. + kind : str, optional + Only records of this kind (`"simulated"`, `"gmm"` or `"observed"`). + gmm_key : str, optional + Only records from this GMM. record_int_ids : list of int, optional Only these `record_int_id` values. Returns ------- pd.DataFrame - Indexed by `record_int_id`, with `event_id`, `rel_id`, `site_id` and - `component` columns. + Indexed by `record_int_id`, with `event_id`, `rel_id`, `site_id`, + `component`, `kind` and `gmm_key` columns. """ events = self.con.table("events").select("event_int_id", "event_id") realisations = self.con.table("realisations").select("rel_int_id", "rel_id") @@ -461,17 +511,32 @@ def get_records( t = t.filter(t.site_id.isin(site_ids)) if component is not None: t = t.filter(t.component == component) + if kind is not None: + t = t.filter(t.kind == kind) + if gmm_key is not None: + t = t.filter(t.gmm_key == gmm_key) if record_int_ids is not None: ids = ibis.memtable({"record_int_id": list(record_int_ids)}) t = t.semi_join(ids, "record_int_id") return ( - t.select("record_int_id", "event_id", "rel_id", "site_id", "component") + t.select( + "record_int_id", + "event_id", + "rel_id", + "site_id", + "component", + "kind", + "gmm_key", + ) .to_pandas() .set_index("record_int_id") ) def get_psa( - self, periods: list[float] | None = None, **filters: Any + self, + periods: list[float] | None = None, + sigma: bool = False, + **filters: Any, ) -> pd.DataFrame: """Return response spectral acceleration. @@ -479,6 +544,9 @@ def get_psa( ---------- periods : list of float, optional Periods to return, in seconds. Defaults to every period on the grid. + sigma : bool + Also return each period's ln-space total sigma, as `_sigma` + columns. **filters Passed to `get_records` to select which records to return. @@ -495,6 +563,13 @@ def get_psa( record_int_ids = self.get_records(**filters).index.tolist() t = self.con.table("psa_ims").filter(_.record_int_id.isin(record_int_ids)) cols = {str(p): t.pSA[int(period_index.loc[p]) - 1] for p in periods} + if sigma: + cols.update( + { + f"{p}_sigma": t.pSA_sigma[int(period_index.loc[p]) - 1] + for p in periods + } + ) return ( t.select(record_int_id=t.record_int_id, **cols) .to_pandas() @@ -502,7 +577,10 @@ def get_psa( ) def get_fas( - self, frequencies: list[float] | None = None, **filters: Any + self, + frequencies: list[float] | None = None, + sigma: bool = False, + **filters: Any, ) -> pd.DataFrame: """Return Fourier amplitude spectra. @@ -510,6 +588,9 @@ def get_fas( ---------- frequencies : list of float, optional Frequencies to return, in Hz. Defaults to every frequency on the grid. + sigma : bool + Also return each frequency's ln-space total sigma, as `_sigma` + columns. **filters Passed to `get_records` to select which records to return. @@ -528,19 +609,30 @@ def get_fas( record_int_ids = self.get_records(**filters).index.tolist() t = self.con.table("fas_ims").filter(_.record_int_id.isin(record_int_ids)) cols = {str(f): t.FAS[int(freq_index.loc[f]) - 1] for f in frequencies} + if sigma: + cols.update( + { + f"{f}_sigma": t.FAS_sigma[int(freq_index.loc[f]) - 1] + for f in frequencies + } + ) return ( t.select(record_int_id=t.record_int_id, **cols) .to_pandas() .set_index("record_int_id") ) - def get_scalars(self, ims: list[str] | None = None, **filters: Any) -> pd.DataFrame: + def get_scalars( + self, ims: list[str] | None = None, sigma: bool = False, **filters: Any + ) -> pd.DataFrame: """Return scalar intensity measures. Parameters ---------- ims : list of str, optional Which scalar IMs to return. Defaults to all of `schema.SCALAR_IMS`. + sigma : bool + Also return each IM's ln-space total sigma, as `_sigma` columns. **filters Passed to `get_records` to select which records to return. @@ -550,6 +642,7 @@ def get_scalars(self, ims: list[str] | None = None, **filters: Any) -> pd.DataFr Indexed by `record_int_id`, one column per requested IM. """ ims = ims or list(schema.SCALAR_IMS) + cols = [*ims, *([f"{im}_sigma" for im in ims] if sigma else [])] record_int_ids = self.get_records(**filters).index.tolist() t = self.con.table("scalars_ims").filter(_.record_int_id.isin(record_int_ids)) - return t.select("record_int_id", *ims).to_pandas().set_index("record_int_id") + return t.select("record_int_id", *cols).to_pandas().set_index("record_int_id") diff --git a/imdb/schema.py b/imdb/schema.py index e5f4aa1..69d123b 100644 --- a/imdb/schema.py +++ b/imdb/schema.py @@ -25,6 +25,8 @@ 'ACTIVE_SHALLOW', 'VOLCANIC', 'SUBDUCTION_INTERFACE', 'SUBDUCTION_SLAB' ); +CREATE TYPE record_kind_t AS ENUM ('simulated', 'gmm', 'observed'); + CREATE TABLE events ( event_int_id INTEGER PRIMARY KEY, event_id VARCHAR NOT NULL UNIQUE, @@ -81,21 +83,30 @@ event_int_id INTEGER NOT NULL, rel_int_id INTEGER NOT NULL, site_int_id INTEGER NOT NULL, - component VARCHAR NOT NULL + component VARCHAR NOT NULL, + kind record_kind_t NOT NULL DEFAULT 'simulated', + gmm_key VARCHAR ); -CREATE TABLE psa_ims (record_int_id BIGINT NOT NULL, pSA FLOAT[]); -CREATE TABLE fas_ims (record_int_id BIGINT NOT NULL, FAS FLOAT[]); +CREATE TABLE psa_ims (record_int_id BIGINT NOT NULL, pSA FLOAT[], pSA_sigma FLOAT[]); +CREATE TABLE fas_ims (record_int_id BIGINT NOT NULL, FAS FLOAT[], FAS_sigma FLOAT[]); CREATE TABLE scalars_ims ( record_int_id BIGINT NOT NULL, - PGA FLOAT, - PGV FLOAT, - PGD FLOAT, - CAV FLOAT, - AI FLOAT, - Ds575 FLOAT, - Ds595 FLOAT + PGA FLOAT, + PGA_sigma FLOAT, + PGV FLOAT, + PGV_sigma FLOAT, + PGD FLOAT, + PGD_sigma FLOAT, + CAV FLOAT, + CAV_sigma FLOAT, + AI FLOAT, + AI_sigma FLOAT, + Ds575 FLOAT, + Ds575_sigma FLOAT, + Ds595 FLOAT, + Ds595_sigma FLOAT ); """ @@ -129,8 +140,16 @@ NOTES = { "logical keys": ( "site_event's logical key is (site_int_id, event_int_id); records' logical key is " - "(rel_int_id, site_int_id, component). Neither is enforced by a constraint; the " - "writer is responsible for not creating duplicates." + "(rel_int_id, site_int_id, component, kind, gmm_key). Neither is enforced by a " + "constraint; the writer is responsible for not creating duplicates." + ), + "record kind": ( + "records.kind is 'simulated' (physics-based simulation), 'gmm' (empirical " + "ground-motion model prediction) or 'observed' (real recorded ground motion). " + "gmm_key identifies the model, e.g. 'Bradley_2013'; NULL unless kind = 'gmm', and " + "not validated against a fixed vocabulary. A gmm/observed record still needs a " + "rel_int_id: for a fault with no per-realisation concept, that is the event's " + "synthetic realisation." ), "array indexing is 1-based": ( "periods.period_index and frequencies.freq_index are 1-based, matching DuckDB list " @@ -141,7 +160,7 @@ "event_id, rel_id and site_id are stable. The integer surrogates event_int_id, " "rel_int_id, site_int_id and record_int_id are assigned at ingest and change on rebuild; " "nothing outside the database may reference them. External references use " - "(rel_id, site_id, component)." + "(rel_id, site_id, component, kind, gmm_key)." ), "component vocabulary": ( "000, 090, ver, geom, rotd0, rotd50, rotd100, following IM_calculation. A database " @@ -155,7 +174,10 @@ "units": ( "Linear, physical units; log is a read-time transform. Every IM unit is a row in " "im_units. Outside the IM tables: distances km, vs30 m/s, z1p0 and z2p5 km, depths " - "km, angles degrees, coordinates WGS84." + "km, angles degrees, coordinates WGS84. Every *_sigma column/array is the " + "exception: a ln-space (natural-log) total standard deviation, dimensionless, not " + "the linear physical unit of its paired IM column. Populated for kind = 'gmm', " + "NULL for 'simulated'/'observed'." ), "metadata columns are JSON": ( "events, realisations, sites and site_event each carry a metadata VARCHAR holding a " diff --git a/tests/test_imdb.py b/tests/test_imdb.py index fc1b99e..24f5a27 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -1,5 +1,7 @@ """Basic tests: the library works for its intended, correct usage.""" +import pandas as pd + from tests.conftest import COMPONENTS, EVENTS, FREQUENCIES, PERIODS, SITES @@ -35,3 +37,32 @@ def test_delete_and_readd(db): assert len(db.get_records(event_ids=["eventA"])) == 0 assert len(db.get_realisations()) == 2 assert db.validate() == [] + + +def test_gmm_records(db): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["gmm"], + "gmm_key": ["TestGMM2020"], + "PGA": [0.5], + "PGA_sigma": [0.6], + } + ) + ) + + assert db.validate() == [] + + gmm_records = db.get_records(kind="gmm") + assert len(gmm_records) == 1 + assert gmm_records["gmm_key"].iloc[0] == "TestGMM2020" + + scalars = db.get_scalars(ims=["PGA"], sigma=True, kind="gmm") + assert scalars["PGA"].iloc[0] == 0.5 + assert scalars["PGA_sigma"].iloc[0] == 0.6 + + simulated_records = db.get_records(kind="simulated") + assert (simulated_records["gmm_key"].isna()).all() From d5be258ee63e273cc26798d74ae3648ccb83e281 Mon Sep 17 00:00:00 2001 From: Claudio Date: Thu, 10 Sep 2026 09:07:14 +1200 Subject: [PATCH 07/25] Code review comments --- README.md | 2 +- imdb/imdb.py | 71 +++++++++++++++++++++++++++++++++------------- imdb/schema.py | 2 +- tests/conftest.py | 1 + tests/test_imdb.py | 37 ++++++++++++++++++++++++ 5 files changed, 92 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index fa30faa..abbfb73 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # imdb -A library for reading and writing intensity measure databases (IMDBs) — DuckDB +A library for reading and writing intensity measure databases (IMDBs), DuckDB databases of simulated ground-motion intensity measures. Schema is documented in `imdb/schema.py`. diff --git a/imdb/imdb.py b/imdb/imdb.py index 5512bc4..dd86b27 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -47,6 +47,15 @@ def open(self) -> Self: """Open the database connection, if not already open.""" if self._con is None: self._con = ibis.duckdb.connect(self.path, read_only=self.read_only) + if "db_meta" in self._con.list_tables(): + found = self.db_meta.get("schema_version") + if found != schema.SCHEMA_VERSION: + self.close() + raise RuntimeError( + f"database schema version {found!r} does not match " + f"this imdb version's {schema.SCHEMA_VERSION!r}; " + "the database needs rebuilding with a matching imdb version" + ) return self @property @@ -226,11 +235,10 @@ def add_records( Parameters ---------- df : pd.DataFrame - Must have `rel_id`, `site_id` and `component` columns, one row per - record. May also have a `kind` column (default `"simulated"`), a - `gmm_key` column (default `NULL`, only meaningful for `kind="gmm"`), - any of the scalar IM columns (`schema.SCALAR_IMS`), and their paired - `_sigma` columns. + Must have `rel_id`, `site_id`, `component` and `kind` columns, one + row per record. May also have a `gmm_key` column (default `NULL`, + only meaningful for `kind="gmm"`), any of the scalar IM columns + (`schema.SCALAR_IMS`), and their paired `_sigma` columns. pSA : np.ndarray, optional Shape `(len(df), n_periods)`. A row of all NaN means that record has no pSA. @@ -251,9 +259,10 @@ def add_records( """ df = df.copy() - ## TODO: Make kind a required function argument. if "kind" not in df: - df["kind"] = "simulated" + raise ValueError( + "df must have a 'kind' column ('simulated', 'gmm' or 'observed')" + ) if "gmm_key" not in df: df["gmm_key"] = None @@ -261,18 +270,18 @@ def add_records( if unknown: raise ValueError(f"components not in this database: {sorted(unknown)}") - assert pSA is None or pSA.shape[0] == len(df), ( - "pSA must have one row per record" - ) - assert FAS is None or FAS.shape[0] == len(df), ( - "FAS must have one row per record" - ) - assert pSA_sigma is None or (pSA is not None and pSA_sigma.shape == pSA.shape), ( - "pSA_sigma must match pSA's shape, and only be given together with pSA" - ) - assert FAS_sigma is None or (FAS is not None and FAS_sigma.shape == FAS.shape), ( - "FAS_sigma must match FAS's shape, and only be given together with FAS" - ) + if pSA is not None and pSA.shape[0] != len(df): + raise ValueError("pSA must have one row per record") + if FAS is not None and FAS.shape[0] != len(df): + raise ValueError("FAS must have one row per record") + if pSA_sigma is not None and not (pSA is not None and pSA_sigma.shape == pSA.shape): + raise ValueError( + "pSA_sigma must match pSA's shape, and only be given together with pSA" + ) + if FAS_sigma is not None and not (FAS is not None and FAS_sigma.shape == FAS.shape): + raise ValueError( + "FAS_sigma must match FAS's shape, and only be given together with FAS" + ) rel_int_id_mapping = self._id_map("realisations", "rel_id", "rel_int_id") rel_event_int_id_mapping = self._id_map( @@ -360,6 +369,8 @@ def delete_event(self, event_id: str) -> None: f"DELETE FROM records WHERE event_int_id = {event_int_id_subquery}", f"DELETE FROM site_event WHERE event_int_id = {event_int_id_subquery}", f"DELETE FROM realisations WHERE event_int_id = {event_int_id_subquery}", + ## QUESTION:Why is this one a different style than the other deletes? Why not f string? + "DELETE FROM events WHERE event_id = ?", ] for statement in statements: self.con.raw_sql(statement, parameters=[event_id]) @@ -393,6 +404,12 @@ def validate(self) -> list[str]: if n: problems.append(f"records: {n} rows with an unknown {col}") + for table in ("psa_ims", "fas_ims", "scalars_ims"): + t = self.con.table(table) + n = t.filter(~t.record_int_id.isin(records.record_int_id)).count().to_pandas() + if n: + problems.append(f"{table}: {n} rows with an unknown record_int_id") + bad_gmm_key = ( records.filter( ((records.kind == "gmm") & records.gmm_key.isnull()) @@ -406,6 +423,22 @@ def validate(self) -> list[str]: f"records: {bad_gmm_key} rows with gmm_key inconsistent with kind" ) + kind_by_record = records.select("record_int_id", "kind") + sigma_cols = [ + ("psa_ims", "pSA_sigma"), + ("fas_ims", "FAS_sigma"), + *[("scalars_ims", f"{im}_sigma") for im in schema.SCALAR_IMS], + ] + for table, col in sigma_cols: + joined = self.con.table(table).join(kind_by_record, "record_int_id") + n = ( + joined.filter((joined.kind != "gmm") & joined[col].notnull()) + .count() + .to_pandas() + ) + if n: + problems.append(f"{table}.{col}: {n} rows populated for kind != 'gmm'") + return problems # ---- read ------------------------------------------------------------- diff --git a/imdb/schema.py b/imdb/schema.py index 69d123b..4958c58 100644 --- a/imdb/schema.py +++ b/imdb/schema.py @@ -4,7 +4,7 @@ else in this library composes column lists by hand. """ -SCHEMA_VERSION = "0" +SCHEMA_VERSION = "1" DDL = """ CREATE TABLE db_meta (key VARCHAR PRIMARY KEY, value VARCHAR NOT NULL); diff --git a/tests/conftest.py b/tests/conftest.py index 4722d07..1b4462e 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -67,6 +67,7 @@ def db(tmp_path): "rel_id": [r[0] for r in rows], "site_id": [r[1] for r in rows], "component": [r[2] for r in rows], + "kind": "simulated", "PGA": rng.uniform(size=n), "PGV": rng.uniform(size=n), "PGD": rng.uniform(size=n), diff --git a/tests/test_imdb.py b/tests/test_imdb.py index 24f5a27..b85cb26 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -1,7 +1,10 @@ """Basic tests: the library works for its intended, correct usage.""" +import ibis import pandas as pd +import pytest +from imdb import IMDB from tests.conftest import COMPONENTS, EVENTS, FREQUENCIES, PERIODS, SITES @@ -37,6 +40,10 @@ def test_delete_and_readd(db): assert len(db.get_records(event_ids=["eventA"])) == 0 assert len(db.get_realisations()) == 2 assert db.validate() == [] + assert "eventA" not in db.get_events().index + + db.add_events(pd.DataFrame({"event_id": ["eventA"], "magnitude": [6.0]})) + assert "eventA" in db.get_events().index def test_gmm_records(db): @@ -66,3 +73,33 @@ def test_gmm_records(db): simulated_records = db.get_records(kind="simulated") assert (simulated_records["gmm_key"].isna()).all() + + +def test_schema_version_mismatch_rejected(db): + path = db.path + db.close() + + con = ibis.duckdb.connect(path, read_only=False) + con.raw_sql("UPDATE db_meta SET value = 'stale' WHERE key = 'schema_version'") + con.disconnect() + + with pytest.raises(RuntimeError, match="schema version"): + IMDB(path, read_only=True).open() + + +def test_validate_catches_sigma_on_non_gmm_record(db): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["simulated"], + "PGA": [0.5], + "PGA_sigma": [0.6], + } + ) + ) + + problems = db.validate() + assert any("scalars_ims.PGA_sigma" in p for p in problems) From ffe5145193115807e5bc51713995d26773e6dc65 Mon Sep 17 00:00:00 2001 From: Claudio Date: Thu, 10 Sep 2026 09:16:48 +1200 Subject: [PATCH 08/25] Additional tests --- tests/test_imdb.py | 53 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/tests/test_imdb.py b/tests/test_imdb.py index b85cb26..f1b192e 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -1,6 +1,7 @@ """Basic tests: the library works for its intended, correct usage.""" import ibis +import numpy as np import pandas as pd import pytest @@ -103,3 +104,55 @@ def test_validate_catches_sigma_on_non_gmm_record(db): problems = db.validate() assert any("scalars_ims.PGA_sigma" in p for p in problems) + + +def test_all_nan_row_not_written_to_psa(db): + record_int_ids = db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0", "eventA_rel0"], + "site_id": ["siteA", "siteB"], + "component": ["000", "000"], + "kind": ["simulated", "simulated"], + } + ), + pSA=np.array([[1.0] * len(PERIODS), [np.nan] * len(PERIODS)]), + ) + + psa = db.get_psa(record_int_ids=list(record_int_ids)) + assert list(psa.index) == [record_int_ids[0]] + + +def test_get_site_event_filters(db): + by_event = db.get_site_event(event_ids=["eventA"]) + assert (by_event["event_id"] == "eventA").all() + assert len(by_event) == len(SITES) + + by_rrup = db.get_site_event(max_rrup=1.5) + assert (by_rrup["rrup"] <= 1.5).all() + assert len(by_rrup) == 2 + + +def test_rotd_component_nulls_undefined_scalars(tmp_path): + db = IMDB.create(tmp_path / "rotd.duckdb", periods=[], components=("rotd50",)) + db.add_events(pd.DataFrame({"event_id": ["e1"]})) + db.add_realisations(pd.DataFrame({"rel_id": ["r1"], "event_id": ["e1"]})) + db.add_sites(pd.DataFrame({"site_id": ["s1"], "lat": [-43.5], "lon": [172.6]})) + + db.add_records( + pd.DataFrame( + { + "rel_id": ["r1"], + "site_id": ["s1"], + "component": ["rotd50"], + "kind": ["simulated"], + "PGA": [0.5], + "CAV": [1.2], + } + ) + ) + + scalars = db.get_scalars(ims=["PGA", "CAV"]) + assert scalars["PGA"].iloc[0] == 0.5 + assert pd.isna(scalars["CAV"].iloc[0]) + db.close() From 5c65b8681d5ebc734d9074f04b65034d322acbe3 Mon Sep 17 00:00:00 2001 From: Claudio Date: Thu, 10 Sep 2026 09:27:00 +1200 Subject: [PATCH 09/25] readme update --- README.md | 136 ++++++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 132 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index abbfb73..6675fa3 100644 --- a/README.md +++ b/README.md @@ -1,8 +1,136 @@ -# imdb +# IMDB A library for reading and writing intensity measure databases (IMDBs), DuckDB -databases of simulated ground-motion intensity measures. Schema is documented in -`imdb/schema.py`. +databases of intensity measures (IMs) from physics-based ground-motion simulation, +empirical ground-motion model (GMM) prediction, and observed ground motion. + +One database per run set. Every database uses the same schema unchanged and is +self-contained. A database may mix `kind`s of record freely, distinguished per row. Schema is documented in full in `imdb/schema.py` (the single source of truth for the DDL); this README summarises it. + +## Schema + +Thirteen tables: four dimensions (`events`, `realisations`, `sites`, `site_event`), +one identity table (`records`), three IM tables (`psa_ims`, `fas_ims`, +`scalars_ims`), two IM vocabulary tables (`periods`, `frequencies`), and three +documentation tables (`db_meta`, `im_units`, `notes`). + +A ground motion is identified by `(rel_id, site_id, component, kind, gmm_key)`. +Response spectra and Fourier spectra are stored as one array per record; scalar IMs +as named columns. Every IM column has a paired `_sigma` column/array (ln-space +total standard deviation), populated for `kind = "gmm"` records and NULL for +`"simulated"`/`"observed"`. + +```mermaid +erDiagram + events ||--o{ realisations : "FK, declared" + events ||--o{ site_event : "logical" + sites ||--o{ site_event : "logical" + realisations ||--o{ records : "logical" + sites ||--o{ records : "logical" + records ||--o| psa_ims : "record_int_id" + records ||--o| fas_ims : "record_int_id" + records ||--o| scalars_ims : "record_int_id" + periods ||--o{ psa_ims : "period_index indexes pSA[]" + frequencies ||--o{ fas_ims : "freq_index indexes FAS[]" + + events { + INTEGER event_int_id PK + VARCHAR event_id UK "stable identity" + FLOAT magnitude + tect_type_t tect_type "ENUM, 4 values" + VARCHAR metadata "JSON" + } + + realisations { + INTEGER rel_int_id PK + VARCHAR rel_id UK "stable identity" + INTEGER event_int_id FK + FLOAT magnitude + FLOAT rake + VARCHAR metadata "JSON" + } + + sites { + INTEGER site_int_id PK + VARCHAR site_id UK "stable identity" + FLOAT lat + FLOAT lon + FLOAT vs30 "m/s" + VARCHAR metadata "JSON" + } + + site_event { + INTEGER site_int_id "logical key" + INTEGER event_int_id "logical key" + FLOAT rrup "km, event level" + VARCHAR metadata "JSON" + } + + records { + BIGINT record_int_id "nextval, file-local, no PK" + INTEGER event_int_id "denormalised, derived from rel_int_id" + INTEGER rel_int_id "logical key" + INTEGER site_int_id "logical key" + VARCHAR component "logical key" + record_kind_t kind "ENUM: simulated, gmm, observed. logical key" + VARCHAR gmm_key "logical key. NULL unless kind=gmm" + } + + psa_ims { + BIGINT record_int_id "no row means no pSA" + FLOAT_ARRAY pSA "one array per record" + FLOAT_ARRAY pSA_sigma "ln-space total sigma, same grid as pSA" + } + + fas_ims { + BIGINT record_int_id "no row means no FAS" + FLOAT_ARRAY FAS "one array per record" + FLOAT_ARRAY FAS_sigma "ln-space total sigma, same grid as FAS" + } + + scalars_ims { + BIGINT record_int_id "no row means no scalars" + FLOAT PGA "g" + FLOAT PGV "cm/s" + FLOAT PGD "cm" + FLOAT CAV "m/s, NULL on rotd" + FLOAT AI "m/s, NULL on rotd" + FLOAT Ds575 "s, NULL on rotd" + FLOAT Ds595 "s, NULL on rotd" + } +``` + +*(`db_meta`, `notes`, `im_units`, `periods` and `frequencies` are omitted from the +diagram above for space; see `imdb/schema.py` for the full DDL, including the +`_sigma` column on every scalar IM.)* + +### Key conventions + +- **Identity**: `event_id`, `rel_id` and `site_id` are stable. The integer + surrogates (`event_int_id`, `rel_int_id`, `site_int_id`, `record_int_id`) are + assigned at ingest and change on rebuild; nothing outside the database may + reference them. +- **Array indexing is 1-based**: `periods.period_index` and `frequencies.freq_index` + match DuckDB list indexing, so `pSA[period_index]` and `FAS[freq_index]` need no + offset. +- **IM coverage is row presence**: a record has at most one row in each of + `psa_ims`, `fas_ims` and `scalars_ims`. A missing row means that IM type is not + held for that record, not NULL. +- **Components**: `000`, `090`, `ver`, `geom`, `rotd0`, `rotd50`, `rotd100`. A + database may hold any subset, listed in `db_meta.components`; the writer + validates against it. `scalars_ims.CAV`, `AI`, `Ds575` and `Ds595` are NULL for + `rotd*` components; `PGA`, `PGV` and `PGD` are populated for every component. +- **Record kind**: `simulated` (physics-based simulation), `gmm` (empirical GMM + prediction) or `observed` (recorded ground motion). `gmm_key` identifies the + model, e.g. `"Bradley_2013"`, and is NULL unless `kind = "gmm"`. +- **Units**: linear, physical units; log is a read-time transform (`g` for pSA/PGA, + `cm/s` for PGV, `cm` for PGD, `m/s` for CAV/AI, `s` for Ds575/Ds595, see + `im_units`). Every `_sigma` column/array is the exception: ln-space total + standard deviation, dimensionless. +- **Constraints**: `PRIMARY KEY`/`UNIQUE`/`FOREIGN KEY` appear only on the four + dimension tables. The large tables (`site_event`, `records`, `psa_ims`, + `fas_ims`, `scalars_ims`) have none; their logical keys are documented in + `notes` and enforced by the writer, not the schema. ## Usage @@ -21,7 +149,7 @@ db.add_events(events_df) db.add_realisations(realisations_df) db.add_sites(sites_df) db.add_site_event(site_event_df) -db.add_records(records_df) # rel_id, site_id, component, pSA, FAS, scalar IM columns +db.add_records(records_df) # rel_id, site_id, component, kind, pSA, FAS, scalar IM columns db.validate() db.close() ``` From 6197b2c82febb89c6a8545b485f4f2fcd4ea4eef Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 07:41:11 +1200 Subject: [PATCH 10/25] Saving --- imdb/imdb.py | 22 ++++++++++++++++++++++ tests/test_imdb.py | 32 ++++++++++++++++++++++++++++++-- 2 files changed, 52 insertions(+), 2 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index dd86b27..6eb0ef2 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -256,6 +256,12 @@ def add_records( ------- np.ndarray The `record_int_id` assigned to each row of `df`, in input order. + + Raises + ------ + ValueError + If `df` contains duplicate `(rel_id, site_id, component, kind, gmm_key)` + rows, or any of them already exist in this database. """ df = df.copy() @@ -292,6 +298,22 @@ def add_records( df["rel_int_id"] = rel_int_id_mapping.loc[df["rel_id"]].to_numpy() df["site_int_id"] = site_int_id_mapping.loc[df["site_id"]].to_numpy() df["event_int_id"] = rel_event_int_id_mapping.loc[df["rel_int_id"]].to_numpy() + + key_cols = ["rel_int_id", "site_int_id", "component", "kind", "gmm_key"] + keys = df[key_cols] + if keys.duplicated().any(): + raise ValueError( + "df contains duplicate records (same rel_id, site_id, component, " + "kind and gmm_key)" + ) + existing = self.con.table("records").select(*key_cols).to_pandas() + collisions = keys.merge(existing, on=key_cols, how="inner") + if not collisions.empty: + raise ValueError( + f"{len(collisions)} record(s) already exist in this database " + "(same rel_id, site_id, component, kind and gmm_key)" + ) + record_int_id = ( self.con.raw_sql( f"SELECT nextval('record_int_id_seq') FROM range({len(df)})" diff --git a/tests/test_imdb.py b/tests/test_imdb.py index f1b192e..00895a3 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -88,6 +88,34 @@ def test_schema_version_mismatch_rejected(db): IMDB(path, read_only=True).open() +def test_add_records_rejects_duplicate_within_df(db): + with pytest.raises(ValueError, match="duplicate"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0", "eventA_rel0"], + "site_id": ["siteA", "siteA"], + "component": ["000", "000"], + "kind": ["observed", "observed"], + } + ) + ) + + +def test_add_records_rejects_duplicate_of_existing_record(db): + with pytest.raises(ValueError, match="already exist"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["simulated"], + } + ) + ) + + def test_validate_catches_sigma_on_non_gmm_record(db): db.add_records( pd.DataFrame( @@ -95,7 +123,7 @@ def test_validate_catches_sigma_on_non_gmm_record(db): "rel_id": ["eventA_rel0"], "site_id": ["siteA"], "component": ["000"], - "kind": ["simulated"], + "kind": ["observed"], "PGA": [0.5], "PGA_sigma": [0.6], } @@ -113,7 +141,7 @@ def test_all_nan_row_not_written_to_psa(db): "rel_id": ["eventA_rel0", "eventA_rel0"], "site_id": ["siteA", "siteB"], "component": ["000", "000"], - "kind": ["simulated", "simulated"], + "kind": ["observed", "observed"], } ), pSA=np.array([[1.0] * len(PERIODS), [np.nan] * len(PERIODS)]), From ea207d8edbe016d6abe79b4331952396c8a58758 Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 07:58:11 +1200 Subject: [PATCH 11/25] Add CS25.6 ingest script. --- scripts/cs25p6_ingest.py | 275 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 275 insertions(+) create mode 100644 scripts/cs25p6_ingest.py diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py new file mode 100644 index 0000000..5b189ef --- /dev/null +++ b/scripts/cs25p6_ingest.py @@ -0,0 +1,275 @@ +#!/usr/bin/env -S uv run --script +# /// script +# requires-python = ">=3.12" +# dependencies = [ +# "numpy>=2", +# "pandas>=3", +# "tqdm", +# "qcore-utils", +# "source-modelling", +# "imdb", +# "oq-wrapper", +# ] +# +# [tool.uv.sources] +# imdb = { git = "ssh://git@github.com/ucgmsim/imdb.git", branch = "emp-gmm-support" } +# /// + +"""Ingest NZ NSHM 2010 fault ruptures (source_data/im_data) into an IMDB. + +Only the faults that have simulated IM output (RuatoriaS1, Thornton01, UrutiR2) are +ingested; base (non-REL) Srf/IM files are skipped, only numbered realisations count. +""" + +import argparse +import json +import re +from pathlib import Path + +import numpy as np +import oq_wrapper as oqw +import pandas as pd +from tqdm import tqdm + +from imdb import IMDB +from qcore import nhm +from source_modelling.sources import Fault + +FAULTS = ["RuatoriaS1", "Thornton01", "UrutiR2"] +SCALAR_COLS = ["PGA", "PGV", "CAV", "AI", "Ds575", "Ds595"] + + +def run_pSA_logic_tree(rupture_df: pd.DataFrame, tect_type: str, periods: list[float]): + """Run the NSHM2022 pSA GM logic tree: the weighted combination plus every individual GMM/branch. + + NSHM2022's logic tree config only defines weights for pSA (not PGA/PGV), so that's + the only IM available through `run_gmm_logic_tree`. + """ + weighted, ind = oqw.run_gmm_logic_tree( + oqw.constants.GMMLogicTree.NSHM2022, + oqw.constants.TectType[tect_type], + rupture_df, + "pSA", + periods=periods, + return_ind_results=True, + ) + mean_cols = [f"pSA_{p}_mean" for p in periods] + sigma_cols = [f"pSA_{p}_std_Total" for p in periods] + + gmm_rows, psa_rows, psa_sigma_rows = [], [], [] + for gmm_key, result in {"NSHM2022": weighted, **{k: df for k, (_, df) in ind.items()}}.items(): + gmm_rows.append( + pd.DataFrame( + { + "rel_id": rupture_df["rel_id"], + "site_id": rupture_df["site_id"], + "component": "rotd50", + "kind": "gmm", + "gmm_key": gmm_key, + } + ) + ) + psa_rows.append(np.exp(result[mean_cols].to_numpy())) + psa_sigma_rows.append(result[sigma_cols].to_numpy()) + + return pd.concat(gmm_rows, ignore_index=True), np.concatenate(psa_rows), np.concatenate(psa_sigma_rows) + + +def load_sites(data: Path) -> pd.DataFrame: + """Load the station lon/lat, vs30 and z1.0/z2.5, indexed by station code.""" + stem = data / "non_uniform_whole_nz_with_real_stations-hh400_v20p3_land" + ll = pd.read_csv(f"{stem}.ll", sep=r"\s+", header=None, names=["lon", "lat", "station"]) + vs30 = pd.read_csv(f"{stem}.vs30", sep=r"\s+", header=None, names=["station", "vs30"]) + z = pd.read_csv(f"{stem}.z").rename( + columns={"Station_Name": "station", "Z_1.0(km)": "z1p0", "Z_2.5(km)": "z2p5"} + ) + return ll.merge(vs30, on="station").merge(z[["station", "z1p0", "z2p5"]], on="station").set_index("station") + + +def rel_number(path: Path) -> int: + """Extract the realisation number from a `_RELnn.csv` filename.""" + match = re.search(r"_REL(\d+)\.csv$", path.name) + assert match is not None + return int(match.group(1)) + + +def main(data: Path, db_path: Path) -> None: + sites_all = load_sites(data) + im_periods = None + events, realisations, site_events, records, psa_rows = [], [], [], [], [] + gmm_records, gmm_psa_rows, gmm_psa_sigma_rows = [], [], [] + sites_used: dict[str, pd.Series] = {} + + for fault_name in tqdm(FAULTS, desc="events"): + srf_dir = data / "source_data" / fault_name / "Srf" + im_dir = data / "im_data" / fault_name / "IM" + base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] + + events.append( + { + "event_id": fault_name, + "magnitude": base["magnitude"], + "tect_type": base["tect_type"], + "dip": base["dip"], + "dip_dir": base["dip_dir"], + "dtop": base["dtop"], + "dbottom": base["dbottom"], + "length": base["length"], + "metadata": json.dumps( + { + "fault_type": base["fault_type"], + "plane_count": int(base["plane_count"]), + "slip_rate": base["slip_rate"], + } + ), + } + ) + + trace = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt"))[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) + fault = Fault.from_trace_points( + trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] + ) + + event_sites: set[str] = set() + rel_contexts = [] + rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) + for rel_path in rel_paths: + rel = pd.read_csv(rel_path).iloc[0] + rel_id = f"{fault_name}_REL{rel_number(rel_path):02d}" + x = 0.5 + rel["shypo"] / fault.length + y = rel["dhypo"] / fault.width + hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) + realisations.append( + { + "rel_id": rel_id, + "event_id": fault_name, + "magnitude": rel["magnitude"], + "rake": rel["rake"], + "hypo_lat": hypo_lat, + "hypo_lon": hypo_lon, + "hypo_depth": hypo_depth_m / 1000, + "metadata": json.dumps( + { + "shypo": rel["shypo"], + "dhypo": rel["dhypo"], + "seed": int(rel["seed"]), + "srfgen_seed": int(rel["srfgen_seed"]), + "sdrop": rel["sdrop"], + } + ), + } + ) + + im = pd.read_csv(im_dir / f"{rel_id}.csv") + if im_periods is None: + im_periods = [float(c.removeprefix("pSA_")) for c in im.columns if c.startswith("pSA_")] + psa_cols = [f"pSA_{p}" for p in im_periods] + + event_sites.update(im["station"]) + for station in im["station"]: + sites_used.setdefault(station, sites_all.loc[station]) + + records.append( + pd.DataFrame( + { + "rel_id": rel_id, + "site_id": im["station"], + "component": im["component"], + "kind": "simulated", + **{col: im[col] for col in SCALAR_COLS}, + } + ) + ) + psa_rows.append(im[psa_cols].to_numpy()) + + rel_contexts.append( + pd.DataFrame( + { + "rel_id": rel_id, + "site_id": im["station"], + "mag": rel["magnitude"], + "rake": rel["rake"], + "hypo_depth": hypo_depth_m / 1000, + } + ) + ) + + # site_event distances, computed once per event, over every site this event references. + sorted_sites = sorted(event_sites) + coords = sites_all.loc[sorted_sites] + latlon = coords[["lat", "lon"]].to_numpy() + latlondepth = np.column_stack([latlon, np.zeros(len(coords))]) + rrup = fault.rrup_distance(latlondepth) + rjb = fault.rjb_distance(latlondepth) + rx, ry = fault.rx_ry_distance(latlon) + site_events.append( + pd.DataFrame( + { + "site_id": sorted_sites, + "event_id": fault_name, + "rrup": np.asarray(rrup) / 1000, + "rjb": np.asarray(rjb) / 1000, + "rx": np.asarray(rx) / 1000, + "ry": np.asarray(ry) / 1000, + } + ) + ) + + # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. + context = pd.concat(rel_contexts, ignore_index=True) + site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) + distances = ( + site_events[-1] + .set_index("site_id") + .loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]] + .reset_index(drop=True) + ) + rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( + columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} + ) + rupture_df["dip"] = base["dip"] + rupture_df["ztor"] = base["dtop"] + rupture_df["zbot"] = base["dbottom"] + rupture_df["vs30measured"] = True + rupture_df["backarc"] = False + + gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) + gmm_records.append(gmm_df) + gmm_psa_rows.append(gmm_psa) + gmm_psa_sigma_rows.append(gmm_psa_sigma) + + sites_df = pd.DataFrame(sites_used.values(), index=sites_used.keys()).reset_index(names="site_id") + records_df = pd.concat(records, ignore_index=True) + psa = np.concatenate(psa_rows, axis=0) + gmm_records_df = pd.concat(gmm_records, ignore_index=True) + gmm_psa = np.concatenate(gmm_psa_rows, axis=0) + gmm_psa_sigma = np.concatenate(gmm_psa_sigma_rows, axis=0) + + db = IMDB.create( + db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} + ) + db.add_events(pd.DataFrame(events)) + db.add_sites(sites_df) + db.add_site_event(pd.concat(site_events, ignore_index=True)) + db.add_realisations(pd.DataFrame(realisations)) + db.add_records(records_df, pSA=psa) + db.add_records(gmm_records_df, pSA=gmm_psa, pSA_sigma=gmm_psa_sigma) + + problems = db.validate() + print("validate():", problems or "clean") + print("events:", len(db.get_events())) + print("realisations:", len(db.get_realisations())) + print("sites:", len(db.get_sites())) + print("records by kind:") + print(db.get_records()["kind"].value_counts()) + db.close() + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("data", type=Path, help="Directory containing source_data/, im_data/ and the site files") + parser.add_argument( + "db-path", type=Path, help="Output IMDB path" + ) + args = parser.parse_args() + main(args.data, args.db_path) From 4cd41578e96dd41ca9be52b01991a0568d9be4ba Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 08:13:17 +1200 Subject: [PATCH 12/25] Saving --- scripts/cs25p6_ingest.py | 46 ++++++++++++++++++++++++++++++++++------ 1 file changed, 40 insertions(+), 6 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index 5b189ef..8f9394a 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -17,13 +17,15 @@ """Ingest NZ NSHM 2010 fault ruptures (source_data/im_data) into an IMDB. -Only the faults that have simulated IM output (RuatoriaS1, Thornton01, UrutiR2) are -ingested; base (non-REL) Srf/IM files are skipped, only numbered realisations count. +Every fault with simulated IM output (a directory under `im_data/`) is ingested; +base (non-REL) Srf/IM files are skipped, only numbered realisations count. A fault +missing its source_data or NZ_FLTmodel_2010.txt entry is skipped with a warning. """ import argparse import json import re +import warnings from pathlib import Path import numpy as np @@ -35,7 +37,12 @@ from qcore import nhm from source_modelling.sources import Fault -FAULTS = ["RuatoriaS1", "Thornton01", "UrutiR2"] +# oq_wrapper warns per-row when falling back to active-shallow GMMs for VOLCANIC tect +# type; that's expected for this dataset (NSHM2022 has no dedicated volcanic models). +warnings.filterwarnings( + "ignore", message="Using active_shallow type model for VOLCANIC tectonic type", category=UserWarning +) + SCALAR_COLS = ["PGA", "PGV", "CAV", "AI", "Ds575", "Ds595"] @@ -93,14 +100,41 @@ def rel_number(path: Path) -> int: return int(match.group(1)) +def discover_faults(data: Path) -> list[str]: + """Faults with simulated IM output: every directory under `im_data/`.""" + return sorted(p.name for p in (data / "im_data").iterdir() if p.is_dir()) + + +def missing_reason(data: Path, fault_name: str, nhm_faults: set[str]) -> str | None: + """Why `fault_name` can't be ingested, or `None` if it has everything required.""" + if fault_name not in nhm_faults: + return "not present in NZ_FLTmodel_2010.txt" + srf_dir = data / "source_data" / fault_name / "Srf" + if not (srf_dir / f"{fault_name}.csv").exists(): + return f"missing {srf_dir / f'{fault_name}.csv'}" + if not any(srf_dir.glob(f"{fault_name}_REL*.csv")): + return f"no {fault_name}_REL*.csv realisation files in {srf_dir}" + return None + + def main(data: Path, db_path: Path) -> None: sites_all = load_sites(data) + nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) + + faults = [] + for fault_name in discover_faults(data): + reason = missing_reason(data, fault_name, set(nhm_faults)) + if reason is not None: + print(f"WARNING: skipping {fault_name}: {reason}") + else: + faults.append(fault_name) + im_periods = None events, realisations, site_events, records, psa_rows = [], [], [], [], [] gmm_records, gmm_psa_rows, gmm_psa_sigma_rows = [], [], [] sites_used: dict[str, pd.Series] = {} - for fault_name in tqdm(FAULTS, desc="events"): + for fault_name in tqdm(faults, desc="events"): srf_dir = data / "source_data" / fault_name / "Srf" im_dir = data / "im_data" / fault_name / "IM" base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] @@ -125,7 +159,7 @@ def main(data: Path, db_path: Path) -> None: } ) - trace = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt"))[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) + trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) fault = Fault.from_trace_points( trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] ) @@ -269,7 +303,7 @@ def main(data: Path, db_path: Path) -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("data", type=Path, help="Directory containing source_data/, im_data/ and the site files") parser.add_argument( - "db-path", type=Path, help="Output IMDB path" + "db_path", type=Path, help="Output IMDB path" ) args = parser.parse_args() main(args.data, args.db_path) From 201a7ed073f5446044349e0a7944e86b0c6ebd67 Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 08:20:42 +1200 Subject: [PATCH 13/25] Inges - write per event --- scripts/cs25p6_ingest.py | 111 +++++++++++++++++++-------------------- 1 file changed, 53 insertions(+), 58 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index 8f9394a..b61ecc4 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -117,6 +117,14 @@ def missing_reason(data: Path, fault_name: str, nhm_faults: set[str]) -> str | N return None +def discover_periods(data: Path, fault_name: str) -> list[float]: + """pSA periods, read from the first realisation's IM file (same across all faults).""" + im_dir = data / "im_data" / fault_name / "IM" + first_rel = sorted(im_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number)[0] + columns = pd.read_csv(first_rel, nrows=0).columns + return [float(c.removeprefix("pSA_")) for c in columns if c.startswith("pSA_")] + + def main(data: Path, db_path: Path) -> None: sites_all = load_sites(data) nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) @@ -129,34 +137,39 @@ def main(data: Path, db_path: Path) -> None: else: faults.append(fault_name) - im_periods = None - events, realisations, site_events, records, psa_rows = [], [], [], [], [] - gmm_records, gmm_psa_rows, gmm_psa_sigma_rows = [], [], [] - sites_used: dict[str, pd.Series] = {} + im_periods = discover_periods(data, faults[0]) + db = IMDB.create( + db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} + ) + sites_seen: set[str] = set() for fault_name in tqdm(faults, desc="events"): srf_dir = data / "source_data" / fault_name / "Srf" im_dir = data / "im_data" / fault_name / "IM" base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] - events.append( - { - "event_id": fault_name, - "magnitude": base["magnitude"], - "tect_type": base["tect_type"], - "dip": base["dip"], - "dip_dir": base["dip_dir"], - "dtop": base["dtop"], - "dbottom": base["dbottom"], - "length": base["length"], - "metadata": json.dumps( + db.add_events( + pd.DataFrame( + [ { - "fault_type": base["fault_type"], - "plane_count": int(base["plane_count"]), - "slip_rate": base["slip_rate"], + "event_id": fault_name, + "magnitude": base["magnitude"], + "tect_type": base["tect_type"], + "dip": base["dip"], + "dip_dir": base["dip_dir"], + "dtop": base["dtop"], + "dbottom": base["dbottom"], + "length": base["length"], + "metadata": json.dumps( + { + "fault_type": base["fault_type"], + "plane_count": int(base["plane_count"]), + "slip_rate": base["slip_rate"], + } + ), } - ), - } + ] + ) ) trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) @@ -165,7 +178,7 @@ def main(data: Path, db_path: Path) -> None: ) event_sites: set[str] = set() - rel_contexts = [] + realisations, rel_contexts, records, psa_rows = [], [], [], [] rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) for rel_path in rel_paths: rel = pd.read_csv(rel_path).iloc[0] @@ -195,13 +208,9 @@ def main(data: Path, db_path: Path) -> None: ) im = pd.read_csv(im_dir / f"{rel_id}.csv") - if im_periods is None: - im_periods = [float(c.removeprefix("pSA_")) for c in im.columns if c.startswith("pSA_")] psa_cols = [f"pSA_{p}" for p in im_periods] event_sites.update(im["station"]) - for station in im["station"]: - sites_used.setdefault(station, sites_all.loc[station]) records.append( pd.DataFrame( @@ -228,6 +237,12 @@ def main(data: Path, db_path: Path) -> None: ) ) + new_sites = sorted(event_sites - sites_seen) + if new_sites: + db.add_sites(sites_all.loc[new_sites].reset_index(names="site_id")) + sites_seen.update(new_sites) + db.add_realisations(pd.DataFrame(realisations)) + # site_event distances, computed once per event, over every site this event references. sorted_sites = sorted(event_sites) coords = sites_all.loc[sorted_sites] @@ -236,25 +251,24 @@ def main(data: Path, db_path: Path) -> None: rrup = fault.rrup_distance(latlondepth) rjb = fault.rjb_distance(latlondepth) rx, ry = fault.rx_ry_distance(latlon) - site_events.append( - pd.DataFrame( - { - "site_id": sorted_sites, - "event_id": fault_name, - "rrup": np.asarray(rrup) / 1000, - "rjb": np.asarray(rjb) / 1000, - "rx": np.asarray(rx) / 1000, - "ry": np.asarray(ry) / 1000, - } - ) + site_event_df = pd.DataFrame( + { + "site_id": sorted_sites, + "event_id": fault_name, + "rrup": np.asarray(rrup) / 1000, + "rjb": np.asarray(rjb) / 1000, + "rx": np.asarray(rx) / 1000, + "ry": np.asarray(ry) / 1000, + } ) + db.add_site_event(site_event_df) + db.add_records(pd.concat(records, ignore_index=True), pSA=np.concatenate(psa_rows, axis=0)) # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. context = pd.concat(rel_contexts, ignore_index=True) site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) distances = ( - site_events[-1] - .set_index("site_id") + site_event_df.set_index("site_id") .loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]] .reset_index(drop=True) ) @@ -268,26 +282,7 @@ def main(data: Path, db_path: Path) -> None: rupture_df["backarc"] = False gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) - gmm_records.append(gmm_df) - gmm_psa_rows.append(gmm_psa) - gmm_psa_sigma_rows.append(gmm_psa_sigma) - - sites_df = pd.DataFrame(sites_used.values(), index=sites_used.keys()).reset_index(names="site_id") - records_df = pd.concat(records, ignore_index=True) - psa = np.concatenate(psa_rows, axis=0) - gmm_records_df = pd.concat(gmm_records, ignore_index=True) - gmm_psa = np.concatenate(gmm_psa_rows, axis=0) - gmm_psa_sigma = np.concatenate(gmm_psa_sigma_rows, axis=0) - - db = IMDB.create( - db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} - ) - db.add_events(pd.DataFrame(events)) - db.add_sites(sites_df) - db.add_site_event(pd.concat(site_events, ignore_index=True)) - db.add_realisations(pd.DataFrame(realisations)) - db.add_records(records_df, pSA=psa) - db.add_records(gmm_records_df, pSA=gmm_psa, pSA_sigma=gmm_psa_sigma) + db.add_records(gmm_df, pSA=gmm_psa, pSA_sigma=gmm_psa_sigma) problems = db.validate() print("validate():", problems or "clean") From 8ff41486d67162526613ca910105893a4d5789c4 Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 08:37:21 +1200 Subject: [PATCH 14/25] ingest - multiprocessing --- scripts/cs25p6_ingest.py | 313 ++++++++++++++++++++++----------------- 1 file changed, 176 insertions(+), 137 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index b61ecc4..e86b50d 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -24,8 +24,11 @@ import argparse import json +import os import re import warnings +from concurrent.futures import ProcessPoolExecutor +from dataclasses import dataclass from pathlib import Path import numpy as np @@ -125,164 +128,199 @@ def discover_periods(data: Path, fault_name: str) -> list[float]: return [float(c.removeprefix("pSA_")) for c in columns if c.startswith("pSA_")] -def main(data: Path, db_path: Path) -> None: - sites_all = load_sites(data) - nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) +@dataclass +class FaultResult: + event: dict + realisations: list[dict] + event_sites: list[str] + site_event_df: pd.DataFrame + records_df: pd.DataFrame + psa: np.ndarray + gmm_df: pd.DataFrame + gmm_psa: np.ndarray + gmm_psa_sigma: np.ndarray - faults = [] - for fault_name in discover_faults(data): - reason = missing_reason(data, fault_name, set(nhm_faults)) - if reason is not None: - print(f"WARNING: skipping {fault_name}: {reason}") - else: - faults.append(fault_name) - im_periods = discover_periods(data, faults[0]) - db = IMDB.create( - db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} - ) - sites_seen: set[str] = set() +_ctx: dict = {} - for fault_name in tqdm(faults, desc="events"): - srf_dir = data / "source_data" / fault_name / "Srf" - im_dir = data / "im_data" / fault_name / "IM" - base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] - db.add_events( - pd.DataFrame( - [ +def _init_worker(data: Path, sites_all: pd.DataFrame, nhm_faults: dict, im_periods: list[float]) -> None: + _ctx.update(data=data, sites_all=sites_all, nhm_faults=nhm_faults, im_periods=im_periods) + + +def process_fault(fault_name: str) -> FaultResult: + """Compute everything for one fault/event: no db access, safe to run in a worker process.""" + data, sites_all, nhm_faults, im_periods = _ctx["data"], _ctx["sites_all"], _ctx["nhm_faults"], _ctx["im_periods"] + + srf_dir = data / "source_data" / fault_name / "Srf" + im_dir = data / "im_data" / fault_name / "IM" + base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] + + event = { + "event_id": fault_name, + "magnitude": base["magnitude"], + "tect_type": base["tect_type"], + "dip": base["dip"], + "dip_dir": base["dip_dir"], + "dtop": base["dtop"], + "dbottom": base["dbottom"], + "length": base["length"], + "metadata": json.dumps( + { + "fault_type": base["fault_type"], + "plane_count": int(base["plane_count"]), + "slip_rate": base["slip_rate"], + } + ), + } + + trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) + fault = Fault.from_trace_points( + trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] + ) + + event_sites: set[str] = set() + realisations, rel_contexts, records, psa_rows = [], [], [], [] + rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) + for rel_path in rel_paths: + rel = pd.read_csv(rel_path).iloc[0] + rel_id = f"{fault_name}_REL{rel_number(rel_path):02d}" + x = 0.5 + rel["shypo"] / fault.length + y = rel["dhypo"] / fault.width + hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) + realisations.append( + { + "rel_id": rel_id, + "event_id": fault_name, + "magnitude": rel["magnitude"], + "rake": rel["rake"], + "hypo_lat": hypo_lat, + "hypo_lon": hypo_lon, + "hypo_depth": hypo_depth_m / 1000, + "metadata": json.dumps( { - "event_id": fault_name, - "magnitude": base["magnitude"], - "tect_type": base["tect_type"], - "dip": base["dip"], - "dip_dir": base["dip_dir"], - "dtop": base["dtop"], - "dbottom": base["dbottom"], - "length": base["length"], - "metadata": json.dumps( - { - "fault_type": base["fault_type"], - "plane_count": int(base["plane_count"]), - "slip_rate": base["slip_rate"], - } - ), + "shypo": rel["shypo"], + "dhypo": rel["dhypo"], + "seed": int(rel["seed"]), + "srfgen_seed": int(rel["srfgen_seed"]), + "sdrop": rel["sdrop"], } - ] - ) + ), + } ) - trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) - fault = Fault.from_trace_points( - trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] + im = pd.read_csv(im_dir / f"{rel_id}.csv") + psa_cols = [f"pSA_{p}" for p in im_periods] + + event_sites.update(im["station"]) + + records.append( + pd.DataFrame( + { + "rel_id": rel_id, + "site_id": im["station"], + "component": im["component"], + "kind": "simulated", + **{col: im[col] for col in SCALAR_COLS}, + } + ) ) + psa_rows.append(im[psa_cols].to_numpy()) - event_sites: set[str] = set() - realisations, rel_contexts, records, psa_rows = [], [], [], [] - rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) - for rel_path in rel_paths: - rel = pd.read_csv(rel_path).iloc[0] - rel_id = f"{fault_name}_REL{rel_number(rel_path):02d}" - x = 0.5 + rel["shypo"] / fault.length - y = rel["dhypo"] / fault.width - hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) - realisations.append( + rel_contexts.append( + pd.DataFrame( { "rel_id": rel_id, - "event_id": fault_name, - "magnitude": rel["magnitude"], + "site_id": im["station"], + "mag": rel["magnitude"], "rake": rel["rake"], - "hypo_lat": hypo_lat, - "hypo_lon": hypo_lon, "hypo_depth": hypo_depth_m / 1000, - "metadata": json.dumps( - { - "shypo": rel["shypo"], - "dhypo": rel["dhypo"], - "seed": int(rel["seed"]), - "srfgen_seed": int(rel["srfgen_seed"]), - "sdrop": rel["sdrop"], - } - ), } ) + ) - im = pd.read_csv(im_dir / f"{rel_id}.csv") - psa_cols = [f"pSA_{p}" for p in im_periods] + # site_event distances, computed once per event, over every site this event references. + sorted_sites = sorted(event_sites) + coords = sites_all.loc[sorted_sites] + latlon = coords[["lat", "lon"]].to_numpy() + latlondepth = np.column_stack([latlon, np.zeros(len(coords))]) + rrup = fault.rrup_distance(latlondepth) + rjb = fault.rjb_distance(latlondepth) + rx, ry = fault.rx_ry_distance(latlon) + site_event_df = pd.DataFrame( + { + "site_id": sorted_sites, + "event_id": fault_name, + "rrup": np.asarray(rrup) / 1000, + "rjb": np.asarray(rjb) / 1000, + "rx": np.asarray(rx) / 1000, + "ry": np.asarray(ry) / 1000, + } + ) + records_df = pd.concat(records, ignore_index=True) + psa = np.concatenate(psa_rows, axis=0) + + # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. + context = pd.concat(rel_contexts, ignore_index=True) + site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) + distances = ( + site_event_df.set_index("site_id").loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]].reset_index(drop=True) + ) + rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( + columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} + ) + rupture_df["dip"] = base["dip"] + rupture_df["ztor"] = base["dtop"] + rupture_df["zbot"] = base["dbottom"] + rupture_df["vs30measured"] = True + rupture_df["backarc"] = False + + gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) + + return FaultResult( + event=event, + realisations=realisations, + event_sites=sorted_sites, + site_event_df=site_event_df, + records_df=records_df, + psa=psa, + gmm_df=gmm_df, + gmm_psa=gmm_psa, + gmm_psa_sigma=gmm_psa_sigma, + ) - event_sites.update(im["station"]) - records.append( - pd.DataFrame( - { - "rel_id": rel_id, - "site_id": im["station"], - "component": im["component"], - "kind": "simulated", - **{col: im[col] for col in SCALAR_COLS}, - } - ) - ) - psa_rows.append(im[psa_cols].to_numpy()) +def main(data: Path, db_path: Path, workers: int) -> None: + sites_all = load_sites(data) + nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) - rel_contexts.append( - pd.DataFrame( - { - "rel_id": rel_id, - "site_id": im["station"], - "mag": rel["magnitude"], - "rake": rel["rake"], - "hypo_depth": hypo_depth_m / 1000, - } - ) - ) + faults = [] + for fault_name in discover_faults(data): + reason = missing_reason(data, fault_name, set(nhm_faults)) + if reason is not None: + print(f"WARNING: skipping {fault_name}: {reason}") + else: + faults.append(fault_name) - new_sites = sorted(event_sites - sites_seen) - if new_sites: - db.add_sites(sites_all.loc[new_sites].reset_index(names="site_id")) - sites_seen.update(new_sites) - db.add_realisations(pd.DataFrame(realisations)) - - # site_event distances, computed once per event, over every site this event references. - sorted_sites = sorted(event_sites) - coords = sites_all.loc[sorted_sites] - latlon = coords[["lat", "lon"]].to_numpy() - latlondepth = np.column_stack([latlon, np.zeros(len(coords))]) - rrup = fault.rrup_distance(latlondepth) - rjb = fault.rjb_distance(latlondepth) - rx, ry = fault.rx_ry_distance(latlon) - site_event_df = pd.DataFrame( - { - "site_id": sorted_sites, - "event_id": fault_name, - "rrup": np.asarray(rrup) / 1000, - "rjb": np.asarray(rjb) / 1000, - "rx": np.asarray(rx) / 1000, - "ry": np.asarray(ry) / 1000, - } - ) - db.add_site_event(site_event_df) - db.add_records(pd.concat(records, ignore_index=True), pSA=np.concatenate(psa_rows, axis=0)) - - # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. - context = pd.concat(rel_contexts, ignore_index=True) - site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) - distances = ( - site_event_df.set_index("site_id") - .loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]] - .reset_index(drop=True) - ) - rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( - columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} - ) - rupture_df["dip"] = base["dip"] - rupture_df["ztor"] = base["dtop"] - rupture_df["zbot"] = base["dbottom"] - rupture_df["vs30measured"] = True - rupture_df["backarc"] = False + im_periods = discover_periods(data, faults[0]) + db = IMDB.create( + db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} + ) + sites_seen: set[str] = set() - gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) - db.add_records(gmm_df, pSA=gmm_psa, pSA_sigma=gmm_psa_sigma) + with ProcessPoolExecutor( + max_workers=workers, initializer=_init_worker, initargs=(data, sites_all, nhm_faults, im_periods) + ) as pool: + for result in tqdm(pool.map(process_fault, faults, chunksize=1), total=len(faults), desc="events"): + db.add_events(pd.DataFrame([result.event])) + new_sites = sorted(set(result.event_sites) - sites_seen) + if new_sites: + db.add_sites(sites_all.loc[new_sites].reset_index(names="site_id")) + sites_seen.update(new_sites) + db.add_realisations(pd.DataFrame(result.realisations)) + db.add_site_event(result.site_event_df) + db.add_records(result.records_df, pSA=result.psa) + db.add_records(result.gmm_df, pSA=result.gmm_psa, pSA_sigma=result.gmm_psa_sigma) problems = db.validate() print("validate():", problems or "clean") @@ -300,5 +338,6 @@ def main(data: Path, db_path: Path) -> None: parser.add_argument( "db_path", type=Path, help="Output IMDB path" ) + parser.add_argument("--workers", type=int, default=os.cpu_count(), help="Number of worker processes") args = parser.parse_args() - main(args.data, args.db_path) + main(args.data, args.db_path, args.workers) From 02701aecb900bb06701db7d2f39cdbe734756178 Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 12:17:18 +1200 Subject: [PATCH 15/25] Saving --- imdb/imdb.py | 7 ++++- scripts/cs25p6_ingest.py | 64 +++++++++++++++++++++++++++------------- 2 files changed, 49 insertions(+), 22 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index 6eb0ef2..1a36fd1 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -306,7 +306,12 @@ def add_records( "df contains duplicate records (same rel_id, site_id, component, " "kind and gmm_key)" ) - existing = self.con.table("records").select(*key_cols).to_pandas() + existing = ( + self.con.table("records") + .filter(_.rel_int_id.isin(df["rel_int_id"].unique())) + .select(*key_cols) + .to_pandas() + ) collisions = keys.merge(existing, on=key_cols, how="inner") if not collisions.empty: raise ValueError( diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index e86b50d..d476f48 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -24,10 +24,10 @@ import argparse import json +import multiprocessing import os import re import warnings -from concurrent.futures import ProcessPoolExecutor from dataclasses import dataclass from pathlib import Path @@ -47,6 +47,7 @@ ) SCALAR_COLS = ["PGA", "PGV", "CAV", "AI", "Ds575", "Ds595"] +GMM_BATCH_RELS = 8 # realisations per oq_wrapper call: bounds peak memory, amortises per-call overhead def run_pSA_logic_tree(rupture_df: pd.DataFrame, tect_type: str, periods: list[float]): @@ -79,8 +80,8 @@ def run_pSA_logic_tree(rupture_df: pd.DataFrame, tect_type: str, periods: list[f } ) ) - psa_rows.append(np.exp(result[mean_cols].to_numpy())) - psa_sigma_rows.append(result[sigma_cols].to_numpy()) + psa_rows.append(np.exp(result[mean_cols].to_numpy(dtype=np.float32))) + psa_sigma_rows.append(result[sigma_cols].to_numpy(dtype=np.float32)) return pd.concat(gmm_rows, ignore_index=True), np.concatenate(psa_rows), np.concatenate(psa_sigma_rows) @@ -225,7 +226,7 @@ def process_fault(fault_name: str) -> FaultResult: } ) ) - psa_rows.append(im[psa_cols].to_numpy()) + psa_rows.append(im[psa_cols].to_numpy(dtype=np.float32)) rel_contexts.append( pd.DataFrame( @@ -261,21 +262,36 @@ def process_fault(fault_name: str) -> FaultResult: psa = np.concatenate(psa_rows, axis=0) # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. - context = pd.concat(rel_contexts, ignore_index=True) - site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) - distances = ( - site_event_df.set_index("site_id").loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]].reset_index(drop=True) - ) - rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( - columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} - ) - rupture_df["dip"] = base["dip"] - rupture_df["ztor"] = base["dtop"] - rupture_df["zbot"] = base["dbottom"] - rupture_df["vs30measured"] = True - rupture_df["backarc"] = False + # Run a batch of realisations at a time rather than building one rupture_df for the whole + # fault: oq_wrapper's GMM logic tree evaluates rows independently, so this is numerically + # identical, but bounds the peak memory of a single run_gmm_logic_tree call to GMM_BATCH_RELS + # realisations' worth of sites instead of (all realisations x sites), which for a large fault + # is the dominant memory driver. oq_wrapper has real fixed per-call overhead (~1s, likely GSIM + # /logic-tree setup), so batching a few realisations per call rather than one at a time keeps + # the added run time small while still bounding memory. + site_distances = site_event_df.set_index("site_id")[["rrup", "rjb", "rx", "ry"]] + gmm_dfs, gmm_psa_rows, gmm_psa_sigma_rows = [], [], [] + for batch_start in range(0, len(rel_contexts), GMM_BATCH_RELS): + context = pd.concat(rel_contexts[batch_start : batch_start + GMM_BATCH_RELS], ignore_index=True) + site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) + distances = site_distances.loc[context["site_id"]].reset_index(drop=True) + rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( + columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} + ) + rupture_df["dip"] = base["dip"] + rupture_df["ztor"] = base["dtop"] + rupture_df["zbot"] = base["dbottom"] + rupture_df["vs30measured"] = True + rupture_df["backarc"] = False - gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) + rel_gmm_df, rel_gmm_psa, rel_gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) + gmm_dfs.append(rel_gmm_df) + gmm_psa_rows.append(rel_gmm_psa) + gmm_psa_sigma_rows.append(rel_gmm_psa_sigma) + + gmm_df = pd.concat(gmm_dfs, ignore_index=True) + gmm_psa = np.concatenate(gmm_psa_rows, axis=0) + gmm_psa_sigma = np.concatenate(gmm_psa_sigma_rows, axis=0) return FaultResult( event=event, @@ -291,6 +307,9 @@ def process_fault(fault_name: str) -> FaultResult: def main(data: Path, db_path: Path, workers: int) -> None: + if db_path.exists(): + raise FileExistsError(f"{db_path} already exists") + sites_all = load_sites(data) nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) @@ -308,10 +327,13 @@ def main(data: Path, db_path: Path, workers: int) -> None: ) sites_seen: set[str] = set() - with ProcessPoolExecutor( - max_workers=workers, initializer=_init_worker, initargs=(data, sites_all, nhm_faults, im_periods) + with multiprocessing.Pool( + processes=workers, + initializer=_init_worker, + initargs=(data, sites_all, nhm_faults, im_periods), + maxtasksperchild=1, ) as pool: - for result in tqdm(pool.map(process_fault, faults, chunksize=1), total=len(faults), desc="events"): + for result in tqdm(pool.imap_unordered(process_fault, faults), total=len(faults), desc="events"): db.add_events(pd.DataFrame([result.event])) new_sites = sorted(set(result.event_sites) - sites_seen) if new_sites: From 67ebcb170428ded81112d7362c47ffb6bf08d5aa Mon Sep 17 00:00:00 2001 From: Claudio Date: Fri, 11 Sep 2026 17:01:15 +1200 Subject: [PATCH 16/25] Ingest script, memory issue fix. --- scripts/cs25p6_ingest.py | 50 +++++++++++++++++++++++++++++----------- 1 file changed, 36 insertions(+), 14 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index d476f48..eef5687 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -24,11 +24,12 @@ import argparse import json -import multiprocessing import os import re import warnings +from concurrent.futures import FIRST_COMPLETED, ProcessPoolExecutor, wait from dataclasses import dataclass +from itertools import islice from pathlib import Path import numpy as np @@ -48,6 +49,7 @@ SCALAR_COLS = ["PGA", "PGV", "CAV", "AI", "Ds575", "Ds595"] GMM_BATCH_RELS = 8 # realisations per oq_wrapper call: bounds peak memory, amortises per-call overhead +DB_MEMORY_LIMIT = "8GB" # DuckDB buffer pool cap, leaving RAM for the workers def run_pSA_logic_tree(rupture_df: pd.DataFrame, tect_type: str, periods: list[float]): @@ -325,24 +327,44 @@ def main(data: Path, db_path: Path, workers: int) -> None: db = IMDB.create( db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} ) + # DuckDB defaults memory_limit to 80% of RAM; left alone its buffer pool grows with the + # database and crowds out the worker processes. Capped, it spills to disk instead. + db.con.raw_sql(f"PRAGMA memory_limit='{DB_MEMORY_LIMIT}'") sites_seen: set[str] = set() - with multiprocessing.Pool( - processes=workers, + # Keep only a small window of faults in flight. Pool.imap_unordered/submit-all run every + # task as fast as the workers allow and buffer each finished result in the parent, with no + # backpressure; since the workers (parallel) outrun the database writer (serial), that + # backlog grows without bound, and a result here is GBs. + with ProcessPoolExecutor( + max_workers=workers, initializer=_init_worker, initargs=(data, sites_all, nhm_faults, im_periods), - maxtasksperchild=1, + max_tasks_per_child=1, ) as pool: - for result in tqdm(pool.imap_unordered(process_fault, faults), total=len(faults), desc="events"): - db.add_events(pd.DataFrame([result.event])) - new_sites = sorted(set(result.event_sites) - sites_seen) - if new_sites: - db.add_sites(sites_all.loc[new_sites].reset_index(names="site_id")) - sites_seen.update(new_sites) - db.add_realisations(pd.DataFrame(result.realisations)) - db.add_site_event(result.site_event_df) - db.add_records(result.records_df, pSA=result.psa) - db.add_records(result.gmm_df, pSA=result.gmm_psa, pSA_sigma=result.gmm_psa_sigma) + queued = iter(faults) + pending = {pool.submit(process_fault, name) for name in islice(queued, workers + 1)} + progress = tqdm(total=len(faults), desc="events") + while pending: + done, pending = wait(pending, return_when=FIRST_COMPLETED) + pending |= {pool.submit(process_fault, name) for name in islice(queued, len(done))} + for future in done: + result = future.result() + db.add_events(pd.DataFrame([result.event])) + new_sites = sorted(set(result.event_sites) - sites_seen) + if new_sites: + db.add_sites(sites_all.loc[new_sites].reset_index(names="site_id")) + sites_seen.update(new_sites) + db.add_realisations(pd.DataFrame(result.realisations)) + db.add_site_event(result.site_event_df) + db.add_records(result.records_df, pSA=result.psa) + db.add_records(result.gmm_df, pSA=result.gmm_psa, pSA_sigma=result.gmm_psa_sigma) + del result + progress.update() + # A Future caches its result until collected, so drop the finished ones before + # waiting again, or the window holds more than it looks like it does. + done.clear() + progress.close() problems = db.validate() print("validate():", problems or "clean") From 3106894936890bbfbdcc05374bfacbbb5d25db59 Mon Sep 17 00:00:00 2001 From: Claudio Date: Mon, 14 Sep 2026 10:13:11 +1200 Subject: [PATCH 17/25] Saving --- scripts/cs25p6_ingest.py | 83 ++++++++++++++-------------------------- 1 file changed, 29 insertions(+), 54 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index eef5687..d15f434 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -48,44 +48,34 @@ ) SCALAR_COLS = ["PGA", "PGV", "CAV", "AI", "Ds575", "Ds595"] -GMM_BATCH_RELS = 8 # realisations per oq_wrapper call: bounds peak memory, amortises per-call overhead DB_MEMORY_LIMIT = "8GB" # DuckDB buffer pool cap, leaving RAM for the workers def run_pSA_logic_tree(rupture_df: pd.DataFrame, tect_type: str, periods: list[float]): - """Run the NSHM2022 pSA GM logic tree: the weighted combination plus every individual GMM/branch. + """Run the NSHM2022 pSA GM logic tree, keeping only the weighted combination. NSHM2022's logic tree config only defines weights for pSA (not PGA/PGV), so that's the only IM available through `run_gmm_logic_tree`. """ - weighted, ind = oqw.run_gmm_logic_tree( + result = oqw.run_gmm_logic_tree( oqw.constants.GMMLogicTree.NSHM2022, oqw.constants.TectType[tect_type], rupture_df, "pSA", periods=periods, - return_ind_results=True, ) - mean_cols = [f"pSA_{p}_mean" for p in periods] - sigma_cols = [f"pSA_{p}_std_Total" for p in periods] - - gmm_rows, psa_rows, psa_sigma_rows = [], [], [] - for gmm_key, result in {"NSHM2022": weighted, **{k: df for k, (_, df) in ind.items()}}.items(): - gmm_rows.append( - pd.DataFrame( - { - "rel_id": rupture_df["rel_id"], - "site_id": rupture_df["site_id"], - "component": "rotd50", - "kind": "gmm", - "gmm_key": gmm_key, - } - ) - ) - psa_rows.append(np.exp(result[mean_cols].to_numpy(dtype=np.float32))) - psa_sigma_rows.append(result[sigma_cols].to_numpy(dtype=np.float32)) - - return pd.concat(gmm_rows, ignore_index=True), np.concatenate(psa_rows), np.concatenate(psa_sigma_rows) + gmm_df = pd.DataFrame( + { + "rel_id": rupture_df["rel_id"], + "site_id": rupture_df["site_id"], + "component": "rotd50", + "kind": "gmm", + "gmm_key": "NSHM2022", + } + ) + psa = np.exp(result[[f"pSA_{p}_mean" for p in periods]].to_numpy(dtype=np.float32)) + psa_sigma = result[[f"pSA_{p}_std_Total" for p in periods]].to_numpy(dtype=np.float32) + return gmm_df, psa, psa_sigma def load_sites(data: Path) -> pd.DataFrame: @@ -264,36 +254,21 @@ def process_fault(fault_name: str) -> FaultResult: psa = np.concatenate(psa_rows, axis=0) # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. - # Run a batch of realisations at a time rather than building one rupture_df for the whole - # fault: oq_wrapper's GMM logic tree evaluates rows independently, so this is numerically - # identical, but bounds the peak memory of a single run_gmm_logic_tree call to GMM_BATCH_RELS - # realisations' worth of sites instead of (all realisations x sites), which for a large fault - # is the dominant memory driver. oq_wrapper has real fixed per-call overhead (~1s, likely GSIM - # /logic-tree setup), so batching a few realisations per call rather than one at a time keeps - # the added run time small while still bounding memory. - site_distances = site_event_df.set_index("site_id")[["rrup", "rjb", "rx", "ry"]] - gmm_dfs, gmm_psa_rows, gmm_psa_sigma_rows = [], [], [] - for batch_start in range(0, len(rel_contexts), GMM_BATCH_RELS): - context = pd.concat(rel_contexts[batch_start : batch_start + GMM_BATCH_RELS], ignore_index=True) - site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) - distances = site_distances.loc[context["site_id"]].reset_index(drop=True) - rupture_df = pd.concat([context.reset_index(drop=True), site_attrs, distances], axis=1).rename( - columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} - ) - rupture_df["dip"] = base["dip"] - rupture_df["ztor"] = base["dtop"] - rupture_df["zbot"] = base["dbottom"] - rupture_df["vs30measured"] = True - rupture_df["backarc"] = False - - rel_gmm_df, rel_gmm_psa, rel_gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) - gmm_dfs.append(rel_gmm_df) - gmm_psa_rows.append(rel_gmm_psa) - gmm_psa_sigma_rows.append(rel_gmm_psa_sigma) - - gmm_df = pd.concat(gmm_dfs, ignore_index=True) - gmm_psa = np.concatenate(gmm_psa_rows, axis=0) - gmm_psa_sigma = np.concatenate(gmm_psa_sigma_rows, axis=0) + context = pd.concat(rel_contexts, ignore_index=True) + site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) + distances = ( + site_event_df.set_index("site_id").loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]].reset_index(drop=True) + ) + rupture_df = pd.concat([context, site_attrs, distances], axis=1).rename( + columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} + ) + rupture_df["dip"] = base["dip"] + rupture_df["ztor"] = base["dtop"] + rupture_df["zbot"] = base["dbottom"] + rupture_df["vs30measured"] = True + rupture_df["backarc"] = False + + gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) return FaultResult( event=event, From e45fb00eaf04d51101f2b2d17f30a1bc9f13234d Mon Sep 17 00:00:00 2001 From: Claudio Date: Mon, 14 Sep 2026 13:14:47 +1200 Subject: [PATCH 18/25] Saving --- scripts/cs25p6_ingest.py | 264 ++++++++++++++++++++------------------- 1 file changed, 138 insertions(+), 126 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index d15f434..d01bbe0 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -26,6 +26,7 @@ import json import os import re +import traceback import warnings from concurrent.futures import FIRST_COMPLETED, ProcessPoolExecutor, wait from dataclasses import dataclass @@ -141,146 +142,154 @@ def _init_worker(data: Path, sites_all: pd.DataFrame, nhm_faults: dict, im_perio _ctx.update(data=data, sites_all=sites_all, nhm_faults=nhm_faults, im_periods=im_periods) -def process_fault(fault_name: str) -> FaultResult: +def process_fault(fault_name: str) -> FaultResult | None: """Compute everything for one fault/event: no db access, safe to run in a worker process.""" - data, sites_all, nhm_faults, im_periods = _ctx["data"], _ctx["sites_all"], _ctx["nhm_faults"], _ctx["im_periods"] + try: + data, sites_all = _ctx["data"], _ctx["sites_all"] + nhm_faults, im_periods = _ctx["nhm_faults"], _ctx["im_periods"] - srf_dir = data / "source_data" / fault_name / "Srf" - im_dir = data / "im_data" / fault_name / "IM" - base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] - - event = { - "event_id": fault_name, - "magnitude": base["magnitude"], - "tect_type": base["tect_type"], - "dip": base["dip"], - "dip_dir": base["dip_dir"], - "dtop": base["dtop"], - "dbottom": base["dbottom"], - "length": base["length"], - "metadata": json.dumps( - { - "fault_type": base["fault_type"], - "plane_count": int(base["plane_count"]), - "slip_rate": base["slip_rate"], - } - ), - } - - trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) - fault = Fault.from_trace_points( - trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] - ) + srf_dir = data / "source_data" / fault_name / "Srf" + im_dir = data / "im_data" / fault_name / "IM" + base = pd.read_csv(srf_dir / f"{fault_name}.csv").iloc[0] - event_sites: set[str] = set() - realisations, rel_contexts, records, psa_rows = [], [], [], [] - rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) - for rel_path in rel_paths: - rel = pd.read_csv(rel_path).iloc[0] - rel_id = f"{fault_name}_REL{rel_number(rel_path):02d}" - x = 0.5 + rel["shypo"] / fault.length - y = rel["dhypo"] / fault.width - hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) - realisations.append( - { - "rel_id": rel_id, - "event_id": fault_name, - "magnitude": rel["magnitude"], - "rake": rel["rake"], - "hypo_lat": hypo_lat, - "hypo_lon": hypo_lon, - "hypo_depth": hypo_depth_m / 1000, - "metadata": json.dumps( - { - "shypo": rel["shypo"], - "dhypo": rel["dhypo"], - "seed": int(rel["seed"]), - "srfgen_seed": int(rel["srfgen_seed"]), - "sdrop": rel["sdrop"], - } - ), - } - ) - - im = pd.read_csv(im_dir / f"{rel_id}.csv") - psa_cols = [f"pSA_{p}" for p in im_periods] - - event_sites.update(im["station"]) - - records.append( - pd.DataFrame( + event = { + "event_id": fault_name, + "magnitude": base["magnitude"], + "tect_type": base["tect_type"], + "dip": base["dip"], + "dip_dir": base["dip_dir"], + "dtop": base["dtop"], + "dbottom": base["dbottom"], + "length": base["length"], + "metadata": json.dumps( { - "rel_id": rel_id, - "site_id": im["station"], - "component": im["component"], - "kind": "simulated", - **{col: im[col] for col in SCALAR_COLS}, + "fault_type": base["fault_type"], + "plane_count": int(base["plane_count"]), + "slip_rate": base["slip_rate"], } - ) + ), + } + + trace = nhm_faults[fault_name].trace[:, ::-1] # (lon,lat) -> (lat,lon) + fault = Fault.from_trace_points( + trace, dtop=base["dtop"], dbottom=base["dbottom"], dip=base["dip"], dip_dir=base["dip_dir"] ) - psa_rows.append(im[psa_cols].to_numpy(dtype=np.float32)) - rel_contexts.append( - pd.DataFrame( + event_sites: set[str] = set() + realisations, rel_contexts, records, psa_rows = [], [], [], [] + rel_paths = sorted(srf_dir.glob(f"{fault_name}_REL*.csv"), key=rel_number) + for rel_path in rel_paths: + rel = pd.read_csv(rel_path).iloc[0] + rel_id = f"{fault_name}_REL{rel_number(rel_path):02d}" + x = 0.5 + rel["shypo"] / fault.length + y = rel["dhypo"] / fault.width + hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) + realisations.append( { "rel_id": rel_id, - "site_id": im["station"], - "mag": rel["magnitude"], + "event_id": fault_name, + "magnitude": rel["magnitude"], "rake": rel["rake"], + "hypo_lat": hypo_lat, + "hypo_lon": hypo_lon, "hypo_depth": hypo_depth_m / 1000, + "metadata": json.dumps( + { + "shypo": rel["shypo"], + "dhypo": rel["dhypo"], + "seed": int(rel["seed"]), + "srfgen_seed": int(rel["srfgen_seed"]), + "sdrop": rel["sdrop"], + } + ), } ) - ) - # site_event distances, computed once per event, over every site this event references. - sorted_sites = sorted(event_sites) - coords = sites_all.loc[sorted_sites] - latlon = coords[["lat", "lon"]].to_numpy() - latlondepth = np.column_stack([latlon, np.zeros(len(coords))]) - rrup = fault.rrup_distance(latlondepth) - rjb = fault.rjb_distance(latlondepth) - rx, ry = fault.rx_ry_distance(latlon) - site_event_df = pd.DataFrame( - { - "site_id": sorted_sites, - "event_id": fault_name, - "rrup": np.asarray(rrup) / 1000, - "rjb": np.asarray(rjb) / 1000, - "rx": np.asarray(rx) / 1000, - "ry": np.asarray(ry) / 1000, - } - ) - records_df = pd.concat(records, ignore_index=True) - psa = np.concatenate(psa_rows, axis=0) - - # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. - context = pd.concat(rel_contexts, ignore_index=True) - site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) - distances = ( - site_event_df.set_index("site_id").loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]].reset_index(drop=True) - ) - rupture_df = pd.concat([context, site_attrs, distances], axis=1).rename( - columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} - ) - rupture_df["dip"] = base["dip"] - rupture_df["ztor"] = base["dtop"] - rupture_df["zbot"] = base["dbottom"] - rupture_df["vs30measured"] = True - rupture_df["backarc"] = False - - gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) - - return FaultResult( - event=event, - realisations=realisations, - event_sites=sorted_sites, - site_event_df=site_event_df, - records_df=records_df, - psa=psa, - gmm_df=gmm_df, - gmm_psa=gmm_psa, - gmm_psa_sigma=gmm_psa_sigma, - ) + im = pd.read_csv(im_dir / f"{rel_id}.csv") + psa_cols = [f"pSA_{p}" for p in im_periods] + + event_sites.update(im["station"]) + + records.append( + pd.DataFrame( + { + "rel_id": rel_id, + "site_id": im["station"], + "component": im["component"], + "kind": "simulated", + **{col: im[col] for col in SCALAR_COLS}, + } + ) + ) + psa_rows.append(im[psa_cols].to_numpy(dtype=np.float32)) + + rel_contexts.append( + pd.DataFrame( + { + "rel_id": rel_id, + "site_id": im["station"], + "mag": rel["magnitude"], + "rake": rel["rake"], + "hypo_depth": hypo_depth_m / 1000, + } + ) + ) + + # site_event distances, computed once per event, over every site this event references. + sorted_sites = sorted(event_sites) + coords = sites_all.loc[sorted_sites] + latlon = coords[["lat", "lon"]].to_numpy() + latlondepth = np.column_stack([latlon, np.zeros(len(coords))]) + rrup = fault.rrup_distance(latlondepth) + rjb = fault.rjb_distance(latlondepth) + rx, ry = fault.rx_ry_distance(latlon) + site_event_df = pd.DataFrame( + { + "site_id": sorted_sites, + "event_id": fault_name, + "rrup": np.asarray(rrup) / 1000, + "rjb": np.asarray(rjb) / 1000, + "rx": np.asarray(rx) / 1000, + "ry": np.asarray(ry) / 1000, + } + ) + records_df = pd.concat(records, ignore_index=True) + psa = np.concatenate(psa_rows, axis=0) + + # empirical GMM predictions, over the same (rel_id, site_id) pairs as the simulated records. + context = pd.concat(rel_contexts, ignore_index=True) + site_attrs = sites_all.loc[context["site_id"], ["vs30", "z1p0", "z2p5"]].reset_index(drop=True) + distances = ( + site_event_df.set_index("site_id") + .loc[context["site_id"], ["rrup", "rjb", "rx", "ry"]] + .reset_index(drop=True) + ) + rupture_df = pd.concat([context, site_attrs, distances], axis=1).rename( + columns={"z1p0": "z1pt0", "z2p5": "z2pt5"} + ) + rupture_df["dip"] = base["dip"] + rupture_df["ztor"] = base["dtop"] + rupture_df["zbot"] = base["dbottom"] + rupture_df["vs30measured"] = True + rupture_df["backarc"] = False + + gmm_df, gmm_psa, gmm_psa_sigma = run_pSA_logic_tree(rupture_df, base["tect_type"], im_periods) + + return FaultResult( + event=event, + realisations=realisations, + event_sites=sorted_sites, + site_event_df=site_event_df, + records_df=records_df, + psa=psa, + gmm_df=gmm_df, + gmm_psa=gmm_psa, + gmm_psa_sigma=gmm_psa_sigma, + ) + except Exception: + print(f"ERROR: skipping {fault_name}", flush=True) + traceback.print_exc() + return None def main(data: Path, db_path: Path, workers: int) -> None: @@ -325,6 +334,9 @@ def main(data: Path, db_path: Path, workers: int) -> None: pending |= {pool.submit(process_fault, name) for name in islice(queued, len(done))} for future in done: result = future.result() + if result is None: + progress.update() + continue db.add_events(pd.DataFrame([result.event])) new_sites = sorted(set(result.event_sites) - sites_seen) if new_sites: From 0eeab454eb58df0939b0ceeb756cbca4650955ec Mon Sep 17 00:00:00 2001 From: Claudio Date: Tue, 15 Sep 2026 11:15:34 +1200 Subject: [PATCH 19/25] Update readme --- README.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/README.md b/README.md index 6675fa3..a307f86 100644 --- a/README.md +++ b/README.md @@ -38,6 +38,14 @@ erDiagram VARCHAR event_id UK "stable identity" FLOAT magnitude tect_type_t tect_type "ENUM, 4 values" + FLOAT dip + FLOAT dip_dir + FLOAT dtop + FLOAT dbottom + FLOAT length + VARCHAR source_wkt + VARCHAR trace_wkt + VARCHAR domain_wkt VARCHAR metadata "JSON" } @@ -47,6 +55,9 @@ erDiagram INTEGER event_int_id FK FLOAT magnitude FLOAT rake + FLOAT hypo_lat + FLOAT hypo_lon + FLOAT hypo_depth VARCHAR metadata "JSON" } @@ -56,6 +67,8 @@ erDiagram FLOAT lat FLOAT lon FLOAT vs30 "m/s" + FLOAT z1p0 "km" + FLOAT z2p5 "km" VARCHAR metadata "JSON" } @@ -63,6 +76,9 @@ erDiagram INTEGER site_int_id "logical key" INTEGER event_int_id "logical key" FLOAT rrup "km, event level" + FLOAT rjb "km, event level" + FLOAT rx "km, event level" + FLOAT ry "km, event level" VARCHAR metadata "JSON" } From 2124f24fcae3635fc1457ed663b44a446698f29c Mon Sep 17 00:00:00 2001 From: Claudio Date: Tue, 15 Sep 2026 11:57:00 +1200 Subject: [PATCH 20/25] Extra tests, tidy up. --- imdb/imdb.py | 92 ++++++++++++++++-- pyproject.toml | 9 ++ tests/conftest.py | 2 +- tests/test_imdb.py | 236 +++++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 329 insertions(+), 10 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index 1a36fd1..ea694bb 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -29,14 +29,28 @@ class IMDB: """ def __init__(self, path: Path, read_only: bool = True) -> None: - """Set up the database path; does not open a connection.""" + """Set up the database path; does not open a connection. + + Parameters + ---------- + path : Path + Path to the database file. + read_only : bool + Open the database read-only. + """ self.path = Path(path) self.read_only = read_only self._con: DuckDBBackend | None = None @property def con(self) -> DuckDBBackend: - """The underlying ibis connection. Raises if the database is not open.""" + """The underlying ibis connection. Raises if the database is not open. + + Returns + ------- + DuckDBBackend + The open ibis connection. + """ if self._con is None: raise RuntimeError( "database is not open; call .open() or use as a context manager" @@ -44,7 +58,13 @@ def con(self) -> DuckDBBackend: return self._con def open(self) -> Self: - """Open the database connection, if not already open.""" + """Open the database connection, if not already open. + + Returns + ------- + Self + This database, open for use. + """ if self._con is None: self._con = ibis.duckdb.connect(self.path, read_only=self.read_only) if "db_meta" in self._con.list_tables(): @@ -60,7 +80,13 @@ def open(self) -> Self: @property def db_meta(self) -> dict[str, str]: - """The `db_meta` table, as a dict.""" + """The `db_meta` table, as a dict. + + Returns + ------- + dict of str to str + The `db_meta` table's `key`/`value` rows. + """ df = self.con.table("db_meta").to_pandas() return dict(zip(df["key"], df["value"], strict=True)) @@ -152,13 +178,43 @@ def create( return db def _next_ids(self, table: str, int_col: str, n: int) -> np.ndarray: - """Return `n` new contiguous integer ids for `table`, starting after the current max.""" + """Return `n` new contiguous integer ids for `table`, starting after the current max. + + Parameters + ---------- + table : str + Table to find the current max id in. + int_col : str + The integer id column of `table`. + n : int + How many new ids to return. + + Returns + ------- + np.ndarray + `n` new contiguous integer ids. + """ current = self.con.table(table)[int_col].max().to_pandas() start = 0 if pd.isna(current) else int(current) + 1 # ty: ignore[invalid-argument-type] return np.arange(start, start + n) def _id_map(self, table: str, id_col: str, int_col: str) -> pd.Series: - """Return a `pd.Series` mapping string id to int id for `table`.""" + """Return a `pd.Series` mapping string id to int id for `table`. + + Parameters + ---------- + table : str + Table to read the id mapping from. + id_col : str + The stable string id column of `table`. + int_col : str + The integer surrogate id column of `table`. + + Returns + ------- + pd.Series + Indexed by `id_col`, valued by `int_col`. + """ df = self.con.table(table).select(id_col, int_col).to_pandas() return df.set_index(id_col)[int_col] @@ -471,15 +527,33 @@ def validate(self) -> list[str]: # ---- read ------------------------------------------------------------- def get_events(self) -> pd.DataFrame: - """Return all events, indexed by `event_id`.""" + """Return all events, indexed by `event_id`. + + Returns + ------- + pd.DataFrame + One row per event. + """ return self.con.table("events").to_pandas().set_index("event_id") def get_realisations(self) -> pd.DataFrame: - """Return all realisations, indexed by `rel_id`.""" + """Return all realisations, indexed by `rel_id`. + + Returns + ------- + pd.DataFrame + One row per realisation. + """ return self.con.table("realisations").to_pandas().set_index("rel_id") def get_sites(self) -> pd.DataFrame: - """Return all sites, indexed by `site_id`.""" + """Return all sites, indexed by `site_id`. + + Returns + ------- + pd.DataFrame + One row per site. + """ return self.con.table("sites").to_pandas().set_index("site_id") def get_site_event( diff --git a/pyproject.toml b/pyproject.toml index 348efb1..4d7370d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -27,11 +27,17 @@ types = [ ] dev = ["ruff", "deptry", "ty", "numpydoc"] +[tool.deptry] +extend_exclude = ["scripts"] + [tool.setuptools_scm] [tool.setuptools.packages.find] include = ["imdb*"] +[tool.ruff] +extend-exclude = ["scripts"] + [tool.ruff.lint] extend-select = [ # isort imports @@ -69,6 +75,9 @@ known-first-party = ["imdb", "qcore", "IM", "workflow", "source_modelling"] # Ignore docstring errors and fixture annotations in tests folder "tests/**.py" = ["D", "ANN001"] +[tool.ty.src] +exclude = ["scripts"] + [tool.numpydoc_validation] checks = [ "GL05", diff --git a/tests/conftest.py b/tests/conftest.py index 1b4462e..1228831 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -8,7 +8,7 @@ PERIODS = [0.1, 0.2, 0.5, 1.0, 2.0] FREQUENCIES = [1.0, 5.0, 10.0] -COMPONENTS = ["000", "090"] +COMPONENTS = ("000", "090") EVENTS = ["eventA", "eventB"] SITES = ["siteA", "siteB", "siteC"] diff --git a/tests/test_imdb.py b/tests/test_imdb.py index 00895a3..fd5c09b 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -161,6 +161,242 @@ def test_get_site_event_filters(db): assert len(by_rrup) == 2 +def test_con_raises_when_not_open(tmp_path): + db = IMDB(tmp_path / "unopened.duckdb") + with pytest.raises(RuntimeError, match="not open"): + _ = db.con + + +def test_context_manager(tmp_path): + path = tmp_path / "ctx.duckdb" + with IMDB.create(path, periods=[]) as db: + assert db.con is not None + assert db._con is None + + +def test_add_records_requires_kind_column(db): + with pytest.raises(ValueError, match="kind"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + } + ) + ) + + +def test_add_records_rejects_unknown_component(db): + with pytest.raises(ValueError, match="components not in this database"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["ver"], + "kind": ["observed"], + } + ) + ) + + +def test_add_records_rejects_bad_psa_shape(db): + with pytest.raises(ValueError, match="pSA must have one row per record"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + } + ), + pSA=np.zeros((2, len(PERIODS))), + ) + + +def test_add_records_rejects_bad_fas_shape(db): + with pytest.raises(ValueError, match="FAS must have one row per record"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + } + ), + FAS=np.zeros((2, len(FREQUENCIES))), + ) + + +def test_add_records_rejects_psa_sigma_without_psa(db): + with pytest.raises(ValueError, match="pSA_sigma must match pSA"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + } + ), + pSA_sigma=np.zeros((1, len(PERIODS))), + ) + + +def test_add_records_rejects_fas_sigma_without_fas(db): + with pytest.raises(ValueError, match="FAS_sigma must match FAS"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + } + ), + FAS_sigma=np.zeros((1, len(FREQUENCIES))), + ) + + +def test_add_records_masks_rotd_sigma(tmp_path): + db = IMDB.create(tmp_path / "rotd_sigma.duckdb", periods=[], components=("rotd50",)) + db.add_events(pd.DataFrame({"event_id": ["e1"]})) + db.add_realisations(pd.DataFrame({"rel_id": ["r1"], "event_id": ["e1"]})) + db.add_sites(pd.DataFrame({"site_id": ["s1"], "lat": [-43.5], "lon": [172.6]})) + + db.add_records( + pd.DataFrame( + { + "rel_id": ["r1"], + "site_id": ["s1"], + "component": ["rotd50"], + "kind": ["gmm"], + "gmm_key": ["TestGMM2020"], + "PGA": [0.5], + "CAV": [1.2], + "CAV_sigma": [0.3], + } + ) + ) + + scalars = db.get_scalars(ims=["PGA", "CAV"], sigma=True) + assert scalars["PGA"].iloc[0] == 0.5 + assert pd.isna(scalars["CAV"].iloc[0]) + assert pd.isna(scalars["CAV_sigma"].iloc[0]) + db.close() + + +def test_add_records_rolls_back_on_failure(db, monkeypatch): + def broken_insert(table, data): + if table == "scalars_ims": + raise RuntimeError("boom") + return real_insert(table, data) + + real_insert = db.con.insert + monkeypatch.setattr(db.con, "insert", broken_insert) + + with pytest.raises(RuntimeError, match="boom"): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + "PGA": [0.5], + } + ) + ) + + monkeypatch.undo() + assert len(db.get_records(rel_ids=["eventA_rel0"], site_ids=["siteA"])) == 2 + + +def test_validate_detects_duplicate_record_int_id(db): + some_id = db.get_records().index[0] + db.con.raw_sql(f"INSERT INTO psa_ims (record_int_id, pSA) VALUES ({some_id}, NULL)") + problems = db.validate() + assert any("record_int_id is not unique" in p for p in problems) + + +def test_validate_detects_unknown_foreign_key(db): + db.con.raw_sql( + "INSERT INTO records (record_int_id, event_int_id, rel_int_id, site_int_id, " + "component, kind) VALUES (999999, 999999, 999999, 999999, '000', 'observed')" + ) + problems = db.validate() + assert any("unknown event_int_id" in p for p in problems) + + +def test_validate_detects_orphan_im_row(db): + db.con.raw_sql("INSERT INTO psa_ims (record_int_id, pSA) VALUES (999999, NULL)") + problems = db.validate() + assert any("psa_ims: 1 rows with an unknown record_int_id" in p for p in problems) + + +def test_validate_detects_bad_gmm_key(db): + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["000"], + "kind": ["observed"], + "gmm_key": ["shouldnt be set"], + } + ) + ) + problems = db.validate() + assert any("gmm_key inconsistent with kind" in p for p in problems) + + +def test_get_sites(db): + sites = db.get_sites() + assert set(sites.index) == set(SITES) + + +def test_get_site_event_filters_by_site_ids(db): + by_site = db.get_site_event(site_ids=["siteA"]) + assert (by_site["site_id"] == "siteA").all() + assert len(by_site) == len(EVENTS) + + +def test_get_records_filters(db): + by_rel = db.get_records(rel_ids=["eventA_rel0"]) + assert (by_rel["rel_id"] == "eventA_rel0").all() + + by_site = db.get_records(site_ids=["siteA"]) + assert (by_site["site_id"] == "siteA").all() + + by_component = db.get_records(component="000") + assert (by_component["component"] == "000").all() + + db.add_records( + pd.DataFrame( + { + "rel_id": ["eventA_rel0"], + "site_id": ["siteA"], + "component": ["090"], + "kind": ["gmm"], + "gmm_key": ["TestGMM2020"], + } + ) + ) + by_gmm_key = db.get_records(gmm_key="TestGMM2020") + assert (by_gmm_key["gmm_key"] == "TestGMM2020").all() + + +def test_get_psa_and_fas_sigma(db): + psa = db.get_psa(sigma=True) + assert all(f"{p}_sigma" in psa.columns for p in PERIODS) + + fas = db.get_fas(sigma=True) + assert all(f"{f}_sigma" in fas.columns for f in FREQUENCIES) + + def test_rotd_component_nulls_undefined_scalars(tmp_path): db = IMDB.create(tmp_path / "rotd.duckdb", periods=[], components=("rotd50",)) db.add_events(pd.DataFrame({"event_id": ["e1"]})) From 72d8d4fb7b0434d0b5a63233b36f066d930f0e80 Mon Sep 17 00:00:00 2001 From: Claudio Date: Tue, 15 Sep 2026 12:04:20 +1200 Subject: [PATCH 21/25] ruff fixes. --- README.md | 4 +++- imdb/imdb.py | 14 +++++++++++--- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index a307f86..cf661e0 100644 --- a/README.md +++ b/README.md @@ -165,7 +165,9 @@ db.add_events(events_df) db.add_realisations(realisations_df) db.add_sites(sites_df) db.add_site_event(site_event_df) -db.add_records(records_df) # rel_id, site_id, component, kind, pSA, FAS, scalar IM columns +db.add_records( + records_df +) # rel_id, site_id, component, kind, pSA, FAS, scalar IM columns db.validate() db.close() ``` diff --git a/imdb/imdb.py b/imdb/imdb.py index ea694bb..3360b2b 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -336,11 +336,15 @@ def add_records( raise ValueError("pSA must have one row per record") if FAS is not None and FAS.shape[0] != len(df): raise ValueError("FAS must have one row per record") - if pSA_sigma is not None and not (pSA is not None and pSA_sigma.shape == pSA.shape): + if pSA_sigma is not None and not ( + pSA is not None and pSA_sigma.shape == pSA.shape + ): raise ValueError( "pSA_sigma must match pSA's shape, and only be given together with pSA" ) - if FAS_sigma is not None and not (FAS is not None and FAS_sigma.shape == FAS.shape): + if FAS_sigma is not None and not ( + FAS is not None and FAS_sigma.shape == FAS.shape + ): raise ValueError( "FAS_sigma must match FAS's shape, and only be given together with FAS" ) @@ -489,7 +493,11 @@ def validate(self) -> list[str]: for table in ("psa_ims", "fas_ims", "scalars_ims"): t = self.con.table(table) - n = t.filter(~t.record_int_id.isin(records.record_int_id)).count().to_pandas() + n = ( + t.filter(~t.record_int_id.isin(records.record_int_id)) + .count() + .to_pandas() + ) if n: problems.append(f"{table}: {n} rows with an unknown record_int_id") From bf9e55dd923f61bf7ca2149ffd5923cc10d16c41 Mon Sep 17 00:00:00 2001 From: Claudio Date: Tue, 15 Sep 2026 13:14:11 +1200 Subject: [PATCH 22/25] Make CS25.6 ingest script more robust --- scripts/cs25p6_ingest.py | 27 +++++++++++++++++++-------- 1 file changed, 19 insertions(+), 8 deletions(-) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index d01bbe0..bf4271b 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -184,6 +184,7 @@ def process_fault(fault_name: str) -> FaultResult | None: x = 0.5 + rel["shypo"] / fault.length y = rel["dhypo"] / fault.width hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) + srfgen_seed = rel.get("srfgen_seed") realisations.append( { "rel_id": rel_id, @@ -198,7 +199,7 @@ def process_fault(fault_name: str) -> FaultResult | None: "shypo": rel["shypo"], "dhypo": rel["dhypo"], "seed": int(rel["seed"]), - "srfgen_seed": int(rel["srfgen_seed"]), + "srfgen_seed": None if pd.isna(srfgen_seed) else int(srfgen_seed), "sdrop": rel["sdrop"], } ), @@ -293,9 +294,6 @@ def process_fault(fault_name: str) -> FaultResult | None: def main(data: Path, db_path: Path, workers: int) -> None: - if db_path.exists(): - raise FileExistsError(f"{db_path} already exists") - sites_all = load_sites(data) nhm_faults = nhm.load_nhm(str(data / "NZ_FLTmodel_2010.txt")) @@ -308,13 +306,26 @@ def main(data: Path, db_path: Path, workers: int) -> None: faults.append(fault_name) im_periods = discover_periods(data, faults[0]) - db = IMDB.create( - db_path, periods=im_periods, components=["geom", "rotd50"], db_meta={"dataset_id": "nz_nshm_2010_fault_ims"} - ) + + if db_path.exists(): + db = IMDB(db_path, read_only=False).open() + done_events = set(db.get_events().index) + skipped = [f for f in faults if f in done_events] + faults = [f for f in faults if f not in done_events] + if skipped: + print(f"resuming: skipping {len(skipped)} faults already in {db_path}") + sites_seen = set(db.get_sites().index) + else: + db = IMDB.create( + db_path, + periods=im_periods, + components=("geom", "rotd50"), + db_meta={"dataset_id": "nz_nshm_2010_fault_ims"}, + ) + sites_seen = set() # DuckDB defaults memory_limit to 80% of RAM; left alone its buffer pool grows with the # database and crowds out the worker processes. Capped, it spills to disk instead. db.con.raw_sql(f"PRAGMA memory_limit='{DB_MEMORY_LIMIT}'") - sites_seen: set[str] = set() # Keep only a small window of faults in flight. Pool.imap_unordered/submit-all run every # task as fast as the workers allow and buffer each finished result in the parent, with no From 6eacbc17090c1f6a744484a62d5becf19f23c406 Mon Sep 17 00:00:00 2001 From: Claudio Date: Tue, 15 Sep 2026 15:16:37 +1200 Subject: [PATCH 23/25] Example --- examples/read_imdb.py | 79 ++++++++++++++++++++++++++++++++++++++++ scripts/cs25p6_ingest.py | 3 +- 2 files changed, 81 insertions(+), 1 deletion(-) create mode 100644 examples/read_imdb.py diff --git a/examples/read_imdb.py b/examples/read_imdb.py new file mode 100644 index 0000000..3a34e0b --- /dev/null +++ b/examples/read_imdb.py @@ -0,0 +1,79 @@ +"""Read the different kinds of data an IMDB holds: dimensions, records, spectra and scalars. + +Run against a real database, e.g.: + + uv run python examples/read_imdb.py /path/to/cs25p6_imdb.duckdb +""" + +import argparse +from pathlib import Path + +from imdb import IMDB + + +def main(db_path: Path) -> None: + """Print each kind of data an IMDB holds, for a single event. + + Parameters + ---------- + db_path : Path + Path to an IMDB. + """ + with IMDB(db_path) as db: + # db_meta: dataset-level metadata (schema version, components held, when it was built). + print("db_meta:", db.db_meta) + + # Dimensions: events, realisations, sites. Each is indexed by its stable id. + events = db.get_events() + print(f"\n{len(events)} events, columns: {list(events.columns)}") + event_id = events.index[0] + print(events.loc[event_id]) + + realisations = db.get_realisations() + print( + f"\n{len(realisations)} realisations, columns: {list(realisations.columns)}" + ) + + sites = db.get_sites() + print(f"\n{len(sites)} sites, columns: {list(sites.columns)}") + + # site_event: per (site, event) distance measures (rrup, rjb, rx, ry). + site_event = db.get_site_event(event_ids=[event_id]) + print(f"\nsite_event rows for {event_id}: {len(site_event)}") + + # Records: one row per (rel_id, site_id, component, kind, gmm_key). `kind` distinguishes + # physics-based simulation, empirical GMM prediction, and observed ground motion. + records = db.get_records(event_ids=[event_id]) + print(f"\nrecord kinds for {event_id}:") + print(records["kind"].value_counts()) + + # Response spectra (pSA) and scalar IMs, filtered the same way as get_records: any of its + # keyword filters (event_ids, rel_ids, site_ids, component, kind, gmm_key) can be passed + # straight through via **filters. + simulated_psa = db.get_psa( + event_ids=[event_id], kind="simulated", component="geom" + ) + print(f"\nsimulated pSA (geom component), {len(simulated_psa)} records:") + print(simulated_psa.iloc[:3, :5]) + + # GMM predictions carry a ln-space total sigma alongside each IM (`sigma=True`) + gmm_psa = db.get_psa( + event_ids=[event_id], sigma=True, kind="gmm", component="rotd50" + ) + print(f"\nGMM pSA with sigma (rotd50 component), {len(gmm_psa)} records:") + print(gmm_psa.iloc[:3, :4]) + + simulated_scalars = db.get_scalars( + ims=["PGA", "PGV"], event_ids=[event_id], kind="simulated", component="geom" + ) + print( + f"\nsimulated scalar IMs (geom component), {len(simulated_scalars)} records:" + ) + print(simulated_scalars.head(3)) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("db_path", type=Path, help="Path to an IMDB") + args = parser.parse_args() + main(args.db_path) diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index bf4271b..d83e6ac 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -185,6 +185,7 @@ def process_fault(fault_name: str) -> FaultResult | None: y = rel["dhypo"] / fault.width hypo_lat, hypo_lon, hypo_depth_m = fault.fault_coordinates_to_wgs_depth_coordinates(np.array([x, y])) srfgen_seed = rel.get("srfgen_seed") + sdrop = rel.get("sdrop") realisations.append( { "rel_id": rel_id, @@ -200,7 +201,7 @@ def process_fault(fault_name: str) -> FaultResult | None: "dhypo": rel["dhypo"], "seed": int(rel["seed"]), "srfgen_seed": None if pd.isna(srfgen_seed) else int(srfgen_seed), - "sdrop": rel["sdrop"], + "sdrop": None if pd.isna(sdrop) else sdrop, } ), } From 9f21eb6eee9008b5b887b7833d8696a646904617 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 16 Sep 2026 07:55:19 +1200 Subject: [PATCH 24/25] Tidy up, rename package to ucgmsim-imdb for pypi --- .github/workflows/publish-PyPI.yml | 63 ++++++++++++++++++ imdb/imdb.py | 3 + pyproject.toml | 2 +- scripts/cs25p6_ingest.py | 4 +- tests/test_imdb.py | 8 +++ uv.lock | 100 ++++++++++++++--------------- 6 files changed, 127 insertions(+), 53 deletions(-) create mode 100644 .github/workflows/publish-PyPI.yml diff --git a/.github/workflows/publish-PyPI.yml b/.github/workflows/publish-PyPI.yml new file mode 100644 index 0000000..b1a6f5b --- /dev/null +++ b/.github/workflows/publish-PyPI.yml @@ -0,0 +1,63 @@ +name: Publish to PyPI +run-name: Publish Python Package to PyPI for release ${{ github.event.release.tag_name || inputs.tag_name || github.ref_name }} + +on: + release: + types: [published] + workflow_dispatch: + inputs: + tag_name: + description: "Git tag to checkout and publish" + required: false + type: string + +jobs: + build: + name: Build distribution 📦 + runs-on: ubuntu-latest + + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + ref: ${{ inputs.tag_name || github.ref }} + fetch-depth: 0 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + - name: Install pypa/build + run: >- + python3 -m + pip install + build + --user + - name: Build a binary wheel and a source tarball + run: python3 -m build + - name: Store the distribution packages + uses: actions/upload-artifact@v4 + with: + name: python-package-distributions + path: dist/ + publish-to-pypi: + name: Publish to PyPI + needs: + - build + runs-on: ubuntu-latest + + environment: + name: pypi + url: https://pypi.org/p/ucgmsim-imdb + + permissions: + id-token: write + + steps: + - name: Download all the dists + uses: actions/download-artifact@v4 + with: + name: python-package-distributions + path: dist/ + - name: Publish distribution to PyPI + uses: pypa/gh-action-pypi-publish@release/v1 diff --git a/imdb/imdb.py b/imdb/imdb.py index 3360b2b..e59dd64 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -133,6 +133,9 @@ def create( IMDB The newly created database, open for writing. """ + if Path(path).exists(): + raise FileExistsError(f"{path} already exists") + frequencies = frequencies or [] db = cls(path, read_only=False).open() con = db.con diff --git a/pyproject.toml b/pyproject.toml index 4d7370d..956b333 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -3,7 +3,7 @@ requires = ["setuptools", "setuptools-scm"] build-backend = "setuptools.build_meta" [project] -name = "imdb" +name = "ucgmsim-imdb" authors = [{name="ucgmsim"}] description = "A library for reading and writing intensity measure databases" readme = "README.md" diff --git a/scripts/cs25p6_ingest.py b/scripts/cs25p6_ingest.py index d83e6ac..0f1b737 100644 --- a/scripts/cs25p6_ingest.py +++ b/scripts/cs25p6_ingest.py @@ -7,12 +7,12 @@ # "tqdm", # "qcore-utils", # "source-modelling", -# "imdb", +# "ucgmsim-imdb", # "oq-wrapper", # ] # # [tool.uv.sources] -# imdb = { git = "ssh://git@github.com/ucgmsim/imdb.git", branch = "emp-gmm-support" } +# ucgmsim-imdb = { git = "ssh://git@github.com/ucgmsim/imdb.git", branch = "emp-gmm-support" } # /// """Ingest NZ NSHM 2010 fault ruptures (source_data/im_data) into an IMDB. diff --git a/tests/test_imdb.py b/tests/test_imdb.py index fd5c09b..6baaa6d 100644 --- a/tests/test_imdb.py +++ b/tests/test_imdb.py @@ -9,6 +9,14 @@ from tests.conftest import COMPONENTS, EVENTS, FREQUENCIES, PERIODS, SITES +def test_create_rejects_existing_path(tmp_path): + path = tmp_path / "existing.duckdb" + IMDB.create(path, periods=[]).close() + + with pytest.raises(FileExistsError): + IMDB.create(path, periods=[]) + + def test_round_trip(db): records = db.get_records() n = len(EVENTS) * 2 * len(SITES) * len(COMPONENTS) diff --git a/uv.lock b/uv.lock index 53e4d1b..f58d639 100644 --- a/uv.lock +++ b/uv.lock @@ -402,56 +402,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/01/f9/575c8d760eae1fc99651b7cc5efd96ad5379ca4d6b53750b0fb4fe983f34/imagesize-2.0.1-py3-none-any.whl", hash = "sha256:ea0c9a0384df69ed86a943a15cde37d0360b82491b3910dc2215e202e62b5b02", size = 14794, upload-time = "2026-08-24T12:35:12.548Z" }, ] -[[package]] -name = "imdb" -source = { editable = "." } -dependencies = [ - { name = "ibis-framework", extra = ["duckdb"] }, - { name = "numpy" }, - { name = "pandas" }, -] - -[package.dev-dependencies] -dev = [ - { name = "deptry" }, - { name = "numpydoc" }, - { name = "ruff" }, - { name = "ty" }, -] -test = [ - { name = "coverage" }, - { name = "pytest" }, - { name = "pytest-cov" }, -] -types = [ - { name = "pandas-stubs" }, - { name = "ty" }, -] - -[package.metadata] -requires-dist = [ - { name = "ibis-framework", extras = ["duckdb"], specifier = ">=9" }, - { name = "numpy", specifier = ">=2" }, - { name = "pandas", specifier = ">=3" }, -] - -[package.metadata.requires-dev] -dev = [ - { name = "deptry" }, - { name = "numpydoc" }, - { name = "ruff" }, - { name = "ty" }, -] -test = [ - { name = "coverage", extras = ["toml"] }, - { name = "pytest" }, - { name = "pytest-cov" }, -] -types = [ - { name = "pandas-stubs" }, - { name = "ty" }, -] - [[package]] name = "iniconfig" version = "2.3.0" @@ -1103,6 +1053,56 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e5/6d/b53b99a9f2766d095985947a5782f1702cabb129a34f7a802d7197af832f/tzdata-2026.3-py2.py3-none-any.whl", hash = "sha256:dc096730c87af6cab1b171c9d532be840741ff5d459015e7f6947bd7d7e54931", size = 348168, upload-time = "2026-07-10T08:50:36.46Z" }, ] +[[package]] +name = "ucgmsim-imdb" +source = { editable = "." } +dependencies = [ + { name = "ibis-framework", extra = ["duckdb"] }, + { name = "numpy" }, + { name = "pandas" }, +] + +[package.dev-dependencies] +dev = [ + { name = "deptry" }, + { name = "numpydoc" }, + { name = "ruff" }, + { name = "ty" }, +] +test = [ + { name = "coverage" }, + { name = "pytest" }, + { name = "pytest-cov" }, +] +types = [ + { name = "pandas-stubs" }, + { name = "ty" }, +] + +[package.metadata] +requires-dist = [ + { name = "ibis-framework", extras = ["duckdb"], specifier = ">=9" }, + { name = "numpy", specifier = ">=2" }, + { name = "pandas", specifier = ">=3" }, +] + +[package.metadata.requires-dev] +dev = [ + { name = "deptry" }, + { name = "numpydoc" }, + { name = "ruff" }, + { name = "ty" }, +] +test = [ + { name = "coverage", extras = ["toml"] }, + { name = "pytest" }, + { name = "pytest-cov" }, +] +types = [ + { name = "pandas-stubs" }, + { name = "ty" }, +] + [[package]] name = "urllib3" version = "2.7.0" From 2187db37b2d12372d20a7765caccc7e6ae648cc7 Mon Sep 17 00:00:00 2001 From: Claudio Date: Wed, 16 Sep 2026 07:57:25 +1200 Subject: [PATCH 25/25] PR check fixes --- imdb/imdb.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/imdb/imdb.py b/imdb/imdb.py index e59dd64..23599a7 100644 --- a/imdb/imdb.py +++ b/imdb/imdb.py @@ -135,7 +135,7 @@ def create( """ if Path(path).exists(): raise FileExistsError(f"{path} already exists") - + frequencies = frequencies or [] db = cls(path, read_only=False).open() con = db.con @@ -172,7 +172,7 @@ def create( "n_periods": str(len(periods)), "n_frequencies": str(len(frequencies)), "created_at": datetime.datetime.now(datetime.UTC).isoformat(), - "imdb_version": version("imdb"), + "imdb_version": version("ucgmsim-imdb"), **(db_meta or {}), } con.insert(