From 11baccdc71ec15bc557808e4e7f81748f33c8352 Mon Sep 17 00:00:00 2001 From: "google-labs-jules[bot]" <161369871+google-labs-jules[bot]@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:41:26 +0000 Subject: [PATCH] perf: replace pandas iterrows with to_dict for faster iteration Iterating over Pandas DataFrames with `iterrows()` is notoriously slow because it wraps every row into a Pandas Series object. By converting the DataFrame to a list of Python dictionaries upfront via `to_dict('records')`, we bypass Series creation overhead and significantly speed up the iteration required for structural alignment verification. Co-authored-by: alinelena <3306823+alinelena@users.noreply.github.com> --- .jules/bolt.md | 4 ++++ src/lavello_mlips/verify_processed_omol25.py | 8 +++++--- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/.jules/bolt.md b/.jules/bolt.md index d87bd6d..f6757f7 100644 --- a/.jules/bolt.md +++ b/.jules/bolt.md @@ -8,3 +8,7 @@ ## 2024-03-29 - ASE Custom JSON encoding vs standard JSON **Learning:** ASE's custom JSON encoder (`ase.io.jsonio.encode`) will generate dicts with special keys like `__ndarray__` or `__complex__` (e.g. `{"__ndarray__": [[5], "int64", ...]}`). When optimizing JSON deserialization using faster alternatives like `orjson`, it's critical to realize that a normal `json.loads` or `orjson.loads` will deserialize this into a Python dictionary, while ASE's custom `decode` will properly reconstruct the underlying numpy array. Bypassing ASE's decoder without checking for these keys leads to downstream type errors (e.g. `KeyError: '__ndarray__'`). **Action:** When replacing or wrapping ASE's jsonio with `orjson`, always fall back to ASE's `decode` if the payload string contains `__ndarray__` or `__complex__` markers, to ensure custom objects are correctly reconstructed. + +## 2024-05-19 - Replacing Pandas `df.iterrows()` with `df.to_dict('records')` +**Learning:** Iterating over a Pandas DataFrame using `df.iterrows()` creates massive performance bottlenecks because it wraps every row in a full Pandas Series object. Profiling showed that for large DataFrames, converting the entire DataFrame to a list of dictionaries via `df.to_dict('records')` first, and then iterating over the resulting native Python dictionaries, is significantly faster. +**Action:** Always replace `df.iterrows()` with `df.to_dict('records')` when processing data row-by-row. Make sure to update any downstream code that relies on Pandas Series-specific methods (like `row.to_dict()`) to use native dict features instead (like `dict(row)`). diff --git a/src/lavello_mlips/verify_processed_omol25.py b/src/lavello_mlips/verify_processed_omol25.py index 9d1c47c..a400711 100644 --- a/src/lavello_mlips/verify_processed_omol25.py +++ b/src/lavello_mlips/verify_processed_omol25.py @@ -77,8 +77,10 @@ def main() -> None: ) logger.info(f"Loaded {len(df)} records from Parquet.") - parquet_by_sha = {row["geom_sha1"]: row for _, row in df.iterrows()} - parquet_by_argone_rel = {row["argonne_rel"]: row for _, row in df.iterrows()} + # Optimize DataFrame iteration: avoid df.iterrows() + records = df.to_dict("records") + parquet_by_sha = {row["geom_sha1"]: row for row in records} + parquet_by_argone_rel = {row["argonne_rel"]: row for row in records} logger.info(f"Loading ExtXYZ file from {args.extxyz} (this may take a moment)...") all_atoms = read(str(args.extxyz), index=":") if not isinstance(all_atoms, list): @@ -108,7 +110,7 @@ def get_dump_entry(at): info = dict(at.info) rel = info.get("argonne_rel") pq_row = parquet_by_argone_rel.get(rel) - pq_data = pq_row.to_dict() if pq_row is not None else None + pq_data = dict(pq_row) if pq_row is not None else None return {"xyz": info, "parquet": pq_data} duplicates = {