Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions LICENSE
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
Copyright (c) 2026 Anonymous Authors

This work is licensed under CC BY-NC-ND 4.0.
https://creativecommons.org/licenses/by-nc-nd/4.0/

This license is temporary and applies only during the review period.
The license will be updated upon public release.
67 changes: 39 additions & 28 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,10 +4,30 @@ URSA is a framework for evaluating retrosynthetic routes: it checks the structur

## Installation

ChemCensor is not published. Clone the ChemCensor repository into `./chemcensor` at the repo root (this directory is gitignored), then install dependencies:
### 1. Install ChemCensor dependency
ChemCensor is not published on PyPI. Download the repository archive from
https://anonymous.4open.science/r/ChemCensor-81B0/ (use the **Download ZIP**
button), then unpack it into `./chemcensor` at the URSA repo root
(this directory is gitignored):

```bash
unzip /path/to/downloads/ChemCensor-81B0.zip
mv ChemCensor-81B0 chemcensor
```

### 2. Install uv

We recommend using uv for fast, reliable dependency management.

```bash
curl -LsSf https://astral.sh/uv/install.sh | sh
echo 'export PATH="$HOME/.local/bin:$PATH"' >> ~/.bashrc
source ~/.bashrc
```

### 3. Install URSA dependencies

```bash
git clone https://github.com/<ORG_OR_USER>/chemcensor.git chemcensor
uv sync --extra dev
```

Expand All @@ -18,7 +38,22 @@ Before working with benchmarks or the full scoring pipeline locally, download th
- `data/building_blocks/URSA_BBs_v1_0_0.csv` — building-block catalog for `BuildingBlockChecker`
- `data/chemcensor_db/ChemCensor_DB_v1_0_0.sqlite` — ChemCensor SQLite database for scoring

Download links: *to be added.*
Download links: https://osf.io/wms6r/overview?view_only=073f80629f674f9084d95b7efc9e01ba

```bash
mkdir -p data/building_blocks data/chemcensor_db
unzip /path/to/downloads/'D3. URSA_BBs.csv.zip' -d data/building_blocks
unzip /path/to/downloads/'D6. URSA-minor-0.5.2-U2_database.zip' -d data/chemcensor_db
```

Once the zip contents are extracted, build the ChemCensor SQLite database from
the Parquet export:

```bash
python scripts/import_parquet_to_sqlite.py \
--parquet-dir data/chemcensor_db/uspto_full_parquet \
--out-sqlite data/chemcensor_db/ChemCensor_DB_v1_0_0.sqlite
```

## Example run

Expand All @@ -31,7 +66,7 @@ Download links: *to be added.*
Score the bundle:

```bash
ursa-bench \
uv run ursa-bench \
--input data/example_data/bundled.json.gz \
--adapter retrochimera \
--benchmark EXPERT_2026 \
Expand Down Expand Up @@ -72,24 +107,6 @@ retrochimera, retrostar, synplanner, syntheseus, synllama

Skip `--adapter` and use `--routes` when you already have RetroCast-formatted routes on disk — useful for caching the adaptation step across multiple benchmark runs.

## Programmatic API

```python
from ursa import Ursa, RetrosyntheticPath

ursa = Ursa()

# single path
result = ursa.score(path)

# dataset
dataset_result = ursa.score_dataset(paths, target_smiles=target_smiles)
print(dataset_result.metrics.solv_2)
dataset_result.save("data/results", stem="example")
```

`Ursa` orchestrates the pipeline: `PathConsistencyChecker` → `BuildingBlockChecker` → `PathCollapser` → `PathScorer` → `BestPathSelector` → `DatasetMetricsCalculator`.

## Built-in benchmark sets

Located in `data/URSA_benchmarking_sets/`:
Expand All @@ -99,9 +116,3 @@ Located in `data/URSA_benchmarking_sets/`:
- `USPTO_190`

A custom set of target molecules can be supplied by passing a CSV path instead of a preset name (default columns: `Structure ID`, `SMILES`; overridable via `--id-col` / `--smiles-col`).

## Tests

```bash
pytest
```
211 changes: 211 additions & 0 deletions scripts/import_parquet_to_sqlite.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,211 @@
#!/usr/bin/env python3
"""
Import the ChemCensor reaction-center Parquet exports back into a SQLite DB.

Expected Parquet files (produced by scripts/export_uspto_full_sqlite_to_csv.py):
- reaction_centers.parquet
- centers_to_reactions.parquet
- reactions.parquet

This recreates the original schema:
- reaction_centers(
reaction_center_smiles PK,
fg_signature BLOB,
is_multi_component INT,
components TEXT
)
- centers_to_reactions(
PK(reaction_center_smiles,reaction_smiles),
fg_signature BLOB,
sear_signature BLOB
)
- reactions(PK(reaction_smiles,document_id))
and the index:
- idx_ctr_reaction_smiles ON centers_to_reactions(reaction_smiles)
"""
from __future__ import annotations

import argparse
import sqlite3
from pathlib import Path
from typing import Iterable


def _require_pyarrow():
try:
import pyarrow.parquet as pq # type: ignore
except ModuleNotFoundError as e: # pragma: no cover
raise ModuleNotFoundError(
"pyarrow is required for Parquet import. Install it with "
"`pip install pyarrow` (or add it to your environment)."
) from e
return pq


SCHEMA_SQL = """
CREATE TABLE reaction_centers (
reaction_center_smiles TEXT PRIMARY KEY,
fg_signature BLOB NOT NULL DEFAULT (x''),
is_multi_component INTEGER NOT NULL DEFAULT 0,
components TEXT NOT NULL DEFAULT '[]'
);
CREATE TABLE centers_to_reactions (
reaction_center_smiles TEXT NOT NULL
REFERENCES reaction_centers(reaction_center_smiles),
reaction_smiles TEXT NOT NULL,
fg_signature BLOB NOT NULL DEFAULT (x''),
sear_signature BLOB NOT NULL DEFAULT (x''),
PRIMARY KEY (reaction_center_smiles, reaction_smiles)
);
CREATE TABLE reactions (
reaction_smiles TEXT NOT NULL,
document_id TEXT NOT NULL,
PRIMARY KEY (reaction_smiles, document_id)
);
CREATE INDEX idx_ctr_reaction_smiles
ON centers_to_reactions(reaction_smiles);
""".strip()


def _connect(out_sqlite: Path) -> sqlite3.Connection:
out_sqlite.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(out_sqlite))
conn.execute("PRAGMA journal_mode=WAL;")
conn.execute("PRAGMA synchronous=NORMAL;")
conn.execute("PRAGMA temp_store=MEMORY;")
return conn


def _init_schema(conn: sqlite3.Connection) -> None:
conn.executescript(
"""
DROP TABLE IF EXISTS centers_to_reactions;
DROP TABLE IF EXISTS reaction_centers;
DROP TABLE IF EXISTS reactions;
"""
)
conn.executescript(SCHEMA_SQL)


def _iter_batches(parquet_path: Path, batch_size: int):
pq = _require_pyarrow()
pf = pq.ParquetFile(str(parquet_path))
# iter_batches yields RecordBatch with columnar arrays (often zero-copy).
yield from pf.iter_batches(batch_size=batch_size)


def _executemany(conn: sqlite3.Connection, sql: str, rows: Iterable[tuple]) -> None:
conn.executemany(sql, rows)


def import_reaction_centers(
conn: sqlite3.Connection,
parquet_path: Path,
batch_size: int,
) -> None:
insert_sql = """
INSERT OR REPLACE INTO reaction_centers
(reaction_center_smiles, fg_signature, is_multi_component, components)
VALUES (?, ?, ?, ?)
"""
for batch in _iter_batches(parquet_path, batch_size=batch_size):
cols = batch.to_pydict()
# Export uses `components_json` column name; map back to `components`.
rows = zip(
cols["reaction_center_smiles"],
cols["fg_signature"],
cols["is_multi_component"],
cols["components_json"],
strict=True,
)
_executemany(conn, insert_sql, rows)


def import_reactions(
conn: sqlite3.Connection, parquet_path: Path, batch_size: int
) -> None:
insert_sql = """
INSERT OR REPLACE INTO reactions
(reaction_smiles, document_id)
VALUES (?, ?)
"""
for batch in _iter_batches(parquet_path, batch_size=batch_size):
cols = batch.to_pydict()
rows = zip(cols["reaction_smiles"], cols["document_id"], strict=True)
_executemany(conn, insert_sql, rows)


def import_centers_to_reactions(
conn: sqlite3.Connection, parquet_path: Path, batch_size: int
) -> None:
insert_sql = """
INSERT OR REPLACE INTO centers_to_reactions
(reaction_center_smiles, reaction_smiles, fg_signature, sear_signature)
VALUES (?, ?, ?, ?)
"""
for batch in _iter_batches(parquet_path, batch_size=batch_size):
cols = batch.to_pydict()
rows = zip(
cols["reaction_center_smiles"],
cols["reaction_smiles"],
cols["fg_signature"],
cols["sear_signature"],
strict=True,
)
_executemany(conn, insert_sql, rows)


def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument(
"--parquet-dir",
default="data/chemcensor_db/uspto_full_parquet",
help="Directory containing the Parquet exports.",
)
parser.add_argument(
"--out-sqlite",
default="uspto_full_roundtrip.sqlite",
help="Output SQLite path to create/overwrite.",
)
parser.add_argument(
"--batch-size",
type=int,
default=50_000,
help="Row batch size for Parquet -> SQLite inserts.",
)
args = parser.parse_args()

parquet_dir = Path(args.parquet_dir)
out_sqlite = Path(args.out_sqlite)

reaction_centers_pq = parquet_dir / "reaction_centers.parquet"
centers_to_reactions_pq = parquet_dir / "centers_to_reactions.parquet"
reactions_pq = parquet_dir / "reactions.parquet"

for p in [reaction_centers_pq, centers_to_reactions_pq, reactions_pq]:
if not p.exists():
raise FileNotFoundError(p)

conn = _connect(out_sqlite)
try:
_init_schema(conn)
with conn:
# Order matters because of FK: reaction_centers first.
import_reaction_centers(
conn,
reaction_centers_pq,
batch_size=args.batch_size,
)
import_reactions(conn, reactions_pq, batch_size=args.batch_size)
import_centers_to_reactions(
conn, centers_to_reactions_pq, batch_size=args.batch_size
)
finally:
conn.close()

print(str(out_sqlite))
return 0


if __name__ == "__main__":
raise SystemExit(main())
Loading