diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..705131e --- /dev/null +++ b/LICENSE @@ -0,0 +1,7 @@ +Copyright (c) 2026 Anonymous Authors + +This work is licensed under CC BY-NC-ND 4.0. +https://creativecommons.org/licenses/by-nc-nd/4.0/ + +This license is temporary and applies only during the review period. +The license will be updated upon public release. \ No newline at end of file diff --git a/README.md b/README.md index d7363f7..db1bd98 100644 --- a/README.md +++ b/README.md @@ -4,10 +4,30 @@ URSA is a framework for evaluating retrosynthetic routes: it checks the structur ## Installation -ChemCensor is not published. Clone the ChemCensor repository into `./chemcensor` at the repo root (this directory is gitignored), then install dependencies: +### 1. Install ChemCensor dependency +ChemCensor is not published on PyPI. Download the repository archive from +https://anonymous.4open.science/r/ChemCensor-81B0/ (use the **Download ZIP** +button), then unpack it into `./chemcensor` at the URSA repo root +(this directory is gitignored): + +```bash +unzip /path/to/downloads/ChemCensor-81B0.zip +mv ChemCensor-81B0 chemcensor +``` + +### 2. Install uv + +We recommend using uv for fast, reliable dependency management. + +```bash +curl -LsSf https://astral.sh/uv/install.sh | sh +echo 'export PATH="$HOME/.local/bin:$PATH"' >> ~/.bashrc +source ~/.bashrc +``` + +### 3. Install URSA dependencies ```bash -git clone https://github.com//chemcensor.git chemcensor uv sync --extra dev ``` @@ -18,7 +38,22 @@ Before working with benchmarks or the full scoring pipeline locally, download th - `data/building_blocks/URSA_BBs_v1_0_0.csv` — building-block catalog for `BuildingBlockChecker` - `data/chemcensor_db/ChemCensor_DB_v1_0_0.sqlite` — ChemCensor SQLite database for scoring -Download links: *to be added.* +Download links: https://osf.io/wms6r/overview?view_only=073f80629f674f9084d95b7efc9e01ba + +```bash +mkdir -p data/building_blocks data/chemcensor_db +unzip /path/to/downloads/'D3. URSA_BBs.csv.zip' -d data/building_blocks +unzip /path/to/downloads/'D6. URSA-minor-0.5.2-U2_database.zip' -d data/chemcensor_db +``` + +Once the zip contents are extracted, build the ChemCensor SQLite database from +the Parquet export: + +```bash +python scripts/import_parquet_to_sqlite.py \ + --parquet-dir data/chemcensor_db/uspto_full_parquet \ + --out-sqlite data/chemcensor_db/ChemCensor_DB_v1_0_0.sqlite +``` ## Example run @@ -31,7 +66,7 @@ Download links: *to be added.* Score the bundle: ```bash -ursa-bench \ +uv run ursa-bench \ --input data/example_data/bundled.json.gz \ --adapter retrochimera \ --benchmark EXPERT_2026 \ @@ -72,24 +107,6 @@ retrochimera, retrostar, synplanner, syntheseus, synllama Skip `--adapter` and use `--routes` when you already have RetroCast-formatted routes on disk — useful for caching the adaptation step across multiple benchmark runs. -## Programmatic API - -```python -from ursa import Ursa, RetrosyntheticPath - -ursa = Ursa() - -# single path -result = ursa.score(path) - -# dataset -dataset_result = ursa.score_dataset(paths, target_smiles=target_smiles) -print(dataset_result.metrics.solv_2) -dataset_result.save("data/results", stem="example") -``` - -`Ursa` orchestrates the pipeline: `PathConsistencyChecker` → `BuildingBlockChecker` → `PathCollapser` → `PathScorer` → `BestPathSelector` → `DatasetMetricsCalculator`. - ## Built-in benchmark sets Located in `data/URSA_benchmarking_sets/`: @@ -99,9 +116,3 @@ Located in `data/URSA_benchmarking_sets/`: - `USPTO_190` A custom set of target molecules can be supplied by passing a CSV path instead of a preset name (default columns: `Structure ID`, `SMILES`; overridable via `--id-col` / `--smiles-col`). - -## Tests - -```bash -pytest -``` diff --git a/scripts/import_parquet_to_sqlite.py b/scripts/import_parquet_to_sqlite.py new file mode 100644 index 0000000..a00f291 --- /dev/null +++ b/scripts/import_parquet_to_sqlite.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +""" +Import the ChemCensor reaction-center Parquet exports back into a SQLite DB. + +Expected Parquet files (produced by scripts/export_uspto_full_sqlite_to_csv.py): +- reaction_centers.parquet +- centers_to_reactions.parquet +- reactions.parquet + +This recreates the original schema: +- reaction_centers( + reaction_center_smiles PK, + fg_signature BLOB, + is_multi_component INT, + components TEXT + ) +- centers_to_reactions( + PK(reaction_center_smiles,reaction_smiles), + fg_signature BLOB, + sear_signature BLOB + ) +- reactions(PK(reaction_smiles,document_id)) +and the index: +- idx_ctr_reaction_smiles ON centers_to_reactions(reaction_smiles) +""" +from __future__ import annotations + +import argparse +import sqlite3 +from pathlib import Path +from typing import Iterable + + +def _require_pyarrow(): + try: + import pyarrow.parquet as pq # type: ignore + except ModuleNotFoundError as e: # pragma: no cover + raise ModuleNotFoundError( + "pyarrow is required for Parquet import. Install it with " + "`pip install pyarrow` (or add it to your environment)." + ) from e + return pq + + +SCHEMA_SQL = """ +CREATE TABLE reaction_centers ( + reaction_center_smiles TEXT PRIMARY KEY, + fg_signature BLOB NOT NULL DEFAULT (x''), + is_multi_component INTEGER NOT NULL DEFAULT 0, + components TEXT NOT NULL DEFAULT '[]' +); +CREATE TABLE centers_to_reactions ( + reaction_center_smiles TEXT NOT NULL + REFERENCES reaction_centers(reaction_center_smiles), + reaction_smiles TEXT NOT NULL, + fg_signature BLOB NOT NULL DEFAULT (x''), + sear_signature BLOB NOT NULL DEFAULT (x''), + PRIMARY KEY (reaction_center_smiles, reaction_smiles) +); +CREATE TABLE reactions ( + reaction_smiles TEXT NOT NULL, + document_id TEXT NOT NULL, + PRIMARY KEY (reaction_smiles, document_id) +); +CREATE INDEX idx_ctr_reaction_smiles + ON centers_to_reactions(reaction_smiles); +""".strip() + + +def _connect(out_sqlite: Path) -> sqlite3.Connection: + out_sqlite.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(out_sqlite)) + conn.execute("PRAGMA journal_mode=WAL;") + conn.execute("PRAGMA synchronous=NORMAL;") + conn.execute("PRAGMA temp_store=MEMORY;") + return conn + + +def _init_schema(conn: sqlite3.Connection) -> None: + conn.executescript( + """ + DROP TABLE IF EXISTS centers_to_reactions; + DROP TABLE IF EXISTS reaction_centers; + DROP TABLE IF EXISTS reactions; + """ + ) + conn.executescript(SCHEMA_SQL) + + +def _iter_batches(parquet_path: Path, batch_size: int): + pq = _require_pyarrow() + pf = pq.ParquetFile(str(parquet_path)) + # iter_batches yields RecordBatch with columnar arrays (often zero-copy). + yield from pf.iter_batches(batch_size=batch_size) + + +def _executemany(conn: sqlite3.Connection, sql: str, rows: Iterable[tuple]) -> None: + conn.executemany(sql, rows) + + +def import_reaction_centers( + conn: sqlite3.Connection, + parquet_path: Path, + batch_size: int, +) -> None: + insert_sql = """ + INSERT OR REPLACE INTO reaction_centers + (reaction_center_smiles, fg_signature, is_multi_component, components) + VALUES (?, ?, ?, ?) + """ + for batch in _iter_batches(parquet_path, batch_size=batch_size): + cols = batch.to_pydict() + # Export uses `components_json` column name; map back to `components`. + rows = zip( + cols["reaction_center_smiles"], + cols["fg_signature"], + cols["is_multi_component"], + cols["components_json"], + strict=True, + ) + _executemany(conn, insert_sql, rows) + + +def import_reactions( + conn: sqlite3.Connection, parquet_path: Path, batch_size: int +) -> None: + insert_sql = """ + INSERT OR REPLACE INTO reactions + (reaction_smiles, document_id) + VALUES (?, ?) + """ + for batch in _iter_batches(parquet_path, batch_size=batch_size): + cols = batch.to_pydict() + rows = zip(cols["reaction_smiles"], cols["document_id"], strict=True) + _executemany(conn, insert_sql, rows) + + +def import_centers_to_reactions( + conn: sqlite3.Connection, parquet_path: Path, batch_size: int +) -> None: + insert_sql = """ + INSERT OR REPLACE INTO centers_to_reactions + (reaction_center_smiles, reaction_smiles, fg_signature, sear_signature) + VALUES (?, ?, ?, ?) + """ + for batch in _iter_batches(parquet_path, batch_size=batch_size): + cols = batch.to_pydict() + rows = zip( + cols["reaction_center_smiles"], + cols["reaction_smiles"], + cols["fg_signature"], + cols["sear_signature"], + strict=True, + ) + _executemany(conn, insert_sql, rows) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument( + "--parquet-dir", + default="data/chemcensor_db/uspto_full_parquet", + help="Directory containing the Parquet exports.", + ) + parser.add_argument( + "--out-sqlite", + default="uspto_full_roundtrip.sqlite", + help="Output SQLite path to create/overwrite.", + ) + parser.add_argument( + "--batch-size", + type=int, + default=50_000, + help="Row batch size for Parquet -> SQLite inserts.", + ) + args = parser.parse_args() + + parquet_dir = Path(args.parquet_dir) + out_sqlite = Path(args.out_sqlite) + + reaction_centers_pq = parquet_dir / "reaction_centers.parquet" + centers_to_reactions_pq = parquet_dir / "centers_to_reactions.parquet" + reactions_pq = parquet_dir / "reactions.parquet" + + for p in [reaction_centers_pq, centers_to_reactions_pq, reactions_pq]: + if not p.exists(): + raise FileNotFoundError(p) + + conn = _connect(out_sqlite) + try: + _init_schema(conn) + with conn: + # Order matters because of FK: reaction_centers first. + import_reaction_centers( + conn, + reaction_centers_pq, + batch_size=args.batch_size, + ) + import_reactions(conn, reactions_pq, batch_size=args.batch_size) + import_centers_to_reactions( + conn, centers_to_reactions_pq, batch_size=args.batch_size + ) + finally: + conn.close() + + print(str(out_sqlite)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())