diff --git a/.gitignore b/.gitignore index 04ce48a..914b9d5 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,10 @@ +flask_tools/pipette/_java_build/ +.podman-run/ +.podman-root/ +.podman-runroot/ +.podman-tmp/ +.rdt-podman.env + # Logs logs *.log diff --git a/flask_tools/pipette/README.md b/flask_tools/pipette/README.md index 9aec781..3f31409 100644 --- a/flask_tools/pipette/README.md +++ b/flask_tools/pipette/README.md @@ -8,6 +8,9 @@ The current pipeline includes: - reaction SMILES parsing - basic SMILES validation - exact-match checker interfaces for reaction databases +- graph based balancing - Attempt to balance reaction by adding copies of reactants. Good for dimerization reactions + - If this is enabled, the reaction fixing LLM is called here instead of later +- atom mapping - reaction fixing LLM call - Run if no exact match is found - If new reaction is returned, goes back to start. Only allowed to run once in a pipeline @@ -150,20 +153,14 @@ The tools can also be disabled by setting `tool_list: null` in the the config. "skipped_reason": null }, { - "name": "llm_reaction_fix", + "name": "RDTAtomMapper", "status": "pass", "data": { - "original_reaction_smiles": "Cn1cnc2c1c(=O)[nH]c(=O)n2C.CI>>CN1C=NC2=C1C(=O)N(C(=O)N2C)C", - "fixed_reaction_smiles": "CI.Cn1cnc2c1c(=O)[nH]c(=O)n2C>>Cn1c(=O)c2c(ncn2C)n(C)c1=O.[H+].[I-]", - "removed_agents": [], - "added_reactants": [], - "added_products": [ - "[H+]", - "[I-]" - ] + "input_reaction_smiles": "CI.Cn1cnc2c1c(=O)[nH]c(=O)n2C>>Cn1c(=O)c2c(ncn2C)n(C)c1=O.I", + "mapped_reaction_smiles": "[O:10]=[C:9]1[NH:11][C:12](=[O:13])[N:14]([C:7]=2[N:6]=[CH:5][N:4]([C:8]12)[CH3:3])[CH3:15].[I:2][CH3:1]>>[N:14]1([C:7]=2[N:6]=[CH:5][N:4]([CH3:3])[C:8]2[C:9]([N:11]([CH3:1])[C:12]1=[O:13])=[O:10])[CH3:15].[IH:2]", + "product_to_reactant": [] }, - "comment": "N-methylation of the xanthine NH with methyl iodide requires HI as the byproduct, represented as [H+] and [I-]. No agents were present to remove.", - "skipped_reason": null + "comment": "RDT atom mapping completed." }, { "name": "basic_smiles_validation", @@ -236,3 +233,46 @@ config = PipetteConfig.from_yaml("my-config.yaml") `pytest` Or `pytest -m llm_query` to run the tests that use LLM + +# ReactionDecoder / RDT + +`pipette` includes a Python wrapper around the Java-based [ReactionDecoder Tool](https://github.com/asad/ReactionDecoder) (RDT). There is accomplished with a short Java wrapper script that calls ReactionDecoder, and gets called by the Python wrapper script. + +The wrapper uses `RDT`'s built-in defaults for mapping options. + +To compile the Java wrapper script, you must first build a fat jar of RDT. + +You must have Java 25 installed. Check your existing java with `java --version`. The `install_rdt.sh` script will install Java and Maven, and then compile the RDT java wrapper. + +```bash +./scripts/install_rdt.sh +export PIPETTE_RDT_JAR=/absolute/path/to/ReactionDecoder/target/rdt-4.0.0-jar-with-dependencies.jar +export PIPETTE_RDT_HELPER_BUILD_DIR=/absolute/path/to/flask-tools/flask_tools/pipette/_java_build +``` + +Use it from Python: + +```python +from flask_tools.pipette.verifiers.rdt import ( + map_reaction_smiles_with_rdt, + map_reaction_smiles_list_with_rdt, +) + +mapped = map_reaction_smiles_with_rdt("CC(=O)O.OCC>>CC(=O)OCC.O") +mapped_many = map_reaction_smiles_list_with_rdt( + [ + "CC(=O)O.OCC>>CC(=O)OCC.O", + "CCO>>CC=O", + ] +) +``` + +Or from the CLI: + +```bash +pipette-rdt --rxn-smi 'CC(=O)O.OCC>>CC(=O)OCC.O' +# Outputs [O:3]=[C:2]([OH:4])[CH3:1].[OH:5][CH2:6][CH3:7]>>[O:5]([C:2]([CH3:1])=[O:4])[CH2:6][CH3:7].[OH2:3] +pipette-rdt --file reactions.txt --json +``` + +If the input uses `reactants>agents>products`, the wrapper strips agents for RDT, maps the core reaction, and then reinserts the original agents into the returned reaction SMILES. diff --git a/flask_tools/pipette/__init__.py b/flask_tools/pipette/__init__.py index 995f32c..23b4fdb 100644 --- a/flask_tools/pipette/__init__.py +++ b/flask_tools/pipette/__init__.py @@ -5,7 +5,6 @@ ## SPDX-License-Identifier: Apache-2.0 ############################################################################### -from .grade_rxn import grade_reaction from .constants import FinalGrade, ReactionGrade, ToolResult, ToolStatus __all__ = [ @@ -15,3 +14,11 @@ "ToolStatus", "grade_reaction", ] + + +def __getattr__(name: str): + if name == "grade_reaction": + from .grade_rxn import grade_reaction + + return grade_reaction + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") diff --git a/flask_tools/pipette/assets/ai_judge_no_dft.yaml b/flask_tools/pipette/assets/ai_judge_no_dft.yaml index 5980521..1dbc260 100644 --- a/flask_tools/pipette/assets/ai_judge_no_dft.yaml +++ b/flask_tools/pipette/assets/ai_judge_no_dft.yaml @@ -1,4 +1,5 @@ llm_judge: + enable_atom_mapping_dict_in_prompt: false allow_fail: - exact_match settings: diff --git a/flask_tools/pipette/assets/ai_judge_no_dft_atom_mapper.yaml b/flask_tools/pipette/assets/ai_judge_no_dft_atom_mapper.yaml new file mode 100644 index 0000000..f1997dd --- /dev/null +++ b/flask_tools/pipette/assets/ai_judge_no_dft_atom_mapper.yaml @@ -0,0 +1,14 @@ +llm_judge: + allow_fail: + - exact_match +tool_list: + - "basic_smiles_validation" + - "exact_match" + - "charge_conservation" + - "mass_conservation" + - "reaction_energy" +settings: + stop_on_hard_fail: true + mass_tolerance_atoms: 0 + reaction_energy_max_ev_mol: 0.2 + use_dft: false diff --git a/flask_tools/pipette/assets/ai_judge_with_dft.yaml b/flask_tools/pipette/assets/ai_judge_with_dft.yaml index fa9b92a..994bc35 100644 --- a/flask_tools/pipette/assets/ai_judge_with_dft.yaml +++ b/flask_tools/pipette/assets/ai_judge_with_dft.yaml @@ -1,4 +1,9 @@ -tool_list: all +tool_list: + - "basic_smiles_validation" + - "exact_match" + - "charge_conservation" + - "mass_conservation" + - "reaction_energy" tools_settings: reaction_energy: database: null diff --git a/flask_tools/pipette/config.py b/flask_tools/pipette/config.py index b48c903..0ff0de0 100644 --- a/flask_tools/pipette/config.py +++ b/flask_tools/pipette/config.py @@ -27,6 +27,11 @@ def package_config_path(filename: str) -> Path: return Path(__file__).with_name("assets") / filename +def graph_rxn_mapper_prompt_path(filename: str) -> Path: + # Relative to top level of pipette module / graph_rxn_mapper / prmopts + return Path(__file__).with_name("graph_rxn_mapper") / "prompts" / filename + + def _validate_mapping_format(data: object, *, name: str) -> dict[str, Any]: if data is None: return {} @@ -47,6 +52,23 @@ def _resolve_optional_path(path_value: object, *, base_dir: Path) -> Path | None return candidate +def _resolve_defaultable_cwd_path( + path_value: object, + *, + default_path: Path, + name: str, +) -> Path: + if path_value is None or path_value == "default": + return default_path + if not isinstance(path_value, str): + raise ValueError(f"{name} must be a string, 'default', or null.") + + candidate = Path(path_value).expanduser() + if not candidate.is_absolute(): + candidate = (Path.cwd() / candidate).resolve() + return candidate + + @dataclass class PipelineConfig: stop_on_hard_fail: bool = True @@ -136,6 +158,7 @@ def _llm_kwargs_from_mapping( @dataclass class LLMJudgeConfig(LLMConfig): allow_fail: Literal["all"] | list[str] = field(default_factory=list) + enable_atom_mapping_dict_in_prompt: bool = False prompt_path: Path = field( default_factory=lambda: package_config_path("judge-prompt.txt") ) @@ -156,8 +179,17 @@ def from_mapping( raise ValueError( "llm_judge.allow_fail must be 'all' or a list of tool names." ) + enable_atom_mapping_dict_in_prompt = mapping.get( + "enable_atom_mapping_dict_in_prompt", + cls.enable_atom_mapping_dict_in_prompt, + ) + if not isinstance(enable_atom_mapping_dict_in_prompt, bool): + raise ValueError( + "llm_judge.enable_atom_mapping_dict_in_prompt must be a boolean." + ) return cls( allow_fail=allow_fail if allow_fail == "all" else list(allow_fail), + enable_atom_mapping_dict_in_prompt=enable_atom_mapping_dict_in_prompt, **cls._llm_kwargs_from_mapping( mapping, name="llm_judge", @@ -214,9 +246,81 @@ def from_mapping( ) +@dataclass +class LLMAtomMappingConfig: + url: str = DEFAULT_LLM_BASE_URL + model: str = "gpt-5.4" + reasoning_effort: ReasoningEffort = "medium" + api_key: str | None = None + system_prompt_path: Path = field( + default_factory=lambda: graph_rxn_mapper_prompt_path("atom_mapping_system.md") + ) + user_prompt_path: Path = field( + default_factory=lambda: graph_rxn_mapper_prompt_path("atom_mapping_user.md") + ) + skill_prompt_path: Path = field( + default_factory=lambda: graph_rxn_mapper_prompt_path("atom_mapping_skill.md") + ) + + @classmethod + def from_mapping( + cls, + data: object, + *, + base_dir: Path, + ) -> LLMAtomMappingConfig: + mapping = _validate_mapping_format(data, name="tools_settings.llm_atom_mapping") + del base_dir + + url = mapping.get("url") + if url is not None and not isinstance(url, str): + raise ValueError( + "tools_settings.llm_atom_mapping.url must be a string when provided." + ) + + model = mapping.get("model", cls.model) + if not isinstance(model, str): + raise ValueError("tools_settings.llm_atom_mapping.model must be a string.") + + reasoning_effort = mapping.get("reasoning_effort", cls.reasoning_effort) + if reasoning_effort not in {"low", "medium", "high"}: + raise ValueError( + "tools_settings.llm_atom_mapping.reasoning_effort must be 'low', 'medium', or 'high'." + ) + + api_key = mapping.get("api_key") + if api_key is not None and not isinstance(api_key, str): + raise ValueError( + "tools_settings.llm_atom_mapping.api_key must be a string when provided." + ) + + return cls( + url=resolve_llm_base_url(url), + model=model, + reasoning_effort=reasoning_effort, + api_key=api_key, + system_prompt_path=_resolve_defaultable_cwd_path( + mapping.get("system_prompt_path"), + default_path=graph_rxn_mapper_prompt_path("atom_mapping_system.md"), + name="tools_settings.llm_atom_mapping.system_prompt_path", + ), + user_prompt_path=_resolve_defaultable_cwd_path( + mapping.get("user_prompt_path"), + default_path=graph_rxn_mapper_prompt_path("atom_mapping_user.md"), + name="tools_settings.llm_atom_mapping.user_prompt_path", + ), + skill_prompt_path=_resolve_defaultable_cwd_path( + mapping.get("skill_prompt_path"), + default_path=graph_rxn_mapper_prompt_path("atom_mapping_skill.md"), + name="tools_settings.llm_atom_mapping.skill_prompt_path", + ), + ) + + @dataclass class ToolsConfig: reaction_energy: ReactionEnergyConfig = field(default_factory=ReactionEnergyConfig) + llm_atom_mapping: LLMAtomMappingConfig = field(default_factory=LLMAtomMappingConfig) @classmethod def from_mapping( @@ -226,11 +330,16 @@ def from_mapping( base_dir: Path, ) -> ToolsConfig: mapping = _validate_mapping_format(data, name="tools_settings") + return cls( reaction_energy=ReactionEnergyConfig.from_mapping( mapping.get("reaction_energy"), base_dir=base_dir, - ) + ), + llm_atom_mapping=LLMAtomMappingConfig.from_mapping( + mapping.get("llm_atom_mapping"), + base_dir=base_dir, + ), ) diff --git a/flask_tools/pipette/constants.py b/flask_tools/pipette/constants.py index f089242..c6d3088 100644 --- a/flask_tools/pipette/constants.py +++ b/flask_tools/pipette/constants.py @@ -52,6 +52,22 @@ def resolve_llm_base_url(explicit_base_url: str | None = None) -> str: DEFAULT_LLM_BASE_URL = resolve_llm_base_url() +class SmilesContainer(str): + """A class so most tools can assume rxn_smiles passed to tool.run() is a plain string, but some can play around + with the presence of the reagents + """ + + def __new__(cls, value, original_smiles=None): + instance = super().__new__(cls, value) + return instance + + def __init__(self, value, reagents_smi: str | None = None): + self.reagents_smi = reagents_smi + + def __repr__(self): + return f"SmilesContainer({str.__repr__(self)}, reagents_smiles={self.reagents_smi!r})" + + class ToolStatus(str, Enum): PASS = "pass" # Reaction passed this tool FAIL = "fail" # Reaction failed to pass this tool @@ -72,7 +88,7 @@ class ToolResult(BaseModel): status: ToolStatus data: ( SerializeAsAny[ToolResultDetails] | None - ) # None if tool had an error or wasn't run. SerializeAsAny or else model_dump only outputs the parent class ToolResultDetails' fields which are nothing. + ) # None if tool had an error or wasn't run. Must use SerializeAsAny or else model_dump only outputs the parent class ToolResultDetails' fields which are no fields. comment: str = "" skipped_reason: str | None = ( None # If a priority checker skipped this tool, like in an exact rule pipeline, or a traceback if there was an error diff --git a/flask_tools/pipette/grade_rxn.py b/flask_tools/pipette/grade_rxn.py index 31dc8fe..a42c842 100644 --- a/flask_tools/pipette/grade_rxn.py +++ b/flask_tools/pipette/grade_rxn.py @@ -32,12 +32,13 @@ from typing import TYPE_CHECKING from flask_tools.pipette.config import PipetteConfig, load_config, ConfigType -from flask_tools.pipette.constants import ReactionGrade, ToolResult +from flask_tools.pipette.constants import ReactionGrade from flask_tools.pipette.pipeline import build_default_pipeline from flask_tools.pipette.reaction_fixer import ReactionFixResultDetails if TYPE_CHECKING: from .judge import AsyncLLMJudge + from flask_tools.pipette.constants import ToolResult REACTION_SMILES_COLUMNS = ( "rxn_smiles", @@ -52,8 +53,9 @@ def _get_possible_fixed_rxn_smi(reaction_grade: ReactionGrade) -> str | None: tool_res: ToolResult for tool_res in reaction_grade.results: if tool_res.name == "llm_reaction_fix": - d: ReactionFixResultDetails = tool_res.data # noqa - return d.fixed_reaction_smiles + d: ReactionFixResultDetails | None = tool_res.data + if d: + return d.fixed_reaction_smiles return None @@ -78,7 +80,7 @@ def _build_output_records( { "rxn_smiles": rxn_smiles, "cleaned_rxn_smiles": _get_possible_fixed_rxn_smi(result) or rxn_smiles, - "grade": result.model_dump(mode="json"), + "grade": result.model_dump(mode="json", exclude_none=True), } for rxn_smiles, result in zip(rxn_smiles_list, results, strict=True) ] @@ -187,6 +189,7 @@ def main() -> list[dict]: f"or '{ConfigType.LLM_JUDGE_NO_DFT}.", ) parser.add_argument( + "-v", "--verbose", action="store_true", help="Prints out json object", diff --git a/flask_tools/pipette/graph_rxn_mapper/llm_benchmark_reactions.py b/flask_tools/pipette/graph_rxn_mapper/llm_benchmark_reactions.py new file mode 100644 index 0000000..a6560a0 --- /dev/null +++ b/flask_tools/pipette/graph_rxn_mapper/llm_benchmark_reactions.py @@ -0,0 +1,645 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +#!/usr/bin/env python3 +"""Benchmark an LLM atom mapper against mapped RDF reactions.""" + +from __future__ import annotations + +import argparse +import json +import os +import time +import urllib.error +import urllib.request +from collections import Counter +from concurrent.futures import ThreadPoolExecutor, as_completed +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple + +from rdkit import Chem, RDLogger +from tqdm import tqdm + +from .benchmark_reactions import ( + BenchmarkRecord, + ReactionTask, + collect_tasks, + default_worker_count, + format_record, + normalize_mapped_reaction_parts, + print_summary, + reaction_without_agents, + split_reaction_smiles, + write_json_report, +) +from flask_tools.pipette.smiles import mol_from_side, clear_atom_maps_from_reaction + +RDLogger.DisableLog("rdApp.warning") + +PROMPT_DIR = Path(__file__).resolve().parent / "prompts" +DEFAULT_SYSTEM_PROMPT = PROMPT_DIR / "atom_mapping_system.md" +DEFAULT_USER_PROMPT = PROMPT_DIR / "atom_mapping_user.md" +DEFAULT_SKILL_PROMPT = PROMPT_DIR / "atom_mapping_skill.md" + + +@dataclass +class LLMRecord: + benchmark: BenchmarkRecord + model: str + raw_response: str + reasoning_summary: str = "" + confidence: Optional[float] = None + + +def load_text(path: str) -> str: + return Path(path).read_text() + + +def build_system_prompt(args: argparse.Namespace) -> str: + system_prompt = load_text(args.system_prompt) + if args.use_skill_prompt and args.skill_prompt: + skill_prompt = load_text(args.skill_prompt) + system_prompt = f"{system_prompt}\n\nAdditional atom-mapping skill instructions:\n{skill_prompt}" + return system_prompt + + +def side_graph_json(side: str) -> str: + mol = mol_from_side(side) + atoms = [] + atom_to_fragment: Dict[int, int] = {} + for frag_id, atom_ids in enumerate( + Chem.GetMolFrags(mol, asMols=False, sanitizeFrags=True) + ): + for atom_id in atom_ids: + atom_to_fragment[int(atom_id)] = frag_id + + for atom in mol.GetAtoms(): + atoms.append( + { + "id": atom.GetIdx(), + "fragment": atom_to_fragment.get(atom.GetIdx(), 0), + "element": atom.GetSymbol(), + "atomic_num": atom.GetAtomicNum(), + "formal_charge": atom.GetFormalCharge(), + "is_aromatic": atom.GetIsAromatic(), + "isotope": atom.GetIsotope(), + "neighbors": sorted(n.GetIdx() for n in atom.GetNeighbors()), + } + ) + + bonds = [] + for bond in mol.GetBonds(): + bonds.append( + { + "begin": bond.GetBeginAtomIdx(), + "end": bond.GetEndAtomIdx(), + "order": float(bond.GetBondTypeAsDouble()), + "is_aromatic": bond.GetIsAromatic(), + "in_ring": bond.IsInRing(), + } + ) + + return json.dumps( + { + "atom_count": mol.GetNumAtoms(), + "atoms": atoms, + "bonds": bonds, + }, + separators=(",", ":"), + sort_keys=True, + ) + + +def response_schema() -> Dict[str, Any]: + return { + "type": "object", + "additionalProperties": False, + "properties": { + "product_to_reactant": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": False, + "properties": { + "product_atom": {"type": "integer"}, + "reactant_atom": {"type": "integer"}, + }, + "required": ["product_atom", "reactant_atom"], + }, + }, + "confidence": {"type": "number"}, + "reasoning_summary": {"type": "string"}, + }, + "required": ["product_to_reactant", "confidence", "reasoning_summary"], + } + + +def build_user_prompt(template: str, task: ReactionTask, unmapped_smiles: str) -> str: + reactants, _agents, products = split_reaction_smiles(unmapped_smiles) + return template.format( + reaction_index=task.index, + unmapped_reaction_smiles=unmapped_smiles, + reactant_graph_json=side_graph_json(reactants), + product_graph_json=side_graph_json(products), + ) + + +def http_json( + url: str, headers: Mapping[str, str], payload: Mapping[str, Any], timeout: float +) -> Dict[str, Any]: + data = json.dumps(payload).encode("utf-8") + request = urllib.request.Request( + url, data=data, headers=dict(headers), method="POST" + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + return json.loads(response.read().decode("utf-8")) + + +def extract_responses_text(data: Mapping[str, Any]) -> str: + if isinstance(data.get("output_text"), str): + return str(data["output_text"]) + chunks: List[str] = [] + for item in data.get("output", []) or []: + for content in item.get("content", []) or []: + if isinstance(content.get("text"), str): + chunks.append(str(content["text"])) + return "".join(chunks) + + +def extract_json_object(text: str) -> Dict[str, Any]: + text = text.strip() + try: + return json.loads(text) + except json.JSONDecodeError: + start = text.find("{") + end = text.rfind("}") + if start < 0 or end <= start: + raise + return json.loads(text[start : end + 1]) + + +def call_openai( + system_prompt: str, user_prompt: str, args: argparse.Namespace +) -> Tuple[Dict[str, Any], str]: + api_key = os.environ.get(args.api_key_env) + if not api_key: + raise RuntimeError(f"Missing API key in ${args.api_key_env}.") + + headers = { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + } + base_url = args.base_url.rstrip("/") + schema = response_schema() + + if args.api == "responses": + payload: Dict[str, Any] = { + "model": args.model, + "input": [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt}, + ], + "max_output_tokens": args.max_output_tokens, + "text": { + "format": { + "type": "json_schema", + "name": "atom_mapping_response", + "schema": schema, + "strict": True, + } + }, + } + if args.temperature is not None: + payload["temperature"] = args.temperature + if args.reasoning_effort: + payload["reasoning"] = {"effort": args.reasoning_effort} + url = f"{base_url}/responses" + data = http_json(url, headers, payload, args.timeout) + text = extract_responses_text(data) + else: + payload = { + "model": args.model, + "messages": [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt}, + ], + "max_tokens": args.max_output_tokens, + "response_format": { + "type": "json_schema", + "json_schema": { + "name": "atom_mapping_response", + "schema": schema, + "strict": True, + }, + }, + } + if args.temperature is not None: + payload["temperature"] = args.temperature + url = f"{base_url}/chat/completions" + data = http_json(url, headers, payload, args.timeout) + text = data["choices"][0]["message"]["content"] + + return extract_json_object(text), text + + +def call_openai_with_retries( + system_prompt: str, user_prompt: str, args: argparse.Namespace +) -> Tuple[Dict[str, Any], str]: + last_error: Optional[BaseException] = None + for attempt in range(args.retries + 1): + try: + return call_openai(system_prompt, user_prompt, args) + except ( + urllib.error.HTTPError, + urllib.error.URLError, + TimeoutError, + RuntimeError, + json.JSONDecodeError, + ) as exc: + last_error = exc + if attempt >= args.retries: + break + time.sleep(args.retry_delay * (2**attempt)) + raise RuntimeError( + f"LLM request failed after {args.retries + 1} attempts: {last_error}" + ) + + +def parse_product_to_reactant(payload: Mapping[str, Any]) -> List[Tuple[int, int]]: + raw = payload.get("product_to_reactant") + if isinstance(raw, dict): + return [(int(p), int(r)) for p, r in raw.items()] + if not isinstance(raw, list): + raise ValueError("Response field product_to_reactant must be a list or object.") + + pairs: List[Tuple[int, int]] = [] + for item in raw: + if not isinstance(item, Mapping): + raise ValueError("Each product_to_reactant entry must be an object.") + pairs.append((int(item["product_atom"]), int(item["reactant_atom"]))) + return pairs + + +def mapped_reaction_from_pairs( + unmapped_smiles: str, pairs: Sequence[Tuple[int, int]], keep_agents: bool = False +) -> str: + reactants, agents, products = split_reaction_smiles(unmapped_smiles) + reactant_mol = mol_from_side(reactants) + agent_mol = mol_from_side(agents) + product_mol = mol_from_side(products) + + for mol in (reactant_mol, agent_mol, product_mol): + for atom in mol.GetAtoms(): + atom.SetAtomMapNum(0) + + product_to_reactant = {int(p): int(r) for p, r in pairs} + if len(product_to_reactant) != len(pairs): + raise ValueError("Duplicate product_atom ids in response.") + if set(product_to_reactant) != set(range(product_mol.GetNumAtoms())): + missing = sorted( + set(range(product_mol.GetNumAtoms())) - set(product_to_reactant) + ) + extra = sorted(set(product_to_reactant) - set(range(product_mol.GetNumAtoms()))) + raise ValueError( + f"Product atom coverage mismatch; missing={missing}, extra={extra}." + ) + if len(set(product_to_reactant.values())) != len(product_to_reactant): + raise ValueError("Duplicate reactant_atom ids in response.") + + for product_atom, reactant_atom in product_to_reactant.items(): + if reactant_atom < 0 or reactant_atom >= reactant_mol.GetNumAtoms(): + raise ValueError(f"Reactant atom id out of range: {reactant_atom}.") + pa = product_mol.GetAtomWithIdx(product_atom) + ra = reactant_mol.GetAtomWithIdx(reactant_atom) + if pa.GetAtomicNum() != ra.GetAtomicNum(): + raise ValueError( + f"Element mismatch for product atom {product_atom} ({pa.GetSymbol()}) " + f"and reactant atom {reactant_atom} ({ra.GetSymbol()})." + ) + + next_map = 1 + for product_atom in sorted(product_to_reactant): + reactant_atom = product_to_reactant[product_atom] + reactant_mol.GetAtomWithIdx(reactant_atom).SetAtomMapNum(next_map) + product_mol.GetAtomWithIdx(product_atom).SetAtomMapNum(next_map) + next_map += 1 + + for atom in reactant_mol.GetAtoms(): + if atom.GetAtomMapNum() == 0: + atom.SetAtomMapNum(next_map) + next_map += 1 + if keep_agents: + for atom in agent_mol.GetAtoms(): + if atom.GetAtomMapNum() == 0: + atom.SetAtomMapNum(next_map) + next_map += 1 + + lhs = Chem.MolToSmiles(reactant_mol, canonical=True, isomericSmiles=True) + rhs = Chem.MolToSmiles(product_mol, canonical=True, isomericSmiles=True) + if keep_agents: + middle = Chem.MolToSmiles(agent_mol, canonical=True, isomericSmiles=True) + return f"{lhs}>{middle}>{rhs}" + return f"{lhs}>>{rhs}" + + +def expected_normalized( + source_smiles: str, keep_agents: bool +) -> Tuple[str, str, str, str]: + expected_smiles = ( + source_smiles if keep_agents else reaction_without_agents(source_smiles) + ) + reactants, agents, products = normalize_mapped_reaction_parts(expected_smiles) + normalized = ( + f"{reactants}>{agents}>{products}" if agents else f"{reactants}>>{products}" + ) + return expected_smiles, normalized, reactants, products + + +def run_one_llm( + task: ReactionTask, + system_prompt: str, + user_template: str, + args: argparse.Namespace, +) -> LLMRecord: + started = time.perf_counter() + source_smiles = task.smiles + unmapped_smiles = clear_atom_maps_from_reaction( + source_smiles, keep_agents=args.keep_agents + ) + expected_smiles, expected_norm, expected_reactants, expected_products = ( + expected_normalized(source_smiles, args.keep_agents) + ) + + user_prompt = build_user_prompt(user_template, task, unmapped_smiles) + payload, raw_text = call_openai_with_retries(system_prompt, user_prompt, args) + pairs = parse_product_to_reactant(payload) + predicted_smiles = mapped_reaction_from_pairs( + unmapped_smiles, pairs, keep_agents=args.keep_agents + ) + predicted_reactants, predicted_agents, predicted_products = ( + normalize_mapped_reaction_parts(predicted_smiles) + ) + predicted_norm = ( + f"{predicted_reactants}>{predicted_agents}>{predicted_products}" + if predicted_agents + else f"{predicted_reactants}>>{predicted_products}" + ) + reactants_matched = expected_reactants == predicted_reactants + products_matched = expected_products == predicted_products + + record = BenchmarkRecord( + index=task.index, + rdf_line=task.rdf_line, + source_smiles=source_smiles, + unmapped_smiles=unmapped_smiles, + expected_smiles=expected_smiles, + predicted_smiles=predicted_smiles, + expected_normalized=expected_norm, + predicted_normalized=predicted_norm, + expected_reactants_normalized=expected_reactants, + predicted_reactants_normalized=predicted_reactants, + expected_products_normalized=expected_products, + predicted_products_normalized=predicted_products, + reactants_matched=reactants_matched, + products_matched=products_matched, + matched=reactants_matched and products_matched, + mapper_status="llm", + elapsed_seconds=time.perf_counter() - started, + topology_counts={}, + ) + return LLMRecord( + benchmark=record, + model=args.model, + raw_response=raw_text, + reasoning_summary=str(payload.get("reasoning_summary", "")), + confidence=float(payload["confidence"]) if "confidence" in payload else None, + ) + + +def error_llm_record( + task: ReactionTask, exc: BaseException, args: argparse.Namespace +) -> LLMRecord: + record = BenchmarkRecord( + index=task.index, + rdf_line=task.rdf_line, + source_smiles=task.smiles, + unmapped_smiles="", + expected_smiles=task.smiles, + predicted_smiles="", + expected_normalized="", + predicted_normalized="", + expected_reactants_normalized="", + predicted_reactants_normalized="", + expected_products_normalized="", + predicted_products_normalized="", + reactants_matched=False, + products_matched=False, + matched=False, + mapper_status="error", + elapsed_seconds=0.0, + topology_counts={}, + error=f"{type(exc).__name__}: {exc}", + ) + return LLMRecord(benchmark=record, model=args.model, raw_response="") + + +def update_progress_postfix( + progress: tqdm, records: Sequence[BenchmarkRecord], errors: int +) -> None: + completed = len(records) + matched = sum(1 for record in records if record.matched) + mismatched = max(0, completed - matched - errors) + accuracy = matched / completed if completed else 0.0 + progress.set_postfix( + {"acc": f"{accuracy:.1%}", "pass": matched, "fail": mismatched, "err": errors}, + refresh=False, + ) + + +def write_llm_json_report( + path: str, records: Sequence[LLMRecord], skipped: int, errors: int, workers: int +) -> None: + benchmark_records = [record.benchmark for record in records] + payload = { + "summary": { + "completed": len(records), + "matched": sum(1 for record in benchmark_records if record.matched), + "mismatched": sum(1 for record in benchmark_records if not record.matched), + "reactant_matches": sum( + 1 for record in benchmark_records if record.reactants_matched + ), + "product_matches": sum( + 1 for record in benchmark_records if record.products_matched + ), + "skipped": skipped, + "errors": errors, + "workers": workers, + }, + "records": [ + { + **asdict(record.benchmark), + "model": record.model, + "confidence": record.confidence, + "reasoning_summary": record.reasoning_summary, + "raw_response": record.raw_response, + } + for record in records + ], + } + with open(path, "w") as out: + json.dump(payload, out, indent=2, sort_keys=True) + out.write("\n") + + +def build_arg_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.ArgumentDefaultsHelpFormatter + ) + parser.add_argument( + "rdf_file", nargs="?", default="reactions.rdf", help="RDF file to benchmark." + ) + parser.add_argument( + "--limit", type=int, help="Stop after this many valid RDF reactions." + ) + parser.add_argument( + "--start", + type=int, + default=1, + help="One-based RDF reaction index to start from.", + ) + parser.add_argument( + "-j", + "--workers", + type=int, + default=default_worker_count(), + help="Worker threads/API calls to use.", + ) + parser.add_argument( + "--debug", + action="store_true", + help="Print expected/predicted mappings for every completed reaction.", + ) + parser.add_argument( + "--fail-on-mismatch", + action="store_true", + help="Exit nonzero when any completed reaction mismatches.", + ) + parser.add_argument( + "--json-report", help="Write a detailed JSON report to this path." + ) + parser.add_argument( + "--keep-agents", + action="store_true", + help="Keep the RDF agent field in mapper input.", + ) + parser.add_argument("--model", default=os.environ.get("OPENAI_MODEL", "gpt-5.5")) + parser.add_argument("--api", choices=["responses", "chat"], default="responses") + parser.add_argument( + "--base-url", default=os.environ.get("BASE_URL", "https://livai-api.llnl.gov/") + ) + parser.add_argument("--api-key-env", default="LIVAI_API_KEY") + parser.add_argument("--system-prompt", default=str(DEFAULT_SYSTEM_PROMPT)) + parser.add_argument( + "--skill-prompt", + default=str(DEFAULT_SKILL_PROMPT), + help="Additional skill/instruction file appended to the system prompt.", + ) + parser.add_argument( + "--no-skill-prompt", + dest="use_skill_prompt", + action="store_false", + help="Do not append the skill prompt.", + ) + parser.set_defaults(use_skill_prompt=True) + parser.add_argument("--user-prompt-template", default=str(DEFAULT_USER_PROMPT)) + parser.add_argument("--max-output-tokens", type=int, default=4096) + parser.add_argument( + "--reasoning-effort", + choices=["minimal", "low", "medium", "high"], + help="Responses API reasoning effort for models that support it.", + ) + parser.add_argument("--temperature", type=float, default=None) + parser.add_argument("--timeout", type=float, default=120.0) + parser.add_argument("--retries", type=int, default=2) + parser.add_argument("--retry-delay", type=float, default=2.0) + return parser + + +def main(argv: Optional[Sequence[str]] = None) -> int: + args = build_arg_parser().parse_args(argv) + if args.workers < 1: + raise SystemExit("--workers must be at least 1.") + + system_prompt = build_system_prompt(args) + user_template = load_text(args.user_prompt_template) + tasks, skipped = collect_tasks(args) + records: List[LLMRecord] = [] + errors = 0 + started = time.perf_counter() + + with tqdm(total=len(tasks), unit="rxn", desc="LLM mapping") as progress: + if tasks: + with ThreadPoolExecutor(max_workers=args.workers) as executor: + futures = { + executor.submit( + run_one_llm, task, system_prompt, user_template, args + ): task + for task in tasks + } + for future in as_completed(futures): + task = futures[future] + try: + record = future.result() + except Exception as exc: + errors += 1 + record = error_llm_record(task, exc, args) + + records.append(record) + benchmark = record.benchmark + if args.debug or not benchmark.matched or benchmark.error: + tqdm.write(format_record(benchmark, debug=args.debug)) + if args.debug and record.reasoning_summary: + tqdm.write(f" llm reasoning: {record.reasoning_summary}") + update_progress_postfix( + progress, [r.benchmark for r in records], errors + ) + progress.update(1) + + records.sort(key=lambda record: record.benchmark.index) + benchmark_records = [record.benchmark for record in records] + print_summary( + benchmark_records, + skipped=skipped, + errors=errors, + started=started, + workers=args.workers, + ) + print(f" model: {args.model}") + print(f" api: {args.api}") + + if args.json_report: + write_llm_json_report( + args.json_report, + records, + skipped=skipped, + errors=errors, + workers=args.workers, + ) + print(f" json: {args.json_report}") + + if errors: + return 1 + if args.fail_on_mismatch and any( + not record.benchmark.matched for record in records + ): + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_pipette_tool.py b/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_pipette_tool.py new file mode 100644 index 0000000..3064c3d --- /dev/null +++ b/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_pipette_tool.py @@ -0,0 +1,361 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from pydantic import BaseModel + +from flask_tools.pipette.smiles import ( + split_reaction_smiles, + clear_atom_maps_from_reaction, +) +from flask_tools.pipette.verifiers import ReactionChecker +from . import llm_benchmark_reactions +from .llm_benchmark_reactions import mapped_reaction_from_pairs +from .subtractive_reaction_mapper_v3 import ( + ReactionAtomMapperConfig as SubtractiveReactionAtomMapperConfig, + subtractive_map_reaction, + SubtractiveMappingResult, +) +from ..constants import ( + SmilesContainer, + ToolResult, + ToolResultDetails, + ToolResultsDict, + ToolStatus, + resolve_llm_api_key, +) +from ..llm_query import _run_coroutine_sync, query_task_async +from ..verifiers.rdt import map_reaction_smiles_with_rdt + +if TYPE_CHECKING: + from flask_tools.pipette.config import PipetteConfig, ReasoningEffort + +""" +Overall flow + ``` + O.A > B > AA + | rm reagent (New GraphBasedBalancer tool) + V + O.A >> AA + | Algorithmic balancing + atom mapping (New GraphBasedBalancer tool) + V + O.A[:1].A[:2] >> A[:1]A[:2] (New GraphBasedBalancer tool) + | rm atom map + V + O.A.A >> AA + | LLM balance (Existing LLMFixer tool) + V + O.A.A >> AA.O + | Add back reagents (New LLMAtomMapper tool) (in case reagent label was incorrect) + V + O.A.A > B > AA.O + | LLM atom mapping (New LLMAtomMapper tool) + V + O[:3].A[:1].A[:2] > B > A[:1]A[:2].O[:3] + ``` +""" + + +class GraphBasedBalancer(ReactionChecker): + name = "graph_based_balancing" + + def __init__(self, config: PipetteConfig) -> None: + self.config = config + self.atom_map_config = SubtractiveReactionAtomMapperConfig() + + def run( + self, rxn_smiles: str, context: ToolResultsDict | None = None + ) -> ToolResult: + return _run_coroutine_sync(self.arun(rxn_smiles, context)) + + async def arun( + self, rxn_smiles: str, context: ToolResultsDict | None = None + ) -> ToolResult: + """ + The subtractive_mapping is used for balancing dimerization. + then LLM atom mapping + Args: + rxn_smiles: + context: + + Returns: + + """ + context = context if context is not None else {} + # Rm agents + reactants_smi, agents_smi, products_smi = split_reaction_smiles(rxn_smiles) + reactants_products_smi = reactants_smi + ">>" + products_smi + # Balance + res: SubtractiveMappingResult = subtractive_map_reaction( + reactants_products_smi, config=self.atom_map_config + ) + # Rm atom mapping + graph_mapped_smi = res.atom_mapped_reaction_smiles() + initial_balanced_smi = clear_atom_maps_from_reaction(graph_mapped_smi) + if initial_balanced_smi is None: + print(f"{rxn_smiles=}\n{graph_mapped_smi=}\n{initial_balanced_smi=}") + return ToolResult( + name=self.name, + status=ToolStatus.ERROR, + data=None, + comment="Graph-based mapper produced an invalid atom-mapped reaction SMILES.", + ) + + ## Call the llm balancing (llm fixing) already in pipette + # reaction_fixer = AsyncLLMReactionFixer.from_config(self.config) + # llm_balanced_smi = None + # if reaction_fixer is not None: + # llm_balanced_smi: str | None + # llm_balanced_tool_res, llm_balanced_smi = ( + # await reaction_fixer.attempt_llm_fix_async( + # reaction_fixer, initial_balanced_smi, context + # ) + # ) + # context[(initial_balanced_smi, BaseLLMReactionFixer.name)] = ( + # llm_balanced_tool_res + # ) + # + # if llm_balanced_smi is None: # Error, no change, or balancer not enabled + # llm_balanced_smi = initial_balanced_smi + # + # new_reactants_smi, new_agents_smi, new_products_smi = split_reaction_smiles( + # llm_balanced_smi + # ) + # if new_agents_smi: + # raise ValueError( + # f"LLM balancing unexpectedly produced an agent. Input {initial_balanced_smi}, output {llm_balanced_smi} " + # ) + # + # balanced_smi = f"{new_reactants_smi}>{agents_smi}>{new_products_smi}" + balanced_smi = "CCCC" # todo remove + return ToolResult( + name=self.name, + status=ToolStatus.PASS, + data=GraphBasedBalancerResultDetails( + original_reaction_smiles=rxn_smiles, + graph_mapped_reaction_smiles=graph_mapped_smi, + graph_balanced_reaction_smiles=initial_balanced_smi, + final_balanced_reaction_smiles=balanced_smi, + objective_value=res.objective_value, + mapper_status=res.status, + reasoning_summary=( + "Subtractive graph mapping balanced the reactant/product sides, " + "then the LLM reaction fixer was used to optionally add missing species." + ), + ), + comment="Graph-based balancing completed.", + ) + + +class GraphBasedBalancerResultDetails(ToolResultDetails): + original_reaction_smiles: str + graph_mapped_reaction_smiles: str + graph_balanced_reaction_smiles: str + final_balanced_reaction_smiles: str + objective_value: float + mapper_status: str + reasoning_summary: str + + +class AtomMapping(BaseModel): + product_atom: int + reactant_atom: int + + +class ReactionMapping(BaseModel): + product_to_reactant: list[AtomMapping] + confidence: float + reasoning_summary: str + + +class AtomMappingResultDetails(ToolResultDetails): + input_reaction_smiles: str + mapped_reaction_smiles: str + product_to_reactant: list[AtomMapping] + confidence: float | None + reasoning_summary: str | None + + +class LLMAtomMapper(ReactionChecker): + name = "llm_atom_mapping" + + def __init__( + self, + config: PipetteConfig, + url: str, + model: str, + reasoning_effort: ReasoningEffort, + api_key: str, + user_prompt_path: Path, + system_prompt_path: Path, + skill_prompt_path: Path, + ) -> None: + self.config = config + self.url = url + self.model = model + self.reasoning_effort = reasoning_effort + self.api_key = api_key + self.user_prompt_path = user_prompt_path + self.system_prompt_path = system_prompt_path + self.skill_prompt_path = skill_prompt_path + + @classmethod + def from_config(cls, config: PipetteConfig) -> "LLMAtomMapper": + atom_mapping_config = config.tools_settings.llm_atom_mapping + api_key = resolve_llm_api_key(atom_mapping_config.api_key) + if not api_key: + raise ValueError( + "LLM atom mapper requires an API key via " + "tools_settings.llm_atom_mapping.api_key or the standard LLM env vars." + ) + return cls( + config=config, + url=atom_mapping_config.url, + model=atom_mapping_config.model, + reasoning_effort=atom_mapping_config.reasoning_effort, + api_key=api_key, + user_prompt_path=atom_mapping_config.user_prompt_path, + system_prompt_path=atom_mapping_config.system_prompt_path, + skill_prompt_path=atom_mapping_config.skill_prompt_path, + ) + + def _build_user_payload( + self, + rxn_smiles: str, + results: list[ToolResult], + ) -> dict[str, Any]: + serialized_results = [r.model_dump(exclude_none=True) for r in results] + for s in serialized_results: + if "skipped_reason" in s: + del s["skipped_reason"] + + user_template = self.user_prompt_path.read_text(encoding="utf-8") + reactants_smi, _agents_smi, products_smi = split_reaction_smiles(rxn_smiles) + user_prompt = user_template.format( + unmapped_reaction_smiles=rxn_smiles, + reactant_graph_json=llm_benchmark_reactions.side_graph_json(reactants_smi), + product_graph_json=llm_benchmark_reactions.side_graph_json(products_smi), + ) + + return { + "reaction_smiles": rxn_smiles, + "tool_results": serialized_results, + "instructions": user_prompt, + } + + def run( + self, rxn_smiles: str, context: ToolResultsDict | None = None + ) -> ToolResult: + return _run_coroutine_sync(self.arun(rxn_smiles, context)) + + async def arun( + self, rxn_smiles: str | SmilesContainer, context: ToolResultsDict | None = None + ) -> ToolResult: + """ + LLM atom mapping + Args: + rxn_smiles: + context: + + Returns: + + """ + context = context if context is not None else {} + # Add reagents back in + if isinstance(rxn_smiles, SmilesContainer): + reactants, agents, products = split_reaction_smiles(rxn_smiles) + if agents and agents != rxn_smiles.reagents_smi: + raise ValueError( + f"How did these diverge?" + ) # Just in case. Nothing in current code would do this + agents = rxn_smiles.reagents_smi or "" + rxn_smiles = f"{reactants}>{agents}>{products}" + + # Call LLM atom mapper + user_prompt = json.dumps( + self._build_user_payload(rxn_smiles, []), # list(context.values())), + indent=2, + sort_keys=True, + ) + system_prompt = self.system_prompt_path.read_text(encoding="utf-8") + skill_prompt = self.skill_prompt_path.read_text(encoding="utf-8") + system_prompt = f"{system_prompt}\n\nAdditional atom-mapping skill instructions:\n{skill_prompt}" + response_text = await query_task_async( + system_prompt=system_prompt, + user_prompt=user_prompt, + model=self.model, + api_key=self.api_key, + url=self.url, + reasoning_effort=self.reasoning_effort, + structured_output_schema=ReactionMapping, + agent_name="PipetteAtomMapper", + ) + return self._parse_output(rxn_smiles, response_text) + + def _parse_output(self, rxn_smiles: str, response_text: str) -> ToolResult: + try: + parsed = ReactionMapping.model_validate_json(response_text) + except Exception as exc: + raise ValueError( + f"LLM atom mapper did not return valid JSON: {response_text}" + ) from exc + + mapped_reaction_smiles = mapped_reaction_from_pairs( + rxn_smiles, + [ + (mapping.product_atom, mapping.reactant_atom) + for mapping in parsed.product_to_reactant + ], + keep_agents=True, + ) + return ToolResult( + name=self.name, + status=ToolStatus.PASS, + data=AtomMappingResultDetails( + input_reaction_smiles=rxn_smiles, + mapped_reaction_smiles=mapped_reaction_smiles, + product_to_reactant=parsed.product_to_reactant, + confidence=parsed.confidence, + reasoning_summary=parsed.reasoning_summary, + ), + comment="LLM atom mapping completed.", + ) + + +class RDTAtomMapper(ReactionChecker): + name = "RDTAtomMapper" + + def run( + self, rxn_smiles: str | SmilesContainer, context: ToolResultsDict | None = None + ) -> ToolResult: + try: + atom_mapped_str = map_reaction_smiles_with_rdt(rxn_smiles) + except Exception as e: + return ToolResult( + name=self.name, + status=ToolStatus.FAIL, + data=None, + comment="LLM atom mapping failed.", # todo: find the code for getting traceback + ) + return ToolResult( + name=self.name, + status=ToolStatus.PASS, + data=AtomMappingResultDetails( + input_reaction_smiles=rxn_smiles, + mapped_reaction_smiles=atom_mapped_str, + product_to_reactant=[], # Not readily available + confidence=None, + reasoning_summary=None, + ), + comment="RDT atom mapping completed.", + ) diff --git a/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_v3.py b/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_v3.py new file mode 100644 index 0000000..0e1421c --- /dev/null +++ b/flask_tools/pipette/graph_rxn_mapper/subtractive_reaction_mapper_v3.py @@ -0,0 +1,2434 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +#!/usr/bin/env python3 +""" +Subtractive common-subgraph atom mapper for reaction topology analysis. + +This module implements the approach discussed in the conversation: + + product graph - common subgraph occurrences of reactant copies = residual + +The subtraction unit is NOT necessarily a full reactant. It is a connected +common subgraph occurrence between a reactant component and the product side. +The selector can be an ILP (via scipy.optimize.milp) or a greedy heuristic. + +Typical use: + + python subtractive_reaction_mapper.py 'CC.CNC>>CCNCC' + python subtractive_reaction_mapper.py '[C:1]C[C:2]>>[C:1]CC[C:2]' + python subtractive_reaction_mapper.py '[C:1][C:2].[C:3][N:4][C:5]>>[C:1][C:3][N:4][C:5][C:2]' + +Python API: + + from subtractive_reaction_mapper import subtractive_map_reaction + result = subtractive_map_reaction('CC.CNC>>CCNCC') + print(result.to_jsonable()) + +Dependencies: + rdkit, networkx +Optional dependency: + scipy, for ILP selection. If scipy MILP is unavailable, selector='ilp' + falls back to greedy selection unless fallback=False is passed. + +Important modeling notes: + * Reactant components are copied virtually up to max_copies. + * Candidates are connected common subgraph occurrences. + * Multiple candidates may be chosen from the same reactant copy, which is + how true split lineages are represented and penalized. + * Atom maps, when present on both sides, are treated as hard anchors by + default. This allows examples such as [C:1]C[C:2]>>[C:1]CC[C:2] to + report a stretched/split lineage rather than remapping to a contiguous + product subgraph. + * After subtraction, connected residual fragments are reported on both the + product and reactant sides. Whole uncovered product components are + flagged as byproduct/missing-source candidates. +""" + +from __future__ import annotations + +import argparse +import dataclasses +import json +import math +from collections import Counter, defaultdict, deque +from dataclasses import dataclass +from typing import ( + Any, + Dict, + FrozenSet, + Iterable, + List, + Mapping, + Optional, + Sequence, + Set, + Tuple, +) + +import networkx as nx +from rdkit import Chem +from rdkit.Chem import rdFMCS + +try: + from scipy.optimize import Bounds, LinearConstraint, milp + import scipy.sparse as sp + import numpy as np + + SCIPY_MILP_AVAILABLE = True +except Exception: # pragma: no cover - import availability depends on env + Bounds = None # type: ignore + LinearConstraint = None # type: ignore + milp = None # type: ignore + sp = None # type: ignore + np = None # type: ignore + SCIPY_MILP_AVAILABLE = False + + +INF = 10**12 + + +@dataclass(frozen=True) +class ReactionAtomMapperConfig: + """Configuration for candidate generation, selection, and diagnostics.""" + + max_copies: int = 3 + min_fragment_atoms: int = 1 + max_fragment_atoms: int = 8 + max_fragments_per_reactant: int = 2500 + max_matches_per_fragment: int = 128 + max_base_candidates_per_reactant: int = 6000 + include_rdkit_mcs_candidates: bool = True + max_mcs_matches: int = 256 + respect_atom_maps: Optional[bool] = None # None means auto-detect. + require_atom_map_match_when_present: bool = True + mapped_reactants_single_copy: bool = True + compare_formal_charge: bool = True + compare_aromaticity: bool = False + compare_isotope: bool = False + compare_bond_order: bool = True + allow_extra_product_edges_in_candidate: bool = True + selector: str = "ilp" # ilp or greedy + fallback_to_greedy: bool = True + + # Linear objective terms. These are deliberately simple; detailed topology + # diagnostics are computed after selection. + atom_reward: float = 10.0 + preserved_bond_reward: float = 5.0 + atom_map_anchor_bonus: float = 25.0 + candidate_piece_penalty: float = 2.0 + active_copy_penalty: float = 1.0 + unused_reactant_atom_penalty_active_copy: float = 6.0 + extra_product_edge_penalty: float = 2.0 + single_atom_piece_penalty: float = 4.0 + broken_bond_environment_penalty: float = 1.0 + bond_environment_objective: str = "off" # off, integrated, or rerank + bond_environment_rank_tolerance: float = 1.0e-6 + stable_single_bond_break_penalty: float = 1.0 + unsaturated_endpoint_break_credit: float = 0.75 + ring_bond_break_penalty: float = 2.0 + max_broken_bond_pair_penalty_terms: int = 25000 + + # Diagnostic distances. + max_segment_distance: int = 8 + + @classmethod + def from_mapping( + cls, + data: object, + *, + base_dir: Path, + ) -> ReactionEnergyConfig: + mapping = _validate_mapping_format( + data, name="tools_settings.reaction_atom_mapper" + ) + return cls(todo) # todo + + +@dataclass(frozen=True) +class ReactantComponent: + rid: int + mol: Chem.Mol + graph: nx.Graph + smiles: str + + +@dataclass(frozen=True) +class Candidate: + """A connected common-subgraph subtraction candidate before copy expansion.""" + + cid: int + reactant_id: int + reactant_atoms: Tuple[int, ...] + product_atoms: Tuple[int, ...] + r_to_p: Tuple[Tuple[int, int], ...] + preserved_bonds: int + extra_product_edges: int + atom_map_matches: int + score: float + source: str + + def mapping_dict(self) -> Dict[int, int]: + return dict(self.r_to_p) + + def product_atom_set(self) -> FrozenSet[int]: + return frozenset(self.product_atoms) + + def reactant_atom_set(self) -> FrozenSet[int]: + return frozenset(self.reactant_atoms) + + +@dataclass(frozen=True) +class ExpandedCandidate: + xid: int + base: Candidate + copy_id: int + + @property + def reactant_id(self) -> int: + return self.base.reactant_id + + @property + def score(self) -> float: + return self.base.score + + @property + def product_atoms(self) -> Tuple[int, ...]: + return self.base.product_atoms + + @property + def reactant_atoms(self) -> Tuple[int, ...]: + return self.base.reactant_atoms + + @property + def r_to_p(self) -> Tuple[Tuple[int, int], ...]: + return self.base.r_to_p + + +@dataclass +class SelectedPiece: + # todo: add doc + reactant_id: int + copy_id: int + candidate_id: int + source: str + reactant_atoms: Tuple[int, ...] + product_atoms: Tuple[int, ...] + r_to_p: Dict[int, int] + preserved_bonds: int + extra_product_edges: int + score: float + + +@dataclass +class SubtractiveMappingResult: + reaction_smiles: str + selector: str # todo what is this + objective_value: float + status: str + reactant_components: List[ReactantComponent] + product_mol: Chem.Mol + product_graph: nx.Graph + selected_pieces: List[SelectedPiece] + diagnostics: Dict[str, Any] + config: ReactionAtomMapperConfig + + def atom_mapped_reaction_smiles(self) -> str: + return build_atom_mapped_reaction_smiles(self) + + def to_jsonable(self) -> Dict[str, Any]: + reactants = [ + { + "reactant_id": rc.rid, + "smiles": rc.smiles, + "atom_count": rc.mol.GetNumAtoms(), + "bond_count": rc.mol.GetNumBonds(), + } + for rc in self.reactant_components + ] + pieces = [ + { + "reactant_id": p.reactant_id, + "copy_id": p.copy_id, + "candidate_id": p.candidate_id, + "source": p.source, + "reactant_atoms": list(p.reactant_atoms), + "product_atoms": list(p.product_atoms), + "r_to_p": {str(k): v for k, v in sorted(p.r_to_p.items())}, + "preserved_bonds": p.preserved_bonds, + "extra_product_edges": p.extra_product_edges, + "score": p.score, + } + for p in self.selected_pieces + ] + return { + "reaction_smiles": self.reaction_smiles, + "atom_mapped_reaction_smiles": self.atom_mapped_reaction_smiles(), + "selector": self.selector, + "status": self.status, + "objective_value": self.objective_value, + "reactants": reactants, + "product_smiles": ( + Chem.MolToSmiles(self.product_mol, canonical=True) + if self.product_mol is not None + else "" + ), + "selected_pieces": pieces, + "diagnostics": self.diagnostics, + "config": dataclasses.asdict(self.config), + } + + def to_json(self, indent: int = 2) -> str: + return json.dumps(self.to_jsonable(), indent=indent, sort_keys=True) + + +def _copy_mol_with_fresh_atom_maps( + mol: Chem.Mol, atom_to_map_num: Mapping[int, int] +) -> Chem.Mol: + """Return a copy of mol with only atom_to_map_num atom maps set. + + Existing atom-map numbers are cleared first so copied reactants get unique + map numbers, which is important when a reactant is reused via virtual copies. + """ + out = Chem.Mol(mol) + for atom in out.GetAtoms(): + atom.SetAtomMapNum(0) + for atom_idx, map_num in atom_to_map_num.items(): + if 0 <= int(atom_idx) < out.GetNumAtoms(): + out.GetAtomWithIdx(int(atom_idx)).SetAtomMapNum(int(map_num)) + return out + + +def _next_available_atom_map(used: Set[int]) -> int: + value = 1 + while value in used: + value += 1 + return value + + +def build_atom_mapped_reaction_smiles(result: "SubtractiveMappingResult") -> str: + """Construct a partially atom-mapped reaction SMILES from selected pieces. + + The left side contains one molecule for each active virtual reactant copy; + unused original reactant components are included once without atom maps. + Product atoms not covered by selected subtraction pieces remain unmapped. + + Existing atom-map numbers are preserved when they are unambiguous. This + keeps anchored inputs such as [C:1]C[C:2]>>[C:1]CC[C:2] readable, while + still assigning fresh unique map numbers for unmapped or copied atoms. + """ + pairs: List[Tuple[int, int, int, int]] = [] + for piece in result.selected_pieces: + for reactant_atom, product_atom in piece.r_to_p.items(): + pairs.append( + (piece.reactant_id, piece.copy_id, reactant_atom, product_atom) + ) + pairs = sorted(set(pairs), key=lambda x: (x[0], x[1], x[2], x[3])) + + # Prefer existing map numbers only when that preference is unique across + # all selected pairs. This avoids duplicate atom-map labels when a mapped + # reactant is virtually copied. + preferred_by_pair: Dict[Tuple[int, int, int, int], int] = {} + preferred_counts: Counter[int] = Counter() + for rid, copy_id, reactant_atom, product_atom in pairs: + rc = result.reactant_components[rid] + rm = int(rc.mol.GetAtomWithIdx(reactant_atom).GetAtomMapNum()) + pm = int(result.product_mol.GetAtomWithIdx(product_atom).GetAtomMapNum()) + preferred = 0 + if rm > 0 and pm > 0 and rm == pm: + preferred = rm + elif rm > 0 and pm == 0: + preferred = rm + elif pm > 0 and rm == 0: + preferred = pm + if preferred > 0: + preferred_by_pair[(rid, copy_id, reactant_atom, product_atom)] = preferred + preferred_counts[preferred] += 1 + + rcopy_atom_to_map: Dict[Tuple[int, int, int], int] = {} + product_atom_to_map: Dict[int, int] = {} + used_maps: Set[int] = set() + + def assign_pair( + rid: int, copy_id: int, reactant_atom: int, product_atom: int, map_num: int + ) -> None: + rcopy_atom_to_map[(rid, copy_id, reactant_atom)] = map_num + product_atom_to_map[product_atom] = map_num + used_maps.add(map_num) + + # First assign unambiguous existing atom maps. + for rid, copy_id, reactant_atom, product_atom in pairs: + pair = (rid, copy_id, reactant_atom, product_atom) + preferred = preferred_by_pair.get(pair, 0) + if preferred > 0 and preferred_counts[preferred] == 1: + assign_pair(rid, copy_id, reactant_atom, product_atom, preferred) + + # Then assign fresh map numbers for everything else. Sort by product atom so + # the generated labels are stable and easy to inspect on the product side. + for rid, copy_id, reactant_atom, product_atom in sorted( + pairs, key=lambda x: (x[3], x[0], x[1], x[2]) + ): + rkey = (rid, copy_id, reactant_atom) + if rkey in rcopy_atom_to_map and product_atom in product_atom_to_map: + continue + if rkey in rcopy_atom_to_map: + map_num = rcopy_atom_to_map[rkey] + elif product_atom in product_atom_to_map: + map_num = product_atom_to_map[product_atom] + else: + map_num = _next_available_atom_map(used_maps) + assign_pair(rid, copy_id, reactant_atom, product_atom, map_num) + + active_copies_by_reactant: Dict[int, Set[int]] = defaultdict(set) + for rid, copy_id, _atom in rcopy_atom_to_map: + active_copies_by_reactant[rid].add(copy_id) + + reactant_smiles_parts: List[str] = [] + for rc in sorted(result.reactant_components, key=lambda x: x.rid): + copy_ids = sorted(active_copies_by_reactant.get(rc.rid, set())) + if not copy_ids: + copy_ids = [0] + for copy_id in copy_ids: + atom_maps = { + atom_idx: map_num + for (rid, k, atom_idx), map_num in rcopy_atom_to_map.items() + if rid == rc.rid and k == copy_id + } + reactant_mol = _copy_mol_with_fresh_atom_maps(rc.mol, atom_maps) + reactant_smiles_parts.append(Chem.MolToSmiles(reactant_mol, canonical=True)) + + product_mol = _copy_mol_with_fresh_atom_maps( + result.product_mol, product_atom_to_map + ) + product_smiles = Chem.MolToSmiles(product_mol, canonical=True) + return f"{'.'.join(reactant_smiles_parts)}>>{product_smiles}" + + +# --------------------------------------------------------------------------- +# RDKit / graph helpers +# --------------------------------------------------------------------------- + + +def parse_reaction_smiles_to_halves(reaction_smiles: str) -> Tuple[str, str]: + """Return reactant_side, product_side for SMILES or reaction SMILES.""" + if ">>" in reaction_smiles: + left, right = reaction_smiles.split(">>", 1) + return left.strip(), right.strip() + parts = reaction_smiles.split(">") + if len(parts) == 3: + return parts[0].strip(), parts[2].strip() + raise ValueError( + "Expected reaction SMILES containing '>>' or 'reactants>agents>products'." + ) + + +def mol_from_side(side: str) -> Chem.Mol: + if side == "": + return Chem.Mol() + mol = Chem.MolFromSmiles(side, sanitize=True) + if mol is None: + raise ValueError(f"Could not parse SMILES side: {side!r}") + return mol + + +def split_reactant_components(reactant_side: str) -> List[ReactantComponent]: + mol = mol_from_side(reactant_side) + frags = Chem.GetMolFrags(mol, asMols=True, sanitizeFrags=True) + comps: List[ReactantComponent] = [] + for rid, frag in enumerate(frags): + # Remove atom-map numbers from canonical component SMILES? Keep them, + # since they are useful when inspecting anchored examples. + smi = Chem.MolToSmiles(frag, canonical=True) + comps.append( + ReactantComponent(rid=rid, mol=frag, graph=mol_to_nx(frag), smiles=smi) + ) + return comps + + +def atom_attrs(atom: Chem.Atom) -> Dict[str, Any]: + return { + "atomic_num": atom.GetAtomicNum(), + "symbol": atom.GetSymbol(), + "formal_charge": atom.GetFormalCharge(), + "isotope": atom.GetIsotope(), + "is_aromatic": atom.GetIsAromatic(), + "atom_map": atom.GetAtomMapNum(), + } + + +def bond_order_value(bond: Chem.Bond) -> float: + # RDKit BondTypeAsDouble handles aromatic as 1.5. + try: + return float(bond.GetBondTypeAsDouble()) + except Exception: + return float(bond.GetBondType()) + + +def bond_attrs(bond: Chem.Bond) -> Dict[str, Any]: + return { + "bond_order": bond_order_value(bond), + "bond_type": str(bond.GetBondType()), + "is_aromatic": bond.GetIsAromatic(), + } + + +def mol_to_nx(mol: Chem.Mol) -> nx.Graph: + g = nx.Graph() + for atom in mol.GetAtoms(): + g.add_node(atom.GetIdx(), **atom_attrs(atom)) + for bond in mol.GetBonds(): + g.add_edge(bond.GetBeginAtomIdx(), bond.GetEndAtomIdx(), **bond_attrs(bond)) + return g + + +def side_has_any_atom_maps(mol: Chem.Mol) -> bool: + return any(atom.GetAtomMapNum() > 0 for atom in mol.GetAtoms()) + + +def auto_respect_atom_maps( + reactants: Sequence[ReactantComponent], + product_mol: Chem.Mol, + cfg: ReactionAtomMapperConfig, +) -> bool: + if cfg.respect_atom_maps is not None: + return bool(cfg.respect_atom_maps) + left = any(side_has_any_atom_maps(rc.mol) for rc in reactants) + right = side_has_any_atom_maps(product_mol) + return left and right + + +def atom_compatible_attrs( + a: Mapping[str, Any], + b: Mapping[str, Any], + cfg: ReactionAtomMapperConfig, + respect_maps: bool, +) -> bool: + if a["atomic_num"] != b["atomic_num"]: + return False + if cfg.compare_formal_charge and a.get("formal_charge") != b.get("formal_charge"): + return False + if cfg.compare_aromaticity and a.get("is_aromatic") != b.get("is_aromatic"): + return False + if cfg.compare_isotope and a.get("isotope") != b.get("isotope"): + return False + if respect_maps and cfg.require_atom_map_match_when_present: + ma = int(a.get("atom_map") or 0) + mb = int(b.get("atom_map") or 0) + # Hard-anchor semantics: if either endpoint is mapped, both must have the + # same nonzero map number. Unmapped atoms can only match unmapped atoms. + if ma != mb: + return False + return True + + +def bond_compatible_attrs( + a: Mapping[str, Any], b: Mapping[str, Any], cfg: ReactionAtomMapperConfig +) -> bool: + if ( + cfg.compare_bond_order + and abs(float(a.get("bond_order", 0.0)) - float(b.get("bond_order", 0.0))) + > 1.0e-6 + ): + return False + return True + + +def atom_compatible_mol( + ra: Chem.Atom, pa: Chem.Atom, cfg: ReactionAtomMapperConfig, respect_maps: bool +) -> bool: + return atom_compatible_attrs(atom_attrs(ra), atom_attrs(pa), cfg, respect_maps) + + +def bond_compatible_mol( + rb: Chem.Bond, pb: Chem.Bond, cfg: ReactionAtomMapperConfig +) -> bool: + return bond_compatible_attrs(bond_attrs(rb), bond_attrs(pb), cfg) + + +# --------------------------------------------------------------------------- +# Candidate generation +# --------------------------------------------------------------------------- + + +def connected_subsets_limited( + g: nx.Graph, min_size: int, max_size: int, max_subsets: int +) -> List[Tuple[int, ...]]: + """Generate connected node subsets up to max_size, largest first. + + This is intentionally bounded. Exhaustive connected-subgraph enumeration is + exponential, so this helper produces a diverse bounded set suitable for + common-subgraph candidate generation. + """ + if g.number_of_nodes() == 0: + return [] + max_size = min(max_size, g.number_of_nodes()) + min_size = max(1, min_size) + + seen: Set[FrozenSet[int]] = set() + q: deque[FrozenSet[int]] = deque() + for n in sorted(g.nodes): + fs = frozenset([n]) + seen.add(fs) + q.append(fs) + + out: List[Tuple[int, ...]] = [] + while q and len(seen) <= max_subsets * 8: + cur = q.popleft() + if len(cur) >= min_size: + out.append(tuple(sorted(cur))) + if len(out) >= max_subsets: + break + if len(cur) >= max_size: + continue + boundary: Set[int] = set() + for u in cur: + boundary.update(g.neighbors(u)) + for v in sorted(boundary - set(cur)): + nxt = frozenset(set(cur) | {v}) + if nxt not in seen: + seen.add(nxt) + q.append(nxt) + out.sort(key=lambda xs: (-len(xs), xs)) + return out[:max_subsets] + + +def count_internal_edges(g: nx.Graph, nodes: Iterable[int]) -> int: + s = set(nodes) + return sum(1 for u, v in g.edges if u in s and v in s) + + +def validate_mapping_edges( + r_graph: nx.Graph, + p_graph: nx.Graph, + r_to_p: Mapping[int, int], + cfg: ReactionAtomMapperConfig, +) -> Tuple[bool, int, int]: + """Check reactant edges are present/compatible in product. + + Returns (valid, preserved_bonds, extra_product_edges_among_selected_atoms). + If allow_extra_product_edges_in_candidate is False, extra product edges make + the candidate invalid. Otherwise they are allowed and penalized. + """ + preserved = 0 + for ru, rv, rdata in r_graph.edges(data=True): + if ru not in r_to_p or rv not in r_to_p: + continue + pu, pv = r_to_p[ru], r_to_p[rv] + if not p_graph.has_edge(pu, pv): + return False, 0, 0 + if not bond_compatible_attrs(rdata, p_graph.edges[pu, pv], cfg): + return False, 0, 0 + preserved += 1 + + inv = {p: r for r, p in r_to_p.items()} + extra = 0 + selected_p = set(inv) + for pu, pv in p_graph.subgraph(selected_p).edges: + ru, rv = inv[pu], inv[pv] + if not r_graph.has_edge(ru, rv): + extra += 1 + if extra and not cfg.allow_extra_product_edges_in_candidate: + return False, 0, 0 + return True, preserved, extra + + +def candidate_score( + r_graph: nx.Graph, + p_graph: nx.Graph, + r_to_p: Mapping[int, int], + preserved_bonds: int, + extra_product_edges: int, + cfg: ReactionAtomMapperConfig, +) -> Tuple[float, int]: + atom_count = len(r_to_p) + anchor_matches = 0 + for r, p in r_to_p.items(): + rm = int(r_graph.nodes[r].get("atom_map") or 0) + pm = int(p_graph.nodes[p].get("atom_map") or 0) + if rm > 0 and rm == pm: + anchor_matches += 1 + score = ( + cfg.atom_reward * atom_count + + cfg.preserved_bond_reward * preserved_bonds + + cfg.atom_map_anchor_bonus * anchor_matches + - cfg.extra_product_edge_penalty * extra_product_edges + ) + if atom_count == 1: + score -= cfg.single_atom_piece_penalty + return score, anchor_matches + + +def mapping_key(reactant_id: int, r_to_p: Mapping[int, int]) -> Tuple[Any, ...]: + return (reactant_id, tuple(sorted(r_to_p.items()))) + + +def add_candidate_if_valid( + candidates: Dict[Tuple[Any, ...], Candidate], + reactant_id: int, + r_graph: nx.Graph, + p_graph: nx.Graph, + r_to_p: Mapping[int, int], + cfg: ReactionAtomMapperConfig, + source: str, + next_id: List[int], +) -> None: + if not r_to_p: + return + if len(set(r_to_p.values())) != len(r_to_p): + return + # Candidate must be connected on the reactant side and on the product side. + # This prevents one candidate from hiding a connectivity break. A split + # lineage must be represented as multiple selected pieces. + r_nodes = tuple(sorted(r_to_p)) + p_nodes = tuple(sorted(r_to_p.values())) + if len(r_nodes) > 1: + if not nx.is_connected(r_graph.subgraph(r_nodes)): + return + if not nx.is_connected(p_graph.subgraph(p_nodes)): + return + valid, preserved, extra = validate_mapping_edges(r_graph, p_graph, r_to_p, cfg) + if not valid: + return + score, anchors = candidate_score(r_graph, p_graph, r_to_p, preserved, extra, cfg) + key = mapping_key(reactant_id, r_to_p) + existing = candidates.get(key) + if existing is not None and existing.score >= score: + return + cid = next_id[0] + next_id[0] += 1 + candidates[key] = Candidate( + cid=cid, + reactant_id=reactant_id, + reactant_atoms=tuple(sorted(r_to_p)), + product_atoms=tuple(sorted(r_to_p.values())), + r_to_p=tuple(sorted(r_to_p.items())), + preserved_bonds=preserved, + extra_product_edges=extra, + atom_map_matches=anchors, + score=score, + source=source, + ) + + +def generate_fragment_candidates_for_reactant( + rc: ReactantComponent, + product_graph: nx.Graph, + cfg: ReactionAtomMapperConfig, + respect_maps: bool, + next_id: List[int], +) -> List[Candidate]: + """Generate connected reactant-fragment candidates via NetworkX matching.""" + r_graph = rc.graph + candidates: Dict[Tuple[Any, ...], Candidate] = {} + fragments = connected_subsets_limited( + r_graph, + min_size=cfg.min_fragment_atoms, + max_size=cfg.max_fragment_atoms, + max_subsets=cfg.max_fragments_per_reactant, + ) + + def nm(p_attrs: Mapping[str, Any], r_attrs: Mapping[str, Any]) -> bool: + return atom_compatible_attrs(r_attrs, p_attrs, cfg, respect_maps) + + def em(p_attrs: Mapping[str, Any], r_attrs: Mapping[str, Any]) -> bool: + return bond_compatible_attrs(r_attrs, p_attrs, cfg) + + for frag_nodes in fragments: + r_sub = r_graph.subgraph(frag_nodes).copy() + gm = nx.algorithms.isomorphism.GraphMatcher( + product_graph, r_sub, node_match=nm, edge_match=em + ) + if cfg.allow_extra_product_edges_in_candidate and hasattr( + gm, "subgraph_monomorphisms_iter" + ): + iterator = gm.subgraph_monomorphisms_iter() + else: + iterator = gm.subgraph_isomorphisms_iter() + + n_matches = 0 + for p_to_r in iterator: + # p_to_r maps product node -> reactant node. Invert. + r_to_p = {r: p for p, r in p_to_r.items()} + # GraphMatcher can return mappings larger than the query for some + # monomorphism variants; filter to the fragment atom set. + r_to_p = {r: p for r, p in r_to_p.items() if r in r_sub.nodes} + if set(r_to_p) != set(r_sub.nodes): + continue + add_candidate_if_valid( + candidates, + rc.rid, + r_graph, + product_graph, + r_to_p, + cfg, + "nx_fragment", + next_id, + ) + n_matches += 1 + if n_matches >= cfg.max_matches_per_fragment: + break + if len(candidates) >= cfg.max_base_candidates_per_reactant: + break + + vals = list(candidates.values()) + vals.sort( + key=lambda c: ( + -c.score, + -len(c.reactant_atoms), + c.reactant_atoms, + c.product_atoms, + ) + ) + return vals[: cfg.max_base_candidates_per_reactant] + + +def generate_rdkit_mcs_candidates_for_reactant( + rc: ReactantComponent, + product_mol: Chem.Mol, + product_graph: nx.Graph, + cfg: ReactionAtomMapperConfig, + respect_maps: bool, + next_id: List[int], +) -> List[Candidate]: + """Generate large candidates from RDKit FindMCS.""" + if rc.mol.GetNumAtoms() == 0 or product_mol.GetNumAtoms() == 0: + return [] + candidates: Dict[Tuple[Any, ...], Candidate] = {} + try: + params = rdFMCS.MCSParameters() + params.AtomTyper = rdFMCS.AtomCompare.CompareElements + params.BondTyper = ( + rdFMCS.BondCompare.CompareOrder + if cfg.compare_bond_order + else rdFMCS.BondCompare.CompareAny + ) + params.RingMatchesRingOnly = True + params.CompleteRingsOnly = False + params.Timeout = 5 + mcs = rdFMCS.FindMCS([rc.mol, product_mol], params) + except Exception: + return [] + if mcs.canceled or not mcs.smartsString: + return [] + query = Chem.MolFromSmarts(mcs.smartsString) + if query is None or query.GetNumAtoms() == 0: + return [] + try: + r_matches = list( + rc.mol.GetSubstructMatches( + query, uniquify=True, maxMatches=cfg.max_mcs_matches + ) + ) + p_matches = list( + product_mol.GetSubstructMatches( + query, uniquify=True, maxMatches=cfg.max_mcs_matches + ) + ) + except TypeError: + # Older RDKit versions may not accept maxMatches as keyword. + r_matches = list(rc.mol.GetSubstructMatches(query, True))[: cfg.max_mcs_matches] + p_matches = list(product_mol.GetSubstructMatches(query, True))[ + : cfg.max_mcs_matches + ] + + for r_match in r_matches[: cfg.max_mcs_matches]: + for p_match in p_matches[: cfg.max_mcs_matches]: + r_to_p = {int(r): int(p) for r, p in zip(r_match, p_match)} + ok = True + for r, p in r_to_p.items(): + if not atom_compatible_mol( + rc.mol.GetAtomWithIdx(r), + product_mol.GetAtomWithIdx(p), + cfg, + respect_maps, + ): + ok = False + break + if not ok: + continue + add_candidate_if_valid( + candidates, + rc.rid, + rc.graph, + product_graph, + r_to_p, + cfg, + "rdkit_mcs", + next_id, + ) + if len(candidates) >= cfg.max_base_candidates_per_reactant: + break + if len(candidates) >= cfg.max_base_candidates_per_reactant: + break + vals = list(candidates.values()) + vals.sort( + key=lambda c: ( + -c.score, + -len(c.reactant_atoms), + c.reactant_atoms, + c.product_atoms, + ) + ) + return vals[: cfg.max_base_candidates_per_reactant] + + +def generate_base_candidates( + reactants: Sequence[ReactantComponent], + product_mol: Chem.Mol, + product_graph: nx.Graph, + cfg: ReactionAtomMapperConfig, + respect_maps: bool, +) -> List[Candidate]: + """todo: doc. what does base candidates mean? + * Candidates are connected common subgraph occurrences. + """ + next_id = [0] + all_candidates: Dict[Tuple[Any, ...], Candidate] = {} + for rc in reactants: + per_reactant: List[Candidate] = [] + if cfg.include_rdkit_mcs_candidates: + per_reactant.extend( + generate_rdkit_mcs_candidates_for_reactant( + rc, product_mol, product_graph, cfg, respect_maps, next_id + ) + ) + per_reactant.extend( + generate_fragment_candidates_for_reactant( + rc, product_graph, cfg, respect_maps, next_id + ) + ) + # Deduplicate across MCS and fragment generation. + local: Dict[Tuple[Any, ...], Candidate] = {} + for c in per_reactant: + key = mapping_key(c.reactant_id, c.mapping_dict()) + if key not in local or c.score > local[key].score: + local[key] = c + vals = list(local.values()) + vals.sort( + key=lambda c: ( + -c.score, + -len(c.reactant_atoms), + c.reactant_atoms, + c.product_atoms, + ) + ) + vals = vals[: cfg.max_base_candidates_per_reactant] + for c in vals: + key = mapping_key(c.reactant_id, c.mapping_dict()) + all_candidates[key] = c + # Reassign candidate IDs densely for readability. + vals = list(all_candidates.values()) + vals.sort( + key=lambda c: ( + c.reactant_id, + -c.score, + -len(c.reactant_atoms), + c.reactant_atoms, + c.product_atoms, + ) + ) + dense: List[Candidate] = [] + for cid, c in enumerate(vals): + dense.append(dataclasses.replace(c, cid=cid)) + return dense + + +# --------------------------------------------------------------------------- +# Candidate selection: ILP and greedy +# --------------------------------------------------------------------------- + + +def reactant_has_mapped_atoms(rc: ReactantComponent) -> bool: + return any(int(a.GetAtomMapNum()) > 0 for a in rc.mol.GetAtoms()) + + +def copy_count_for_reactant( + rc: ReactantComponent, cfg: ReactionAtomMapperConfig +) -> int: + if cfg.mapped_reactants_single_copy and reactant_has_mapped_atoms(rc): + return 1 + return cfg.max_copies + + +def expand_candidates( + base_candidates: Sequence[Candidate], + cfg: ReactionAtomMapperConfig, + reactants: Optional[Sequence[ReactantComponent]] = None, +) -> List[ExpandedCandidate]: + expanded: List[ExpandedCandidate] = [] + xid = 0 + copy_counts: Dict[int, int] = defaultdict(lambda: cfg.max_copies) + if reactants is not None: + copy_counts = defaultdict( + lambda: cfg.max_copies, + {rc.rid: copy_count_for_reactant(rc, cfg) for rc in reactants}, + ) + for c in base_candidates: + for k in range(copy_counts[c.reactant_id]): + expanded.append(ExpandedCandidate(xid=xid, base=c, copy_id=k)) + xid += 1 + return expanded + + +def select_candidates_greedy( + base_candidates: Sequence[Candidate], + reactants: Sequence[ReactantComponent], + cfg: ReactionAtomMapperConfig, +) -> Tuple[List[ExpandedCandidate], float, str]: + expanded = expand_candidates(base_candidates, cfg, reactants) + expanded.sort( + key=lambda x: ( + -( + x.score + - cfg.candidate_piece_penalty + + cfg.unused_reactant_atom_penalty_active_copy * len(x.reactant_atoms) + ), + -len(x.product_atoms), + x.reactant_id, + x.copy_id, + ) + ) + used_product: Set[int] = set() + used_reactant_by_copy: Set[Tuple[int, int, int]] = set() + active_copies: Set[Tuple[int, int]] = set() + chosen: List[ExpandedCandidate] = [] + objective = 0.0 + for x in expanded: + marginal = ( + x.score + - cfg.candidate_piece_penalty + + cfg.unused_reactant_atom_penalty_active_copy * len(x.reactant_atoms) + ) + if (x.reactant_id, x.copy_id) not in active_copies: + rc_atoms = reactants[x.reactant_id].graph.number_of_nodes() + marginal -= ( + cfg.active_copy_penalty + + cfg.unused_reactant_atom_penalty_active_copy * rc_atoms + ) + if marginal <= 0: + continue + if any(p in used_product for p in x.product_atoms): + continue + if any( + (x.reactant_id, x.copy_id, r) in used_reactant_by_copy + for r in x.reactant_atoms + ): + continue + chosen.append(x) + objective += marginal + used_product.update(x.product_atoms) + for r in x.reactant_atoms: + used_reactant_by_copy.add((x.reactant_id, x.copy_id, r)) + active_copies.add((x.reactant_id, x.copy_id)) + return chosen, objective, "greedy" + + +def atom_has_multiple_bond_to_hetero(atom: Chem.Atom) -> bool: + """Generic local electronic environment test used for bond-break scoring.""" + for bond in atom.GetBonds(): + if bond_order_value(bond) < 1.5: + continue + other = bond.GetOtherAtom(atom) + if other.GetAtomicNum() not in {1, 6}: + return True + return False + + +def atom_is_saturated_carbon(atom: Chem.Atom) -> bool: + if atom.GetAtomicNum() != 6 or atom.GetIsAromatic(): + return False + return all(bond_order_value(bond) <= 1.1 for bond in atom.GetBonds()) + + +def bond_environment_break_penalty( + mol: Chem.Mol, begin_atom: int, end_atom: int, cfg: ReactionAtomMapperConfig +) -> float: + """Return a local-environment penalty for breaking a reactant bond. + + This intentionally uses generic atom/bond features rather than named + functional groups. Single bonds from saturated carbon to hetero atoms are + treated as harder to break, while bonds attached to an atom with a multiple + bond to a hetero atom are treated as more plausible reaction-center breaks. + """ + scale = max(0.0, float(cfg.broken_bond_environment_penalty)) + if scale == 0.0: + return 0.0 + + bond = mol.GetBondBetweenAtoms(int(begin_atom), int(end_atom)) + if bond is None: + return 0.0 + + a1 = mol.GetAtomWithIdx(int(begin_atom)) + a2 = mol.GetAtomWithIdx(int(end_atom)) + order = bond_order_value(bond) + penalty = scale + + if order > 1.1: + penalty += scale * (order - 1.0) + if bond.IsInRing(): + penalty += cfg.ring_bond_break_penalty + if a1.GetIsAromatic() or a2.GetIsAromatic(): + penalty += 0.5 * scale + + has_unsaturated_endpoint = atom_has_multiple_bond_to_hetero( + a1 + ) or atom_has_multiple_bond_to_hetero(a2) + if has_unsaturated_endpoint: + penalty -= cfg.unsaturated_endpoint_break_credit + + atomic_nums = {a1.GetAtomicNum(), a2.GetAtomicNum()} + has_hetero = any(z not in {1, 6} for z in atomic_nums) + has_saturated_carbon = atom_is_saturated_carbon(a1) or atom_is_saturated_carbon(a2) + if ( + order <= 1.1 + and has_hetero + and has_saturated_carbon + and not has_unsaturated_endpoint + ): + penalty += cfg.stable_single_bond_break_penalty + + return max(0.05 * scale, penalty) + + +def broken_reactant_bond_pair_penalties( + expanded: Sequence[ExpandedCandidate], + reactants: Sequence[ReactantComponent], + product_graph: nx.Graph, + cfg: ReactionAtomMapperConfig, +) -> List[Tuple[int, int, float]]: + """Build pairwise penalties for selected pieces that break reactant bonds.""" + if cfg.broken_bond_environment_penalty <= 0.0: + return [] + + by_reactant_copy_atom: Dict[Tuple[int, int, int], List[Tuple[int, int]]] = ( + defaultdict(list) + ) + for i, x in enumerate(expanded): + r_to_p = dict(x.r_to_p) + for r in x.reactant_atoms: + by_reactant_copy_atom[(x.reactant_id, x.copy_id, r)].append((i, r_to_p[r])) + + penalties: Dict[Tuple[int, int], float] = defaultdict(float) + reactant_sets = [set(x.reactant_atoms) for x in expanded] + product_sets = [set(x.product_atoms) for x in expanded] + for rc in reactants: + n_copies = copy_count_for_reactant(rc, cfg) + for ru, rv, rdata in rc.graph.edges(data=True): + bond_penalty = bond_environment_break_penalty(rc.mol, ru, rv, cfg) + if bond_penalty <= 0.0: + continue + for copy_id in range(n_copies): + left = by_reactant_copy_atom.get((rc.rid, copy_id, ru), []) + right = by_reactant_copy_atom.get((rc.rid, copy_id, rv), []) + for i, pu in left: + for j, pv in right: + if i == j: + continue + # These pairs are already mutually exclusive by atom + # coverage constraints, so a pairwise break variable + # would only enlarge the ILP without changing feasible + # solutions or the objective. + if reactant_sets[i] & reactant_sets[j]: + continue + if product_sets[i] & product_sets[j]: + continue + if product_graph.has_edge(pu, pv) and bond_compatible_attrs( + rdata, product_graph.edges[pu, pv], cfg + ): + continue + penalties[tuple(sorted((i, j)))] += bond_penalty + + items = list(penalties.items()) + if ( + cfg.max_broken_bond_pair_penalty_terms > 0 + and len(items) > cfg.max_broken_bond_pair_penalty_terms + ): + items = sorted(items, key=lambda kv: (-kv[1], kv[0]))[ + : cfg.max_broken_bond_pair_penalty_terms + ] + return [(i, j, penalty) for (i, j), penalty in sorted(items)] + + +def select_candidates_ilp( + base_candidates: Sequence[Candidate], + reactants: Sequence[ReactantComponent], + product_graph: nx.Graph, + cfg: ReactionAtomMapperConfig, +) -> Tuple[List[ExpandedCandidate], float, str]: + if not SCIPY_MILP_AVAILABLE: + if cfg.fallback_to_greedy: + return select_candidates_greedy(base_candidates, reactants, cfg) + raise RuntimeError("scipy.optimize.milp is not available.") + if ( + np is None + or sp is None + or milp is None + or LinearConstraint is None + or Bounds is None + ): + raise RuntimeError("scipy.optimize.milp is not available.") + + mode = cfg.bond_environment_objective.lower().strip() + if mode not in {"off", "integrated", "rerank"}: + raise ValueError( + "bond_environment_objective must be 'off', 'integrated', or 'rerank'." + ) + + expanded = expand_candidates(base_candidates, cfg, reactants) + n_x = len(expanded) + copy_keys: List[Tuple[int, int]] = [] + for rc in reactants: + for k in range(cfg.max_copies): + copy_keys.append((rc.rid, k)) + copy_index = {ck: i for i, ck in enumerate(copy_keys)} + n_y = len(copy_keys) + if n_x == 0: + return [], 0.0, "ilp_no_candidates" + + primary_c_base = np.zeros(n_x + n_y, dtype=float) + for i, x in enumerate(expanded): + # scipy minimizes, so negate the maximization coefficient. + coeff = ( + x.score + - cfg.candidate_piece_penalty + + cfg.unused_reactant_atom_penalty_active_copy * len(x.reactant_atoms) + ) + primary_c_base[i] = -coeff + for ck, yi in copy_index.items(): + rid, _copy_id = ck + primary_c_base[n_x + yi] = ( + cfg.active_copy_penalty + + cfg.unused_reactant_atom_penalty_active_copy + * reactants[rid].graph.number_of_nodes() + ) + + def solve( + include_bond_environment: bool, + primary_floor: Optional[float] = None, + secondary_only: bool = False, + ) -> Any: + broken_pair_penalties = ( + broken_reactant_bond_pair_penalties(expanded, reactants, product_graph, cfg) + if include_bond_environment + else [] + ) + n_z = len(broken_pair_penalties) + z_start = n_x + n_y + n_vars = n_x + n_y + n_z + + primary_c = np.zeros(n_vars, dtype=float) + primary_c[: n_x + n_y] = primary_c_base + c = np.zeros(n_vars, dtype=float) if secondary_only else primary_c.copy() + if include_bond_environment: + for zi, (_i, _j, penalty) in enumerate(broken_pair_penalties): + c[z_start + zi] = penalty + + constraint_rows: List[int] = [] + constraint_cols: List[int] = [] + constraint_data: List[float] = [] + lower_bounds: List[float] = [] + upper_bounds: List[float] = [] + + def add_sparse_constraint( + coeffs: Mapping[int, float], lower: float, upper: float + ) -> None: + row_idx = len(lower_bounds) + for col_idx, value in coeffs.items(): + if value != 0.0: + constraint_rows.append(row_idx) + constraint_cols.append(int(col_idx)) + constraint_data.append(float(value)) + lower_bounds.append(float(lower)) + upper_bounds.append(float(upper)) + + # Product atom covered at most once. + for p in product_graph.nodes: + coeffs: Dict[int, float] = {} + for i, x in enumerate(expanded): + if p in x.product_atoms: + coeffs[i] = 1.0 + if coeffs: + add_sparse_constraint(coeffs, -np.inf, 1.0) + + # Reactant atom per copy used at most once. + for rc in reactants: + for k in range(cfg.max_copies): + for r in rc.graph.nodes: + coeffs = {} + for i, x in enumerate(expanded): + if ( + x.reactant_id == rc.rid + and x.copy_id == k + and r in x.reactant_atoms + ): + coeffs[i] = 1.0 + if coeffs: + add_sparse_constraint(coeffs, -np.inf, 1.0) + + # x_j <= y_{reactant,copy} + for i, x in enumerate(expanded): + add_sparse_constraint( + {i: 1.0, n_x + copy_index[(x.reactant_id, x.copy_id)]: -1.0}, + -np.inf, + 0.0, + ) + + # y_{reactant,copy} <= sum selected pieces using that copy. + for ck, yi in copy_index.items(): + coeffs = {n_x + yi: 1.0} + any_piece = False + for i, x in enumerate(expanded): + if (x.reactant_id, x.copy_id) == ck: + coeffs[i] = coeffs.get(i, 0.0) - 1.0 + any_piece = True + if any_piece: + add_sparse_constraint(coeffs, -np.inf, 0.0) + else: + add_sparse_constraint(coeffs, 0.0, 0.0) + + if include_bond_environment: + # z_ij is forced on when both selected pieces are present and their + # mapped endpoints imply a broken reactant bond. + for zi, (i, j, _penalty) in enumerate(broken_pair_penalties): + add_sparse_constraint( + {i: 1.0, j: 1.0, z_start + zi: -1.0}, -np.inf, 1.0 + ) + + # Symmetry breaking: y_{i,k+1} <= y_{i,k} + for rc in reactants: + for k in range(cfg.max_copies - 1): + add_sparse_constraint( + { + n_x + copy_index[(rc.rid, k + 1)]: 1.0, + n_x + copy_index[(rc.rid, k)]: -1.0, + }, + -np.inf, + 0.0, + ) + + if primary_floor is not None: + coeffs = {i: float(v) for i, v in enumerate(primary_c) if v != 0.0} + add_sparse_constraint(coeffs, -np.inf, -float(primary_floor)) + + constraints: List[LinearConstraint] = [] + if lower_bounds: + a = sp.coo_matrix( + (constraint_data, (constraint_rows, constraint_cols)), + shape=(len(lower_bounds), n_vars), + ).tocsr() + constraints.append( + LinearConstraint(a, np.array(lower_bounds), np.array(upper_bounds)) + ) + + return milp( + c=c, + constraints=constraints, + bounds=Bounds(0.0, 1.0), + integrality=np.ones(n_vars, dtype=int), + options={"time_limit": 30.0}, + ) + + def finalize( + res: Any, objective: float, status: str + ) -> Tuple[List[ExpandedCandidate], float, str]: + xval = res.x[:n_x] + return [expanded[i] for i, v in enumerate(xval) if v > 0.5], objective, status + + try: + if mode == "integrated": + res = solve(include_bond_environment=True) + if getattr(res, "success", False) and getattr(res, "x", None) is not None: + return finalize(res, -float(res.fun), "ilp") + else: + res = solve(include_bond_environment=False) + if getattr(res, "success", False) and getattr(res, "x", None) is not None: + primary_objective = -float(res.fun) + if mode == "rerank" and cfg.broken_bond_environment_penalty > 0.0: + floor = primary_objective - max( + 0.0, cfg.bond_environment_rank_tolerance + ) + rerank = solve( + include_bond_environment=True, + primary_floor=floor, + secondary_only=True, + ) + if ( + getattr(rerank, "success", False) + and getattr(rerank, "x", None) is not None + ): + return finalize(rerank, primary_objective, "ilp_rerank") + return finalize( + res, + primary_objective, + f"ilp_rerank_primary_only_after_status:{getattr(rerank, 'message', 'unknown')}", + ) + return finalize(res, primary_objective, "ilp") + except Exception as e: + if cfg.fallback_to_greedy: + chosen, obj, status = select_candidates_greedy( + base_candidates, reactants, cfg + ) + return chosen, obj, f"greedy_fallback_after_ilp_error:{e}" + raise + + if cfg.fallback_to_greedy: + chosen, obj, status = select_candidates_greedy(base_candidates, reactants, cfg) + return ( + chosen, + obj, + f"greedy_fallback_after_ilp_status:{getattr(res, 'message', 'unknown')}", + ) + raise RuntimeError(f"MILP failed: {getattr(res, 'message', 'unknown')}") + + +def selected_pieces_from_expanded( + chosen: Sequence[ExpandedCandidate], +) -> List[SelectedPiece]: + pieces: List[SelectedPiece] = [] + for x in chosen: + pieces.append( + SelectedPiece( + reactant_id=x.reactant_id, + copy_id=x.copy_id, + candidate_id=x.base.cid, + source=x.base.source, + reactant_atoms=x.reactant_atoms, + product_atoms=x.product_atoms, + r_to_p=dict(x.r_to_p), + preserved_bonds=x.base.preserved_bonds, + extra_product_edges=x.base.extra_product_edges, + score=x.score, + ) + ) + pieces.sort( + key=lambda p: (p.reactant_id, p.copy_id, -len(p.product_atoms), p.product_atoms) + ) + return pieces + + +# --------------------------------------------------------------------------- +# Topology diagnostics +# --------------------------------------------------------------------------- + + +def lineage_label(lineage: Tuple[int, int]) -> str: + rid, copy = lineage + return f"R{rid}/copy{copy}" + + +def source_to_jsonable(src: Any) -> str: + if ( + isinstance(src, tuple) + and len(src) == 2 + and all(isinstance(x, int) for x in src) + ): + return lineage_label(src) + return str(src) + + +def collapse_consecutive(xs: Sequence[Any]) -> List[Any]: + out: List[Any] = [] + for x in xs: + if not out or out[-1] != x: + out.append(x) + return out + + +def safe_shortest_path( + g: nx.Graph, source: int, target: int +) -> Tuple[float, List[int]]: + try: + path = nx.shortest_path(g, source, target) + return float(len(path) - 1), list(path) + except (nx.NetworkXNoPath, nx.NodeNotFound): + return math.inf, [] + + +def build_mapping_indexes( + pieces: Sequence[SelectedPiece], product_graph: nx.Graph +) -> Tuple[ + Dict[int, Tuple[int, int]], + Dict[Tuple[int, int, int], int], + Dict[Tuple[int, int, int], int], + Dict[Tuple[int, int], Set[int]], +]: + """Return product source, reactant-copy->product mapping, inverse, lineage atoms.""" + product_source: Dict[int, Tuple[int, int]] = {} + rcopy_to_product: Dict[Tuple[int, int, int], int] = {} + product_to_rcopy_atom: Dict[Tuple[int, int, int], int] = {} + atoms_by_lineage: Dict[Tuple[int, int], Set[int]] = defaultdict(set) + for piece in pieces: + lin = (piece.reactant_id, piece.copy_id) + for r, p in piece.r_to_p.items(): + product_source[p] = lin + rcopy_to_product[(piece.reactant_id, piece.copy_id, r)] = p + product_to_rcopy_atom[(piece.reactant_id, piece.copy_id, p)] = r + atoms_by_lineage[lin].add(p) + return product_source, rcopy_to_product, product_to_rcopy_atom, atoms_by_lineage + + +def mapped_anchor_segments( + r_graph: nx.Graph, + mapped_atoms: Set[int], + max_distance: int, +) -> List[Dict[str, Any]]: + """Return compressed segments between mapped reactant anchor atoms. + + A segment is a pair of mapped atoms whose shortest path in the reactant has + no mapped interior atom. This catches partial mappings such as + [C:1]C[C:2] where only the endpoints are anchors. + """ + mapped = sorted(mapped_atoms) + segments: List[Dict[str, Any]] = [] + seen: Set[Tuple[int, int]] = set() + for i, u in enumerate(mapped): + for v in mapped[i + 1 :]: + try: + path = nx.shortest_path(r_graph, u, v) + except nx.NetworkXNoPath: + continue + d = len(path) - 1 + if d > max_distance: + continue + interior = path[1:-1] + if any(x in mapped_atoms for x in interior): + continue + key = (u, v) + if key in seen: + continue + seen.add(key) + segments.append( + { + "reactant_atoms": [u, v], + "reactant_distance": d, + "reactant_path": path, + "interior_unmapped_reactant_atoms": interior, + } + ) + segments.sort(key=lambda e: (e["reactant_distance"], e["reactant_atoms"])) + return segments + + +def product_lineage_blocks( + product_graph: nx.Graph, + product_source: Mapping[int, Tuple[int, int]], +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Contract maximal connected blocks with the same product source.""" + # Blocks are computed over all product atoms. Uncovered atoms get source + # ('uncovered', -1) but will be printed as 'uncovered'. + node_source: Dict[int, Any] = { + p: product_source.get(p, "uncovered") for p in product_graph.nodes + } + visited: Set[int] = set() + blocks: List[Dict[str, Any]] = [] + block_id_of_node: Dict[int, int] = {} + for start in product_graph.nodes: + if start in visited: + continue + src = node_source[start] + stack = [start] + visited.add(start) + atoms: List[int] = [] + while stack: + u = stack.pop() + atoms.append(u) + for v in product_graph.neighbors(u): + if v not in visited and node_source[v] == src: + visited.add(v) + stack.append(v) + bid = len(blocks) + for a in atoms: + block_id_of_node[a] = bid + blocks.append( + { + "block_id": bid, + "source": source_to_jsonable(src), + "atoms": sorted(atoms), + "size": len(atoms), + } + ) + + q_edges_set: Set[Tuple[int, int]] = set() + for u, v in product_graph.edges: + bu, bv = block_id_of_node[u], block_id_of_node[v] + if bu != bv: + q_edges_set.add(tuple(sorted((bu, bv)))) + q_edges = [{"block_1": a, "block_2": b} for a, b in sorted(q_edges_set)] + return blocks, q_edges + + +def _copy_mol_clearing_atom_maps(mol: Chem.Mol) -> Chem.Mol: + """Return a shallow molecule copy with atom-map numbers removed.""" + out = Chem.Mol(mol) + for atom in out.GetAtoms(): + atom.SetAtomMapNum(0) + return out + + +def fragment_smiles_for_atoms( + mol: Chem.Mol, atoms: Iterable[int], clear_atom_maps: bool = False +) -> str: + """Return canonical SMILES for a fragment induced by atoms. + + The atom set is expected to be connected, but RDKit can also render a + disconnected set. For residual reporting we call this on connected + components so summaries prioritize fragments over individual atoms. + """ + atom_list = sorted(int(a) for a in atoms) + if not atom_list: + return "" + use_mol = _copy_mol_clearing_atom_maps(mol) if clear_atom_maps else mol + return Chem.MolFragmentToSmiles(use_mol, atomsToUse=atom_list, canonical=True) + + +def connected_atom_components(graph: nx.Graph, atoms: Iterable[int]) -> List[List[int]]: + """Connected components of graph induced by atoms, largest first.""" + atom_set = set(int(a) for a in atoms) + if not atom_set: + return [] + sub = graph.subgraph(atom_set) + comps = ( + [sorted(c) for c in nx.connected_components(sub)] + if sub.number_of_nodes() + else [] + ) + comps.sort(key=lambda xs: (-len(xs), xs)) + return comps + + +def edge_count_in_atom_set(graph: nx.Graph, atoms: Iterable[int]) -> int: + atom_set = set(int(a) for a in atoms) + return sum(1 for u, v in graph.edges if u in atom_set and v in atom_set) + + +def boundary_bonds_for_atom_set( + graph: nx.Graph, atoms: Iterable[int] +) -> List[List[int]]: + atom_set = set(int(a) for a in atoms) + bonds: Set[Tuple[int, int]] = set() + for u in atom_set: + for v in graph.neighbors(u): + if v not in atom_set: + bonds.add(tuple(sorted((u, v)))) + return [list(b) for b in sorted(bonds)] + + +def product_component_lookup( + product_graph: nx.Graph, +) -> Tuple[Dict[int, int], Dict[int, List[int]]]: + """Return atom->component and component->atoms for product connected components.""" + atom_to_component: Dict[int, int] = {} + component_atoms: Dict[int, List[int]] = {} + comps = ( + [sorted(c) for c in nx.connected_components(product_graph)] + if product_graph.number_of_nodes() + else [] + ) + comps.sort(key=lambda xs: (xs[0] if xs else INF, xs)) + for cid, atoms in enumerate(comps): + component_atoms[cid] = atoms + for atom in atoms: + atom_to_component[atom] = cid + return atom_to_component, component_atoms + + +def compute_residual_fragments( + reactants: Sequence[ReactantComponent], + product_mol: Chem.Mol, + product_graph: nx.Graph, + pieces: Sequence[SelectedPiece], +) -> Dict[str, Any]: + """Summarize product and reactant remainders after selected subtractions. + + Product residuals are connected components of product atoms not covered by + any selected common-subgraph piece. A residual that is an entire disconnected + product molecule/component is flagged as a likely byproduct or missing-source + product; a residual that touches selected mapped material is flagged as a + partial unmapped product fragment. + + Reactant residuals are unused connected fragments from active virtual + reactant copies, plus whole original reactant components that were never + selected at all. We do not list every inactive virtual copy, because those + are just optional copies that were not needed. + """ + product_source, rcopy_to_product, _product_to_rcopy_atom, atoms_by_lineage = ( + build_mapping_indexes(pieces, product_graph) + ) + covered_product_atoms = set(product_source) + uncovered_product_atoms = set(product_graph.nodes) - covered_product_atoms + + product_atom_to_component, product_components = product_component_lookup( + product_graph + ) + product_residuals: List[Dict[str, Any]] = [] + for ridx, atoms in enumerate( + connected_atom_components(product_graph, uncovered_product_atoms) + ): + atom_set = set(atoms) + parent_ids = sorted( + { + product_atom_to_component[a] + for a in atoms + if a in product_atom_to_component + } + ) + is_whole_component = False + if len(parent_ids) == 1: + parent_atoms = set(product_components[parent_ids[0]]) + is_whole_component = atom_set == parent_atoms + boundary = boundary_bonds_for_atom_set(product_graph, atoms) + adjacent_sources = sorted( + { + source_to_jsonable(product_source[v]) + for u, v in (tuple(b) for b in boundary) + if v in product_source and u in atom_set + } + | { + source_to_jsonable(product_source[u]) + for u, v in (tuple(b) for b in boundary) + if u in product_source and v in atom_set + } + ) + classification = ( + "whole_uncovered_product_component" + if is_whole_component + else "partial_uncovered_product_fragment" + ) + product_residuals.append( + { + "residual_id": ridx, + "classification": classification, + "atoms": atoms, + "size": len(atoms), + "bond_count": edge_count_in_atom_set(product_graph, atoms), + "smiles": fragment_smiles_for_atoms( + product_mol, atoms, clear_atom_maps=False + ), + "unmapped_smiles": fragment_smiles_for_atoms( + product_mol, atoms, clear_atom_maps=True + ), + "parent_product_components": parent_ids, + "is_whole_product_component": is_whole_component, + "touches_selected_mapping": bool(adjacent_sources), + "boundary_bonds_to_nonresidual_atoms": boundary, + "adjacent_selected_lineages": adjacent_sources, + } + ) + + product_residuals.sort(key=lambda e: (-e["size"], e["classification"], e["atoms"])) + for i, entry in enumerate(product_residuals): + entry["residual_id"] = i + + byproduct_candidates = [ + e + for e in product_residuals + if e["classification"] == "whole_uncovered_product_component" + ] + partial_product_residuals = [ + e + for e in product_residuals + if e["classification"] == "partial_uncovered_product_fragment" + ] + + active_lineages = sorted(atoms_by_lineage) + active_reactants = {rid for rid, _k in active_lineages} + reactant_residuals: List[Dict[str, Any]] = [] + + for rid, k in active_lineages: + rc = reactants[rid] + used_r = { + r for (rr, kk, r), _p in rcopy_to_product.items() if rr == rid and kk == k + } + unused_r = set(rc.graph.nodes) - used_r + for atoms in connected_atom_components(rc.graph, unused_r): + boundary = boundary_bonds_for_atom_set(rc.graph, atoms) + reactant_residuals.append( + { + "classification": "unused_fragment_in_active_reactant_copy", + "reactant_id": rid, + "copy_id": k, + "lineage": lineage_label((rid, k)), + "atoms": atoms, + "size": len(atoms), + "bond_count": edge_count_in_atom_set(rc.graph, atoms), + "smiles": fragment_smiles_for_atoms( + rc.mol, atoms, clear_atom_maps=False + ), + "unmapped_smiles": fragment_smiles_for_atoms( + rc.mol, atoms, clear_atom_maps=True + ), + "boundary_bonds_to_selected_reactant_atoms": boundary, + } + ) + + for rc in reactants: + if rc.rid in active_reactants: + continue + atoms = sorted(rc.graph.nodes) + if not atoms: + continue + reactant_residuals.append( + { + "classification": "unused_reactant_component", + "reactant_id": rc.rid, + "copy_id": None, + "lineage": f"R{rc.rid}/unused_component", + "atoms": atoms, + "size": len(atoms), + "bond_count": edge_count_in_atom_set(rc.graph, atoms), + "smiles": Chem.MolToSmiles(rc.mol, canonical=True), + "unmapped_smiles": Chem.MolToSmiles( + _copy_mol_clearing_atom_maps(rc.mol), canonical=True + ), + "boundary_bonds_to_selected_reactant_atoms": [], + } + ) + + reactant_residuals.sort( + key=lambda e: ( + -e["size"], + e["classification"], + e["reactant_id"], + -1 if e["copy_id"] is None else e["copy_id"], + e["atoms"], + ) + ) + for i, entry in enumerate(reactant_residuals): + entry["residual_id"] = i + + product_residual_smiles = ".".join( + e["unmapped_smiles"] for e in product_residuals if e["unmapped_smiles"] + ) + product_byproduct_candidate_smiles = ".".join( + e["unmapped_smiles"] for e in byproduct_candidates if e["unmapped_smiles"] + ) + reactant_residual_smiles = ".".join( + e["unmapped_smiles"] for e in reactant_residuals if e["unmapped_smiles"] + ) + + return { + "product_residual_fragments": product_residuals, + "product_byproduct_candidates": byproduct_candidates, + "product_partial_unmapped_fragments": partial_product_residuals, + "reactant_residual_fragments": reactant_residuals, + "product_residual_smiles": product_residual_smiles, + "product_byproduct_candidate_smiles": product_byproduct_candidate_smiles, + "reactant_residual_smiles": reactant_residual_smiles, + "counts": { + "product_residual_fragment_count": len(product_residuals), + "product_residual_atom_count": len(uncovered_product_atoms), + "product_byproduct_candidate_count": len(byproduct_candidates), + "product_partial_unmapped_fragment_count": len(partial_product_residuals), + "reactant_residual_fragment_count": len(reactant_residuals), + "reactant_residual_atom_count": sum( + int(e["size"]) for e in reactant_residuals + ), + "active_reactant_residual_fragment_count": sum( + 1 + for e in reactant_residuals + if e["classification"] == "unused_fragment_in_active_reactant_copy" + ), + "unused_reactant_component_count": sum( + 1 + for e in reactant_residuals + if e["classification"] == "unused_reactant_component" + ), + }, + } + + +def compute_diagnostics( + reactants: Sequence[ReactantComponent], + product_mol: Chem.Mol, + product_graph: nx.Graph, + pieces: Sequence[SelectedPiece], + cfg: ReactionAtomMapperConfig, +) -> Dict[str, Any]: + product_source, rcopy_to_product, product_to_rcopy_atom, atoms_by_lineage = ( + build_mapping_indexes(pieces, product_graph) + ) + + covered_product_atoms = set(product_source) + uncovered_product_atoms = sorted(set(product_graph.nodes) - covered_product_atoms) + + active_copies_counter = Counter((p.reactant_id, p.copy_id) for p in pieces) + active_copies_by_reactant: Dict[str, int] = defaultdict(int) + pieces_by_lineage: Dict[Tuple[int, int], List[SelectedPiece]] = defaultdict(list) + for p in pieces: + active_copies_by_reactant[str(p.reactant_id)] = max( + active_copies_by_reactant[str(p.reactant_id)], p.copy_id + 1 + ) + pieces_by_lineage[(p.reactant_id, p.copy_id)].append(p) + + # Unused reactant atoms by active copy. For inactive copies, every atom is + # unused by definition, but we usually care about active lineages. + unused_reactant_atoms_active: Dict[str, List[int]] = {} + for lin in sorted(atoms_by_lineage): + rid, k = lin + used_r = { + r for (rr, kk, r), p in rcopy_to_product.items() if rr == rid and kk == k + } + all_r = set(reactants[rid].graph.nodes) + unused_reactant_atoms_active[lineage_label(lin)] = sorted(all_r - used_r) + + # Lineage split: same lineage product atoms induce multiple connected blocks. + lineage_split_events: List[Dict[str, Any]] = [] + for lin, p_atoms in sorted(atoms_by_lineage.items()): + if not p_atoms: + continue + sub = product_graph.subgraph(p_atoms) + comps = ( + [sorted(c) for c in nx.connected_components(sub)] + if sub.number_of_nodes() + else [] + ) + if len(comps) > 1: + lineage_split_events.append( + { + "lineage": lineage_label(lin), + "num_product_blocks": len(comps), + "extra_blocks": len(comps) - 1, + "blocks": comps, + "piece_count_for_lineage": len(pieces_by_lineage.get(lin, [])), + } + ) + + # Reactant bond preservation/breakage/deletion. + reactant_bond_events: List[Dict[str, Any]] = [] + for lin in sorted(atoms_by_lineage): + rid, k = lin + rc = reactants[rid] + for ru, rv, rdata in rc.graph.edges(data=True): + key_u = (rid, k, ru) + key_v = (rid, k, rv) + mu = rcopy_to_product.get(key_u) + mv = rcopy_to_product.get(key_v) + if mu is None or mv is None: + reactant_bond_events.append( + { + "event": "reactant_bond_deleted_or_unmapped", + "lineage": lineage_label(lin), + "reactant_bond": [ru, rv], + "mapped_product_atoms": [mu, mv], + } + ) + elif product_graph.has_edge(mu, mv): + compatible = bond_compatible_attrs( + rdata, product_graph.edges[mu, mv], cfg + ) + reactant_bond_events.append( + { + "event": ( + "reactant_bond_preserved" + if compatible + else "reactant_bond_order_changed" + ), + "lineage": lineage_label(lin), + "reactant_bond": [ru, rv], + "product_bond": [mu, mv], + } + ) + else: + reactant_bond_events.append( + { + "event": "reactant_bond_broken", + "lineage": lineage_label(lin), + "reactant_bond": [ru, rv], + "mapped_product_atoms": [mu, mv], + } + ) + + # Product bond provenance. + product_bond_events: List[Dict[str, Any]] = [] + for pu, pv, pdata in product_graph.edges(data=True): + su = product_source.get(pu) + sv = product_source.get(pv) + if su is None or sv is None: + product_bond_events.append( + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [pu, pv], + "source_1": source_to_jsonable( + su if su is not None else "uncovered" + ), + "source_2": source_to_jsonable( + sv if sv is not None else "uncovered" + ), + } + ) + elif su != sv: + product_bond_events.append( + { + "event": "interlineage_product_bond_formed", + "product_bond": [pu, pv], + "source_1": source_to_jsonable(su), + "source_2": source_to_jsonable(sv), + } + ) + else: + rid, k = su + ru = product_to_rcopy_atom.get((rid, k, pu)) + rv = product_to_rcopy_atom.get((rid, k, pv)) + if ru is None or rv is None: + continue + if reactants[rid].graph.has_edge(ru, rv): + compatible = bond_compatible_attrs( + reactants[rid].graph.edges[ru, rv], pdata, cfg + ) + product_bond_events.append( + { + "event": ( + "product_bond_explained_by_reactant_bond" + if compatible + else "product_bond_order_changed_from_reactant" + ), + "product_bond": [pu, pv], + "lineage": lineage_label(su), + "reactant_bond": [ru, rv], + } + ) + else: + product_bond_events.append( + { + "event": "intralineage_product_bond_formed", + "product_bond": [pu, pv], + "lineage": lineage_label(su), + "reactant_atoms": [ru, rv], + } + ) + + # Segment / path diagnostics. + segment_events: Dict[str, List[Dict[str, Any]]] = { + "segments": [], + "lineage_restricted_breaks": [], + "foreign_or_unknown_bridged_breaks": [], + "stretches": [], + "contractions": [], + } + for lin in sorted(atoms_by_lineage): + rid, k = lin + rc = reactants[rid] + mapped_r_atoms = { + r for (rr, kk, r), p in rcopy_to_product.items() if rr == rid and kk == k + } + if len(mapped_r_atoms) < 2: + continue + segments = mapped_anchor_segments( + rc.graph, mapped_r_atoms, cfg.max_segment_distance + ) + same_lineage_product_atoms = atoms_by_lineage[lin] + same_subgraph = product_graph.subgraph(same_lineage_product_atoms).copy() + for seg in segments: + ru, rv = seg["reactant_atoms"] + pu = rcopy_to_product[(rid, k, ru)] + pv = rcopy_to_product[(rid, k, rv)] + d_full, path_full = safe_shortest_path(product_graph, pu, pv) + d_same, path_same = safe_shortest_path(same_subgraph, pu, pv) + event = { + "lineage": lineage_label(lin), + "reactant_atoms": [ru, rv], + "product_atoms": [pu, pv], + "reactant_distance": seg["reactant_distance"], + "reactant_path": seg["reactant_path"], + "product_distance_full": None if math.isinf(d_full) else int(d_full), + "product_path_full": path_full, + "product_distance_same_lineage": ( + None if math.isinf(d_same) else int(d_same) + ), + "product_path_same_lineage": path_same, + } + segment_events["segments"].append(event) + if math.isinf(d_same): + segment_events["lineage_restricted_breaks"].append(event) + if not math.isinf(d_full): + source_sequence = [ + product_source.get(a, "uncovered") for a in path_full + ] + bridge_sources = [s for s in source_sequence[1:-1] if s != lin] + bridged_event = dict(event) + bridged_event.update( + { + "source_sequence": [ + source_to_jsonable(s) for s in source_sequence + ], + "collapsed_source_sequence": [ + source_to_jsonable(s) + for s in collapse_consecutive(source_sequence) + ], + "bridge_sources": sorted( + {source_to_jsonable(s) for s in bridge_sources} + ), + "foreign_or_unknown_bridge_atom_count": len(bridge_sources), + } + ) + segment_events["foreign_or_unknown_bridged_breaks"].append( + bridged_event + ) + d_r = int(seg["reactant_distance"]) + if not math.isinf(d_full): + if d_full > d_r: + stretch_event = dict(event) + stretch_event["stretch"] = int(d_full - d_r) + segment_events["stretches"].append(stretch_event) + elif d_full < d_r: + contract_event = dict(event) + contract_event["contraction"] = int(d_r - d_full) + segment_events["contractions"].append(contract_event) + + blocks, q_edges = product_lineage_blocks(product_graph, product_source) + residuals = compute_residual_fragments( + reactants, product_mol, product_graph, pieces + ) + + # Counts for easy logging/filtering. + rb_counts = Counter(e["event"] for e in reactant_bond_events) + pb_counts = Counter(e["event"] for e in product_bond_events) + topology_counts = { + "selected_piece_count": len(pieces), + "active_lineage_count": len(atoms_by_lineage), + "covered_product_atom_count": len(covered_product_atoms), + "uncovered_product_atom_count": len(uncovered_product_atoms), + "lineage_split_event_count": len(lineage_split_events), + "lineage_extra_block_count": sum( + e["extra_blocks"] for e in lineage_split_events + ), + "reactant_bond_preserved_count": rb_counts.get("reactant_bond_preserved", 0), + "reactant_bond_broken_count": rb_counts.get("reactant_bond_broken", 0), + "reactant_bond_deleted_or_unmapped_count": rb_counts.get( + "reactant_bond_deleted_or_unmapped", 0 + ), + "interlineage_product_bond_formed_count": pb_counts.get( + "interlineage_product_bond_formed", 0 + ), + "intralineage_product_bond_formed_count": pb_counts.get( + "intralineage_product_bond_formed", 0 + ), + "product_bond_touches_uncovered_atom_count": pb_counts.get( + "product_bond_touches_uncovered_atom", 0 + ), + "lineage_restricted_break_count": len( + segment_events["lineage_restricted_breaks"] + ), + "foreign_or_unknown_bridged_break_count": len( + segment_events["foreign_or_unknown_bridged_breaks"] + ), + "foreign_or_unknown_bridge_atom_count": sum( + e.get("foreign_or_unknown_bridge_atom_count", 0) + for e in segment_events["foreign_or_unknown_bridged_breaks"] + ), + "segment_stretch_count": len(segment_events["stretches"]), + "segment_stretch_total": sum( + e.get("stretch", 0) for e in segment_events["stretches"] + ), + "segment_contraction_count": len(segment_events["contractions"]), + "segment_contraction_total": sum( + e.get("contraction", 0) for e in segment_events["contractions"] + ), + "product_residual_fragment_count": residuals["counts"][ + "product_residual_fragment_count" + ], + "product_residual_atom_count": residuals["counts"][ + "product_residual_atom_count" + ], + "product_byproduct_candidate_count": residuals["counts"][ + "product_byproduct_candidate_count" + ], + "product_partial_unmapped_fragment_count": residuals["counts"][ + "product_partial_unmapped_fragment_count" + ], + "reactant_residual_fragment_count": residuals["counts"][ + "reactant_residual_fragment_count" + ], + "reactant_residual_atom_count": residuals["counts"][ + "reactant_residual_atom_count" + ], + } + + return { + "active_copies_by_reactant": dict(active_copies_by_reactant), + "selected_piece_count_by_lineage": { + lineage_label(k): len(v) for k, v in sorted(pieces_by_lineage.items()) + }, + "uncovered_product_atoms": uncovered_product_atoms, + "unused_reactant_atoms_by_active_lineage": unused_reactant_atoms_active, + "lineage_split_events": lineage_split_events, + "reactant_bond_events": reactant_bond_events, + "product_bond_events": product_bond_events, + "segment_events": segment_events, + "product_lineage_quotient": {"blocks": blocks, "edges": q_edges}, + "residuals": residuals, + "topology_counts": topology_counts, + } + + +# --------------------------------------------------------------------------- +# Public API +# --------------------------------------------------------------------------- + + +def subtractive_map_reaction( + reaction_smiles: str, + config: Optional[ReactionAtomMapperConfig] = None, + **config_overrides: Any, +) -> SubtractiveMappingResult: + """Run subtractive common-subgraph mapping and topology diagnostics. + + Parameters + ---------- + reaction_smiles: + Reaction SMILES, either reactants>>products or reactants>agents>products. + config: + Optional MapperConfig. Keyword overrides can also be supplied. + + Returns + ------- + SubtractiveMappingResult + """ + cfg = config or ReactionAtomMapperConfig() + if config_overrides: + cfg = dataclasses.replace(cfg, **config_overrides) + if cfg.max_copies < 1: + raise ValueError("max_copies must be at least 1.") + left, right = parse_reaction_smiles_to_halves(reaction_smiles) + reactants = split_reactant_components(left) + product_mol = mol_from_side(right) + product_graph = mol_to_nx(product_mol) + respect_maps = auto_respect_atom_maps(reactants, product_mol, cfg) + + base_candidates = generate_base_candidates( + reactants, product_mol, product_graph, cfg, respect_maps + ) + + if cfg.selector == "ilp": + chosen, objective, status = select_candidates_ilp( + base_candidates, reactants, product_graph, cfg + ) + elif cfg.selector == "greedy": + chosen, objective, status = select_candidates_greedy( + base_candidates, reactants, cfg + ) + else: + raise ValueError("selector must be 'ilp' or 'greedy'.") + + pieces = selected_pieces_from_expanded(chosen) + diagnostics = compute_diagnostics( + reactants, product_mol, product_graph, pieces, cfg + ) + diagnostics["candidate_generation"] = { + "base_candidate_count": len(base_candidates), + "expanded_candidate_count": len( + expand_candidates(base_candidates, cfg, reactants) + ), + "respect_atom_maps": respect_maps, + } + return SubtractiveMappingResult( + reaction_smiles=reaction_smiles, + selector=cfg.selector, + objective_value=objective, + status=status, + reactant_components=list(reactants), + product_mol=product_mol, + product_graph=product_graph, + selected_pieces=pieces, + diagnostics=diagnostics, + config=cfg, + ) + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + + +def _parse_bool_auto(value: str) -> Optional[bool]: + v = value.lower().strip() + if v in {"auto", "none"}: + return None + if v in {"1", "true", "yes", "y"}: + return True + if v in {"0", "false", "no", "n"}: + return False + raise argparse.ArgumentTypeError("Expected auto, true, or false.") + + +def build_arg_parser() -> argparse.ArgumentParser: + p = argparse.ArgumentParser( + description="Subtractive common-subgraph reaction mapper." + ) + p.add_argument("reaction", help="Reaction SMILES, e.g. 'CC.CNC>>CCNCC'.") + p.add_argument("--selector", choices=["ilp", "greedy"], default="ilp") + p.add_argument("--max-copies", type=int, default=3) + p.add_argument("--min-fragment-atoms", type=int, default=1) + p.add_argument("--max-fragment-atoms", type=int, default=8) + p.add_argument("--max-fragments-per-reactant", type=int, default=2500) + p.add_argument("--max-matches-per-fragment", type=int, default=128) + p.add_argument("--max-base-candidates-per-reactant", type=int, default=6000) + p.add_argument( + "--respect-atom-maps", + type=_parse_bool_auto, + default=None, + help="auto, true, or false; default auto", + ) + p.add_argument( + "--no-rdkit-mcs", action="store_true", help="Disable RDKit MCS seed candidates." + ) + p.add_argument( + "--allow-mapped-reactant-copies", + action="store_true", + help="Allow reactant components containing atom-map anchors to use multiple virtual copies.", + ) + p.add_argument( + "--unused-reactant-atom-penalty", + type=float, + default=6.0, + help="Penalty per unused atom in each active reactant copy.", + ) + p.add_argument( + "--broken-bond-environment-penalty", + type=float, + default=1.0, + help="Scale for local-environment penalties on broken reactant bonds.", + ) + p.add_argument( + "--bond-environment-objective", + choices=["off", "integrated", "rerank"], + default="off", + help="How to use local bond-environment penalties in ILP selection.", + ) + p.add_argument( + "--bond-environment-rank-tolerance", + type=float, + default=1.0e-6, + help="Primary-objective tolerance for reranking near-tied ILP solutions.", + ) + p.add_argument( + "--stable-single-bond-break-penalty", + type=float, + default=1.0, + help="Extra penalty for breaking saturated-carbon/hetero single bonds.", + ) + p.add_argument( + "--unsaturated-endpoint-break-credit", + type=float, + default=0.75, + help="Credit for breaking bonds attached to atoms with multiple bonds to hetero atoms.", + ) + p.add_argument( + "--ring-bond-break-penalty", + type=float, + default=2.0, + help="Extra penalty for breaking ring bonds.", + ) + p.add_argument( + "--max-broken-bond-pair-penalty-terms", + type=int, + default=25000, + help="Maximum pairwise broken-bond ILP terms to keep; 0 means no cap.", + ) + p.add_argument("--ignore-bond-order", action="store_true") + p.add_argument("--json-indent", type=int, default=2) + p.add_argument( + "--summary", + action="store_true", + help="Print a concise summary instead of full JSON.", + ) + return p + + +def main(argv: Optional[Sequence[str]] = None) -> int: + args = build_arg_parser().parse_args(argv) + cfg = ReactionAtomMapperConfig( + selector=args.selector, + max_copies=args.max_copies, + min_fragment_atoms=args.min_fragment_atoms, + max_fragment_atoms=args.max_fragment_atoms, + max_fragments_per_reactant=args.max_fragments_per_reactant, + max_matches_per_fragment=args.max_matches_per_fragment, + max_base_candidates_per_reactant=args.max_base_candidates_per_reactant, + respect_atom_maps=args.respect_atom_maps, + include_rdkit_mcs_candidates=not args.no_rdkit_mcs, + compare_bond_order=not args.ignore_bond_order, + mapped_reactants_single_copy=not args.allow_mapped_reactant_copies, + unused_reactant_atom_penalty_active_copy=args.unused_reactant_atom_penalty, + broken_bond_environment_penalty=args.broken_bond_environment_penalty, + bond_environment_objective=args.bond_environment_objective, + bond_environment_rank_tolerance=args.bond_environment_rank_tolerance, + stable_single_bond_break_penalty=args.stable_single_bond_break_penalty, + unsaturated_endpoint_break_credit=args.unsaturated_endpoint_break_credit, + ring_bond_break_penalty=args.ring_bond_break_penalty, + max_broken_bond_pair_penalty_terms=args.max_broken_bond_pair_penalty_terms, + ) + result = subtractive_map_reaction(args.reaction, cfg) + if args.summary: + print( + json.dumps( + { + "reaction_smiles": result.reaction_smiles, + "atom_mapped_reaction_smiles": result.atom_mapped_reaction_smiles(), + "status": result.status, + "objective_value": result.objective_value, + "selected_pieces": [ + { + "lineage": lineage_label((p.reactant_id, p.copy_id)), + "reactant_atoms": list(p.reactant_atoms), + "product_atoms": list(p.product_atoms), + "source": p.source, + } + for p in result.selected_pieces + ], + "topology_counts": result.diagnostics["topology_counts"], + "residual_summary": { + "product_residual_smiles": result.diagnostics["residuals"][ + "product_residual_smiles" + ], + "product_byproduct_candidate_smiles": result.diagnostics[ + "residuals" + ]["product_byproduct_candidate_smiles"], + "reactant_residual_smiles": result.diagnostics["residuals"][ + "reactant_residual_smiles" + ], + "counts": result.diagnostics["residuals"]["counts"], + }, + "product_residual_fragments": [ + { + "classification": f["classification"], + "smiles": f["smiles"], + "unmapped_smiles": f["unmapped_smiles"], + "atoms": f["atoms"], + "size": f["size"], + "touches_selected_mapping": f["touches_selected_mapping"], + } + for f in result.diagnostics["residuals"][ + "product_residual_fragments" + ] + ], + "reactant_residual_fragments": [ + { + "classification": f["classification"], + "lineage": f["lineage"], + "smiles": f["smiles"], + "unmapped_smiles": f["unmapped_smiles"], + "atoms": f["atoms"], + "size": f["size"], + } + for f in result.diagnostics["residuals"][ + "reactant_residual_fragments" + ] + ], + "candidate_generation": result.diagnostics["candidate_generation"], + }, + indent=args.json_indent, + sort_keys=True, + ) + ) + else: + print(result.to_json(indent=args.json_indent)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/flask_tools/pipette/java/PipetteAtomMapperCli.java b/flask_tools/pipette/java/PipetteAtomMapperCli.java new file mode 100644 index 0000000..d042454 --- /dev/null +++ b/flask_tools/pipette/java/PipetteAtomMapperCli.java @@ -0,0 +1,228 @@ +package flask_tools.pipette.java; + +import com.bioinceptionlabs.reactionblast.api.RDT; +import com.bioinceptionlabs.reactionblast.api.ReactionResult; +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStreamReader; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; + +/** + * Batch-friendly CLI for mapping reaction SMILES from stdin or a file. + * + *

Each non-empty input line must be a single reaction SMILES containing + * {@code >>}. The output is one JSON object per line, preserving input order. + */ +public final class PipetteAtomMapperCli { + + private PipetteAtomMapperCli() {} + + public static void main(String[] args) throws Exception { + Options options = Options.parse(args); + if (options.showHelp) { + printUsage(); + return; + } + + List reactions = readReactions(options); + int emitted = 0; + for (String rawReaction : reactions) { + String reaction = rawReaction == null ? "" : rawReaction.trim(); + if (reaction.isEmpty()) { + continue; + } + + try { + ReactionResult result = RDT.map(reaction); + String mappedSmiles = result.getMappedSmiles(); + if (mappedSmiles == null || mappedSmiles.isBlank()) { + throw new IllegalStateException("RDT returned no mapped SMILES"); + } + System.out.println( + buildRecord( + emitted, + reaction, + mappedSmiles, + result.getAlgorithm(), + result.getFormedCleavedCount(), + result.getOrderChangeCount(), + result.getStereoChangeCount(), + null + ) + ); + } catch (Exception exc) { + System.out.println( + buildRecord( + emitted, + reaction, + null, + null, + null, + null, + null, + rootCauseMessage(exc) + ) + ); + } + emitted++; + } + } + + private static List readReactions(Options options) throws IOException { + if (!options.positionalReactions.isEmpty()) { + return options.positionalReactions; + } + + if (options.inputPath != null) { + return Files.readAllLines(options.inputPath, StandardCharsets.UTF_8); + } + + List reactions = new ArrayList<>(); + try (BufferedReader reader = new BufferedReader( + new InputStreamReader(System.in, StandardCharsets.UTF_8))) { + String line; + while ((line = reader.readLine()) != null) { + reactions.add(line); + } + } + return reactions; + } + + private static String rootCauseMessage(Throwable throwable) { + Throwable current = throwable; + while (current.getCause() != null) { + current = current.getCause(); + } + String message = current.getMessage(); + if (message == null || message.isBlank()) { + message = current.toString(); + } + return current.getClass().getSimpleName() + ": " + message; + } + + private static String buildRecord( + int index, + String inputSmiles, + String mappedSmiles, + String algorithm, + Integer formedCleavedCount, + Integer orderChangeCount, + Integer stereoChangeCount, + String error + ) { + return "{" + + "\"index\":" + index + "," + + "\"input_smiles\":" + jsonString(inputSmiles) + "," + + "\"mapped_smiles\":" + jsonString(mappedSmiles) + "," + + "\"algorithm\":" + jsonString(algorithm) + "," + + "\"formed_cleaved_count\":" + jsonNumber(formedCleavedCount) + "," + + "\"order_change_count\":" + jsonNumber(orderChangeCount) + "," + + "\"stereo_change_count\":" + jsonNumber(stereoChangeCount) + "," + + "\"error\":" + jsonString(error) + + "}"; + } + + private static String jsonNumber(Integer value) { + return value == null ? "null" : Integer.toString(value); + } + + private static String jsonString(String value) { + if (value == null) { + return "null"; + } + + StringBuilder builder = new StringBuilder(); + builder.append('"'); + for (int i = 0; i < value.length(); i++) { + char c = value.charAt(i); + switch (c) { + case '"': + builder.append("\\\""); + break; + case '\\': + builder.append("\\\\"); + break; + case '\b': + builder.append("\\b"); + break; + case '\f': + builder.append("\\f"); + break; + case '\n': + builder.append("\\n"); + break; + case '\r': + builder.append("\\r"); + break; + case '\t': + builder.append("\\t"); + break; + default: + if (c < 0x20) { + builder.append(String.format("\\u%04x", (int) c)); + } else { + builder.append(c); + } + } + } + builder.append('"'); + return builder.toString(); + } + + private static void printUsage() { + System.err.println("Usage: java -cp : " + + "flask_tools.pipette.java.PipetteAtomMapperCli " + + "[--input reactions.txt] [reaction ...]"); + System.err.println( + "Reads one reaction SMILES per line from stdin when no file or positional reactions are provided." + ); + } + + private static final class Options { + private final Path inputPath; + private final boolean showHelp; + private final List positionalReactions; + + private Options( + Path inputPath, + boolean showHelp, + List positionalReactions + ) { + this.inputPath = inputPath; + this.showHelp = showHelp; + this.positionalReactions = positionalReactions; + } + + private static Options parse(String[] args) { + Path inputPath = null; + boolean showHelp = false; + List positional = new ArrayList<>(); + + for (int i = 0; i < args.length; i++) { + String arg = args[i]; + switch (arg) { + case "-h": + case "--help": + showHelp = true; + break; + case "-i": + case "--input": + if (i + 1 >= args.length) { + throw new IllegalArgumentException("--input requires a file path"); + } + inputPath = Path.of(args[++i]); + break; + default: + positional.add(arg); + break; + } + } + + return new Options(inputPath, showHelp, positional); + } + } +} diff --git a/flask_tools/pipette/judge.py b/flask_tools/pipette/judge.py index b2077e3..43b97f3 100644 --- a/flask_tools/pipette/judge.py +++ b/flask_tools/pipette/judge.py @@ -43,6 +43,7 @@ def __init__( api_key: str, prompt_path: Path, prompt: str | None = None, + enable_atom_mapping_dict_in_prompt: bool = False, ) -> None: self.url = url self.model = model @@ -50,6 +51,7 @@ def __init__( self.api_key = api_key self.prompt_path = prompt_path self._prompt = prompt + self.enable_atom_mapping_dict_in_prompt = enable_atom_mapping_dict_in_prompt @classmethod def from_config(cls: type[_JudgeT], config: PipetteConfig) -> _JudgeT: @@ -67,6 +69,9 @@ def from_config(cls: type[_JudgeT], config: PipetteConfig) -> _JudgeT: api_key=api_key, prompt_path=config.llm_judge.prompt_path, prompt=config.llm_judge.prompt, + enable_atom_mapping_dict_in_prompt=( + config.llm_judge.enable_atom_mapping_dict_in_prompt + ), ) @property @@ -75,16 +80,24 @@ def system_prompt(self) -> str: return self._prompt return self.prompt_path.read_text(encoding="utf-8") - @staticmethod def _build_user_payload( - rxn_smiles: str, results: list[ToolResult] + self, + rxn_smiles: str, + results: list[ToolResult], ) -> dict[str, Any]: serialized_results: list[dict[str, object]] = [ r.model_dump(exclude_none=True) for r in results ] - for s in serialized_results: + for result, serialized_result in zip(results, serialized_results): + s = serialized_result if "skipped_reason" in s: del s["skipped_reason"] + if ( + not self.enable_atom_mapping_dict_in_prompt + and result.name == "llm_atom_mapping" + and isinstance(s.get("data"), dict) + ): + s["data"].pop("product_to_reactant", None) user_payload = { "reaction_smiles": rxn_smiles, "tool_results": serialized_results, diff --git a/flask_tools/pipette/llm_query.py b/flask_tools/pipette/llm_query.py index 7348bb7..69be1af 100644 --- a/flask_tools/pipette/llm_query.py +++ b/flask_tools/pipette/llm_query.py @@ -10,7 +10,7 @@ import asyncio import os import threading -from typing import TYPE_CHECKING, Awaitable, Literal +from typing import TYPE_CHECKING, Awaitable, Literal, Any from urllib.parse import urlsplit, urlunsplit from charge.clients.agentframework import AgentFrameworkBackend @@ -152,7 +152,7 @@ async def query_task_async( return str(result) -def _run_coroutine_sync(coro: Awaitable[str]) -> str: +def _run_coroutine_sync(coro: Awaitable) -> Any: try: asyncio.get_running_loop() except RuntimeError: diff --git a/flask_tools/pipette/pipeline.py b/flask_tools/pipette/pipeline.py index d96264a..df5649d 100644 --- a/flask_tools/pipette/pipeline.py +++ b/flask_tools/pipette/pipeline.py @@ -11,6 +11,11 @@ import traceback from .config import PipetteConfig +from .graph_rxn_mapper.subtractive_reaction_mapper_pipette_tool import ( + GraphBasedBalancer, + GraphBasedBalancerResultDetails, + RDTAtomMapper, +) from .verifiers import ( BasicSmilesValidationChecker, ChargeConservationChecker, @@ -22,7 +27,11 @@ from .verifiers.base import CacheableReactionChecker from .judge import AsyncLLMJudge from .constants import ReactionGrade, ToolResult, ToolStatus, ToolResultsDict -from .reaction_fixer import AsyncLLMReactionFixer, ReactionFixResultDetails +from .reaction_fixer import ( + AsyncLLMReactionFixer, + ReactionFixResultDetails, + BaseLLMReactionFixer, +) from .smiles import canonicalize_reaction_smiles from .llm_query import _run_coroutine_sync @@ -153,55 +162,6 @@ def _with_prefix_results( comment=result.comment, ) - @staticmethod - def _build_fix_result(fix: ReactionFixResultDetails) -> ToolResult: - return ToolResult( - name="llm_reaction_fix", - status=ToolStatus.PASS, - data=fix, - comment="LLM proposed a corrected reaction", - ) - - async def _attempt_llm_fix_async( - self, - rxn_smiles: str, - tool_results: ToolResultsDict, - ) -> tuple[ToolResult, str | None] | None: - if self.reaction_fixer is None: - return None - - try: - fix_result = self.reaction_fixer.fix( - rxn_smiles, - list(tool_results.values()), - ) - fix = await fix_result if inspect.isawaitable(fix_result) else fix_result - except Exception as exc: - return ( - ToolResult( - name="llm_reaction_fix", - status=ToolStatus.ERROR, - data=None, - comment=f"LLM reaction fixer failed: {exc}", - ), - None, - ) - - if fix.fixed_reaction_smiles == canonicalize_reaction_smiles( - rxn_smiles, include_agents=True - ): - return ( - ToolResult( - name="llm_reaction_fix", - status=ToolStatus.UNKNOWN, - data=fix, - comment="LLM reaction fixer did not propose a changed reaction.", - ), - None, - ) - - return self._build_fix_result(fix), fix.fixed_reaction_smiles - async def _finalize_grade_async( self, rxn_smiles: str, @@ -230,6 +190,8 @@ async def grade_one_async( fix_attempted: bool = False, previous_tool_results: ToolResultsDict | None = None, ) -> ReactionGrade: + run_at_most_once = [GraphBasedBalancer.name] + async def maybe_call_fixer() -> ReactionGrade | None: if not (self.config.settings.use_fixing and not fix_attempted): # and self._should_try_llm_fix(context) @@ -237,7 +199,8 @@ async def maybe_call_fixer() -> ReactionGrade | None: if (rxn_smiles, "llm_reaction_fix") in previous_tool_results: raise RuntimeError("llm_reaction_fix already ran") - fix_attempt = await self._attempt_llm_fix_async( + fix_attempt = await BaseLLMReactionFixer.attempt_llm_fix_async( + self.reaction_fixer, rxn_smiles, all_tool_results, ) @@ -252,14 +215,37 @@ async def maybe_call_fixer() -> ReactionGrade | None: ) return None + def is_tool_to_call_fixer_after(checker_name: str) -> bool: + if checker_name in (ExactMatchChecker.name, GraphBasedBalancer.name): + if ( + GraphBasedBalancer.name in checker_names + and checker_name == ExactMatchChecker.name + ): + # Two natural points to call the LLM balancer at. Only call one. + return False + return True + return False + previous_tool_results = previous_tool_results or {} + previous_tool_names = [ + checker_name for (_rxn_smi, checker_name) in previous_tool_results.keys() + ] all_tool_results: ToolResultsDict = previous_tool_results.copy() should_skip_remaining = False skip_reason = "" + checker_names = [c.name for c in self.checkers] + for checker in self.checkers: if (rxn_smiles, checker.name) in previous_tool_results: continue + if checker.name in run_at_most_once and checker.name in previous_tool_names: + # Run fixer for steps that + if is_tool_to_call_fixer_after(checker.name): + none_or_fixed_and_graded = await maybe_call_fixer() + if none_or_fixed_and_graded is not None: + return none_or_fixed_and_graded + continue if should_skip_remaining: result = checker.skipped(skip_reason) else: @@ -269,7 +255,7 @@ async def maybe_call_fixer() -> ReactionGrade | None: if is_cacheable: result = checker.check_cache(rxn_smiles) # noqa if result is None: - result = checker.run(rxn_smiles, all_tool_results) + result = await checker.arun(rxn_smiles, all_tool_results) except Exception as exc: result = checker.errored( f"{checker.name} raised an unexpected error: {exc}", @@ -283,10 +269,26 @@ async def maybe_call_fixer() -> ReactionGrade | None: # prefix.append((rxn_smiles, checker.name, result)) all_tool_results[(rxn_smiles, checker.name)] = result - # Reaction fixing / infilling of byproducts - # If LLM fixing returned a value, call new grade_one with new rxn and + # Actions that change the smiles, calling grade_one_async again. + # IE, balancing / fixing. + # If fixing/balancing returned a value, call new grade_one with new rxn and # it's main tool result list will start from the new rxn - if checker.name == "exact_match": + + # Graph based balancer + if checker.name == GraphBasedBalancer.name: + result: ToolResult + if result.status == ToolStatus.PASS: + d: GraphBasedBalancerResultDetails = result.data + if ( + d.original_reaction_smiles != d.graph_balanced_reaction_smiles + ): # Does this need canonicalization? + return await self.grade_one_async( + d.graph_balanced_reaction_smiles, + fix_attempted=False, + previous_tool_results=all_tool_results, + ) + # LLM fixer + if is_tool_to_call_fixer_after(checker.name): none_or_fixed_and_graded = await maybe_call_fixer() if none_or_fixed_and_graded is not None: return none_or_fixed_and_graded @@ -335,10 +337,18 @@ def build_default_pipeline( attribute of `ReactionChecker`). """ # Edit this function when adding new `ReactionChecker`s + from .graph_rxn_mapper.subtractive_reaction_mapper_pipette_tool import ( + GraphBasedBalancer, + LLMAtomMapper, + ) + config = config or PipetteConfig() possible_checker_factories = checker_factories or { "basic_smiles_validation": lambda _: BasicSmilesValidationChecker(), "exact_match": lambda _: ExactMatchChecker(), + "graph_based_balancing": lambda config: GraphBasedBalancer(config), + # "llm_atom_mapping": lambda config: LLMAtomMapper.from_config(config), + "rdt_atom_mapping": lambda _: RDTAtomMapper(), "charge_conservation": lambda _: ChargeConservationChecker(), "mass_conservation": lambda config: MassConservationChecker(config), "reaction_energy": lambda config: ReactionEnergyChecker( diff --git a/flask_tools/pipette/reaction_fixer.py b/flask_tools/pipette/reaction_fixer.py index 8d9be0c..affc0b1 100644 --- a/flask_tools/pipette/reaction_fixer.py +++ b/flask_tools/pipette/reaction_fixer.py @@ -9,6 +9,7 @@ from collections import Counter from dataclasses import dataclass +import inspect import json from pathlib import Path from typing import Any, Literal, TypeVar @@ -16,7 +17,13 @@ from pydantic import BaseModel from .config import PipetteConfig -from .constants import ToolResult, resolve_llm_api_key, ToolResultDetails +from .constants import ( + ToolStatus, + ToolResult, + ToolResultsDict, + resolve_llm_api_key, + ToolResultDetails, +) from .llm_query import query_task, query_task_async from .smiles import ( canonicalize_reaction_smiles, @@ -47,6 +54,8 @@ class ReactionFixResponse(BaseModel): class BaseLLMReactionFixer: + name = "llm_reaction_fix" + def __init__( self, *, @@ -147,6 +156,53 @@ def _parse_reaction_fix( reasoning_summary=parsed.comment, ) + @staticmethod + async def attempt_llm_fix_async( + reaction_fixer: BaseLLMReactionFixer, + rxn_smiles: str, + tool_results: ToolResultsDict, + ) -> tuple[ToolResult, str | None] | None: + if reaction_fixer is None: + return None + + try: + fix_result = reaction_fixer.fix( + rxn_smiles, + list(tool_results.values()), + ) + fix = await fix_result if inspect.isawaitable(fix_result) else fix_result + except Exception as exc: + return ( + ToolResult( + name="llm_reaction_fix", + status=ToolStatus.ERROR, + data=None, + comment=f"LLM reaction fixer failed: {exc}", + ), + None, + ) + + if fix.fixed_reaction_smiles == canonicalize_reaction_smiles( + rxn_smiles, include_agents=True + ): + return ( + ToolResult( + name="llm_reaction_fix", + status=ToolStatus.UNKNOWN, + data=fix, + comment="LLM reaction fixer did not propose a changed reaction.", + ), + None, + ) + + tool_res = ToolResult( + name="llm_reaction_fix", + status=ToolStatus.PASS, + data=fix, + comment="LLM proposed a corrected reaction", + ) + return tool_res, fix.fixed_reaction_smiles + @staticmethod def _build_user_payload( rxn_smiles: str, diff --git a/flask_tools/pipette/smiles.py b/flask_tools/pipette/smiles.py index 66c9d35..6b69e78 100644 --- a/flask_tools/pipette/smiles.py +++ b/flask_tools/pipette/smiles.py @@ -7,8 +7,6 @@ from __future__ import annotations -from typing import Any - from rdkit import Chem from rdkit.Chem.rdchem import Mol @@ -98,3 +96,43 @@ def smiles_to_inchi(smiles: str) -> str: f"Could not generate an InChIKey for SMILES component: {smiles!r}" ) return inchi + + +def mol_from_side(side: str) -> Chem.Mol: + if not side: + return Chem.Mol() + mol = Chem.MolFromSmiles(side, sanitize=True) + if mol is None: + raise ValueError(f"Could not parse reaction side: {side!r}") + return mol + + +def clear_atom_maps_from_side(side: str) -> str: + mol = mol_from_side(side) + for atom in mol.GetAtoms(): + atom.SetAtomMapNum(0) + return canonical_side_smiles(mol) + + +def clear_atom_maps_from_reaction( + reaction_smiles: str, keep_agents: bool = False +) -> str: + reactants, agents, products = split_reaction_smiles(reaction_smiles) + cleared_reactants = clear_atom_maps_from_side(reactants) + cleared_products = clear_atom_maps_from_side(products) + if keep_agents: + cleared_agents = clear_atom_maps_from_side(agents) + return f"{cleared_reactants}>{cleared_agents}>{cleared_products}" + return f"{cleared_reactants}>>{cleared_products}" + + +def canonical_side_smiles(mol: Chem.Mol) -> str: + """Canonicalize a reaction side as a sorted multiset of mapped fragments.""" + if mol.GetNumAtoms() == 0: + return "" + fragments = Chem.GetMolFrags(mol, asMols=True, sanitizeFrags=True) + smiles = [ + Chem.MolToSmiles(fragment, canonical=True, isomericSmiles=True) + for fragment in fragments + ] + return ".".join(sorted(smiles)) diff --git a/flask_tools/pipette/verifiers/base.py b/flask_tools/pipette/verifiers/base.py index 5d04914..75e6406 100644 --- a/flask_tools/pipette/verifiers/base.py +++ b/flask_tools/pipette/verifiers/base.py @@ -21,6 +21,19 @@ class ReactionChecker(ABC): def run(self, rxn_smiles: str, context: ToolResultsDict) -> ToolResult: raise NotImplementedError + async def arun(self, rxn_smiles: str, context: ToolResultsDict) -> ToolResult: + """A function to be overwritten by tools that benefit from async. Actually what's called in pipeline. + This calls the sync run method by default. + Leaving `run()` in ReactionChecker makes writing non async tools simpler, and there are more non async tools. + For async classes, you can define run like this: + ``` + def run(self, rxn_smiles: str, context: ToolResultsDict) -> ToolResult: + return _run_coroutine_sync(self.arun(rxn_smiles, context)) + ``` + See LLMAtomMapper for an example of a class that uses this. + """ + return self.run(rxn_smiles, context) + def skipped(self, reason: str) -> ToolResult: return ToolResult( name=self.name, diff --git a/flask_tools/pipette/verifiers/mass.py b/flask_tools/pipette/verifiers/mass.py index 235008f..6767e54 100644 --- a/flask_tools/pipette/verifiers/mass.py +++ b/flask_tools/pipette/verifiers/mass.py @@ -198,7 +198,7 @@ def __init__(self, config: PipetteConfig) -> None: self.missing_product_rules = load_solvent_rules(config.solvent_catalog_path) def run( - self, rxn_smiles: str, context: dict[str, ToolResult] | None = None + self, rxn_smiles: str, context: ToolResultsDict | None = None ) -> ToolResult: try: delta = element_delta(rxn_smiles, explicit_hydrogens=True) diff --git a/flask_tools/pipette/verifiers/rdt.py b/flask_tools/pipette/verifiers/rdt.py new file mode 100644 index 0000000..0db4088 --- /dev/null +++ b/flask_tools/pipette/verifiers/rdt.py @@ -0,0 +1,368 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +""" +A python wrapper to call Reaction Decoder Tool (RDT) for atom mapping, through a java wrapper script. +Will compile against the RDT jar if the pipette java wrapper is not already compiled. See env vars. +""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Sequence + +from flask_tools.pipette.smiles import split_reaction_smiles + +RDT_JAR_ENV_VAR = "PIPETTE_RDT_JAR" +RDT_REPO_ENV_VAR = "PIPETTE_RDT_REPO" +RDT_HELPER_BUILD_ENV_VAR = "PIPETTE_RDT_HELPER_BUILD_DIR" +RDT_JAVA_BIN_ENV_VAR = "PIPETTE_RDT_JAVA_BIN" +RDT_JAVAC_BIN_ENV_VAR = "PIPETTE_RDT_JAVAC_BIN" +RDT_MAIN_CLASS = "flask_tools.pipette.java.PipetteAtomMapperCli" + + +@dataclass(frozen=True) +class _PreparedReaction: + # To remove and reintroduce agents during mapping + original_smiles: str + stripped_smiles: str + agents_smiles: str + + +def _default_rdt_repo_path() -> Path: + # Almost the same lvl as the flask_tools repo, under a lib folder. A really arbitrary default. + repo_root = Path(__file__).resolve().parents[3] + return repo_root.parent / "lib" / "ReactionDecoder" + + +def _helper_source_path() -> Path: + return ( + Path(__file__).resolve().parent.with_name("java") / "PipetteAtomMapperCli.java" + ) + + +def _helper_build_dir() -> Path: + env_path = os.environ.get(RDT_HELPER_BUILD_ENV_VAR) + if env_path: + return Path(env_path).expanduser().resolve() + return Path(__file__).resolve().with_name("_java_build") + + +def _default_java_bin() -> str: + """Return the configured Java executable.""" + return os.environ.get(RDT_JAVA_BIN_ENV_VAR, "java") + + +def _default_javac_bin() -> str: + """Return the configured javac executable.""" + return os.environ.get(RDT_JAVAC_BIN_ENV_VAR, "javac") + + +JAR_GLOB: str = "target/*-jar-with-dependencies.jar" + + +def _resolve_jar_from_repo(repo_path: Path) -> Path | None: + jar_candidates = sorted(repo_path.glob(JAR_GLOB)) + if not jar_candidates: + return None + return jar_candidates[-1] + + +def resolve_rdt_jar_path( + jar_path: str | Path | None = None, + repo_path: str | Path | None = None, +) -> Path: + if jar_path is not None: + resolved = Path(jar_path).expanduser().resolve() + if not resolved.exists(): + raise FileNotFoundError(f"RDT jar does not exist: {resolved}") + return resolved + + env_jar = os.environ.get(RDT_JAR_ENV_VAR) + if env_jar: + return resolve_rdt_jar_path(env_jar) + + candidate_repos: list[Path] = [] + if repo_path is not None: + candidate_repos.append(Path(repo_path).expanduser().resolve()) + + env_repo = os.environ.get(RDT_REPO_ENV_VAR) + if env_repo: + candidate_repos.append(Path(env_repo).expanduser().resolve()) + + candidate_repos.append(_default_rdt_repo_path().resolve()) + + for candidate_repo in candidate_repos: + jar_candidate = _resolve_jar_from_repo(candidate_repo) + if jar_candidate is not None: + return jar_candidate.resolve() + + searched = ", ".join(str(path) for path in candidate_repos) + raise FileNotFoundError( + "Could not locate an RDT fat jar. Set " + f"{RDT_JAR_ENV_VAR}, pass jar_path=..., or build one in a default location ({searched}), matching the glob string {JAR_GLOB}" + ) + + +def _prepare_reaction_smiles(reaction_smiles: str) -> _PreparedReaction: + reactants, agents, products = split_reaction_smiles(reaction_smiles) + return _PreparedReaction( + original_smiles=reaction_smiles, + stripped_smiles=f"{reactants}>>{products}", + agents_smiles=agents, + ) + + +def _restore_agents(mapped_reaction_smiles: str, agents_smiles: str) -> str: + if not agents_smiles: + return mapped_reaction_smiles + reactants, _agents, products = split_reaction_smiles(mapped_reaction_smiles) + return f"{reactants}>{agents_smiles}>{products}" + + +def ensure_rdt_helper_compiled( + *, + jar_path: str | Path | None = None, + repo_path: str | Path | None = None, + javac_bin: str | None = None, + build_dir: str | Path | None = None, +) -> Path: + """Compile the local Java helper against the selected RDT jar.""" + resolved_jar = resolve_rdt_jar_path(jar_path=jar_path, repo_path=repo_path) + source_path = _helper_source_path() + javac_bin = javac_bin or _default_javac_bin() + output_dir = ( + Path(build_dir).expanduser().resolve() + if build_dir is not None + else _helper_build_dir() + ) + class_path = ( + output_dir / "flask_tools" / "pipette" / "java" / "PipetteAtomMapperCli.class" + ) + + if ( + class_path.exists() + and class_path.stat().st_mtime >= source_path.stat().st_mtime + ): + return output_dir + + output_dir.mkdir(parents=True, exist_ok=True) + proc = subprocess.run( + [ + javac_bin, + "-cp", + str(resolved_jar), + "-d", + str(output_dir), + str(source_path), + ], + text=True, + capture_output=True, + check=False, + ) + if proc.returncode != 0: + details = proc.stderr.strip() or proc.stdout.strip() or "no compiler output" + raise RuntimeError(f"Failed to compile the local RDT helper: {details}") + return output_dir + + +def map_reaction_smiles_list_with_rdt( + reaction_smiles_list: Sequence[str], + *, + jar_path: str | Path | None = None, + repo_path: str | Path | None = None, + java_bin: str | None = None, + javac_bin: str | None = None, +) -> list[str]: + """Map a batch of reaction SMILES strings with the RDT helper CLI.""" + if not reaction_smiles_list: + return [] + + prepared = [_prepare_reaction_smiles(smiles) for smiles in reaction_smiles_list] + resolved_jar = resolve_rdt_jar_path(jar_path=jar_path, repo_path=repo_path) + java_bin = java_bin or _default_java_bin() + helper_build_dir = ensure_rdt_helper_compiled( + jar_path=resolved_jar, + javac_bin=javac_bin, + ) + command = [ + java_bin, + "-cp", + os.pathsep.join([str(helper_build_dir), str(resolved_jar)]), + RDT_MAIN_CLASS, + ] + proc = subprocess.run( + command, + input="\n".join(item.stripped_smiles for item in prepared) + "\n", + text=True, + capture_output=True, + check=False, + ) + + stderr = proc.stderr.strip() + stdout_lines = [line for line in proc.stdout.splitlines() if line.strip()] + if proc.returncode != 0: + details = stderr or "no stderr output" + raise RuntimeError( + f"RDT batch process failed with exit code {proc.returncode}: {details}" + ) + + if len(stdout_lines) != len(prepared): + raise RuntimeError( + "RDT returned an unexpected number of records: " + f"expected {len(prepared)}, got {len(stdout_lines)}. stderr={stderr!r}" + ) + + mapped_smiles_list: list[str] = [] + errors: list[str] = [] + for expected_index, (line, original) in enumerate( + zip(stdout_lines, prepared, strict=True) + ): + try: + record = json.loads(line) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"RDT returned invalid JSON on line {expected_index + 1}: {line}" + ) from exc + + actual_index = record.get("index") + if actual_index != expected_index: + raise RuntimeError( + "RDT returned out-of-order records: " + f"expected index {expected_index}, got {actual_index}" + ) + + error = record.get("error") + mapped_smiles = record.get("mapped_smiles") + if error: + errors.append(f"[{expected_index}] {original.original_smiles}: {error}") + continue + if not isinstance(mapped_smiles, str) or not mapped_smiles: + errors.append( + f"[{expected_index}] {original.original_smiles}: missing mapped_smiles" + ) + continue + + mapped_smiles_list.append( + _restore_agents(mapped_smiles, original.agents_smiles) + ) + + if errors: + joined_errors = "\n".join(errors[:10]) + if len(errors) > 10: + joined_errors += f"\n... and {len(errors) - 10} more errors" + raise RuntimeError(f"RDT failed to map one or more reactions:\n{joined_errors}") + + return mapped_smiles_list + + +def map_reaction_smiles_with_rdt( + reaction_smiles: str, + *, + jar_path: str | Path | None = None, + repo_path: str | Path | None = None, + java_bin: str | None = None, + javac_bin: str | None = None, +) -> str: + """Map one reaction SMILES string with the RDT helper CLI.""" + return map_reaction_smiles_list_with_rdt( + [reaction_smiles], + jar_path=jar_path, + repo_path=repo_path, + java_bin=java_bin, + javac_bin=javac_bin, + )[0] + + +def _load_reaction_smiles_file(path: str | Path) -> list[str]: + file_path = Path(path).expanduser().resolve() + return [ + line.strip() + for line in file_path.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Map one or more reaction SMILES with RDT in a single Java batch." + ) + input_group = parser.add_mutually_exclusive_group(required=True) + input_group.add_argument( + "--rxn-smi", + nargs="+", + dest="rxn_smi", + help="One or more reaction SMILES strings.", + ) + input_group.add_argument( + "-f", + "--file", + dest="file", + help="Text file containing one reaction SMILES per line.", + ) + parser.add_argument( + "--jar-path", + help=f"Path to the RDT fat jar. Overrides {RDT_JAR_ENV_VAR}.", + ) + parser.add_argument( + "--repo-path", + help=f"Path to the RDT repository. Used to resolve target/*-jar-with-dependencies.jar. Overrides {RDT_REPO_ENV_VAR}.", + ) + parser.add_argument( + "--java-bin", + default=_default_java_bin(), + help=f"Java executable to use. Defaults to {RDT_JAVA_BIN_ENV_VAR} or 'java'.", + ) + parser.add_argument( + "--javac-bin", + default=_default_javac_bin(), + help=f"javac executable to use for compiling the local helper. Defaults to {RDT_JAVAC_BIN_ENV_VAR} or 'javac'.", + ) + parser.add_argument( + "--json", + action="store_true", + help="Print JSON output records instead of plain mapped SMILES lines.", + ) + args = parser.parse_args() + + reaction_smiles_list = args.rxn_smi or _load_reaction_smiles_file(args.file) + mapped_smiles_list = map_reaction_smiles_list_with_rdt( + reaction_smiles_list, + jar_path=args.jar_path, + repo_path=args.repo_path, + java_bin=args.java_bin, + javac_bin=args.javac_bin, + ) + + if args.json: + print( + json.dumps( + [ + { + "input_reaction_smiles": input_smiles, + "mapped_reaction_smiles": mapped_smiles, + } + for input_smiles, mapped_smiles in zip( + reaction_smiles_list, mapped_smiles_list, strict=True + ) + ], + indent=2, + ) + ) + else: + for mapped_smiles in mapped_smiles_list: + print(mapped_smiles) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/pyproject.toml b/pyproject.toml index f3e54e3..0b4a428 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -30,6 +30,8 @@ include = ["flask_tools*"] # This creates the console script [project.scripts] flask-tools-install = "flask_tools.install:main" +pipette = "flask_tools.pipette.grade_rxn:main" +pipette-rdt = "flask_tools.pipette.rdt.verifiers:main" [project.optional-dependencies] rdkit = ["rdkit>=2025.3.6"] diff --git a/tests/pipette/data/atom_map_rxns.jsonl b/tests/pipette/data/atom_map_rxns.jsonl new file mode 100644 index 0000000..115779a --- /dev/null +++ b/tests/pipette/data/atom_map_rxns.jsonl @@ -0,0 +1,5 @@ +{"id":"wallnut1","expected_output":"wallnut1.json","rxn_smiles":"O=[N+]([O-])c1cnc2no[n+]([O-])c2c1>>O=[N+]([O-])c1cnc(N2OC3C=CC2CC3)c(N2OC3C=CC2CC3)c1"} +{"id":"wallnut2-dimer-fail-balance-fail-atom-mapping","expected_output":"wallnut2.json","rxn_smiles":"Cc1cc(C)n(-c2nnc(-n3nc(C)cc3C)nn2)n1.N*N>>Cc1cc(C)n(-c2nnc(N*Nc3nnc(-n4nc(C)cc4C)nn3)nn2)n1"} +{"id":"wallnut3-dimer-atom-balanced-and-mapped","expected_output":"wallnut3.json","rxn_smiles":"Nc1nonc1-c1nnnn1O>[K+].[O-][Mn](=O)(=O)=O>On1nnnc1-c1nonc1/N=N\\c1nonc1-c1nnnn1O"} +{"id":"golden-rdf1","expected_output":"golden-rdf1.json","rxn_smiles":"C=CCC(N)(c1ccccc1)c1ccccc1.O=CC(C(=O)N1CCOCC1)c1ccccc1>>C=CCC(N=C(c1ccccc1)c1ccccc1)C(C(=O)N1CCOCC1)c1ccccc1.O"} +{"id":"golden-rdf2","expected_output":"golden-rdf2.json","rxn_smiles":"C=CC(O)C(C)N.C=CCSCC(C)=O>>C=C.CC(=O)CSC/C=C/C(O)C(C)N"} diff --git a/tests/pipette/expected_atom_map_res/golden-rdf1.json b/tests/pipette/expected_atom_map_res/golden-rdf1.json new file mode 100644 index 0000000..a100f10 --- /dev/null +++ b/tests/pipette/expected_atom_map_res/golden-rdf1.json @@ -0,0 +1,2618 @@ +{ + "atom_mapped_reaction_smiles": "[CH2:1]=[CH:2][CH2:3][C:4]([NH2:5])([c:19]1[cH:8][cH:9][cH:10][cH:11][cH:12]1)[c:30]1[cH:29][cH:28][cH:33][cH:32][cH:31]1.[CH:6]([CH:7]=[O:34])([c:13]1[cH:14][cH:15][cH:16][cH:17][cH:18]1)[C:20](=[O:21])[N:22]1[CH2:23][CH2:24][O:25][CH2:26][CH2:27]1>>[CH2:1]=[CH:2][CH2:3][CH:4]([N:5]=[C:6]([c:7]1[cH:8][cH:9][cH:10][cH:11][cH:12]1)[c:13]1[cH:14][cH:15][cH:16][cH:17][cH:18]1)[CH:19]([C:20](=[O:21])[N:22]1[CH2:23][CH2:24][O:25][CH2:26][CH2:27]1)[c:28]1[cH:29][cH:30][cH:31][cH:32][cH:33]1.[OH2:34]", + "config": { + "active_copy_penalty": 1.0, + "allow_extra_product_edges_in_candidate": true, + "atom_map_anchor_bonus": 25.0, + "atom_reward": 10.0, + "bond_environment_objective": "off", + "bond_environment_rank_tolerance": 1e-06, + "broken_bond_environment_penalty": 1.0, + "candidate_piece_penalty": 2.0, + "compare_aromaticity": false, + "compare_bond_order": true, + "compare_formal_charge": true, + "compare_isotope": false, + "extra_product_edge_penalty": 2.0, + "fallback_to_greedy": true, + "include_rdkit_mcs_candidates": true, + "mapped_reactants_single_copy": true, + "max_base_candidates_per_reactant": 6000, + "max_broken_bond_pair_penalty_terms": 25000, + "max_copies": 3, + "max_fragment_atoms": 8, + "max_fragments_per_reactant": 2500, + "max_matches_per_fragment": 128, + "max_mcs_matches": 256, + "max_segment_distance": 8, + "min_fragment_atoms": 1, + "preserved_bond_reward": 5.0, + "require_atom_map_match_when_present": true, + "respect_atom_maps": null, + "ring_bond_break_penalty": 2.0, + "selector": "ilp", + "single_atom_piece_penalty": 4.0, + "stable_single_bond_break_penalty": 1.0, + "unsaturated_endpoint_break_credit": 0.75, + "unused_reactant_atom_penalty_active_copy": 6.0 + }, + "diagnostics": { + "active_copies_by_reactant": { + "0": 1, + "1": 1 + }, + "candidate_generation": { + "base_candidate_count": 6898, + "expanded_candidate_count": 20694, + "respect_atom_maps": false + }, + "lineage_split_events": [ + { + "blocks": [ + [ + 0, + 1, + 2, + 3, + 4, + 18, + 27, + 28, + 29, + 30, + 31, + 32 + ], + [ + 7, + 8, + 9, + 10, + 11 + ] + ], + "extra_blocks": 1, + "lineage": "R0/copy0", + "num_product_blocks": 2, + "piece_count_for_lineage": 3 + }, + { + "blocks": [ + [ + 5, + 6, + 12, + 13, + 14, + 15, + 16, + 17 + ], + [ + 19, + 20, + 21, + 22, + 23, + 24, + 25, + 26 + ], + [ + 33 + ] + ], + "extra_blocks": 2, + "lineage": "R1/copy0", + "num_product_blocks": 3, + "piece_count_for_lineage": 3 + } + ], + "product_bond_events": [ + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 18 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 4, + 5 + ], + "source_1": "R0/copy0", + "source_2": "R1/copy0" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 2, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 5, + 12 + ], + "reactant_bond": [ + 2, + 11 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 6, + 7 + ], + "source_1": "R1/copy0", + "source_2": "R0/copy0" + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 6, + 11 + ], + "source_1": "R1/copy0", + "source_2": "R0/copy0" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 8, + 9 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 9, + 10 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 12, + 17 + ], + "reactant_bond": [ + 11, + 16 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 13, + 14 + ], + "reactant_bond": [ + 12, + 13 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 13, + 14 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 14, + 15 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 15, + 16 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 18, + 19 + ], + "source_1": "R0/copy0", + "source_2": "R1/copy0" + }, + { + "event": "intralineage_product_bond_formed", + "lineage": "R0/copy0", + "product_bond": [ + 18, + 27 + ], + "reactant_atoms": [ + 5, + 13 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 19, + 20 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 19, + 21 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 5, + 10 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 21, + 26 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 22, + 23 + ], + "reactant_bond": [ + 10, + 9 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 23, + 24 + ], + "reactant_bond": [ + 9, + 8 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 24, + 25 + ], + "reactant_bond": [ + 8, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 25, + 26 + ], + "reactant_bond": [ + 7, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 27, + 28 + ], + "reactant_bond": [ + 13, + 12 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 27, + 32 + ], + "reactant_bond": [ + 13, + 14 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 28, + 29 + ], + "reactant_bond": [ + 12, + 11 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 29, + 30 + ], + "reactant_bond": [ + 11, + 16 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 30, + 31 + ], + "reactant_bond": [ + 16, + 15 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 31, + 32 + ], + "reactant_bond": [ + 15, + 14 + ] + } + ], + "product_lineage_quotient": { + "blocks": [ + { + "atoms": [ + 0, + 1, + 2, + 3, + 4, + 18, + 27, + 28, + 29, + 30, + 31, + 32 + ], + "block_id": 0, + "size": 12, + "source": "R0/copy0" + }, + { + "atoms": [ + 5, + 6, + 12, + 13, + 14, + 15, + 16, + 17 + ], + "block_id": 1, + "size": 8, + "source": "R1/copy0" + }, + { + "atoms": [ + 7, + 8, + 9, + 10, + 11 + ], + "block_id": 2, + "size": 5, + "source": "R0/copy0" + }, + { + "atoms": [ + 19, + 20, + 21, + 22, + 23, + 24, + 25, + 26 + ], + "block_id": 3, + "size": 8, + "source": "R1/copy0" + }, + { + "atoms": [ + 33 + ], + "block_id": 4, + "size": 1, + "source": "R1/copy0" + } + ], + "edges": [ + { + "block_1": 0, + "block_2": 1 + }, + { + "block_1": 0, + "block_2": 3 + }, + { + "block_1": 1, + "block_2": 2 + } + ] + }, + "reactant_bond_events": [ + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 18 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 3, + 29 + ], + "reactant_bond": [ + 3, + 11 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 18, + 7 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 18, + 11 + ], + "reactant_bond": [ + 5, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 8, + 9 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 9, + 10 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 29, + 28 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 29, + 30 + ], + "reactant_bond": [ + 11, + 16 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 28, + 27 + ], + "reactant_bond": [ + 12, + 13 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 27, + 32 + ], + "reactant_bond": [ + 13, + 14 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 32, + 31 + ], + "reactant_bond": [ + 14, + 15 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 31, + 30 + ], + "reactant_bond": [ + 15, + 16 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R1/copy0", + "mapped_product_atoms": [ + 33, + 6 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 6, + 5 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R1/copy0", + "mapped_product_atoms": [ + 5, + 19 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 5, + 12 + ], + "reactant_bond": [ + 2, + 11 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 19, + 20 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 19, + 21 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 21, + 26 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 5, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 26, + 25 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 25, + 24 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 24, + 23 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 23, + 22 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 12, + 17 + ], + "reactant_bond": [ + 11, + 16 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 13, + 14 + ], + "reactant_bond": [ + 12, + 13 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 13, + 14 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 14, + 15 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 15, + 16 + ] + } + ], + "residuals": { + "counts": { + "active_reactant_residual_fragment_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "unused_reactant_component_count": 0 + }, + "product_byproduct_candidate_smiles": "", + "product_byproduct_candidates": [], + "product_partial_unmapped_fragments": [], + "product_residual_fragments": [], + "product_residual_smiles": "", + "reactant_residual_fragments": [], + "reactant_residual_smiles": "" + }, + "segment_events": { + "contractions": [], + "foreign_or_unknown_bridged_breaks": [ + { + "bridge_sources": [ + "R1/copy0" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "R1/copy0", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 2, + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 7 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0", + "R1/copy0", + "R0/copy0" + ] + }, + { + "bridge_sources": [ + "R1/copy0" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "R1/copy0", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 2, + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 11 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 11 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 10 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0", + "R1/copy0", + "R0/copy0" + ] + }, + { + "bridge_sources": [ + "R0/copy0" + ], + "collapsed_source_sequence": [ + "R1/copy0", + "R0/copy0", + "R1/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 3, + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 19 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 5, + 4, + 3, + 18, + 19 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ], + "source_sequence": [ + "R1/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0" + ] + } + ], + "lineage_restricted_breaks": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 7 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 11 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 11 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 10 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 33, + 6 + ], + "product_distance_full": null, + "product_distance_same_lineage": null, + "product_path_full": [], + "product_path_same_lineage": [], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 19 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 5, + 4, + 3, + 18, + 19 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + } + ], + "segments": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 0, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 0, + 1 + ], + "product_path_same_lineage": [ + 0, + 1 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 2 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 2 + ], + "product_path_same_lineage": [ + 1, + 2 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 2, + 3 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 2, + 3 + ], + "product_path_same_lineage": [ + 2, + 3 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 4 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 4 + ], + "product_path_same_lineage": [ + 3, + 4 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 18 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 18 + ], + "product_path_same_lineage": [ + 3, + 18 + ], + "reactant_atoms": [ + 3, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 29 + ], + "product_distance_full": 4, + "product_distance_same_lineage": 4, + "product_path_full": [ + 3, + 18, + 27, + 28, + 29 + ], + "product_path_same_lineage": [ + 3, + 18, + 27, + 28, + 29 + ], + "reactant_atoms": [ + 3, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 11 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 7 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 11 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 11 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 10 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 8 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 7, + 8 + ], + "product_path_same_lineage": [ + 7, + 8 + ], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 8, + 9 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 8, + 9 + ], + "product_path_same_lineage": [ + 8, + 9 + ], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 9, + 10 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 9, + 10 + ], + "product_path_same_lineage": [ + 9, + 10 + ], + "reactant_atoms": [ + 8, + 9 + ], + "reactant_distance": 1, + "reactant_path": [ + 8, + 9 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 10, + 11 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 11 + ], + "product_path_same_lineage": [ + 10, + 11 + ], + "reactant_atoms": [ + 9, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 10 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 29, + 28 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 29, + 28 + ], + "product_path_same_lineage": [ + 29, + 28 + ], + "reactant_atoms": [ + 11, + 12 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 12 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 29, + 30 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 29, + 30 + ], + "product_path_same_lineage": [ + 29, + 30 + ], + "reactant_atoms": [ + 11, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 16 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 28, + 27 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 28, + 27 + ], + "product_path_same_lineage": [ + 28, + 27 + ], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 27, + 32 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 27, + 32 + ], + "product_path_same_lineage": [ + 27, + 32 + ], + "reactant_atoms": [ + 13, + 14 + ], + "reactant_distance": 1, + "reactant_path": [ + 13, + 14 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 32, + 31 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 32, + 31 + ], + "product_path_same_lineage": [ + 32, + 31 + ], + "reactant_atoms": [ + 14, + 15 + ], + "reactant_distance": 1, + "reactant_path": [ + 14, + 15 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 31, + 30 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 31, + 30 + ], + "product_path_same_lineage": [ + 31, + 30 + ], + "reactant_atoms": [ + 15, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 15, + 16 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 33, + 6 + ], + "product_distance_full": null, + "product_distance_same_lineage": null, + "product_path_full": [], + "product_path_same_lineage": [], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 6, + 5 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 6, + 5 + ], + "product_path_same_lineage": [ + 6, + 5 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 19 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 5, + 4, + 3, + 18, + 19 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 12 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 12 + ], + "product_path_same_lineage": [ + 5, + 12 + ], + "reactant_atoms": [ + 2, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 11 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 19, + 20 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 19, + 20 + ], + "product_path_same_lineage": [ + 19, + 20 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 19, + 21 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 19, + 21 + ], + "product_path_same_lineage": [ + 19, + 21 + ], + "reactant_atoms": [ + 3, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 5 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 21, + 26 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 21, + 26 + ], + "product_path_same_lineage": [ + 21, + 26 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 21, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 21, + 22 + ], + "product_path_same_lineage": [ + 21, + 22 + ], + "reactant_atoms": [ + 5, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 10 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 26, + 25 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 26, + 25 + ], + "product_path_same_lineage": [ + 26, + 25 + ], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 25, + 24 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 25, + 24 + ], + "product_path_same_lineage": [ + 25, + 24 + ], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 24, + 23 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 24, + 23 + ], + "product_path_same_lineage": [ + 24, + 23 + ], + "reactant_atoms": [ + 8, + 9 + ], + "reactant_distance": 1, + "reactant_path": [ + 8, + 9 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 23, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 23, + 22 + ], + "product_path_same_lineage": [ + 23, + 22 + ], + "reactant_atoms": [ + 9, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 10 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 12, + 13 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 12, + 13 + ], + "product_path_same_lineage": [ + 12, + 13 + ], + "reactant_atoms": [ + 11, + 12 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 12 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 12, + 17 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 12, + 17 + ], + "product_path_same_lineage": [ + 12, + 17 + ], + "reactant_atoms": [ + 11, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 16 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 13, + 14 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 13, + 14 + ], + "product_path_same_lineage": [ + 13, + 14 + ], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 14, + 15 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 14, + 15 + ], + "product_path_same_lineage": [ + 14, + 15 + ], + "reactant_atoms": [ + 13, + 14 + ], + "reactant_distance": 1, + "reactant_path": [ + 13, + 14 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 15, + 16 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 15, + 16 + ], + "product_path_same_lineage": [ + 15, + 16 + ], + "reactant_atoms": [ + 14, + 15 + ], + "reactant_distance": 1, + "reactant_path": [ + 14, + 15 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 16, + 17 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 16, + 17 + ], + "product_path_same_lineage": [ + 16, + 17 + ], + "reactant_atoms": [ + 15, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 15, + 16 + ] + } + ], + "stretches": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 29 + ], + "product_distance_full": 4, + "product_distance_same_lineage": 4, + "product_path_full": [ + 3, + 18, + 27, + 28, + 29 + ], + "product_path_same_lineage": [ + 3, + 18, + 27, + 28, + 29 + ], + "reactant_atoms": [ + 3, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 11 + ], + "stretch": 3 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 7 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ], + "stretch": 4 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 11 + ], + "product_distance_full": 5, + "product_distance_same_lineage": null, + "product_path_full": [ + 18, + 3, + 4, + 5, + 6, + 11 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 5, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 10 + ], + "stretch": 4 + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 19 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 5, + 4, + 3, + 18, + 19 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ], + "stretch": 3 + } + ] + }, + "selected_piece_count_by_lineage": { + "R0/copy0": 3, + "R1/copy0": 3 + }, + "topology_counts": { + "active_lineage_count": 2, + "covered_product_atom_count": 34, + "foreign_or_unknown_bridge_atom_count": 7, + "foreign_or_unknown_bridged_break_count": 3, + "interlineage_product_bond_formed_count": 4, + "intralineage_product_bond_formed_count": 1, + "lineage_extra_block_count": 3, + "lineage_restricted_break_count": 4, + "lineage_split_event_count": 2, + "product_bond_touches_uncovered_atom_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_bond_broken_count": 5, + "reactant_bond_deleted_or_unmapped_count": 0, + "reactant_bond_preserved_count": 31, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "segment_contraction_count": 0, + "segment_contraction_total": 0, + "segment_stretch_count": 4, + "segment_stretch_total": 14, + "selected_piece_count": 6, + "uncovered_product_atom_count": 0 + }, + "uncovered_product_atoms": [], + "unused_reactant_atoms_by_active_lineage": { + "R0/copy0": [], + "R1/copy0": [] + } + }, + "objective_value": 477.0000000000002, + "product_smiles": "C=CCC(N=C(c1ccccc1)c1ccccc1)C(C(=O)N1CCOCC1)c1ccccc1.O", + "reactants": [ + { + "atom_count": 17, + "bond_count": 18, + "reactant_id": 0, + "smiles": "C=CCC(N)(c1ccccc1)c1ccccc1" + }, + { + "atom_count": 17, + "bond_count": 18, + "reactant_id": 1, + "smiles": "O=CC(C(=O)N1CCOCC1)c1ccccc1" + } + ], + "reaction_smiles": "C=CCC(N)(c1ccccc1)c1ccccc1.O=CC(C(=O)N1CCOCC1)c1ccccc1>>C=CCC(N=C(c1ccccc1)c1ccccc1)C(C(=O)N1CCOCC1)c1ccccc1.O", + "selected_pieces": [ + { + "candidate_id": 964, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 5, + "product_atoms": [ + 0, + 1, + 2, + 3, + 4, + 18 + ], + "r_to_p": { + "0": 0, + "1": 1, + "2": 2, + "3": 3, + "4": 4, + "5": 18 + }, + "reactant_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5 + ], + "reactant_id": 0, + "score": 85.0, + "source": "nx_fragment" + }, + { + "candidate_id": 956, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 27, + 28, + 29, + 30, + 31, + 32 + ], + "r_to_p": { + "11": 29, + "12": 28, + "13": 27, + "14": 32, + "15": 31, + "16": 30 + }, + "reactant_atoms": [ + 11, + 12, + 13, + 14, + 15, + 16 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + }, + { + "candidate_id": 1695, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 4, + "product_atoms": [ + 7, + 8, + 9, + 10, + 11 + ], + "r_to_p": { + "10": 11, + "6": 7, + "7": 8, + "8": 9, + "9": 10 + }, + "reactant_atoms": [ + 6, + 7, + 8, + 9, + 10 + ], + "reactant_id": 0, + "score": 70.0, + "source": "nx_fragment" + }, + { + "candidate_id": 3999, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 8, + "product_atoms": [ + 5, + 6, + 12, + 13, + 14, + 15, + 16, + 17 + ], + "r_to_p": { + "1": 6, + "11": 12, + "12": 13, + "13": 14, + "14": 15, + "15": 16, + "16": 17, + "2": 5 + }, + "reactant_atoms": [ + 1, + 2, + 11, + 12, + 13, + 14, + 15, + 16 + ], + "reactant_id": 1, + "score": 120.0, + "source": "nx_fragment" + }, + { + "candidate_id": 4013, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 8, + "product_atoms": [ + 19, + 20, + 21, + 22, + 23, + 24, + 25, + 26 + ], + "r_to_p": { + "10": 22, + "3": 19, + "4": 20, + "5": 21, + "6": 26, + "7": 25, + "8": 24, + "9": 23 + }, + "reactant_atoms": [ + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10 + ], + "reactant_id": 1, + "score": 120.0, + "source": "nx_fragment" + }, + { + "candidate_id": 6512, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 0, + "product_atoms": [ + 33 + ], + "r_to_p": { + "0": 33 + }, + "reactant_atoms": [ + 0 + ], + "reactant_id": 1, + "score": 6.0, + "source": "nx_fragment" + } + ], + "selector": "ilp", + "status": "ilp" +} diff --git a/tests/pipette/expected_atom_map_res/golden-rdf2.json b/tests/pipette/expected_atom_map_res/golden-rdf2.json new file mode 100644 index 0000000..7779b3c --- /dev/null +++ b/tests/pipette/expected_atom_map_res/golden-rdf2.json @@ -0,0 +1,957 @@ +{ + "atom_mapped_reaction_smiles": "[CH2:1]=[CH:2][CH:11]([OH:12])[CH:13]([CH3:14])[NH2:15].[CH3:3][C:4](=[O:5])[CH2:6][S:7][CH2:8][CH:9]=[CH2:10]>>[CH2:1]=[CH2:2].[CH3:3][C:4](=[O:5])[CH2:6][S:7][CH2:8]/[CH:9]=[CH:10]/[CH:11]([OH:12])[CH:13]([CH3:14])[NH2:15]", + "config": { + "active_copy_penalty": 1.0, + "allow_extra_product_edges_in_candidate": true, + "atom_map_anchor_bonus": 25.0, + "atom_reward": 10.0, + "bond_environment_objective": "off", + "bond_environment_rank_tolerance": 1e-06, + "broken_bond_environment_penalty": 1.0, + "candidate_piece_penalty": 2.0, + "compare_aromaticity": false, + "compare_bond_order": true, + "compare_formal_charge": true, + "compare_isotope": false, + "extra_product_edge_penalty": 2.0, + "fallback_to_greedy": true, + "include_rdkit_mcs_candidates": true, + "mapped_reactants_single_copy": true, + "max_base_candidates_per_reactant": 6000, + "max_broken_bond_pair_penalty_terms": 25000, + "max_copies": 3, + "max_fragment_atoms": 8, + "max_fragments_per_reactant": 2500, + "max_matches_per_fragment": 128, + "max_mcs_matches": 256, + "max_segment_distance": 8, + "min_fragment_atoms": 1, + "preserved_bond_reward": 5.0, + "require_atom_map_match_when_present": true, + "respect_atom_maps": null, + "ring_bond_break_penalty": 2.0, + "selector": "ilp", + "single_atom_piece_penalty": 4.0, + "stable_single_bond_break_penalty": 1.0, + "unsaturated_endpoint_break_credit": 0.75, + "unused_reactant_atom_penalty_active_copy": 6.0 + }, + "diagnostics": { + "active_copies_by_reactant": { + "0": 1, + "1": 1 + }, + "candidate_generation": { + "base_candidate_count": 300, + "expanded_candidate_count": 900, + "respect_atom_maps": false + }, + "lineage_split_events": [ + { + "blocks": [ + [ + 0, + 1 + ], + [ + 10, + 11, + 12, + 13, + 14 + ] + ], + "extra_blocks": 1, + "lineage": "R0/copy0", + "num_product_blocks": 2, + "piece_count_for_lineage": 2 + } + ], + "product_bond_events": [ + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 6, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 5, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 3, + 5 + ], + "reactant_bond": [ + 5, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 4, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 6, + 7 + ], + "reactant_bond": [ + 3, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 2, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 8, + 9 + ], + "reactant_bond": [ + 1, + 0 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 9, + 10 + ], + "source_1": "R1/copy0", + "source_2": "R0/copy0" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 12 + ], + "reactant_bond": [ + 2, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 12, + 14 + ], + "reactant_bond": [ + 4, + 6 + ] + } + ], + "product_lineage_quotient": { + "blocks": [ + { + "atoms": [ + 0, + 1 + ], + "block_id": 0, + "size": 2, + "source": "R0/copy0" + }, + { + "atoms": [ + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9 + ], + "block_id": 1, + "size": 8, + "source": "R1/copy0" + }, + { + "atoms": [ + 10, + 11, + 12, + 13, + 14 + ], + "block_id": 2, + "size": 5, + "source": "R0/copy0" + } + ], + "edges": [ + { + "block_1": 1, + "block_2": 2 + } + ] + }, + "reactant_bond_events": [ + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 1, + 10 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 12 + ], + "reactant_bond": [ + 2, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 12, + 14 + ], + "reactant_bond": [ + 4, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 9, + 8 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 8, + 7 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 7, + 6 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 6, + 5 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 5, + 3 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 3, + 2 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 5, + 7 + ] + } + ], + "residuals": { + "counts": { + "active_reactant_residual_fragment_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "unused_reactant_component_count": 0 + }, + "product_byproduct_candidate_smiles": "", + "product_byproduct_candidates": [], + "product_partial_unmapped_fragments": [], + "product_residual_fragments": [], + "product_residual_smiles": "", + "reactant_residual_fragments": [], + "reactant_residual_smiles": "" + }, + "segment_events": { + "contractions": [], + "foreign_or_unknown_bridged_breaks": [], + "lineage_restricted_breaks": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 10 + ], + "product_distance_full": null, + "product_distance_same_lineage": null, + "product_path_full": [], + "product_path_same_lineage": [], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + } + ], + "segments": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 0, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 0, + 1 + ], + "product_path_same_lineage": [ + 0, + 1 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 10 + ], + "product_distance_full": null, + "product_distance_same_lineage": null, + "product_path_full": [], + "product_path_same_lineage": [], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 10, + 11 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 11 + ], + "product_path_same_lineage": [ + 10, + 11 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 10, + 12 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 12 + ], + "product_path_same_lineage": [ + 10, + 12 + ], + "reactant_atoms": [ + 2, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 4 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 12, + 13 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 12, + 13 + ], + "product_path_same_lineage": [ + 12, + 13 + ], + "reactant_atoms": [ + 4, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 12, + 14 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 12, + 14 + ], + "product_path_same_lineage": [ + 12, + 14 + ], + "reactant_atoms": [ + 4, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 6 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 9, + 8 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 9, + 8 + ], + "product_path_same_lineage": [ + 9, + 8 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 8, + 7 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 8, + 7 + ], + "product_path_same_lineage": [ + 8, + 7 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 7, + 6 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 7, + 6 + ], + "product_path_same_lineage": [ + 7, + 6 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 6, + 5 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 6, + 5 + ], + "product_path_same_lineage": [ + 6, + 5 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 5, + 3 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 3 + ], + "product_path_same_lineage": [ + 5, + 3 + ], + "reactant_atoms": [ + 4, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 5 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 3, + 2 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 2 + ], + "product_path_same_lineage": [ + 3, + 2 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 3, + 4 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 4 + ], + "product_path_same_lineage": [ + 3, + 4 + ], + "reactant_atoms": [ + 5, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 7 + ] + } + ], + "stretches": [] + }, + "selected_piece_count_by_lineage": { + "R0/copy0": 2, + "R1/copy0": 1 + }, + "topology_counts": { + "active_lineage_count": 2, + "covered_product_atom_count": 15, + "foreign_or_unknown_bridge_atom_count": 0, + "foreign_or_unknown_bridged_break_count": 0, + "interlineage_product_bond_formed_count": 1, + "intralineage_product_bond_formed_count": 0, + "lineage_extra_block_count": 1, + "lineage_restricted_break_count": 1, + "lineage_split_event_count": 1, + "product_bond_touches_uncovered_atom_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_bond_broken_count": 1, + "reactant_bond_deleted_or_unmapped_count": 0, + "reactant_bond_preserved_count": 12, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "segment_contraction_count": 0, + "segment_contraction_total": 0, + "segment_stretch_count": 0, + "segment_stretch_total": 0, + "selected_piece_count": 3, + "uncovered_product_atom_count": 0 + }, + "uncovered_product_atoms": [], + "unused_reactant_atoms_by_active_lineage": { + "R0/copy0": [], + "R1/copy0": [] + } + }, + "objective_value": 202.0, + "product_smiles": "C=C.CC(=O)CSC/C=C/C(O)C(C)N", + "reactants": [ + { + "atom_count": 7, + "bond_count": 6, + "reactant_id": 0, + "smiles": "C=CC(O)C(C)N" + }, + { + "atom_count": 8, + "bond_count": 7, + "reactant_id": 1, + "smiles": "C=CCSCC(C)=O" + } + ], + "reaction_smiles": "C=CC(O)C(C)N.C=CCSCC(C)=O>>C=C.CC(=O)CSC/C=C/C(O)C(C)N", + "selected_pieces": [ + { + "candidate_id": 11, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 4, + "product_atoms": [ + 10, + 11, + 12, + 13, + 14 + ], + "r_to_p": { + "2": 10, + "3": 11, + "4": 12, + "5": 13, + "6": 14 + }, + "reactant_atoms": [ + 2, + 3, + 4, + 5, + 6 + ], + "reactant_id": 0, + "score": 70.0, + "source": "nx_fragment" + }, + { + "candidate_id": 45, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 1, + "product_atoms": [ + 0, + 1 + ], + "r_to_p": { + "0": 0, + "1": 1 + }, + "reactant_atoms": [ + 0, + 1 + ], + "reactant_id": 0, + "score": 25.0, + "source": "nx_fragment" + }, + { + "candidate_id": 145, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 7, + "product_atoms": [ + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9 + ], + "r_to_p": { + "0": 9, + "1": 8, + "2": 7, + "3": 6, + "4": 5, + "5": 3, + "6": 2, + "7": 4 + }, + "reactant_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "reactant_id": 1, + "score": 115.0, + "source": "nx_fragment" + } + ], + "selector": "ilp", + "status": "ilp" +} diff --git a/tests/pipette/expected_atom_map_res/wallnut1.json b/tests/pipette/expected_atom_map_res/wallnut1.json new file mode 100644 index 0000000..43f90c4 --- /dev/null +++ b/tests/pipette/expected_atom_map_res/wallnut1.json @@ -0,0 +1,1444 @@ +{ + "atom_mapped_reaction_smiles": "[O-][n+]1[c:9]2[c:7]([n:6][cH:5][c:4]([N+:2](=[O:1])[O-:3])[cH:11]2)[n:8][o:10]1>>C1=CC2CCC1O[N:8]2[c:7]1[n:6][cH:5][c:4]([N+:2](=[O:1])[O-:3])[cH:11][c:9]1N1C2C=CC(CC2)[O:10]1", + "config": { + "active_copy_penalty": 1.0, + "allow_extra_product_edges_in_candidate": true, + "atom_map_anchor_bonus": 25.0, + "atom_reward": 10.0, + "bond_environment_objective": "off", + "bond_environment_rank_tolerance": 1e-06, + "broken_bond_environment_penalty": 1.0, + "candidate_piece_penalty": 2.0, + "compare_aromaticity": false, + "compare_bond_order": true, + "compare_formal_charge": true, + "compare_isotope": false, + "extra_product_edge_penalty": 2.0, + "fallback_to_greedy": true, + "include_rdkit_mcs_candidates": true, + "mapped_reactants_single_copy": true, + "max_base_candidates_per_reactant": 6000, + "max_broken_bond_pair_penalty_terms": 25000, + "max_copies": 3, + "max_fragment_atoms": 8, + "max_fragments_per_reactant": 2500, + "max_matches_per_fragment": 128, + "max_mcs_matches": 256, + "max_segment_distance": 8, + "min_fragment_atoms": 1, + "preserved_bond_reward": 5.0, + "require_atom_map_match_when_present": true, + "respect_atom_maps": null, + "ring_bond_break_penalty": 2.0, + "selector": "ilp", + "single_atom_piece_penalty": 4.0, + "stable_single_bond_break_penalty": 1.0, + "unsaturated_endpoint_break_credit": 0.75, + "unused_reactant_atom_penalty_active_copy": 6.0 + }, + "diagnostics": { + "active_copies_by_reactant": { + "0": 1 + }, + "candidate_generation": { + "base_candidate_count": 287, + "expanded_candidate_count": 861, + "respect_atom_maps": false + }, + "lineage_split_events": [ + { + "blocks": [ + [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 15, + 24 + ], + [ + 17 + ] + ], + "extra_blocks": 1, + "lineage": "R0/copy0", + "num_product_blocks": 2, + "piece_count_for_lineage": 4 + } + ], + "product_bond_events": [ + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 3 + ], + "reactant_bond": [ + 1, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 24 + ], + "reactant_bond": [ + 3, + 12 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 4, + 5 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "product_bond_order_changed_from_reactant", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 7 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 15 + ], + "reactant_bond": [ + 6, + 11 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 7, + 8 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 7, + 12 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 8, + 9 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 9, + 10 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 9, + 14 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 10, + 11 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 11, + 12 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 12, + 13 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 13, + 14 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 15, + 16 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 15, + 24 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 16, + 17 + ], + "source_1": "uncovered", + "source_2": "R0/copy0" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 16, + 21 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 17, + 18 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 18, + 19 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 18, + 23 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 19, + 20 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 20, + 21 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 21, + 22 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 22, + 23 + ], + "source_1": "uncovered", + "source_2": "uncovered" + } + ], + "product_lineage_quotient": { + "blocks": [ + { + "atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 15, + 24 + ], + "block_id": 0, + "size": 10, + "source": "R0/copy0" + }, + { + "atoms": [ + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "block_id": 1, + "size": 7, + "source": "uncovered" + }, + { + "atoms": [ + 16, + 18, + 19, + 20, + 21, + 22, + 23 + ], + "block_id": 2, + "size": 7, + "source": "uncovered" + }, + { + "atoms": [ + 17 + ], + "block_id": 3, + "size": 1, + "source": "R0/copy0" + } + ], + "edges": [ + { + "block_1": 0, + "block_2": 1 + }, + { + "block_1": 0, + "block_2": 2 + }, + { + "block_1": 2, + "block_2": 3 + } + ] + }, + "reactant_bond_events": [ + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 3 + ], + "reactant_bond": [ + 1, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 24 + ], + "reactant_bond": [ + 3, + 12 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 4, + 5 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_order_changed", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 7 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 15 + ], + "reactant_bond": [ + 6, + 11 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 7, + 17 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_deleted_or_unmapped", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 17, + null + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_deleted_or_unmapped", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + null, + null + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_deleted_or_unmapped", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + null, + 15 + ], + "reactant_bond": [ + 9, + 11 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 15, + 24 + ], + "reactant_bond": [ + 11, + 12 + ] + } + ], + "residuals": { + "counts": { + "active_reactant_residual_fragment_count": 1, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 2, + "product_residual_atom_count": 14, + "product_residual_fragment_count": 2, + "reactant_residual_atom_count": 2, + "reactant_residual_fragment_count": 1, + "unused_reactant_component_count": 0 + }, + "product_byproduct_candidate_smiles": "", + "product_byproduct_candidates": [], + "product_partial_unmapped_fragments": [ + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "bond_count": 7, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 7, + 8 + ], + [ + 7, + 12 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 0, + "size": 7, + "smiles": "OC1C=CCCC1", + "touches_selected_mapping": true, + "unmapped_smiles": "OC1C=CCCC1" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 16, + 18, + 19, + 20, + 21, + 22, + 23 + ], + "bond_count": 7, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 15, + 16 + ], + [ + 16, + 17 + ], + [ + 17, + 18 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 1, + "size": 7, + "smiles": "NC1C=CCCC1", + "touches_selected_mapping": true, + "unmapped_smiles": "NC1C=CCCC1" + } + ], + "product_residual_fragments": [ + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "bond_count": 7, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 7, + 8 + ], + [ + 7, + 12 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 0, + "size": 7, + "smiles": "OC1C=CCCC1", + "touches_selected_mapping": true, + "unmapped_smiles": "OC1C=CCCC1" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 16, + 18, + 19, + 20, + 21, + 22, + 23 + ], + "bond_count": 7, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 15, + 16 + ], + [ + 16, + 17 + ], + [ + 17, + 18 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 1, + "size": 7, + "smiles": "NC1C=CCCC1", + "touches_selected_mapping": true, + "unmapped_smiles": "NC1C=CCCC1" + } + ], + "product_residual_smiles": "OC1C=CCCC1.NC1C=CCCC1", + "reactant_residual_fragments": [ + { + "atoms": [ + 9, + 10 + ], + "bond_count": 1, + "boundary_bonds_to_selected_reactant_atoms": [ + [ + 8, + 9 + ], + [ + 9, + 11 + ] + ], + "classification": "unused_fragment_in_active_reactant_copy", + "copy_id": 0, + "lineage": "R0/copy0", + "reactant_id": 0, + "residual_id": 0, + "size": 2, + "smiles": "[n+][O-]", + "unmapped_smiles": "[n+][O-]" + } + ], + "reactant_residual_smiles": "[n+][O-]" + }, + "segment_events": { + "contractions": [], + "foreign_or_unknown_bridged_breaks": [ + { + "bridge_sources": [ + "uncovered" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "uncovered", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 1, + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 17 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 7, + 6, + 15, + 16, + 17 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "uncovered", + "R0/copy0" + ] + }, + { + "bridge_sources": [ + "uncovered" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "uncovered", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 1, + "lineage": "R0/copy0", + "product_atoms": [ + 17, + 15 + ], + "product_distance_full": 2, + "product_distance_same_lineage": null, + "product_path_full": [ + 17, + 16, + 15 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 8, + 11 + ], + "reactant_distance": 2, + "reactant_path": [ + 8, + 9, + 11 + ], + "source_sequence": [ + "R0/copy0", + "uncovered", + "R0/copy0" + ] + } + ], + "lineage_restricted_breaks": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 17 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 7, + 6, + 15, + 16, + 17 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 17, + 15 + ], + "product_distance_full": 2, + "product_distance_same_lineage": null, + "product_path_full": [ + 17, + 16, + 15 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 8, + 11 + ], + "reactant_distance": 2, + "reactant_path": [ + 8, + 9, + 11 + ] + } + ], + "segments": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 0, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 0, + 1 + ], + "product_path_same_lineage": [ + 0, + 1 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 2 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 2 + ], + "product_path_same_lineage": [ + 1, + 2 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 3 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 3 + ], + "product_path_same_lineage": [ + 1, + 3 + ], + "reactant_atoms": [ + 1, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 3 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 4 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 4 + ], + "product_path_same_lineage": [ + 3, + 4 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 24 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 24 + ], + "product_path_same_lineage": [ + 3, + 24 + ], + "reactant_atoms": [ + 3, + 12 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 12 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 4, + 5 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 4, + 5 + ], + "product_path_same_lineage": [ + 4, + 5 + ], + "reactant_atoms": [ + 4, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 5, + 6 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 6 + ], + "product_path_same_lineage": [ + 5, + 6 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 7 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 6, + 7 + ], + "product_path_same_lineage": [ + 6, + 7 + ], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 15 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 6, + 15 + ], + "product_path_same_lineage": [ + 6, + 15 + ], + "reactant_atoms": [ + 6, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 11 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 17 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 7, + 6, + 15, + 16, + 17 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 15, + 24 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 15, + 24 + ], + "product_path_same_lineage": [ + 15, + 24 + ], + "reactant_atoms": [ + 11, + 12 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 12 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 17, + 15 + ], + "product_distance_full": 2, + "product_distance_same_lineage": null, + "product_path_full": [ + 17, + 16, + 15 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 8, + 11 + ], + "reactant_distance": 2, + "reactant_path": [ + 8, + 9, + 11 + ] + } + ], + "stretches": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 17 + ], + "product_distance_full": 4, + "product_distance_same_lineage": null, + "product_path_full": [ + 7, + 6, + 15, + 16, + 17 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ], + "stretch": 3 + } + ] + }, + "selected_piece_count_by_lineage": { + "R0/copy0": 4 + }, + "topology_counts": { + "active_lineage_count": 1, + "covered_product_atom_count": 11, + "foreign_or_unknown_bridge_atom_count": 2, + "foreign_or_unknown_bridged_break_count": 2, + "interlineage_product_bond_formed_count": 0, + "intralineage_product_bond_formed_count": 0, + "lineage_extra_block_count": 1, + "lineage_restricted_break_count": 2, + "lineage_split_event_count": 1, + "product_bond_touches_uncovered_atom_count": 19, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 2, + "product_residual_atom_count": 14, + "product_residual_fragment_count": 2, + "reactant_bond_broken_count": 1, + "reactant_bond_deleted_or_unmapped_count": 3, + "reactant_bond_preserved_count": 9, + "reactant_residual_atom_count": 2, + "reactant_residual_fragment_count": 1, + "segment_contraction_count": 0, + "segment_contraction_total": 0, + "segment_stretch_count": 1, + "segment_stretch_total": 3, + "selected_piece_count": 4, + "uncovered_product_atom_count": 14 + }, + "uncovered_product_atoms": [ + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 16, + 18, + 19, + 20, + 21, + 22, + 23 + ], + "unused_reactant_atoms_by_active_lineage": { + "R0/copy0": [ + 9, + 10 + ] + } + }, + "objective_value": 121.0, + "product_smiles": "O=[N+]([O-])c1cnc(N2OC3C=CC2CC3)c(N2OC3C=CC2CC3)c1", + "reactants": [ + { + "atom_count": 13, + "bond_count": 14, + "reactant_id": 0, + "smiles": "O=[N+]([O-])c1cnc2no[n+]([O-])c2c1" + } + ], + "reaction_smiles": "O=[N+]([O-])c1cnc2no[n+]([O-])c2c1>>O=[N+]([O-])c1cnc(N2OC3C=CC2CC3)c(N2OC3C=CC2CC3)c1", + "selected_pieces": [ + { + "candidate_id": 28, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 3, + 4, + 5, + 6, + 15, + 24 + ], + "r_to_p": { + "11": 15, + "12": 24, + "3": 3, + "4": 4, + "5": 5, + "6": 6 + }, + "reactant_atoms": [ + 3, + 4, + 5, + 6, + 11, + 12 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + }, + { + "candidate_id": 111, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 2, + "product_atoms": [ + 0, + 1, + 2 + ], + "r_to_p": { + "0": 0, + "1": 1, + "2": 2 + }, + "reactant_atoms": [ + 0, + 1, + 2 + ], + "reactant_id": 0, + "score": 40.0, + "source": "nx_fragment" + }, + { + "candidate_id": 246, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 0, + "product_atoms": [ + 7 + ], + "r_to_p": { + "7": 7 + }, + "reactant_atoms": [ + 7 + ], + "reactant_id": 0, + "score": 6.0, + "source": "nx_fragment" + }, + { + "candidate_id": 250, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 0, + "product_atoms": [ + 17 + ], + "r_to_p": { + "8": 17 + }, + "reactant_atoms": [ + 8 + ], + "reactant_id": 0, + "score": 6.0, + "source": "nx_fragment" + } + ], + "selector": "ilp", + "status": "ilp" +} diff --git a/tests/pipette/expected_atom_map_res/wallnut2.json b/tests/pipette/expected_atom_map_res/wallnut2.json new file mode 100644 index 0000000..700a05f --- /dev/null +++ b/tests/pipette/expected_atom_map_res/wallnut2.json @@ -0,0 +1,2371 @@ +{ + "atom_mapped_reaction_smiles": "[CH3:1][c:2]1[cH:3][c:4]([CH3:5])[n:6](-[c:7]2[n:9][n:8][c:17](-[n:18]3[n:19][c:20]([CH3:10])[cH:21][c:22]3[CH3:14])[n:16][n:15]2)[n:23]1.[NH2:11][*:12][NH2:13]>>C[c:20]1[n:19][n:18](-[c:17]2nn[c:14]([NH:13][*:12][NH:11][c:10]3nn[c:7](-[n:6]4[c:4]([CH3:5])[cH:3][c:2]([CH3:1])[n:23]4)[n:8][n:9]3)[n:15][n:16]2)[c:22](C)[cH:21]1", + "config": { + "active_copy_penalty": 1.0, + "allow_extra_product_edges_in_candidate": true, + "atom_map_anchor_bonus": 25.0, + "atom_reward": 10.0, + "bond_environment_objective": "off", + "bond_environment_rank_tolerance": 1e-06, + "broken_bond_environment_penalty": 1.0, + "candidate_piece_penalty": 2.0, + "compare_aromaticity": false, + "compare_bond_order": true, + "compare_formal_charge": true, + "compare_isotope": false, + "extra_product_edge_penalty": 2.0, + "fallback_to_greedy": true, + "include_rdkit_mcs_candidates": true, + "mapped_reactants_single_copy": true, + "max_base_candidates_per_reactant": 6000, + "max_broken_bond_pair_penalty_terms": 25000, + "max_copies": 3, + "max_fragment_atoms": 8, + "max_fragments_per_reactant": 2500, + "max_matches_per_fragment": 128, + "max_mcs_matches": 256, + "max_segment_distance": 8, + "min_fragment_atoms": 1, + "preserved_bond_reward": 5.0, + "require_atom_map_match_when_present": true, + "respect_atom_maps": null, + "ring_bond_break_penalty": 2.0, + "selector": "ilp", + "single_atom_piece_penalty": 4.0, + "stable_single_bond_break_penalty": 1.0, + "unsaturated_endpoint_break_credit": 0.75, + "unused_reactant_atom_penalty_active_copy": 6.0 + }, + "diagnostics": { + "active_copies_by_reactant": { + "0": 1, + "1": 1 + }, + "candidate_generation": { + "base_candidate_count": 3135, + "expanded_candidate_count": 9405, + "respect_atom_maps": false + }, + "lineage_split_events": [ + { + "blocks": [ + [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 28 + ], + [ + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 21, + 22 + ] + ], + "extra_blocks": 1, + "lineage": "R0/copy0", + "num_product_blocks": 2, + "piece_count_for_lineage": 5 + } + ], + "product_bond_events": [ + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 28 + ], + "reactant_bond": [ + 1, + 19 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 5 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 28 + ], + "reactant_bond": [ + 5, + 19 + ] + }, + { + "event": "intralineage_product_bond_formed", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 7 + ], + "reactant_atoms": [ + 6, + 17 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 6, + 27 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 17, + 18 + ] + }, + { + "event": "intralineage_product_bond_formed", + "lineage": "R0/copy0", + "product_bond": [ + 8, + 9 + ], + "reactant_atoms": [ + 18, + 13 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 9, + 10 + ], + "source_1": "R0/copy0", + "source_2": "R1/copy0" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 9, + 26 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R1/copy0", + "product_bond": [ + 11, + 12 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 12, + 13 + ], + "source_1": "R1/copy0", + "source_2": "R0/copy0" + }, + { + "event": "intralineage_product_bond_formed", + "lineage": "R0/copy0", + "product_bond": [ + 13, + 14 + ], + "reactant_atoms": [ + 16, + 7 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 13, + 25 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 16, + 24 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 17, + 18 + ], + "reactant_bond": [ + 10, + 11 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 17, + 22 + ], + "reactant_bond": [ + 10, + 15 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 18, + 19 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 19, + 20 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 19, + 21 + ], + "reactant_bond": [ + 12, + 14 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 14, + 15 + ] + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 22, + 23 + ], + "source_1": "R0/copy0", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 24, + 25 + ], + "source_1": "uncovered", + "source_2": "uncovered" + }, + { + "event": "product_bond_touches_uncovered_atom", + "product_bond": [ + 26, + 27 + ], + "source_1": "uncovered", + "source_2": "uncovered" + } + ], + "product_lineage_quotient": { + "blocks": [ + { + "atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 28 + ], + "block_id": 0, + "size": 11, + "source": "R0/copy0" + }, + { + "atoms": [ + 10, + 11, + 12 + ], + "block_id": 1, + "size": 3, + "source": "R1/copy0" + }, + { + "atoms": [ + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 21, + 22 + ], + "block_id": 2, + "size": 9, + "source": "R0/copy0" + }, + { + "atoms": [ + 20 + ], + "block_id": 3, + "size": 1, + "source": "uncovered" + }, + { + "atoms": [ + 23 + ], + "block_id": 4, + "size": 1, + "source": "uncovered" + }, + { + "atoms": [ + 24, + 25 + ], + "block_id": 5, + "size": 2, + "source": "uncovered" + }, + { + "atoms": [ + 26, + 27 + ], + "block_id": 6, + "size": 2, + "source": "uncovered" + } + ], + "edges": [ + { + "block_1": 0, + "block_2": 1 + }, + { + "block_1": 0, + "block_2": 6 + }, + { + "block_1": 1, + "block_2": 2 + }, + { + "block_1": 2, + "block_2": 3 + }, + { + "block_1": 2, + "block_2": 4 + }, + { + "block_1": 2, + "block_2": 5 + } + ] + }, + "reactant_bond_events": [ + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 28 + ], + "reactant_bond": [ + 1, + 19 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 5 + ], + "reactant_bond": [ + 3, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 28 + ], + "reactant_bond": [ + 5, + 19 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 6, + 14 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 6, + 8 + ], + "reactant_bond": [ + 6, + 18 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 16, + 7 + ], + "reactant_bond": [ + 9, + 17 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 17, + 18 + ], + "reactant_bond": [ + 10, + 11 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 17, + 22 + ], + "reactant_bond": [ + 10, + 15 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 18, + 19 + ], + "reactant_bond": [ + 11, + 12 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 19, + 9 + ], + "reactant_bond": [ + 12, + 13 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 19, + 21 + ], + "reactant_bond": [ + 12, + 14 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 14, + 15 + ] + }, + { + "event": "reactant_bond_broken", + "lineage": "R0/copy0", + "mapped_product_atoms": [ + 22, + 13 + ], + "reactant_bond": [ + 15, + 16 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 17, + 18 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R1/copy0", + "product_bond": [ + 11, + 12 + ], + "reactant_bond": [ + 1, + 2 + ] + } + ], + "residuals": { + "counts": { + "active_reactant_residual_fragment_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 4, + "product_residual_atom_count": 6, + "product_residual_fragment_count": 4, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "unused_reactant_component_count": 0 + }, + "product_byproduct_candidate_smiles": "", + "product_byproduct_candidates": [], + "product_partial_unmapped_fragments": [ + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 24, + 25 + ], + "bond_count": 1, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 13, + 25 + ], + [ + 16, + 24 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 0, + "size": 2, + "smiles": "nn", + "touches_selected_mapping": true, + "unmapped_smiles": "nn" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 26, + 27 + ], + "bond_count": 1, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 6, + 27 + ], + [ + 9, + 26 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 1, + "size": 2, + "smiles": "nn", + "touches_selected_mapping": true, + "unmapped_smiles": "nn" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 20 + ], + "bond_count": 0, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 19, + 20 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 2, + "size": 1, + "smiles": "C", + "touches_selected_mapping": true, + "unmapped_smiles": "C" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 23 + ], + "bond_count": 0, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 22, + 23 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 3, + "size": 1, + "smiles": "C", + "touches_selected_mapping": true, + "unmapped_smiles": "C" + } + ], + "product_residual_fragments": [ + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 24, + 25 + ], + "bond_count": 1, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 13, + 25 + ], + [ + 16, + 24 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 0, + "size": 2, + "smiles": "nn", + "touches_selected_mapping": true, + "unmapped_smiles": "nn" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 26, + 27 + ], + "bond_count": 1, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 6, + 27 + ], + [ + 9, + 26 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 1, + "size": 2, + "smiles": "nn", + "touches_selected_mapping": true, + "unmapped_smiles": "nn" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 20 + ], + "bond_count": 0, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 19, + 20 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 2, + "size": 1, + "smiles": "C", + "touches_selected_mapping": true, + "unmapped_smiles": "C" + }, + { + "adjacent_selected_lineages": [ + "R0/copy0" + ], + "atoms": [ + 23 + ], + "bond_count": 0, + "boundary_bonds_to_nonresidual_atoms": [ + [ + 22, + 23 + ] + ], + "classification": "partial_uncovered_product_fragment", + "is_whole_product_component": false, + "parent_product_components": [ + 0 + ], + "residual_id": 3, + "size": 1, + "smiles": "C", + "touches_selected_mapping": true, + "unmapped_smiles": "C" + } + ], + "product_residual_smiles": "nn.nn.C.C", + "reactant_residual_fragments": [], + "reactant_residual_smiles": "" + }, + "segment_events": { + "contractions": [], + "foreign_or_unknown_bridged_breaks": [ + { + "bridge_sources": [ + "R1/copy0" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "R1/copy0", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 3, + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 14 + ], + "product_distance_full": 8, + "product_distance_same_lineage": null, + "product_path_full": [ + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0", + "R1/copy0", + "R1/copy0", + "R0/copy0", + "R0/copy0" + ] + }, + { + "bridge_sources": [ + "R1/copy0" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "R1/copy0", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 3, + "lineage": "R0/copy0", + "product_atoms": [ + 16, + 7 + ], + "product_distance_full": 9, + "product_distance_same_lineage": null, + "product_path_full": [ + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9, + 8, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 9, + 17 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 17 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0", + "R1/copy0", + "R1/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0" + ] + }, + { + "bridge_sources": [ + "R1/copy0" + ], + "collapsed_source_sequence": [ + "R0/copy0", + "R1/copy0", + "R0/copy0" + ], + "foreign_or_unknown_bridge_atom_count": 3, + "lineage": "R0/copy0", + "product_atoms": [ + 19, + 9 + ], + "product_distance_full": 10, + "product_distance_same_lineage": null, + "product_path_full": [ + 19, + 18, + 17, + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ], + "source_sequence": [ + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R0/copy0", + "R1/copy0", + "R1/copy0", + "R1/copy0", + "R0/copy0" + ] + } + ], + "lineage_restricted_breaks": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 14 + ], + "product_distance_full": 8, + "product_distance_same_lineage": null, + "product_path_full": [ + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 16, + 7 + ], + "product_distance_full": 9, + "product_distance_same_lineage": null, + "product_path_full": [ + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9, + 8, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 9, + 17 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 17 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 19, + 9 + ], + "product_distance_full": 10, + "product_distance_same_lineage": null, + "product_path_full": [ + 19, + 18, + 17, + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ] + } + ], + "segments": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 0, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 0, + 1 + ], + "product_path_same_lineage": [ + 0, + 1 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 2 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 2 + ], + "product_path_same_lineage": [ + 1, + 2 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 28 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 28 + ], + "product_path_same_lineage": [ + 1, + 28 + ], + "reactant_atoms": [ + 1, + 19 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 19 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 2, + 3 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 2, + 3 + ], + "product_path_same_lineage": [ + 2, + 3 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 4 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 4 + ], + "product_path_same_lineage": [ + 3, + 4 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 5 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 5 + ], + "product_path_same_lineage": [ + 3, + 5 + ], + "reactant_atoms": [ + 3, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 5, + 6 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 6 + ], + "product_path_same_lineage": [ + 5, + 6 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 5, + 28 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 28 + ], + "product_path_same_lineage": [ + 5, + 28 + ], + "reactant_atoms": [ + 5, + 19 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 19 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 14 + ], + "product_distance_full": 8, + "product_distance_same_lineage": null, + "product_path_full": [ + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 8 + ], + "product_distance_full": 2, + "product_distance_same_lineage": 2, + "product_path_full": [ + 6, + 7, + 8 + ], + "product_path_same_lineage": [ + 6, + 7, + 8 + ], + "reactant_atoms": [ + 6, + 18 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 18 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 14, + 15 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 14, + 15 + ], + "product_path_same_lineage": [ + 14, + 15 + ], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 15, + 16 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 15, + 16 + ], + "product_path_same_lineage": [ + 15, + 16 + ], + "reactant_atoms": [ + 8, + 9 + ], + "reactant_distance": 1, + "reactant_path": [ + 8, + 9 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 16, + 17 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 16, + 17 + ], + "product_path_same_lineage": [ + 16, + 17 + ], + "reactant_atoms": [ + 9, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 10 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 16, + 7 + ], + "product_distance_full": 9, + "product_distance_same_lineage": null, + "product_path_full": [ + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9, + 8, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 9, + 17 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 17 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 17, + 18 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 17, + 18 + ], + "product_path_same_lineage": [ + 17, + 18 + ], + "reactant_atoms": [ + 10, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 10, + 11 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 17, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 17, + 22 + ], + "product_path_same_lineage": [ + 17, + 22 + ], + "reactant_atoms": [ + 10, + 15 + ], + "reactant_distance": 1, + "reactant_path": [ + 10, + 15 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 18, + 19 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 18, + 19 + ], + "product_path_same_lineage": [ + 18, + 19 + ], + "reactant_atoms": [ + 11, + 12 + ], + "reactant_distance": 1, + "reactant_path": [ + 11, + 12 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 19, + 9 + ], + "product_distance_full": 10, + "product_distance_same_lineage": null, + "product_path_full": [ + 19, + 18, + 17, + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 19, + 21 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 19, + 21 + ], + "product_path_same_lineage": [ + 19, + 21 + ], + "reactant_atoms": [ + 12, + 14 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 14 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 21, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 21, + 22 + ], + "product_path_same_lineage": [ + 21, + 22 + ], + "reactant_atoms": [ + 14, + 15 + ], + "reactant_distance": 1, + "reactant_path": [ + 14, + 15 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 22, + 13 + ], + "product_distance_full": 5, + "product_distance_same_lineage": 5, + "product_path_full": [ + 22, + 17, + 16, + 15, + 14, + 13 + ], + "product_path_same_lineage": [ + 22, + 17, + 16, + 15, + 14, + 13 + ], + "reactant_atoms": [ + 15, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 15, + 16 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 8 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 7, + 8 + ], + "product_path_same_lineage": [ + 7, + 8 + ], + "reactant_atoms": [ + 17, + 18 + ], + "reactant_distance": 1, + "reactant_path": [ + 17, + 18 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 10, + 11 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 11 + ], + "product_path_same_lineage": [ + 10, + 11 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R1/copy0", + "product_atoms": [ + 11, + 12 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 11, + 12 + ], + "product_path_same_lineage": [ + 11, + 12 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + } + ], + "stretches": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 14 + ], + "product_distance_full": 8, + "product_distance_same_lineage": null, + "product_path_full": [ + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ], + "stretch": 7 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 8 + ], + "product_distance_full": 2, + "product_distance_same_lineage": 2, + "product_path_full": [ + 6, + 7, + 8 + ], + "product_path_same_lineage": [ + 6, + 7, + 8 + ], + "reactant_atoms": [ + 6, + 18 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 18 + ], + "stretch": 1 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 16, + 7 + ], + "product_distance_full": 9, + "product_distance_same_lineage": null, + "product_path_full": [ + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9, + 8, + 7 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 9, + 17 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 17 + ], + "stretch": 8 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 19, + 9 + ], + "product_distance_full": 10, + "product_distance_same_lineage": null, + "product_path_full": [ + 19, + 18, + 17, + 16, + 15, + 14, + 13, + 12, + 11, + 10, + 9 + ], + "product_path_same_lineage": [], + "reactant_atoms": [ + 12, + 13 + ], + "reactant_distance": 1, + "reactant_path": [ + 12, + 13 + ], + "stretch": 9 + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 22, + 13 + ], + "product_distance_full": 5, + "product_distance_same_lineage": 5, + "product_path_full": [ + 22, + 17, + 16, + 15, + 14, + 13 + ], + "product_path_same_lineage": [ + 22, + 17, + 16, + 15, + 14, + 13 + ], + "reactant_atoms": [ + 15, + 16 + ], + "reactant_distance": 1, + "reactant_path": [ + 15, + 16 + ], + "stretch": 4 + } + ] + }, + "selected_piece_count_by_lineage": { + "R0/copy0": 5, + "R1/copy0": 1 + }, + "topology_counts": { + "active_lineage_count": 2, + "covered_product_atom_count": 23, + "foreign_or_unknown_bridge_atom_count": 9, + "foreign_or_unknown_bridged_break_count": 3, + "interlineage_product_bond_formed_count": 2, + "intralineage_product_bond_formed_count": 3, + "lineage_extra_block_count": 1, + "lineage_restricted_break_count": 3, + "lineage_split_event_count": 1, + "product_bond_touches_uncovered_atom_count": 8, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 4, + "product_residual_atom_count": 6, + "product_residual_fragment_count": 4, + "reactant_bond_broken_count": 5, + "reactant_bond_deleted_or_unmapped_count": 0, + "reactant_bond_preserved_count": 19, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "segment_contraction_count": 0, + "segment_contraction_total": 0, + "segment_stretch_count": 5, + "segment_stretch_total": 29, + "selected_piece_count": 6, + "uncovered_product_atom_count": 6 + }, + "uncovered_product_atoms": [ + 20, + 23, + 24, + 25, + 26, + 27 + ], + "unused_reactant_atoms_by_active_lineage": { + "R0/copy0": [], + "R1/copy0": [] + } + }, + "objective_value": 303.0, + "product_smiles": "Cc1cc(C)n(-c2nnc(N*Nc3nnc(-n4nc(C)cc4C)nn3)nn2)n1", + "reactants": [ + { + "atom_count": 20, + "bond_count": 22, + "reactant_id": 0, + "smiles": "Cc1cc(C)n(-c2nnc(-n3nc(C)cc3C)nn2)n1" + }, + { + "atom_count": 3, + "bond_count": 2, + "reactant_id": 1, + "smiles": "N*N" + } + ], + "reaction_smiles": "Cc1cc(C)n(-c2nnc(-n3nc(C)cc3C)nn2)n1.N*N>>Cc1cc(C)n(-c2nnc(N*Nc3nnc(-n4nc(C)cc4C)nn3)nn2)n1", + "selected_pieces": [ + { + "candidate_id": 0, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 8, + "product_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 28 + ], + "r_to_p": { + "0": 0, + "1": 1, + "19": 28, + "2": 2, + "3": 3, + "4": 4, + "5": 5, + "6": 6 + }, + "reactant_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 19 + ], + "reactant_id": 0, + "score": 120.0, + "source": "nx_fragment" + }, + { + "candidate_id": 56, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 8, + "product_atoms": [ + 14, + 15, + 16, + 17, + 18, + 19, + 21, + 22 + ], + "r_to_p": { + "10": 17, + "11": 18, + "12": 19, + "14": 21, + "15": 22, + "7": 14, + "8": 15, + "9": 16 + }, + "reactant_atoms": [ + 7, + 8, + 9, + 10, + 11, + 12, + 14, + 15 + ], + "reactant_id": 0, + "score": 120.0, + "source": "nx_fragment" + }, + { + "candidate_id": 2810, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 1, + "product_atoms": [ + 7, + 8 + ], + "r_to_p": { + "17": 7, + "18": 8 + }, + "reactant_atoms": [ + 17, + 18 + ], + "reactant_id": 0, + "score": 25.0, + "source": "nx_fragment" + }, + { + "candidate_id": 3008, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 0, + "product_atoms": [ + 9 + ], + "r_to_p": { + "13": 9 + }, + "reactant_atoms": [ + 13 + ], + "reactant_id": 0, + "score": 6.0, + "source": "nx_fragment" + }, + { + "candidate_id": 3051, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 0, + "product_atoms": [ + 13 + ], + "r_to_p": { + "16": 13 + }, + "reactant_atoms": [ + 16 + ], + "reactant_id": 0, + "score": 6.0, + "source": "nx_fragment" + }, + { + "candidate_id": 3100, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 2, + "product_atoms": [ + 10, + 11, + 12 + ], + "r_to_p": { + "0": 10, + "1": 11, + "2": 12 + }, + "reactant_atoms": [ + 0, + 1, + 2 + ], + "reactant_id": 1, + "score": 40.0, + "source": "nx_fragment" + } + ], + "selector": "ilp", + "status": "greedy_fallback_after_ilp_status:Time limit reached. (HiGHS Status 13: Time limit reached)" +} diff --git a/tests/pipette/expected_atom_map_res/wallnut3.json b/tests/pipette/expected_atom_map_res/wallnut3.json new file mode 100644 index 0000000..78a3b23 --- /dev/null +++ b/tests/pipette/expected_atom_map_res/wallnut3.json @@ -0,0 +1,1621 @@ +{ + "atom_mapped_reaction_smiles": "[OH:1][n:2]1[n:3][n:4][n:5][c:6]1-[c:7]1[n:8][o:9][n:10][c:11]1[NH2:12].[NH2:13][c:14]1[n:15][o:16][n:17][c:18]1-[c:19]1[n:20][n:21][n:22][n:23]1[OH:24]>>[OH:1][n:2]1[n:3][n:4][n:5][c:6]1-[c:7]1[n:8][o:9][n:10][c:11]1/[N:12]=[N:13]\\[c:14]1[n:15][o:16][n:17][c:18]1-[c:19]1[n:20][n:21][n:22][n:23]1[OH:24]", + "config": { + "active_copy_penalty": 1.0, + "allow_extra_product_edges_in_candidate": true, + "atom_map_anchor_bonus": 25.0, + "atom_reward": 10.0, + "bond_environment_objective": "off", + "bond_environment_rank_tolerance": 1e-06, + "broken_bond_environment_penalty": 1.0, + "candidate_piece_penalty": 2.0, + "compare_aromaticity": false, + "compare_bond_order": true, + "compare_formal_charge": true, + "compare_isotope": false, + "extra_product_edge_penalty": 2.0, + "fallback_to_greedy": true, + "include_rdkit_mcs_candidates": true, + "mapped_reactants_single_copy": true, + "max_base_candidates_per_reactant": 6000, + "max_broken_bond_pair_penalty_terms": 25000, + "max_copies": 3, + "max_fragment_atoms": 8, + "max_fragments_per_reactant": 2500, + "max_matches_per_fragment": 128, + "max_mcs_matches": 256, + "max_segment_distance": 8, + "min_fragment_atoms": 1, + "preserved_bond_reward": 5.0, + "require_atom_map_match_when_present": true, + "respect_atom_maps": null, + "ring_bond_break_penalty": 2.0, + "selector": "ilp", + "single_atom_piece_penalty": 4.0, + "stable_single_bond_break_penalty": 1.0, + "unsaturated_endpoint_break_credit": 0.75, + "unused_reactant_atom_penalty_active_copy": 6.0 + }, + "diagnostics": { + "active_copies_by_reactant": { + "0": 2 + }, + "candidate_generation": { + "base_candidate_count": 1116, + "expanded_candidate_count": 3348, + "respect_atom_maps": false + }, + "lineage_split_events": [], + "product_bond_events": [ + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 0, + 1 + ], + "reactant_bond": [ + 11, + 10 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 2 + ], + "reactant_bond": [ + 10, + 9 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 5 + ], + "reactant_bond": [ + 10, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 3 + ], + "reactant_bond": [ + 9, + 8 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 4 + ], + "reactant_bond": [ + 8, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 4, + 5 + ], + "reactant_bond": [ + 7, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 6 + ], + "reactant_bond": [ + 6, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 7 + ], + "reactant_bond": [ + 5, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 10 + ], + "reactant_bond": [ + 5, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 8 + ], + "reactant_bond": [ + 4, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 8, + 9 + ], + "reactant_bond": [ + 3, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 9, + 10 + ], + "reactant_bond": [ + 2, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 11 + ], + "reactant_bond": [ + 1, + 0 + ] + }, + { + "event": "interlineage_product_bond_formed", + "product_bond": [ + 11, + 12 + ], + "source_1": "R0/copy0", + "source_2": "R0/copy1" + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 13, + 14 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 13, + 17 + ], + "reactant_bond": [ + 1, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 17, + 18 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 18, + 19 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 18, + 22 + ], + "reactant_bond": [ + 6, + 10 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 19, + 20 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 20, + 21 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "product_bond_explained_by_reactant_bond", + "lineage": "R0/copy1", + "product_bond": [ + 22, + 23 + ], + "reactant_bond": [ + 10, + 11 + ] + } + ], + "product_lineage_quotient": { + "blocks": [ + { + "atoms": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11 + ], + "block_id": 0, + "size": 12, + "source": "R0/copy0" + }, + { + "atoms": [ + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22, + 23 + ], + "block_id": 1, + "size": 12, + "source": "R0/copy1" + } + ], + "edges": [ + { + "block_1": 0, + "block_2": 1 + } + ] + }, + "reactant_bond_events": [ + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 11, + 10 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 9 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 10, + 6 + ], + "reactant_bond": [ + 1, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 9, + 8 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 8, + 7 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 7, + 6 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 6, + 5 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 4 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 5, + 1 + ], + "reactant_bond": [ + 6, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 4, + 3 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 3, + 2 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 2, + 1 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy0", + "product_bond": [ + 1, + 0 + ], + "reactant_bond": [ + 10, + 11 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 12, + 13 + ], + "reactant_bond": [ + 0, + 1 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 13, + 14 + ], + "reactant_bond": [ + 1, + 2 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 13, + 17 + ], + "reactant_bond": [ + 1, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 14, + 15 + ], + "reactant_bond": [ + 2, + 3 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 15, + 16 + ], + "reactant_bond": [ + 3, + 4 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 16, + 17 + ], + "reactant_bond": [ + 4, + 5 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 17, + 18 + ], + "reactant_bond": [ + 5, + 6 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 18, + 19 + ], + "reactant_bond": [ + 6, + 7 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 18, + 22 + ], + "reactant_bond": [ + 6, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 19, + 20 + ], + "reactant_bond": [ + 7, + 8 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 20, + 21 + ], + "reactant_bond": [ + 8, + 9 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 21, + 22 + ], + "reactant_bond": [ + 9, + 10 + ] + }, + { + "event": "reactant_bond_preserved", + "lineage": "R0/copy1", + "product_bond": [ + 22, + 23 + ], + "reactant_bond": [ + 10, + 11 + ] + } + ], + "residuals": { + "counts": { + "active_reactant_residual_fragment_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "unused_reactant_component_count": 0 + }, + "product_byproduct_candidate_smiles": "", + "product_byproduct_candidates": [], + "product_partial_unmapped_fragments": [], + "product_residual_fragments": [], + "product_residual_smiles": "", + "reactant_residual_fragments": [], + "reactant_residual_smiles": "" + }, + "segment_events": { + "contractions": [], + "foreign_or_unknown_bridged_breaks": [], + "lineage_restricted_breaks": [], + "segments": [ + { + "lineage": "R0/copy0", + "product_atoms": [ + 11, + 10 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 11, + 10 + ], + "product_path_same_lineage": [ + 11, + 10 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 10, + 9 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 9 + ], + "product_path_same_lineage": [ + 10, + 9 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 10, + 6 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 10, + 6 + ], + "product_path_same_lineage": [ + 10, + 6 + ], + "reactant_atoms": [ + 1, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 9, + 8 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 9, + 8 + ], + "product_path_same_lineage": [ + 9, + 8 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 8, + 7 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 8, + 7 + ], + "product_path_same_lineage": [ + 8, + 7 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 7, + 6 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 7, + 6 + ], + "product_path_same_lineage": [ + 7, + 6 + ], + "reactant_atoms": [ + 4, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 5 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 6, + 5 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 6, + 5 + ], + "product_path_same_lineage": [ + 6, + 5 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 5, + 4 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 4 + ], + "product_path_same_lineage": [ + 5, + 4 + ], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 5, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 5, + 1 + ], + "product_path_same_lineage": [ + 5, + 1 + ], + "reactant_atoms": [ + 6, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 10 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 4, + 3 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 4, + 3 + ], + "product_path_same_lineage": [ + 4, + 3 + ], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 3, + 2 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 3, + 2 + ], + "product_path_same_lineage": [ + 3, + 2 + ], + "reactant_atoms": [ + 8, + 9 + ], + "reactant_distance": 1, + "reactant_path": [ + 8, + 9 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 2, + 1 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 2, + 1 + ], + "product_path_same_lineage": [ + 2, + 1 + ], + "reactant_atoms": [ + 9, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 10 + ] + }, + { + "lineage": "R0/copy0", + "product_atoms": [ + 1, + 0 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 1, + 0 + ], + "product_path_same_lineage": [ + 1, + 0 + ], + "reactant_atoms": [ + 10, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 10, + 11 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 12, + 13 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 12, + 13 + ], + "product_path_same_lineage": [ + 12, + 13 + ], + "reactant_atoms": [ + 0, + 1 + ], + "reactant_distance": 1, + "reactant_path": [ + 0, + 1 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 13, + 14 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 13, + 14 + ], + "product_path_same_lineage": [ + 13, + 14 + ], + "reactant_atoms": [ + 1, + 2 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 2 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 13, + 17 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 13, + 17 + ], + "product_path_same_lineage": [ + 13, + 17 + ], + "reactant_atoms": [ + 1, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 1, + 5 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 14, + 15 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 14, + 15 + ], + "product_path_same_lineage": [ + 14, + 15 + ], + "reactant_atoms": [ + 2, + 3 + ], + "reactant_distance": 1, + "reactant_path": [ + 2, + 3 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 15, + 16 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 15, + 16 + ], + "product_path_same_lineage": [ + 15, + 16 + ], + "reactant_atoms": [ + 3, + 4 + ], + "reactant_distance": 1, + "reactant_path": [ + 3, + 4 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 16, + 17 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 16, + 17 + ], + "product_path_same_lineage": [ + 16, + 17 + ], + "reactant_atoms": [ + 4, + 5 + ], + "reactant_distance": 1, + "reactant_path": [ + 4, + 5 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 17, + 18 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 17, + 18 + ], + "product_path_same_lineage": [ + 17, + 18 + ], + "reactant_atoms": [ + 5, + 6 + ], + "reactant_distance": 1, + "reactant_path": [ + 5, + 6 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 18, + 19 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 18, + 19 + ], + "product_path_same_lineage": [ + 18, + 19 + ], + "reactant_atoms": [ + 6, + 7 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 7 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 18, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 18, + 22 + ], + "product_path_same_lineage": [ + 18, + 22 + ], + "reactant_atoms": [ + 6, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 6, + 10 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 19, + 20 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 19, + 20 + ], + "product_path_same_lineage": [ + 19, + 20 + ], + "reactant_atoms": [ + 7, + 8 + ], + "reactant_distance": 1, + "reactant_path": [ + 7, + 8 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 20, + 21 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 20, + 21 + ], + "product_path_same_lineage": [ + 20, + 21 + ], + "reactant_atoms": [ + 8, + 9 + ], + "reactant_distance": 1, + "reactant_path": [ + 8, + 9 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 21, + 22 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 21, + 22 + ], + "product_path_same_lineage": [ + 21, + 22 + ], + "reactant_atoms": [ + 9, + 10 + ], + "reactant_distance": 1, + "reactant_path": [ + 9, + 10 + ] + }, + { + "lineage": "R0/copy1", + "product_atoms": [ + 22, + 23 + ], + "product_distance_full": 1, + "product_distance_same_lineage": 1, + "product_path_full": [ + 22, + 23 + ], + "product_path_same_lineage": [ + 22, + 23 + ], + "reactant_atoms": [ + 10, + 11 + ], + "reactant_distance": 1, + "reactant_path": [ + 10, + 11 + ] + } + ], + "stretches": [] + }, + "selected_piece_count_by_lineage": { + "R0/copy0": 2, + "R0/copy1": 2 + }, + "topology_counts": { + "active_lineage_count": 2, + "covered_product_atom_count": 24, + "foreign_or_unknown_bridge_atom_count": 0, + "foreign_or_unknown_bridged_break_count": 0, + "interlineage_product_bond_formed_count": 1, + "intralineage_product_bond_formed_count": 0, + "lineage_extra_block_count": 0, + "lineage_restricted_break_count": 0, + "lineage_split_event_count": 0, + "product_bond_touches_uncovered_atom_count": 0, + "product_byproduct_candidate_count": 0, + "product_partial_unmapped_fragment_count": 0, + "product_residual_atom_count": 0, + "product_residual_fragment_count": 0, + "reactant_bond_broken_count": 0, + "reactant_bond_deleted_or_unmapped_count": 0, + "reactant_bond_preserved_count": 26, + "reactant_residual_atom_count": 0, + "reactant_residual_fragment_count": 0, + "segment_contraction_count": 0, + "segment_contraction_total": 0, + "segment_stretch_count": 0, + "segment_stretch_total": 0, + "selected_piece_count": 4, + "uncovered_product_atom_count": 0 + }, + "uncovered_product_atoms": [], + "unused_reactant_atoms_by_active_lineage": { + "R0/copy0": [], + "R0/copy1": [] + } + }, + "objective_value": 350.0, + "product_smiles": "On1nnnc1-c1nonc1/N=N\\c1nonc1-c1nnnn1O", + "reactants": [ + { + "atom_count": 12, + "bond_count": 13, + "reactant_id": 0, + "smiles": "Nc1nonc1-c1nnnn1O" + } + ], + "reaction_smiles": "Nc1nonc1-c1nnnn1O>[K+].[O-][Mn](=O)(=O)=O>On1nnnc1-c1nonc1/N=N\\c1nonc1-c1nnnn1O", + "selected_pieces": [ + { + "candidate_id": 456, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5 + ], + "r_to_p": { + "10": 1, + "11": 0, + "6": 5, + "7": 4, + "8": 3, + "9": 2 + }, + "reactant_atoms": [ + 6, + 7, + 8, + 9, + 10, + 11 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + }, + { + "candidate_id": 448, + "copy_id": 0, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 6, + 7, + 8, + 9, + 10, + 11 + ], + "r_to_p": { + "0": 11, + "1": 10, + "2": 9, + "3": 8, + "4": 7, + "5": 6 + }, + "reactant_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + }, + { + "candidate_id": 449, + "copy_id": 1, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 12, + 13, + 14, + 15, + 16, + 17 + ], + "r_to_p": { + "0": 12, + "1": 13, + "2": 14, + "3": 15, + "4": 16, + "5": 17 + }, + "reactant_atoms": [ + 0, + 1, + 2, + 3, + 4, + 5 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + }, + { + "candidate_id": 457, + "copy_id": 1, + "extra_product_edges": 0, + "preserved_bonds": 6, + "product_atoms": [ + 18, + 19, + 20, + 21, + 22, + 23 + ], + "r_to_p": { + "10": 22, + "11": 23, + "6": 18, + "7": 19, + "8": 20, + "9": 21 + }, + "reactant_atoms": [ + 6, + 7, + 8, + 9, + 10, + 11 + ], + "reactant_id": 0, + "score": 90.0, + "source": "nx_fragment" + } + ], + "selector": "ilp", + "status": "ilp" +} diff --git a/tests/pipette/helpers.py b/tests/pipette/helpers.py index 994bd98..fda604f 100644 --- a/tests/pipette/helpers.py +++ b/tests/pipette/helpers.py @@ -5,6 +5,13 @@ ## SPDX-License-Identifier: Apache-2.0 ############################################################################### +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + from flask_tools.pipette.config import PipetteConfig from flask_tools.pipette.verifiers.reaction_energy import ( DFTExecutor, diff --git a/tests/pipette/test_config.py b/tests/pipette/test_config.py index 4f10f63..a859812 100644 --- a/tests/pipette/test_config.py +++ b/tests/pipette/test_config.py @@ -30,6 +30,7 @@ def test_pipette_config_from_yaml_loads_nested_sections(tmp_path) -> None: llm_judge: allow_fail: - reaction_energy + enable_atom_mapping_dict_in_prompt: true url: https://example.test/v1/ model: custom-model api_key: sk-test @@ -71,6 +72,7 @@ def test_pipette_config_from_yaml_loads_nested_sections(tmp_path) -> None: # Test other attrs are set correctly assert config.tool_list == ["basic_smiles_validation", "reaction_energy"] assert config.llm_judge.allow_fail == ["reaction_energy"] + assert config.llm_judge.enable_atom_mapping_dict_in_prompt is True assert config.llm_judge.url == "https://example.test/v1/" assert config.llm_judge.model == "custom-model" assert config.llm_judge.api_key == "sk-test" diff --git a/tests/pipette/test_reaction_energy.py b/tests/pipette/test_reaction_energy.py deleted file mode 100644 index 8af29bc..0000000 --- a/tests/pipette/test_reaction_energy.py +++ /dev/null @@ -1,42 +0,0 @@ -############################################################################### -## Copyright 2025-2026 Lawrence Livermore National Security, LLC. -## See the top-level LICENSE file for details. -## -## SPDX-License-Identifier: Apache-2.0 -############################################################################### - -from __future__ import annotations - -from textwrap import dedent - -from rdkit.Chem import MolFromSmiles -from rdkit.Chem.inchi import MolToInchiKey, MolToInchi - -from flask_tools.pipette.verifiers.reaction_energy import ( - MoleculeEnergyStore, -) - - -# Should this be removed completely b/c DFT calculation is not implemented rn? -def test_molecule_energy_store_from_csv(tmp_path) -> None: - smiles_to_inchi = lambda x: MolToInchi(MolFromSmiles(x)) - path = tmp_path / "fake_molecule_energies.tsv" - ethanol_inchi = smiles_to_inchi(eth_smi := "CCO") - acetaldehyde_inchi = smiles_to_inchi(acetal_smi := "CC=O") - path.write_text( - dedent( - f""" - inchi\tenergy_ev_mol - {ethanol_inchi}\t-7.0 - {acetaldehyde_inchi}\t-10.0 - """ - ).strip() - + "\n", - encoding="utf-8", - ) - - store = MoleculeEnergyStore.from_csv(path) - reactants_energy = store.lookup(eth_smi).energy_ev - products_energy = store.lookup(acetal_smi).energy_ev - assert reactants_energy == -7.0 - assert products_energy - reactants_energy == -3.0 diff --git a/tests/pipette/test_reactions.py b/tests/pipette/test_reactions.py index fc942b0..3316143 100644 --- a/tests/pipette/test_reactions.py +++ b/tests/pipette/test_reactions.py @@ -25,6 +25,10 @@ from flask_tools.pipette.constants import FinalGrade, ReactionGrade from flask_tools.pipette.grade_rxn import grade_reaction, main from flask_tools.pipette.config import load_config, ConfigType, PipetteConfig +from flask_tools.pipette.graph_rxn_mapper.subtractive_reaction_mapper_pipette_tool import ( + GraphBasedBalancerResultDetails, + AtomMappingResultDetails, +) from flask_tools.pipette.pipeline import GradingPipeline from flask_tools.pipette.reaction_fixer import ReactionFixResultDetails from flask_tools.pipette.verifiers import ChargeConservationChecker, ReactionChecker @@ -107,13 +111,53 @@ def test_calls_fixer_caffeine_llm_judge(rxn_name: str) -> None: ), f"{(result.final_grade, fix_result)}" +expected_tool_call_order = { + (NO_GRAPH_BALANCER_TOOLS := "no_graph_balancer"): [ + "basic_smiles_validation", + "exact_match", + "llm_reaction_fix", + # This assumes llm_reaction_fix produced a changed rxn. Otherwise, it would not start over + "basic_smiles_validation", + "exact_match", + "charge_conservation", + "mass_conservation", + "reaction_energy", + ], + (WITH_GRAPH_BALANCER_TOOLS := "graph_balancer"): [ + "basic_smiles_validation", + "exact_match", + "graph_based_balancing", + "basic_smiles_validation", + "exact_match", + "llm_reaction_fix", # T + "basic_smiles_validation", + "exact_match", + "rdt_atom_mapping", + # "llm_atom_mapping", + "charge_conservation", + "mass_conservation", + "reaction_energy", + ], +} + + @pytest.mark.llm_query +@pytest.mark.parametrize( + "tool_set_name", [NO_GRAPH_BALANCER_TOOLS, WITH_GRAPH_BALANCER_TOOLS] +) def test_pipeline_fixed_reaction( + tool_set_name: str, tests_relative_path, ) -> None: # In-depth test that checks that ever single expected tool is called - original = "CCO>>C=C" - fixed = "CCO>>C=C.O" + # Still calls LLM judge so the rxn has to be reasonable + # A simple rxn without dimerization + # original = "CCO>>C=C" + # fixed = "CCO>>C=C.O" + # aldol condensation of acetaldehyde to crotonaldehyde, which both dimerizes and drops water + original = "CC=O>>CC=CC=O" + graph_balanced_hopefully = "CC=O.CC=O>>CC=CC=O" + fixed = "CC=O.CC=O>>CC=CC=O.O" smiles_validation = SpyChecker( "basic_smiles_validation", @@ -163,6 +207,40 @@ def test_pipeline_fixed_reaction( ), ) + graph_balancer = SpyChecker( + "graph_based_balancing", + lambda rxn_smiles, _: ToolResult( + name="graph_based_balancing", + status=ToolStatus.PASS, + comment="Passed", + data=GraphBasedBalancerResultDetails( + original_reaction_smiles=original, + graph_balanced_reaction_smiles=graph_balanced_hopefully, + graph_mapped_reaction_smiles="", + final_balanced_reaction_smiles="", + objective_value=100, + mapper_status="", + reasoning_summary="", + ), + ), + ) + + llm_atom_mapper = SpyChecker( + "llm_atom_mapping", + lambda rxn_smiles, _: ToolResult( + name="llm_atom_mapping", + status=ToolStatus.PASS, + comment="Passed", + data=AtomMappingResultDetails( + input_reaction_smiles=rxn_smiles, + mapped_reaction_smiles="", + product_to_reactant=[], + confidence=0.9, + reasoning_summary="", + ), + ), + ) + class StubReactionFixer: # Have a fixed LLM fixer step to better test LLM judge step def __init__(self) -> None: @@ -172,12 +250,9 @@ def fix( self, rxn_smiles: str, results: list[ToolResult] ) -> ReactionFixResultDetails: self.calls.append(rxn_smiles) - assert [result.name for result in results] == [ - "basic_smiles_validation", - "exact_match", - # "charge_conservation", - # "mass_conservation", - ] + assert [result.name for result in results] == expected_tool_call_order[ + tool_set_name + ][: expected_tool_call_order[tool_set_name].index("llm_reaction_fix")] return ReactionFixResultDetails( original_reaction_smiles=rxn_smiles, fixed_reaction_smiles=fixed, @@ -185,12 +260,26 @@ def fix( added_reactants=[], removed_products=[], added_products=["O"], - reasoning_summary="Removed the agent and balanced both sides with water.", + reasoning_summary="Balanced both sides with water.", ) fixer = StubReactionFixer() + if tool_set_name == NO_GRAPH_BALANCER_TOOLS: + tool_list = [smiles_validation, exact, charge, mass, reaction_energy] + elif tool_set_name == WITH_GRAPH_BALANCER_TOOLS: + tool_list = [ + smiles_validation, + exact, + graph_balancer, + llm_atom_mapper, + charge, + mass, + reaction_energy, + ] + else: + raise ValueError(f"{tool_set_name=}") pipeline = GradingPipeline( - checkers=[smiles_validation, exact, charge, mass, reaction_energy], + checkers=tool_list, config=PipetteConfig(mode="exact"), reaction_fixer=fixer, # noqa ) @@ -218,24 +307,28 @@ def fix( # # assert res_dict == prev_res_dict # assert json.loads(json.dumps(res_dict)) == prev_res_dict - assert smiles_validation.calls == [original, fixed] - assert exact.calls == [original, fixed] - assert fixer.calls == [original] + assert [tool.name for tool in result.results] == expected_tool_call_order[ + tool_set_name + ] + if tool_set_name == NO_GRAPH_BALANCER_TOOLS: + assert smiles_validation.calls == [original, fixed] + assert exact.calls == [original, fixed] + else: + assert smiles_validation.calls == [original, graph_balanced_hopefully, fixed] + assert exact.calls == [original, graph_balanced_hopefully, fixed] + assert fixer.calls == [graph_balanced_hopefully] assert charge.calls == [fixed] assert mass.calls == [fixed] assert reaction_energy.calls == [fixed] - assert [tool.name for tool in result.results] == [ - "basic_smiles_validation", - "exact_match", - "llm_reaction_fix", - "basic_smiles_validation", - "exact_match", - "charge_conservation", - "mass_conservation", - "reaction_energy", - ] - llm_fix_i = 2 - assert result.results[llm_fix_i].data.original_reaction_smiles == original + + llm_fix_i = expected_tool_call_order[tool_set_name].index("llm_reaction_fix") + if tool_set_name == NO_GRAPH_BALANCER_TOOLS: + assert result.results[llm_fix_i].data.original_reaction_smiles == original + else: + assert ( + result.results[llm_fix_i].data.original_reaction_smiles + == graph_balanced_hopefully + ) assert result.results[llm_fix_i].data.fixed_reaction_smiles == fixed try: assert result.final_grade == FinalGrade.LIKELY, result diff --git a/tests/pipette/test_tools.py b/tests/pipette/test_tools.py new file mode 100644 index 0000000..2c06afe --- /dev/null +++ b/tests/pipette/test_tools.py @@ -0,0 +1,87 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +from __future__ import annotations + +from textwrap import dedent +import json +import sys +from pathlib import Path + +import pytest + + +from rdkit.Chem import MolFromSmiles +from rdkit.Chem.inchi import MolToInchiKey, MolToInchi + +from flask_tools.pipette.verifiers.reaction_energy import ( + MoleculeEnergyStore, +) + + +# Should this be removed completely b/c DFT calculation is not implemented rn? +def test_molecule_energy_store_from_csv(tmp_path) -> None: + smiles_to_inchi = lambda x: MolToInchi(MolFromSmiles(x)) + path = tmp_path / "fake_molecule_energies.tsv" + ethanol_inchi = smiles_to_inchi(eth_smi := "CCO") + acetaldehyde_inchi = smiles_to_inchi(acetal_smi := "CC=O") + path.write_text( + dedent( + f""" + inchi\tenergy_ev_mol + {ethanol_inchi}\t-7.0 + {acetaldehyde_inchi}\t-10.0 + """ + ).strip() + + "\n", + encoding="utf-8", + ) + + store = MoleculeEnergyStore.from_csv(path) + reactants_energy = store.lookup(eth_smi).energy_ev + products_energy = store.lookup(acetal_smi).energy_ev + assert reactants_energy == -7.0 + assert products_energy - reactants_energy == -3.0 + + +# Could move this to conftest or some tests/utils.py if another test needs this. +TESTS_DIR = Path(__file__).resolve().parent +REPO_ROOT = TESTS_DIR.parent +DATA_FILE = TESTS_DIR / "data" / "atom_map_rxns.jsonl" +EXPECTED_DIR = TESTS_DIR / "expected_atom_map_res" + +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from flask_tools.pipette.graph_rxn_mapper.subtractive_reaction_mapper_v3 import main + + +def load_cases() -> list[object]: + cases: list[object] = [] + for line_number, raw_line in enumerate(DATA_FILE.read_text().splitlines(), start=1): + if not raw_line.strip(): + continue + entry = json.loads(raw_line) + rxn_smiles = entry["rxn_smiles"] + expected_output = entry["expected_output"] + case_id = entry.get("id", f"line-{line_number}") + cases.append(pytest.param(rxn_smiles, expected_output, id=case_id)) + return cases + + +@pytest.mark.parametrize(("rxn_smiles", "expected_output"), load_cases()) +def test_atom_mapper_output( + rxn_smiles: str, expected_output: str, capsys: pytest.CaptureFixture[str] +) -> None: + # Tests the raw atom mapper CLI version + expected_path = EXPECTED_DIR / expected_output + assert expected_path.exists(), f"Missing expected output snapshot: {expected_path}" + exit_code = main([rxn_smiles]) + captured = capsys.readouterr() + assert exit_code == 0 + actual_output = captured.out + assert actual_output == expected_path.read_text() diff --git a/tests/pipette/update_graph_balancer_expected_outputs.py b/tests/pipette/update_graph_balancer_expected_outputs.py new file mode 100644 index 0000000..095d344 --- /dev/null +++ b/tests/pipette/update_graph_balancer_expected_outputs.py @@ -0,0 +1,163 @@ +############################################################################### +## Copyright 2025-2026 Lawrence Livermore National Security, LLC. +## See the top-level LICENSE file for details. +## +## SPDX-License-Identifier: Apache-2.0 +############################################################################### + +#!/usr/bin/env python3 +""" +Add a rxn smile's expected output from graph balancer to the test data files + +Modes: + - default: fill missing expected_output fields or missing snapshot files + - --redo-all: refresh every entry + - --rxn-smis '...' ['...']: refresh specific reactions, adding new entries if needed +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import subprocess +import sys +from pathlib import Path +import flask_tools.pipette.graph_rxn_mapper + + +TESTS_DIR = Path(__file__).resolve().parent +DATA_FILE = TESTS_DIR / "data" / "atom_map_rxns.jsonl" +EXPECTED_DIR = TESTS_DIR / "expected_atom_map_res" +SCRIPT_DIR = Path(str(flask_tools.pipette.graph_rxn_mapper.__file__)).resolve().parent +MAPPER_SCRIPT = SCRIPT_DIR / "subtractive_reaction_mapper_v3.py" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Populate missing expected_output fields, refresh all snapshots, " + "or refresh selected rxn_smiles entries." + ) + ) + parser.add_argument( + "--redo-all", + action="store_true", + help="Refresh every expected output in tests/data/rxns.jsonl.", + ) + parser.add_argument( + "--rxn-smis", + "--rxn-smiles", + nargs="*", + default=None, + dest="rxn_smis", + help="Refresh only these reactions; new reactions are appended to tests/data/rxns.jsonl.", + ) + return parser.parse_args() + + +def load_entries() -> list[dict[str, object]]: + entries: list[dict[str, object]] = [] + for raw_line in DATA_FILE.read_text().splitlines(): + if raw_line.strip(): + entries.append(json.loads(raw_line)) + return entries + + +def write_entries(entries: list[dict[str, object]]) -> None: + serialized = "\n".join( + json.dumps(entry, separators=(",", ":")) for entry in entries + ) + DATA_FILE.write_text(f"{serialized}\n") + + +def build_snapshot_path(rxn_smiles: str) -> str: + slug = re.sub(r"[^a-z0-9]+", "_", rxn_smiles.lower()).strip("_") + slug = slug or "reaction" + digest = hashlib.sha256(rxn_smiles.encode("utf-8")).hexdigest()[:8] + return f"{slug[:60]}__{digest}.json" + + +def build_entry_id(rxn_smiles: str) -> str: + snapshot_name = Path(build_snapshot_path(rxn_smiles)).stem + return snapshot_name.replace("__", "_") + + +def run_mapper(rxn_smiles: str) -> str: + completed = subprocess.run( + [sys.executable, str(MAPPER_SCRIPT), rxn_smiles], + cwd=SCRIPT_DIR, + check=True, + capture_output=True, + text=True, + ) + return completed.stdout + + +def should_refresh(entry: dict[str, object], args: argparse.Namespace) -> bool: + rxn_smiles = str(entry["rxn_smiles"]) + expected_output = entry.get("expected_output") + if args.redo_all: + return True + if args.rxn_smis is not None: + return rxn_smiles in args.rxn_smis + if not expected_output: + return True + return not (EXPECTED_DIR / str(expected_output)).exists() + + +def append_missing_entries( + entries: list[dict[str, object]], rxn_smis: list[str] | None +) -> set[str]: + if not rxn_smis: + return set() + + existing = {str(entry["rxn_smiles"]) for entry in entries} + added: set[str] = set() + for rxn_smiles in rxn_smis: + if rxn_smiles in existing: + continue + entries.append( + { + "id": build_entry_id(rxn_smiles), + "rxn_smiles": rxn_smiles, + "expected_output": build_snapshot_path(rxn_smiles), + } + ) + existing.add(rxn_smiles) + added.add(rxn_smiles) + return added + + +def main() -> int: + args = parse_args() + entries = load_entries() + EXPECTED_DIR.mkdir(parents=True, exist_ok=True) + + added_entries = append_missing_entries(entries, args.rxn_smis) + + for entry in entries: + if not should_refresh(entry, args): + continue + + rxn_smiles = str(entry["rxn_smiles"]) + expected_output = str( + entry.get("expected_output") or build_snapshot_path(rxn_smiles) + ) + entry["expected_output"] = expected_output + + snapshot_path = EXPECTED_DIR / expected_output + snapshot_path.parent.mkdir(parents=True, exist_ok=True) + snapshot_path.write_text(run_mapper(rxn_smiles)) + + if rxn_smiles in added_entries: + print(f"added {entry['id']} {rxn_smiles} -> {expected_output}") + else: + print(f"updated {entry['id']} {rxn_smiles} -> {expected_output}") + + write_entries(entries) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())