diff --git a/README.md b/README.md
index 768f8619..def5301e 100644
--- a/README.md
+++ b/README.md
@@ -14,6 +14,7 @@ A Python library for structural cheminformatics.
> This library is still in early stages of development.
+- `compounds.standardization`: standardize chemical records.
- `databases.klifs`: utilities to query the KLIFS database, offline or online.
- `io`: read and write molecules from/to files.
- `structure.pocket`: identification and analysis of protein (sub)pockets.
@@ -34,6 +35,7 @@ The Documentation will be available soon.
- Jaime Rodríguez-Guerra, PhD
- Dominique Sydow
- Dennis Köser, Annie Pham, Enes Kurnaz, Julian Pipart (structural superposition, 2020)
+- Allen Dumler (standardizer, 2021)
# Acknowledgements
diff --git a/docs/tutorials/data/standardization_test_data.csv b/docs/tutorials/data/standardization_test_data.csv
new file mode 100644
index 00000000..eab0c4fc
--- /dev/null
+++ b/docs/tutorials/data/standardization_test_data.csv
@@ -0,0 +1,961 @@
+IDs;Names;SMILEs;HUMANS;RODENTS;NON-RODENTS;;;;;
+1;(R)-Roscovitine;CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1;0;1;0;;;;;
+2;17-Methyltestosterone;CC1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;0;1;0;;;;;
+3;1-alpha-Hydroxycholecalciferol;CC(C)CCCC(C)C1CCC2C(CCCC12C)=CC=C1CC(O)CC(O)C1=C;1;0;0;;;;;
+4;2,3-Dimercaptosuccinic acid;OC(=O)C(S)C(S)C(O)=O;1;1;0;;;;;
+5;2,4,6-Trinitrotoluene;Cc1c(cc(cc1N(=O)=O)N(=O)=O)N(=O)=O;1;0;0;;;;;
+6;2-Deoxy-D-glucose;OCC1OC(O)CC(O)C1O;1;1;0;;;;;
+7;2'-fluoro-5-methylarabinosyluracil;CC1=CN(C2OC(CO)C(O)C2F)C(=O)NC1=O;1;0;0;;;;;
+8;2-Methoxyestradiol;COc1cc2C3CCC4(C)C(O)CCC4C3CCc2cc1O;1;1;0;;;;;
+9;4-aminobenzoic acid;Nc1ccc(cc1)C(O)=O;0;1;0;;;;;
+10;4-Hydroxytamoxifen;CCC(c1ccccc1)=C(c1ccc(O)cc1)c1ccc(OCCN(C)C)cc1;1;1;0;;;;;
+11;5 fluorouracil;FC1=CNC(=O)NC1=O;1;1;1;;;;;
+12;5-Azacitidine;NC1=NC(=O)N(C=N1)C1OC(CO)C(O)C1O;1;1;0;;;;;
+13;5-Bromouracil;OCC1OC(CC1O)N1C=C(Br)C(=O)NC1=O;0;1;0;;;;;
+14;5-fluoro-2'-deoxyuridine;OCC1OC(CC1O)N1C=C(F)C(=O)NC1=O;1;1;0;;;;;
+15;6-Mercaptopurine;Sc1ncnc2[nH]c[nH]c12;1;1;0;;;;;
+16;Acadesine;NC(=O)c1ncn(C2OC(CO)C(O)C2O)c1N;0;1;0;;;;;
+17;Acarbose;CC1OC(OC2C(CO)OC(OC3C(CO)OC(O)C(O)C3O)C(O)C2O)C(O)C(O)C1NC1C=C(CO)C(O)C(O)C1O;1;1;0;;;;;
+18;Acebutolol;CCCC(=O)Nc1ccc(OCC(O)CNC(C)C)c(c1)C(C)=O;1;1;0;;;;;
+19;Acenocoumarol;CC(=O)CC(c1ccc(cc1)N(=O)=O)C1=C(O)Oc2ccccc2C1=O;1;0;0;;;;;
+20;Acetamide;CC(N)=O;0;1;0;;;;;
+21;Acetaminophen;CC(=O)Nc1ccc(O)cc1;1;1;1;;;;;
+22;Acetazolamide;CC(=O)Nc1nnc(s1)S(N)(=O)=O;1;1;1;;;;;
+23;Acetic acid;CC(O)=O;1;1;1;;;;;
+24;Acetohexamide;CC(=O)c1ccc(cc1)S(=O)(=O)NC(=O)NC1CCCCC1;1;0;0;;;;;
+25;Acetohydroxamic acid;CC(=O)NO;0;1;0;;;;;
+26;Acetrizoate Sodium;CC(=O)Nc1c(I)cc(I)c(C(O)=O)c1I;0;1;0;;;;;
+27;Acetylcholine;CC(=O)OCC[N+](C)(C)C;0;1;1;;;;;
+28;Acetylcysteine;CC(=O)NC(CS)C(O)=O;1;1;0;;;;;
+29;Acetyl-L-carnitine;CC(=O)OC(CC(O)=O)C[N+](C)(C)C;0;1;0;;;;;
+30;Acetylsalicylic acid;CC(=O)Oc1ccccc1C(O)=O;1;1;1;;;;;
+31;Acitretin;COc1cc(C)c(C=CC(C)=CC=CC(C)=CC(O)=O)c(C)c1C;1;1;0;;;;;
+32;Acyclovir;NC1=NC(=O)c2ncn(COCCO)c2N1;1;1;0;;;;;
+33;Adefovir dipivoxil;CC(C)(C)C(=O)OCOP(=O)(COCCn1cnc2c(N)ncnc12)OCOC(=O)C(C)(C)C;1;1;0;;;;;
+34;Adenine;Nc1ncnc2[nH]cnc12;0;1;0;;;;;
+35;Adenosine;n2c1c(ncnc1n(c2)[C@@H]3O[C@@H]([C@@H](O)[C@H]3O)CO)N;1;1;1;;;;;
+36;Adenosine triphosphate;Nc1ncnc2n(cnc12)C1OC(COP(O)(=O)OP(O)(=O)OP(O)(O)=O)C(O)C1O;0;1;0;;;;;
+37;AET;NCCSC(N)=N;0;1;0;;;;;
+38;Ajmaline;CCC1C2CC3C4N(C)c5ccccc5C44CC(C2C4O)N3C1O;1;0;0;;;;;
+39;Alanosine;NC(CN(O)N=O)C(O)=O;0;1;0;;;;;
+40;Alatrofloxacin mesylate;CC(N)C(=O)NC(C)C(=O)NC1C2CN(CC12)c1nc2N(C=C(C(O)=O)C(=O)c2cc1F)c1ccc(F)cc1F;1;0;0;;;;;
+41;Albendazole;CCCSc1ccc2nc(NC(=O)OC)[nH]c2c1;1;0;0;;;;;
+42;Alfuzosin;COc1cc2[nH]c(nc(N)c2cc1OC)N(C)CCCNC(=O)C1CCCO1;1;0;0;;;;;
+43;Alitretinoin;CC(C=CC1=C(C)CCCC1(C)C)=CC=CC(C)=CC(O)=O;1;1;0;;;;;
+44;Allantoin;NC(=O)NC1NC(=O)NC1=O;0;1;0;;;;;
+45;Allobarbital;C=CCC1(CC=C)C(=O)NC(=O)NC1=O;0;1;0;;;;;
+46;Allopurinol;O=C1N=CNc2cn[nH]c12;1;1;0;;;;;
+47;Alpha-tocopherol acetate;CC(C)CCCC(C)CCCC(C)CCCC1(C)CCc2c(C)c(OC(C)=O)c(C)c(C)c2O1;0;1;0;;;;;
+48;Alverine;CCN(CCCc1ccccc1)CCCc1ccccc1;1;0;0;;;;;
+49;Amantadine;NC12CC3CC(CC(C3)C1)C2;0;1;0;;;;;
+50;ambrisentan;COC(C(Oc1nc(C)cc(C)n1)C(O)=O)(c1ccccc1)c1ccccc1;0;0;1;;;;;
+51;Ambroxol;Nc1c(Br)cc(Br)cc1CNC1CCC(O)CC1;0;1;0;;;;;
+52;Amikacin;NCCC(O)C(=O)NC1CC(N)C(OC2OC(CN)C(O)C(O)C2O)C(O)C1OC1OC(CO)C(O)C(N)C1O;1;0;1;;;;;
+53;Amiloride;NC(=N)NC(=O)c1nc(Cl)c(N)nc1N;1;1;0;;;;;
+54;Amineptine;OC(=O)CCCCCCNC1c2ccccc2CCc2ccccc12;1;1;0;;;;;
+55;Amino acid;NCC(O)=O;1;1;1;;;;;
+56;Aminoglutethimide;CCC1(CCC(=O)NC1=O)c1ccc(N)cc1;0;1;0;;;;;
+57;Aminoguanidine;N\N=C(\N)N;0;1;0;;;;;
+58;Aminophylline;CN1C(=O)N(C)c2[nH]c[nH]c2C1=O;1;1;0;;;;;
+59;Aminopyrine;CN(C)C1=C(C)N(C)N(c2ccccc2)C1=O;1;1;0;;;;;
+60;Amiodarone;CCCCc1oc2ccccc2c1C(=O)c1cc(I)c(OCCN(CC)CC)c(I)c1;1;1;1;;;;;
+61;Amitriptyline;CN(C)CC\C=C1\c2ccccc2CCc2ccccc12;1;1;0;;;;;
+62;Amlodipine;CCOC(=O)C1=C(COCCN)NC(C)=C(C1c1ccccc1Cl)C(=O)OC;0;1;0;;;;;
+63;Amobarbital;CCC1(CCC(C)C)C(=O)NC(=O)NC1=O;0;1;0;;;;;
+64;Amodiaquin;CCN(CC)Cc1cc(Nc2ccnc3cc(Cl)ccc23)ccc1O;1;1;0;;;;;
+65;Amoxicillin;CC1(C)SC2C(NC(=O)C(N)c3ccc(O)cc3)C(=O)N2C1C(O)=O;1;0;0;;;;;
+66;Amphetamine;CC(N)Cc1ccccc1;1;1;0;;;;;
+67;Amphotericin B;CC1OC(=O)CC(O)CC(O)CCC(O)C(O)CC(O)CC2(O)CC(O)C(C(CC(OC3OC(C)C(O)C(N)C3O)C=CC=CC=CC=CC=CC=CC=CC(C)C(O)C1C)O2)C(O)=O;1;1;0;;;;;
+68;Ampicillin;CC1(C)SC2C(NC(=O)C(N)c3ccccc3)C(=O)N2C1C(O)=O;1;1;0;;;;;
+69;Amprenavir;CC(C)CN(CC(O)C(Cc1ccccc1)NC(=O)OC1CCOC1)S(=O)(=O)c1ccc(N)cc1;1;0;0;;;;;
+70;Amrinone;NC1=CC(=CNC1=O)c1ccncc1;0;1;0;;;;;
+71;Amsacrine;COc1cc(NS(C)(=O)=O)ccc1Nc1c2ccccc2nc2ccccc12;1;0;0;;;;;
+72;Anastrozole;CC(C)(C#N)c1cc(Cn2cncn2)cc(c1)C(C)(C)C#N;1;0;0;;;;;
+73;Anethole Trithione;COc1ccc(cc1)C1=CC(=S)SS1;0;1;0;;;;;
+74;Anthralin;Oc1cccc2cc3cccc(O)c3c(O)c12;1;1;0;;;;;
+75;Apomorphine;CN1CCc2cccc-3c2C1Cc1ccc(O)c(O)c-31;0;1;0;;;;;
+76;Aprobarbital;CC(C)C1(CC=C)C(=O)NC(=O)NC1=O;0;1;0;;;;;
+77;Ascorbic acid;OCC(O)C1OC(O)=C(O)C1=O;1;1;1;;;;;
+78;Astaxanthin;CC(C=CC=C(C)C=CC1=C(C)C(=O)C(O)CC1(C)C)=CC=CC=C(C)C=CC=C(C)C=CC1=C(C)C(=O)C(O)CC1(C)C;0;1;0;;;;;
+79;Atazanavir;COC(=O)NC(C(=O)NC(Cc1ccccc1)C(O)CN(Cc1ccc(cc1)-c1ccccn1)NC(=O)C(NC(=O)OC)C(C)(C)C)C(C)(C)C;1;0;0;;;;;
+80;Atenolol;CC(C)NCC(O)COc1ccc(CC(N)=O)cc1;1;0;0;;;;;
+81;Atiprimod;CCCC1(CCC)CCC2(CCN(CCCN(CC)CC)C2)CC1;1;0;0;;;;;
+82;Atomoxetine hydrochloride;CNCCC(Oc1ccccc1C)c1ccccc1;1;0;0;;;;;
+83;Atorvastatin;CC(C)C1=C(C(=O)Nc2ccccc2)C(c2ccccc2)=C(N1CCC(O)CC(O)CC(O)=O)c1ccc(F)cc1;1;1;1;;;;;
+84;Atropine;CN1C2CCC1CC(C2)OC(=O)C(CO)c1ccccc1;0;1;0;;;;;
+85;avasimibe;CC(C)c1cc(C(C)C)c(CC(=O)NS(=O)(=O)Oc2c(cccc2C(C)C)C(C)C)c(c1)C(C)C;1;1;0;;;;;
+86;Azathioprine;Cn1cnc(c1Sc1ncnc2ncnc12)N(=O)=O;1;1;0;;;;;
+87;AZD6244;Cn1cnc2c(F)c(Nc3ccc(Br)cc3Cl)c(cc12)C(=O)NOCCO;0;1;0;;;;;
+88;Azithromycin;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(O)CC(C)CN(C)C(C)C(O)C1(C)O;1;0;0;;;;;
+89;Azlocillin;CC1(C)SC2C(NC(=O)C(NC(=O)N3CCNC3=O)c3ccccc3)C(=O)N2C1C(O)=O;1;0;0;;;;;
+90;Aztreonam;CC1C(NC(=O)C(=NOC(C)(C)C(O)=O)c2csc(N)n2)C(=O)N1S(O)(=O)=O;1;0;0;;;;;
+91;Bacitracin;CCC(C)C(C)C1=NC(CS1)C(=O)NC(CC(C)C)C(=O)NC(CCC(O)=O)C(=O)NC(C(C)CC)C(=O)NC1CCCCNC(=O)C(CC(O)=O)NC(=O)C(Cc2c[nH]c[nH]2)NC(=O)C(Cc2ccccc2)NC(=O)C(NC(=O)C(CCCN)NC1=O)C(C)CC;0;1;0;;;;;
+92;Baclofen;NCC(CC(O)=O)c1ccc(Cl)cc1;1;0;0;;;;;
+93;Barbital;CCC1(CC)C(=O)NC(=O)NC1=O;0;1;0;;;;;
+94;Bendamustine;Cn1c(CCCC(O)=O)nc2cc(ccc12)N(CCCl)CCCl;1;0;0;;;;;
+95;Bendazac;OC(=O)COc1nn(Cc2ccccc2)c2ccccc12;1;0;0;;;;;
+96;Benorylate;CC(=O)Nc1ccc(OC(=O)c2ccccc2OC(C)=O)cc1;0;1;0;;;;;
+97;Benoxaprofen;CC(C(O)=O)c1ccc2oc(nc2c1)-c1ccc(Cl)cc1;1;1;0;;;;;
+98;Benzarone;CCc1oc2ccccc2c1C(=O)c1ccc(O)cc1;1;1;0;;;;;
+99;Benzbromarone;CCc1oc2ccccc2c1C(=O)c1cc(Br)c(O)c(Br)c1;1;1;0;;;;;
+100;Benziodarone;CCc1oc2ccccc2c1C(=O)c1cc(I)c(O)c(I)c1;0;1;0;;;;;
+101;Benzoyl peroxide;O=C(OOC(=O)c1ccccc1)c1ccccc1;0;1;0;;;;;
+102;Beraprost;CC#CCC(C)C(O)C=CC1C(O)CC2Oc3c(CCCC(O)=O)cccc3C12;0;1;0;;;;;
+103;Beta-Carotene;CC(C=CC=C(C)C=CC1=C(C)CCCC1(C)C)=CC=CC=C(C)C=CC=C(C)C=CC1=C(C)CCCC1(C)C;1;1;1;;;;;
+104;Betaine;C[N+](C)(C)CC(O)=O;0;1;0;;;;;
+105;beta-Lapachone;CC1(C)CCC2=C(O1)c1ccccc1C(=O)C2=O;1;0;0;;;;;
+106;Betamethasone;CC1CC2C3CCC4=CC(=O)C=CC4(C)C3(F)C(O)CC2(C)C1(O)C(=O)COP(O)(O)=O;0;1;0;;;;;
+107;beta-Sitosterol;CCC(CCC(C)C1CCC2C3CC=C4CC(O)CCC4(C)C3CCC12C)C(C)C;1;1;0;;;;;
+108;Betulinic acid;CC(=C)C1CCC2(CCC3(C)C(CCC4C5(C)CCC(O)C(C)(C)C5CCC34C)C12)C(O)=O;1;0;0;;;;;
+109;Bezafibrate;CC(C)(Oc1ccc(CCNC(=O)c2ccc(Cl)cc2)cc1)C(O)=O;1;1;0;;;;;
+110;Bicalutamide;CC(O)(CS(=O)(=O)c1ccc(F)cc1)C(=O)Nc1ccc(C#N)c(c1)C(F)(F)F;1;0;0;;;;;
+111;Bifonazole;c1ccc(cc1)C(c1ccc(cc1)-c1ccccc1)n1ccnc1;0;1;0;;;;;
+112;Biotin;OC(=O)CCCCC1SCC2NC(=O)NC12;1;0;0;;;;;
+113;Bisoprolol;CC(C)NCC(O)COc1ccc(COCCOC(C)C)cc1;1;0;0;;;;;
+114;Bleomycin;CC(O)C(NC(=O)C(C)C(O)C(C)NC(=O)C(NC(=O)c1nc(nc(N)c1C)C(CC(N)=O)NCC(N)C(N)=O)C(OC1OC(CO)C(O)C(O)C1OC1OC(CO)C(O)C(OC(N)=O)C1O)c1cnc[nH]1)C(=O)NCCc1nc(cs1)-c1nc(cs1)C(O)=O;1;1;0;;;;;
+115;Bortezomib;CC(C)CC(NC(=O)C(Cc1ccccc1)NC(=O)c1cnccn1)B(O)O;1;1;0;;;;;
+116;Bosentan;COc1ccccc1Oc1c(NS(=O)(=O)c2ccc(cc2)C(C)(C)C)nc(nc1OCCO)-c1ncccn1;1;1;1;;;;;
+117;Bromfenac;Nc1c(CC(O)=O)cccc1C(=O)c1ccc(Br)cc1;1;0;0;;;;;
+118;Bromisovalum;CC(C)C(Br)C(=O)NC(N)=O;1;1;0;;;;;
+119;Bromocriptine;CC(C)CC1N2C(=O)C(NC(=O)C3CN(C)C4Cc5c(Br)[nH]c6cccc(C4=C3)c56)(OC2(O)C2CCCN2C1=O)C(C)C;0;1;0;;;;;
+120;Brotizolam;Cc1nnc2CN=C(c3ccccc3Cl)c3cc(Br)sc3-n12;0;1;0;;;;;
+121;Bucladesine;CCCC(=O)Nc1ncnc2n(cnc12)C1OC2COP(O)(=O)OC2C1OC(=O)CCC;1;1;1;;;;;
+122;Budesonide;CCCC1OC2CC3C4CCC5=CC(=O)C=CC5(C)C4C(O)CC3(C)C2(O1)C(=O)CO;1;1;0;;;;;
+123;Bumetanide;CCCCNc1cc(cc(c1Oc1ccccc1)S(N)(=O)=O)C(O)=O;0;0;1;;;;;
+124;Bupivacaine;CCCCN1CCCCC1C(=O)Nc1c(C)cccc1C;0;1;0;;;;;
+125;Buprenorphine;COc1c(O)ccc2CC3N(CCC4(CC5(CCC34CC5C(C)(O)C(C)(C)C)OC)c12)CC1CC1;1;1;0;;;;;
+126;Bupropion;CC(NC(C)(C)C)C(=O)c1cccc(Cl)c1;1;0;0;;;;;
+127;Buspirone;O=C1CC2(CCCC2)CC(=O)N1CCCCN1CCN(CC1)c1ncccn1;1;0;0;;;;;
+128;Busulfan;CS(=O)(=O)OCCCCOS(C)(=O)=O;1;1;0;;;;;
+129;Butachlor;CCCCOCN(C(=O)CCl)c1c(CC)cccc1CC;0;1;0;;;;;
+130;Butalbital;CC(C)CC1(CC=C)C(=O)NC(=O)NC1=O;1;0;0;;;;;
+131;Butethal;CCCCC1(CC)C(=O)NC(=O)NC1=O;0;1;0;;;;;
+132;Caffeine;CN1C(=O)N(C)c2ncn(C)c2C1=O;1;1;0;;;;;
+133;Calcitriol;CC(CCCC(C)(C)O)C1CCC2C(CCCC12C)=CC=C1CC(O)CC(O)C1=C;1;1;0;;;;;
+134;Camptothecin;CCC1(O)C(=O)OCC2=C1C=C1N(Cc3cc4ccccc4[nH]c13)C2=O;1;1;0;;;;;
+135;Candesartan cilexetil;CCOc1nc2cccc(C(=O)OC(C)OC(=O)OC3CCCCC3)c2n1Cc1ccc(cc1)-c1ccccc1-c1nn[nH]n1;1;1;0;;;;;
+136;Cannabidiol;CCCCCc1cc(O)c(C2C=C(C)CCC2C(C)=C)c(O)c1;1;1;0;;;;;
+137;Capecitabine;CCCCCOC(=O)NC1=NC(=O)N(C=C1F)C1OC(C)C(O)C1O;1;1;0;;;;;
+138;Capsaicin;COc1cc(CNC(=O)CCCCC=CC(C)C)ccc1O;1;1;0;;;;;
+139;Captopril;CC(CS)C(=O)N1CCCC1C(O)=O;1;1;0;;;;;
+140;Carbamazepine;NC(=O)N1c2ccccc2C=Cc2ccccc12;1;1;0;;;;;
+141;Carbamylcholine;C[N+](C)(C)CCOC(N)=O;0;1;0;;;;;
+142;Carbaryl;CNC(=O)Oc1cccc2ccccc12;0;1;0;;;;;
+143;Carbenicillin disodium;CC1(C)SC2C(NC(=O)C(C(O)=O)c3ccccc3)C(=O)N2C1C(O)=O;1;0;0;;;;;
+144;Carbenoxolone;CC1(C)C(CCC2(C)C1CCC1(C)C2C(=O)C=C2C3CC(C)(CCC3(C)CCC12C)C(O)=O)OC(=O)CCC(O)=O;0;1;0;;;;;
+145;Carbimazole;CCOC(=O)N1C=CN(C)C1=S;1;0;0;;;;;
+146;Carboplatin;OC(=O)C1(CCC1)C(O)=O;1;1;0;;;;;
+147;Carbutamide;CCCCNC(=O)NS(=O)(=O)c1ccc(N)cc1;0;1;0;;;;;
+148;Cariporide;CC(C)c1ccc(cc1S(C)(=O)=O)C(=O)NC(N)=N;0;1;0;;;;;
+149;Carisoprodol;CCCC(C)(COC(N)=O)COC(=O)NC(C)C;0;1;0;;;;;
+150;Carmustine;ClCCNC(=O)N(CCCl)N=O;1;1;0;;;;;
+151;Carnitine;C[N+](C)(C)CC(O)CC(O)=O;1;1;1;;;;;
+152;Carprofen;CC(C(O)=O)c1ccc-2c(Nc3ccc(Cl)cc-23)c1;0;0;1;;;;;
+153;Carvedilol;COc1ccccc1OCCNCC(O)COc1cccc2Nc3ccccc3-c12;1;1;0;;;;;
+154;Catechin;OC1Cc2c(O)cc(O)cc2OC1c1ccc(O)c(O)c1;1;1;1;;;;;
+155;Cefaclor;NC(C(=O)NC1C2SCC(Cl)=C(N2C1=O)C(O)=O)c1ccccc1;1;0;0;;;;;
+156;Cefadroxil;CC1=C(N2C(SC1)C(NC(=O)C(N)c1ccc(O)cc1)C2=O)C(O)=O;1;0;0;;;;;
+157;Cefazolin;Cc1nnc(SCC2=C(N3C(SC2)C(NC(=O)Cn2cnnn2)C3=O)C(O)=O)s1;1;1;0;;;;;
+158;Cefixime;Nc1nc(cs1)C(=NOCC(O)=O)C(=O)NC1C2SCC(C=C)=C(N2C1=O)C(O)=O;1;0;0;;;;;
+159;Cefoperazone;CCN1CCN(C(=O)NC(C(=O)NC2C3SCC(CSc4nnnn4C)=C(N3C2=O)C(O)=O)c2ccc(O)cc2)C(=O)C1=O;1;1;0;;;;;
+160;Cefotaxime;CON=C(C(=O)NC1C2SCC(COC(C)=O)=C(N2C1=O)C(O)=O)c1csc(N)n1;1;0;0;;;;;
+161;Cefotetan;COC1(NC(=O)C2S\C(S2)=C(/C(N)=O)C(O)=O)C2SCC(CSc3nnnn3C)=C(N2C1=O)C(O)=O;1;0;0;;;;;
+162;Cefprozil;CC=CC1=C(N2C(SC1)C(NC(=O)C(N)c1ccc(O)cc1)C2=O)C(O)=O;1;0;0;;;;;
+163;Ceftriaxone;CON=C(C(=O)NC1C2SCC(CSC3=NC(=O)C(=O)NN3C)=C(N2C1=O)C(O)=O)c1csc(N)n1;1;1;1;;;;;
+164;Celecoxib;Cc1ccc(cc1)-c1cc(nn1-c1ccc(cc1)S(N)(=O)=O)C(F)(F)F;1;1;0;;;;;
+165;Cephaloridine;OC(=O)C1=C(CSC2C(NC(=O)Cc3cccs3)C(=O)N12)C[n+]1ccccc1;1;0;0;;;;;
+166;Cephalosporin;CC(=O)OCC1=C(N2C(SC1)C(NC(=O)CCCC(N)C(O)=O)C2=O)C(O)=O;1;0;0;;;;;
+167;Cephalothin;CC(=O)OCC1=C(N2C(SC1)C(NC(=O)Cc1cccs1)C2=O)C(O)=O;1;0;0;;;;;
+168;Cerivastatin;COCc1c(nc(C(C)C)c(C=CC(O)CC(O)CC(O)=O)c1-c1ccc(F)cc1)C(C)C;1;1;1;;;;;
+169;Cerivastatin sodium;COCc1c(nc(C(C)C)c(C=CC(O)CC(O)CC(O)=O)c1-c1ccc(F)cc1)C(C)C;0;1;1;;;;;
+170;Cetirizine;OC(=O)COCCN1CCN(CC1)C(c1ccccc1)c1ccc(Cl)cc1;1;0;0;;;;;
+171;Cetrimonium bromide;CCCCCCCCCCCCCCCC[N+](C)(C)C;0;1;0;;;;;
+172;CF101;CNC(=O)C1OC(C(O)C1O)n1cnc2c(NCc3cccc(I)c3)ncnc12;0;1;0;;;;;
+173;Chenodiol;O=C(O)CC[C@H]([C@H]1CC[C@@H]2[C@]1(C)CC[C@H]4[C@H]2[C@H](O)C[C@@H]3C[C@H](O)CC[C@@]34C)C;1;1;1;;;;;
+174;Chloral hydrate;OC(O)C(Cl)(Cl)Cl;1;1;0;;;;;
+175;Chlorambucil;OC(=O)CCCc1ccc(cc1)N(CCCl)CCCl;1;1;0;;;;;
+176;Chloramine-T;Cc1ccc(cc1)S(=O)(=O)NCl;0;1;1;;;;;
+177;Chloramphenicol;OCC(NC(=O)C(Cl)Cl)C(O)c1ccc(cc1)N(=O)=O;1;1;0;;;;;
+178;Chlordiazepoxide;CN=C1CN(O)C(c2ccccc2)=C2C=C(Cl)C=CC2=N1;0;1;0;;;;;
+179;Chlorguanide;CC(C)NC(=N)NC(=N)Nc1ccc(Cl)cc1;1;0;0;;;;;
+180;Chlormadinone acetate;CC(=O)OC1(CCC2C3C=C(Cl)C4=CC(=O)CCC4(C)C3CCC12C)C(C)=O;1;1;1;;;;;
+181;Chlormethiazole;Cc1ncsc1CCCl;1;1;0;;;;;
+182;Chlormezanone;CN1C(c2ccc(Cl)cc2)S(=O)(=O)CCC1=O;1;0;0;;;;;
+183;Chloroquine;CCN(CC)CCCC(C)Nc1cc[nH]c2cc(Cl)ccc12;1;1;1;;;;;
+184;Chloroxylenol;Cc1cc(O)cc(C)c1Cl;0;1;0;;;;;
+185;Chlorphenamine;CN(C)CCC(c1ccc(Cl)cc1)c1ccccn1;1;0;0;;;;;
+186;Chlorpromazine;CN(C)CCCN1c2ccccc2Sc2ccc(Cl)cc12;1;1;1;;;;;
+187;Chlorpropamide;CCCNC(=O)NS(=O)(=O)c1ccc(Cl)cc1;1;1;0;;;;;
+188;Chlortetracycline;CN(C)C1C2CC3C(C(=O)c4c(O)ccc(Cl)c4C3(C)O)=C(O)C2(O)C(=O)C(C(N)=O)=C1O;0;1;1;;;;;
+189;Chlorzoxazone;Oc1nc2cc(Cl)ccc2o1;1;0;0;;;;;
+190;Cholestyramine;CC(C)(Oc1ccc(cc1)C(=O)c1ccc(Cl)cc1)C(=O)NCCS(O)(=O)=O;1;1;0;;;;;
+191;Chondroitin sulfate;CC(=O)NC1C(O)OC(OS(O)(=O)=O)C(O)C1OC1OC(C(O)C(O)C1O)C(O)=O;0;1;0;;;;;
+192;Cidofovir;NC1=NC(=O)N(CC(CO)OCP(O)(O)=O)C=C1;1;0;0;;;;;
+193;Cimetidine;CN=C(NCCSCc1nc[nH]c1C)NC#N;1;1;1;;;;;
+194;Cinchophen;OC(=O)c1cc(nc2ccccc12)-c1ccccc1;1;0;1;;;;;
+195;Cinoxacin;CCN1N=C(C(O)=O)C(=O)c2cc3OCOc3cc12;1;0;0;;;;;
+196;Ciprofibrate;CC(C)(Oc1ccc(cc1)C1CC1(Cl)Cl)C(O)=O;1;1;1;;;;;
+197;Ciprofloxacin;OC(=O)C1=CN(C2CC2)c2cc(N3CCNCC3)c(F)cc2C1=O;1;1;0;;;;;
+198;Cisapride;COC1CN(CCCOc2ccc(F)cc2)CCC1NC(=O)c1cc(Cl)c(N)cc1OC;1;0;0;;;;;
+199;Citalopram;CN(C)CCCC1(OCc2cc(ccc12)C#N)c1ccc(F)cc1;0;1;0;;;;;
+200;Citric acid;OC(=O)CC(O)(CC(O)=O)C(O)=O;0;1;0;;;;;
+201;Cladribine;Nc1nc(Cl)nc2n(cnc12)C1CC(O)C(CO)O1;1;0;0;;;;;
+202;Clarithromycin;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(CC(C)C(=O)C(C)C(O)C1(C)O)OC;1;1;0;;;;;
+203;Clavulanic acid;OCC=C1OC2CC(=O)N2C1C(O)=O;1;0;0;;;;;
+204;Clindamycin;CCCC1CC(N(C)C1)C(=O)NC(C(C)Cl)C1OC(SC)C(O)C(O)C1O;1;0;0;;;;;
+205;Clioquinol;Oc1c(I)cc(Cl)c2cccnc12;0;1;0;;;;;
+206;Clobazam;CN1C(=O)CC(=O)N(c2ccccc2)c2cc(Cl)ccc12;1;0;0;;;;;
+207;Clodronic Acid;OP(O)(=O)C(Cl)(Cl)P(O)(O)=O;0;1;0;;;;;
+208;Clofarabine;Nc1nc(Cl)nc2n(cnc12)C1OC(CO)C(O)C1F;1;0;0;;;;;
+209;Clofazimine;CC(C)N=C1C=C2N(c3ccc(Cl)cc3)c3ccccc3N=C2C=C1Nc1ccc(Cl)cc1;1;0;0;;;;;
+210;Clofibrate;CCOC(=O)C(C)(C)Oc1ccc(Cl)cc1;1;1;1;;;;;
+211;Clomiphene;CCN(CC)CCOc1ccc(cc1)C(c1ccccc1)=C(Cl)c1ccccc1;1;0;0;;;;;
+212;Clomiphene citrate;CCN(CC)CCOc1ccc(cc1)C(c1ccccc1)=C(Cl)c1ccccc1;0;1;0;;;;;
+213;Clonidine;Clc1cccc(Cl)c1NC1=NCCN1;1;1;0;;;;;
+214;Clopidogrel;COC(=O)C(N1CCc2sccc2C1)c1ccccc1Cl;1;1;0;;;;;
+215;Clotiazepam;CCc1cc2C(=NCC(=O)N(C)c2s1)c1ccccc1Cl;1;0;0;;;;;
+216;Clotrimazole;Clc1ccccc1C(c1ccccc1)(c1ccccc1)n1ccnc1;0;1;0;;;;;
+217;Clozapine;CN1CCN(CC1)C1=C2C=CC=CC2=Nc2ccc(Cl)cc2N1;1;0;0;;;;;
+218;Co-amoxiclav;CC1(C)SC2C(NC(=O)C(N)c3ccc(O)cc3)C(=O)N2C1C(O)=O;1;0;0;;;;;
+219;Cocaine;COC(=O)C1C(CC2CCC1N2C)OC(=O)c1ccccc1;1;1;0;;;;;
+220;Codeine;COc1ccc2CC3C4C=CC(O)C5Oc1c2C45CCN3C;1;1;0;;;;;
+221;Coenzyme Q10;COC1=C(OC)C(=O)C(CC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CCC=C(C)CC\C=C(\C)C)=C(C)C1=O;0;1;0;;;;;
+222;Colchicine;COC1=CC=C2C(=CC1=O)C(CCc1cc(OC)c(OC)c(OC)c21)NC(C)=O;1;1;0;;;;;
+223;Cordycepin;Nc1ncnc2n(cnc12)C1OC(CO)CC1O;0;1;0;;;;;
+224;cortisone acetate;CC(=O)OCC(=O)C1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3C(=O)CC12C;0;1;0;;;;;
+225;Creatine;CN(CC(O)=O)C(N)=N;0;1;0;;;;;
+226;Curcumin;COc1cc(C=CC(=O)CC(=O)C=Cc2ccc(O)c(OC)c2)ccc1O;1;1;0;;;;;
+227;Cyclandelate;CC1CC(CC(C)(C)C1)OC(=O)C(O)c1ccccc1;0;1;0;;;;;
+228;Cyclobarbital;CCC1(C(=O)NC(=O)NC1=O)C1=CCCCC1;0;1;0;;;;;
+229;Cyclobutyrol;CCC(C(O)=O)C1(O)CCCCC1;0;1;0;;;;;
+230;cyclopamine;CC1CNC2C(C1)OC1(CCC3C4CC=C5CC(O)CCC5(C)C4CC3=C1C)C2C;1;0;0;;;;;
+231;Cyclophosphamide;ClCCN(CCCl)P1(=O)NCCCO1;1;1;0;;;;;
+232;Cyclosporine A;CCC1NC(=O)C(C(O)C(C)CC=CC)N(C)C(=O)C(C(C)C)N(C)C(=O)C(CC(C)C)N(C)C(=O)C(CC(C)C)N(C)C(=O)C(C)NC(=O)C(C)NC(=O)C(CC(C)C)N(C)C(=O)C(NC(=O)C(CC(C)C)NC(=O)CN(C)C1=O)C(C)C;1;1;1;;;;;
+233;Cynarin;OC1CC(CC(OC(=O)C=Cc2ccc(O)c(O)c2)C1O)(OC(=O)C=Cc1ccc(O)c(O)c1)C(O)=O;0;1;0;;;;;
+234;Cyproheptadine;CN1CC\C(CC1)=C1/c2ccccc2C=Cc2ccccc12;1;1;0;;;;;
+235;Cyproterone acetate;CC(=O)OC1(CCC2C3C=C(Cl)C4=CC(=O)C5CC5C4(C)C3CCC12C)C(C)=O;1;1;0;;;;;
+236;Cytarabine;NC1=NC(=O)N(C=C1)C1OC(CO)C(O)C1O;1;1;0;;;;;
+237;Dacarbazine;CN(C)N=Nc1[nH]cnc1C(N)=O;1;0;0;;;;;
+238;Dactinomycin;CC(C)C1NC(=O)C(NC(=O)c2ccc(C)c3OC4=C(C)C(=O)C(N)=C(C(=O)NC5C(C)OC(=O)C(C(C)C)N(C)C(=O)CN(C)C(=O)C6CCCN6C(=O)C(NC5=O)C(C)C)C4=Nc23)C(C)OC(=O)C(C(C)C)N(C)C(=O)CN(C)C(=O)C2CCCN2C1=O;1;1;1;;;;;
+239;Dalteparin sodium;CC(=O)N[C@@H]1[C@@H](O)[C@H](O)[C@@H](COS(O)(=O)=O)O[C@@H]1O[C@H]1[C@H](O)[C@@H](O)[C@H](O[C@@H]2[C@@H](O)O[C@H](O[C@H]3[C@H](O)[C@@H](OS(O)(=O)=O)C(O)O[C@H]3C(O)=O)[C@H](OS(O)(=O)=O)[C@H]2CS(O)(=O)=O)O[C@@H]1C(O)=O;0;1;0;;;;;
+240;Danazol;CC12CC3C=NOC3C=C1CCC1C2CCC2(C)C1CCC2(O)C#C;1;0;0;;;;;
+241;Danthron;Oc1cccc2C(=O)c3cccc(O)c3C(=O)c12;0;1;0;;;;;
+242;Dantrolene sodium;OC1=NC(=O)CN1N=Cc1ccc(o1)-c1ccc(cc1)N(=O)=O;0;1;0;;;;;
+243;Dapsone;Nc1ccc(cc1)S(=O)(=O)c1ccc(N)cc1;1;0;0;;;;;
+244;Daunorubicin;COc1cccc2C(=O)c3c(O)c4CC(O)(CC(OC5CC(N)C(O)C(C)O5)c4c(O)c3C(=O)c12)C(C)=O;1;1;0;;;;;
+245;DDT;Clc1ccc(cc1)C(c1ccc(Cl)cc1)C(Cl)(Cl)Cl;0;1;0;;;;;
+246;Decitabine;NC1=NC(=O)N(C=N1)C1CC(O)C(CO)O1;1;0;0;;;;;
+247;Deferasirox;OC(=O)c1ccc(cc1)-n1nc(nc1-c1ccccc1O)-c1ccccc1O;1;0;0;;;;;
+248;Deferoxamine;CC(=O)N(O)CCCCCNC(=O)CCC(=O)N(O)CCCCCNC(=O)CCC(=O)N(O)CCCCCN;1;1;1;;;;;
+249;Dehydrocholic Acid;CC(CCC(O)=O)C1CCC2C3C(CC(=O)C12C)C1(C)CCC(=S)CC1CC3=O;0;1;0;;;;;
+250;Dehydroemetine;CCC1=C(CC2NCCc3cc(OC)c(OC)cc23)CC2N(CCc3cc(OC)c(OC)cc23)C1;0;1;0;;;;;
+251;Delta9-tetrahydrocannabinol;CCCCCc1cc(O)c2C3C=C(C)CCC3C(C)(C)Oc2c1;1;1;0;;;;;
+252;Desflurane;FC(F)OC(F)C(F)(F)F;1;1;0;;;;;
+253;Desogestrel;CCC12CC(=C)C3C(CCC4=CCCCC34)C1CCC2(O)C#C;1;0;0;;;;;
+254;Devazepide;CN1C(=O)C(NC(=O)c2cc3ccccc3[nH]2)N=C(c2ccccc2)c2ccccc12;0;1;0;;;;;
+255;Dexamethasone;CC1CC2C3CCC4=CC(=O)C=CC4(C)C3(F)C(O)CC2(C)C1(O)C(=O)CO;1;1;1;;;;;
+256;Dexfenfluramine;CCNC(C)Cc1cccc(c1)C(F)(F)F;1;1;0;;;;;
+257;Dexrazoxane;CC(CN1CC(=O)NC(=O)C1)N1CC(=O)NC(=O)C1;0;1;0;;;;;
+258;Dextran;OC[C@H]1O[C@H](OC[C@H]2O[C@H](OC[C@@H](O)[C@@H](O)[C@H](O)[C@@H](O)C=O)[C@H](O)[C@@H](O)[C@@H]2O)[C@H](O)[C@@H](O)[C@@H]1O;0;1;1;;;;;
+259;Dextroamphetamine;N[C@H](Cc1ccccc1)C;0;1;0;;;;;
+260;Dextromethorphan;COc1ccc2CC3C4CCCCC4(CCN3C)c2c1;1;0;0;;;;;
+261;Diatrizoate sodium;CC(=O)Nc1c(I)c(NC(C)=O)c(I)c(C(O)=O)c1I;0;1;0;;;;;
+262;Diazepam;CN1C(=O)CN=C(c2ccccc2)c2cc(Cl)ccc12;1;1;1;;;;;
+263;Diazoxide;CC1=Nc2ccc(Cl)cc2S(O)(=O)=N1;1;1;0;;;;;
+264;Diclofenac;OC(=O)Cc1ccccc1Nc1c(Cl)cccc1Cl;1;1;1;;;;;
+265;Dicumarol;OC1=C(CC2=C(O)Oc3ccccc3C2=O)C(=O)c2ccccc2O1;1;1;0;;;;;
+266;Didanosine;OCC1CCC(O1)n1cnc2C(=O)N=CNc12;1;1;0;;;;;
+267;Dienogest;CC12CCC3=C4CCC(=O)C=C4CCC3C1CCC2(O)CC#N;0;1;0;;;;;
+268;Diethylstilbestrol;CCC(c1ccc(OP(O)(O)=O)cc1)=C(CC)c1ccc(OP(O)(O)=O)cc1;1;1;1;;;;;
+269;Diethyltoluamide;CCN(CC)C(=O)c1cccc(C)c1;0;1;0;;;;;
+270;Diflunisal;OC(=O)c1cc(ccc1O)-c1ccc(F)cc1F;1;1;1;;;;;
+271;Digitoxin;CC1OC(CC(O)C1O)OC1CC(O)C(OC1C)OC1C(O)CC(OC2CCC3(C)C(CCC4C3CCC3(C)C(CCC43O)C3=CC(=O)OC3)C2)OC1C;0;1;0;;;;;
+272;Digoxin;CC1OC(CC(O)C1O)OC1CC(O)C(OC1C)OC1C(O)CC(OC2CCC3(C)C(CCC4C3CC(O)C3(C)C(CCC43O)C3=CC(=O)OC3)C2)OC1C;0;1;1;;;;;
+273;Dihydralazine;N=C1NNC(=N)c2ccccc12;1;1;0;;;;;
+274;Dihydroergotamine;CN1CC(CC2C1Cc1c[nH]c3cccc2c13)C(=O)NC1(C)OC2(O)C3CCCN3C(=O)C(Cc3ccccc3)N2C1=O;1;1;0;;;;;
+275;Diltiazem;COc1ccc(cc1)C1Sc2ccccc2N(CCN(C)C)C(=O)C1OC(C)=O;1;1;1;;;;;
+276;Dimethyl sulfoxide;CS(C)=O;1;1;1;;;;;
+277;Diosmin;COc1ccc(cc1O)C1=CC(=O)c2c(O)cc(OC3OC(COC4OC(C)C(O)C(O)C4O)C(O)C(O)C3O)cc2O1;0;1;0;;;;;
+278;Diphenhydramine;CN(C)CCOC(c1ccccc1)c1ccccc1;0;1;0;;;;;
+279;Dipyridamole;OCCN(CCO)c1nc(N2CCCCC2)c2nc(nc(N3CCCCC3)c2n1)N(CCO)CCO;1;1;1;;;;;
+280;Dipyrone;CN(CS(O)(=O)=O)C1=C(C)N(C)N(C1=O)c1ccccc1;1;1;0;;;;;
+281;Discodermolide;CC(CC(C)=CC(C)C(O)C(C)C=CC(O)CC1OC(=O)C(C)C(O)C1C)C(O)C(C)C(OC(N)=O)C(C)C=CC=C;1;0;0;;;;;
+282;Disopyramide;CC(C)N(CCC(C(N)=O)(c1ccccc1)c1ccccn1)C(C)C;1;0;0;;;;;
+283;Disulfiram;CCN(CC)C(=S)SSC(=S)N(CC)CC;1;1;1;;;;;
+284;Ditiocarb sodium;CCN(CC)C(S)=S;0;1;0;;;;;
+285;Dobutamine;CC(CCc1ccc(O)cc1)NCCc1ccc(O)c(O)c1;1;0;1;;;;;
+286;Docetaxel;CC(=O)OC12COC1CC(O)C1(C)C2C(OC(=O)c2ccccc2)C2(O)CC(OC(=O)C(O)C(NC(=O)OC(C)(C)C)c3ccccc3)C(C)=C(C(O)C1=O)C2(C)C;1;1;0;;;;;
+287;Docosahexaenoic acid;CCC=CCC=CCC=CCC=CCC=CCC=CCCC(O)=O;1;1;0;;;;;
+288;Dopamine;NCCc1ccc(O)c(O)c1;1;1;1;;;;;
+289;Dothiepin;CN(C)CCC=C1c2ccccc2CSc2ccccc12;1;0;0;;;;;
+290;Doxapram hydrochloride;CCN1CC(CCN2CCOCC2)C(C1=O)(c1ccccc1)c1ccccc1;1;0;0;;;;;
+291;Doxepin;CN(C)CCC=C1c2ccccc2COc2ccccc12;1;0;0;;;;;
+292;Doxifluridine;CC1OC(C(O)C1O)C1C=C(F)C(=O)NC1=O;1;1;0;;;;;
+293;Doxorubicin;COc1cccc2C(=O)C3C(C(=O)c12)C(O)=C1C(CC(O)(CC1=C3O)C(=O)CO)OC1CC(N)C(O)C(C)O1;1;1;1;;;;;
+294;Doxorubicin hydrochloride;O=C2c1c(O)c5c(c(O)c1C(=O)c3cccc(OC)c23)C[C@@](O)(C(=O)CO)C[C@@H]5O[C@@H]4O[C@H]([C@@H](O)[C@@H](N)C4)C;1;0;0;;;;;
+295;Doxycycline;CC1C2C(O)C3C(N(C)C)C(=O)C(=C(N)O)C(=O)C3(O)C(=O)C2=C(O)c2c(O)cccc12;1;1;1;;;;;
+296;Droloxifene;CCC(c1ccccc1)=C(c1ccc(OCCN(C)C)cc1)c1cccc(O)c1;0;1;0;;;;;
+297;Droperidol;Fc1ccc(cc1)C(=O)CCCN1CCC(=CC1)N1C(=O)Nc2ccccc12;0;1;0;;;;;
+298;Duloxetine;CNCCC(Oc1cccc2ccccc12)c1cccs1;1;0;0;;;;;
+299;Dydrogesterone;CC(=O)C1CCC2C3C=CC4=CC(=O)CCC4(C)C3CCC12C;1;1;0;;;;;
+300;Econazole;Clc1ccc(COC(Cn2ccnc2)c2ccc(Cl)cc2Cl)cc1;0;1;0;;;;;
+301;Ecteinascidin 743;COc1cc2c(CCNC22CSC3C4C5N(C)C(Cc6cc(C)c(OC)c(O)c56)C(O)N4C(COC2=O)c2c4OCOc4c(C)c(OC(C)=O)c32)cc1O;1;1;0;;;;;
+302;Efavirenz;FC(F)(F)C1(OC(=O)Nc2ccc(Cl)cc12)C#CC1CC1;1;1;0;;;;;
+303;Eflornithine;NCCCC(N)(C(F)F)C(O)=O;0;1;0;;;;;
+304;Eflucimibe;CCCCCCCCCCCCCC(=S)Nc1cc(C)c(O)c(C)c1C;0;0;1;;;;;
+305;Eicosapentaenoic acid;CCC=CCC=CCC=CCC=CCC=CCCCC(O)=O;1;1;0;;;;;
+306;Enalapril;CCOC(=O)C(CCc1ccccc1)NC(C)C(=O)N1CCCC1C(O)=O;1;1;0;;;;;
+307;Enalaprilat;CC(NC(CCc1ccccc1)C(O)=O)C(=O)N1CCCC1C(O)=O;1;0;0;;;;;
+308;Encainide;COc1ccc(cc1)C(=O)Nc1ccccc1CCC1CCCCN1C;1;0;0;;;;;
+309;Enflurane;FC(F)OC(F)(F)C(F)Cl;1;1;1;;;;;
+310;Enoxacin;CCN1C=C(C(O)=O)C(=O)c2cc(F)c(nc12)N1CCNCC1;1;1;0;;;;;
+311;Enoximone;CSc1ccc(cc1)C(=O)C1NC(=O)N=C1C;0;1;0;;;;;
+312;Enoxolone;CC1(C)C(O)CCC2(C)C1CCC1(C)C2C(=O)C=C2C3CC(C)(CCC3(C)CCC12C)C(O)=O;1;1;0;;;;;
+313;Entacapone;CCN(CC)C(=O)C(=Cc1cc(O)c(O)c(c1)N(=O)=O)C#N;1;1;0;;;;;
+314;Entecavir;NC1=NC(=O)c2ncn(C3CC(O)C(CO)C3=C)c2N1;1;0;0;;;;;
+315;Ephedrine;CNC(C)C(O)c1ccccc1;1;0;0;;;;;
+316;Epinephrine;CNCC(O)c1ccc(O)c(O)c1;1;1;1;;;;;
+317;Epirubicin;O=C2c1c(O)c5c(c(O)c1C(=O)c3cccc(OC)c23)C[C@@](O)(C(=O)CO)C[C@@H]5O[C@@H]4O[C@H]([C@H](O)[C@@H](N)C4)C;0;1;0;;;;;
+318;Epoprostenol sodium;CCCCCC(O)C=CC1C(O)CC2OC(CC12)=CCCCC(O)=O;0;1;0;;;;;
+319;epsilon-Aminocaproic acid;NCCCCCC(O)=O;1;0;0;;;;;
+320;Erdosteine;OC(=O)CSCC(=O)NC1CCSC1=O;0;1;0;;;;;
+321;Ergotamine;CN1CC(C=C2C1Cc1c[nH]c3cccc2c13)C(=O)NC1(C)OC2(O)C3CCCN3C(=O)C(Cc3ccccc3)N2C1=O;1;0;1;;;;;
+322;Erythromycin;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(O)CC(C)C(=O)C(C)C(O)C1(C)O;1;1;0;;;;;
+323;Erythromycin estolate;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2OC(=O)CC)N(C)C)C(C)(O)CC(C)C(=O)C(C)C(O)C1(C)O;1;1;0;;;;;
+324;Erythromycin ethylsuccinate;CCOC(=O)CCC(=O)OC1C(OC(C)CC1N(C)C)OC1C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(=O)OC(CC)C(C)(O)C(O)C(C)C(=O)C(C)CC1(C)O;1;0;0;;;;;
+325;Erythromycin lactobionate;CCC1OC(=O)C(C)C(CC(OC2OC(C)CC(C2O)N(C)C)C(C)(O)CC(C)C(=O)C(C)CC1(C)O)OC1CC(C)(OC)C(O)C(C)O1;1;0;0;;;;;
+326;Erythromycin stearate;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(O)CC(C)C(=O)C(C)C(O)C1(C)O;1;0;0;;;;;
+327;Esomeprazole;O=S(c2nc1ccc(OC)cc1n2)Cc3ncc(c(OC)c3C)C;1;0;0;;;;;
+328;Estazolam;Clc1ccc-2c(c1)C(=NCc1nncn-21)c1ccccc1;0;1;0;;;;;
+329;Estradiol;CC12CCC3C(CCc4cc(O)ccc34)C1CCC2O;1;1;1;;;;;
+330;Estradiol valerate;CCCCC(=O)OC1CCC2C3CCc4cc(O)ccc4C3CCC12C;0;1;0;;;;;
+331;Estramustine;CC12CCC3C(CCc4cc(OC(=O)N(CCCl)CCCl)ccc34)C1CCC2O;1;0;0;;;;;
+332;Estriol;CC12CCC3C(CCc4cc(O)ccc34)C1CC(O)C2O;0;1;0;;;;;
+333;Estrone;CC12CCC3C(CCc4cc(O)ccc34)C1CCC2=O;1;1;0;;;;;
+334;Estrone sulfate;CC12CCC3C(CCc4cc(OS(O)(=O)=O)ccc34)C1CCC2=O;0;1;0;;;;;
+335;Ethacrynic acid;CCC(=C)C(=O)c1ccc(OCC(O)=O)c(Cl)c1Cl;0;1;1;;;;;
+336;Ethanol;CCO;1;1;1;;;;;
+337;Ethanolamine oleate;CCCCCCCCC=CCCCCCCCC(=O)OCCN;0;1;0;;;;;
+338;Ethinyl estradiol;CC12CCC3C(CCc4cc(O)ccc34)C1CCC2(O)C#C;1;1;1;;;;;
+339;Ethinyl estradiol 3-methyl-ether;O(c1cc4c(cc1)[C@H]3CC[C@]2([C@@H](CC[C@]2(C#C)O)[C@@H]3CC4)C)C;1;1;1;;;;;
+340;Ethionamide;CCc1cc(ccn1)C(N)=S;1;1;0;;;;;
+341;Ethoxzolamide;CCOc1ccc2nc(sc2c1)S(N)(=O)=O;0;1;0;;;;;
+342;Etodolac;CCc1cccc2c3CCOC(CC)(CC(O)=O)c3[nH]c12;1;1;0;;;;;
+343;Etomidate;CCOC(=O)c1cncn1C(C)c1ccccc1;1;1;0;;;;;
+344;Etomoxir;CCOC(=O)C1(CCCCCCOc2ccc(Cl)cc2)CO1;1;1;1;;;;;
+345;Etoposide;COc1cc(cc(OC)c1O)C1C2C(COC2=O)C(OC2OC3COC(C)OC3C(O)C2O)c2cc3OCOc3cc12;1;1;0;;;;;
+346;Etretinate;CCOC(=O)C=C(C)C=CC=C(C)C=Cc1c(C)cc(OC)c(C)c1C;1;0;0;;;;;
+347;Evan's Blue;Cc1cc(ccc1N=Nc1ccc2c(cc(c(N)c2c1O)S(O)(=O)=O)S(O)(=O)=O)-c1ccc(N=Nc2ccc3c(cc(c(N)c3c2O)S(O)(=O)=O)S(O)(=O)=O)c(C)c1;0;1;0;;;;;
+348;Ezetimibe;OC(CCC1C(N(C1=O)c1ccc(F)cc1)c1ccc(O)cc1)c1ccc(F)cc1;1;1;1;;;;;
+349;Famciclovir;CC(=O)OCC(CCn1cnc2cnc(N)nc12)COC(C)=O;1;0;0;;;;;
+350;Famotidine;N\C(N)=N\c1nc(CSCCC(N)=NS(N)(=O)=O)cs1;1;1;0;;;;;
+351;Fasudil;O=S(=O)(N1CCCNCC1)c1cccc2cnccc12;1;1;0;;;;;
+352;Felbamate;NC(=O)OCC(COC(N)=O)c1ccccc1;1;1;0;;;;;
+353;Felodipine;CCOC(=O)C1=C(C)NC(C)=C(C1c1cccc(Cl)c1Cl)C(=O)OC;1;0;0;;;;;
+354;Fenofibrate;CC(C)OC(=O)C(C)(C)Oc1ccc(cc1)C(=O)c1ccc(Cl)cc1;1;1;0;;;;;
+355;Fenretinide;CC(C=CC1=C(C)CCCC1(C)C)=CC=CC(C)=CC(=O)Nc1ccc(O)cc1;1;1;0;;;;;
+356;Fentanyl;CCC(=O)N(C1CCN(CC1)CCc1ccccc1)c1ccccc1;1;1;1;;;;;
+357;Ferrous citrate;OC(=O)CC(O)(CC(O)=O)C(O)=O;0;1;0;;;;;
+358;FK-506;COC1CC(CCC1O)C=C(C)C1OC(=O)C2CCCCN2C(=O)C(=O)C2(O)OC(C(CC(C)CC(C)=CC(CC=C)C(=O)CC(O)C1C)OC)C(CC2C)OC;1;1;0;;;;;
+359;Flavopiridol;CN1CCC(C(O)C1)c1c(O)cc(O)c2C(=O)C=C(Oc12)c1ccccc1Cl;1;0;0;;;;;
+360;Flecainide;FC(F)(F)COc1ccc(OCC(F)(F)F)c(c1)C(=O)NCC1CCCCN1;1;0;0;;;;;
+361;Flosequinan;COSC1=CN(C)c2cc(F)ccc2C1=O;1;0;0;;;;;
+362;Flucloxacillin;Cc1onc(c1C(=O)NC1C2SC(C)(C)C(N2C1=O)C(O)=O)-c1c(F)cccc1Cl;1;1;0;;;;;
+363;Fluconazole;OC(Cn1cncn1)(Cn1cncn1)c1ccc(F)cc1F;1;1;0;;;;;
+364;Flucytosine;NC1=NC(=O)NC=C1F;0;1;0;;;;;
+365;Flufenamic acid;OC(=O)c1ccccc1Nc1cccc(c1)C(F)(F)F;1;1;0;;;;;
+366;Flumazenil;CCOC(=O)c1ncn2-c3ccc(F)cc3C(=O)N(C)Cc12;1;0;0;;;;;
+367;Flunitrazepam;CN1C(=O)CN=C(c2ccccc2F)c2cc(ccc12)N(=O)=O;0;0;1;;;;;
+368;Fluoxetine;CNCCC(Oc1ccc(cc1)C(F)(F)F)c1ccccc1;1;1;0;;;;;
+369;Flurbiprofen;CC(C(O)=O)c1ccc(c(F)c1)-c1ccccc1;0;1;1;;;;;
+370;Flutamide;CC(C)C(=O)Nc1ccc(c(c1)C(F)(F)F)N(=O)=O;1;1;0;;;;;
+371;Fluvastatin;CC(C)n1c(C=CC(O)CC(O)CC(O)=O)c(-c2ccc(F)cc2)c2ccccc12;1;1;0;;;;;
+372;Fluvoxamine;COCCCCC(=NOCCN)c1ccc(cc1)C(F)(F)F;1;1;0;;;;;
+373;Folic acid;NC1=NC(=O)c2nc(CNc3ccc(cc3)C(=O)NC(CCC(O)=O)C(O)=O)cnc2N1;1;1;0;;;;;
+374;Fomepizole;Cc1cn[nH]c1;1;1;0;;;;;
+375;Formaldehyde;C=O;1;1;0;;;;;
+376;Foscarnet Sodium;OC(=O)P(O)(O)=O;0;1;0;;;;;
+377;Fosfomycin;CC1OC1P(O)(O)=O;1;0;0;;;;;
+378;Fosinopril;CCC(=O)OC(OP(=O)(CCCCc1ccccc1)CC(=O)N1CC(CC1C(O)=O)C1CCCCC1)C(C)C;1;0;0;;;;;
+379;Fructose;OCC1OC(O)(CO)C(O)C1O;1;1;1;;;;;
+380;Fructose-1,6-diphosphate;OC1C(O)C(O)(COP(O)(O)=O)OC1COP(O)(O)=O;1;1;0;;;;;
+381;FTY 720;CCCCCCCCc1ccc(CCC(N)(CO)CO)cc1;1;1;0;;;;;
+382;Furan;c1ccoc1;0;1;0;;;;;
+383;Furazolidone;O=C1OCCN1N=Cc1ccc(o1)N(=O)=O;0;1;0;;;;;
+384;Furosemide;NS(=O)(=O)c1cc(C(O)=O)c(NCc2ccco2)cc1Cl;1;1;1;;;;;
+385;Fusidic acid;CC1C(O)CCC2(C)C1CCC1(C)C2C(O)CC2C(C(CC12C)OC(C)=O)=C(CC\C=C(\C)C)C(O)=O;1;0;0;;;;;
+386;Gadobenate dimeglumine;OC(=O)CN(CCN(CC(O)=O)CC(O)=O)CCN(CC(O)=O)C(COCc1ccccc1)C(O)=O;1;0;0;;;;;
+387;Ganciclovir;NC1=NC(=O)c2ncn(COC(CO)CO)c2N1;1;0;0;;;;;
+388;Gatifloxacin;COc1c(N2CCNC(C)C2)c(F)cc2C(=O)C(=CN(C3CC3)c12)C(O)=O;1;0;0;;;;;
+389;Gefitinib;COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1;1;1;0;;;;;
+390;Geldanamycin;COC1CC(C)CC2=C(OC)C(=O)C=C(NC(=O)C(C)=CC=CC(OC)C(OC(N)=O)C(C)=CC(C)C1O)C2=O;1;0;0;;;;;
+391;Gemfibrozil;Cc1ccc(C)c(OCCCC(C)(C)C(O)=O)c1;1;1;0;;;;;
+392;Gentamicin;CNC1C(O)C(OCC1(C)O)OC1C(N)CC(N)C(OC2OC(CN)CCC2N)C1O;0;1;0;;;;;
+393;Gentian Violet;CN(C)c1ccc(cc1)C(\c1ccc(cc1)N(C)C)=C1/C=C\C(\C=C/1)=[N+](\C)C;0;1;0;;;;;
+394;Gepirone;CC1(C)CC(=O)N(CCCCN2CCN(CC2)c2ncccn2)C(=O)C1;1;0;0;;;;;
+395;Gestodene;CCC12CCC3C(CCC4=CC(=O)CCC34)C1C=CC2(O)C#C;1;0;0;;;;;
+396;Glafenine;OCC(O)COC(=O)c1ccccc1Nc1ccnc2cc(Cl)ccc12;1;0;0;;;;;
+397;Glatiramer acetate;NC(Cc1ccc(O)cc1)C(O)=O;1;1;0;;;;;
+398;Gliclazide;Cc1ccc(cc1)S(=O)(=O)NC(=O)NN1CC2CCCC2C1;1;0;0;;;;;
+399;Glucosamine;NC1C(O)OC(CO)C(O)C1O;0;1;0;;;;;
+400;Glucose;OCC1OC(O)C(O)C(O)C1O;1;1;1;;;;;
+401;Glutamic acid;NC(CCC(O)=O)C(O)=O;0;1;0;;;;;
+402;Glutamine;NC(CCC(N)=O)C(O)=O;1;1;1;;;;;
+403;Glutathione Disulfide;NC(CCC(=O)NC(CSSCC(NC(=O)CCC(N)C(O)=O)C(=O)NCC(O)=O)C(=O)NCC(O)=O)C(O)=O;0;1;0;;;;;
+404;Glutethimide;CCC1(CCC(=O)NC1=O)c1ccccc1;1;0;0;;;;;
+405;Glyburide;COc1ccc(Cl)cc1C(=O)NCCc1ccc(cc1)S(=O)(=O)NC(=O)NC1CCCCC1;1;1;0;;;;;
+406;Glycerol;OCC(O)CO;1;1;1;;;;;
+407;Glycine;NCC(O)=O;0;1;1;;;;;
+408;Gold Sodium Thiomalate;OC(=O)CC(S[Au])C(O)=O;1;1;0;;;;;
+409;Gossypol;CC(C)c1c(O)c(O)c(C=O)c2c(O)c(c(C)cc12)-c1c(C)cc2c(C(C)C)c(O)c(O)c(C=O)c2c1O;1;1;0;;;;;
+410;Griseofulvin;COc1cc(OC)c2C(=O)C3(Oc2c1Cl)C(C)CC(=O)C=C3OC;1;1;0;;;;;
+411;Guaiazulene;CC(C)c1ccc(C)c2ccc(C)c2c1;0;1;0;;;;;
+412;Guanabenz acetate;NC(=N)NN=Cc1c(Cl)cccc1Cl;0;1;0;;;;;
+413;Guanidine hydrochloride;NC(N)=N;1;1;0;;;;;
+414;Halofuginone;OC1CCCNC1CC(=O)CN1C=Nc2cc(Br)c(Cl)cc2C1=O;1;1;0;;;;;
+415;Haloperidol;OC1(CCN(CCCC(=O)c2ccc(F)cc2)CC1)c1ccc(Cl)cc1;1;1;1;;;;;
+416;Halothane;FC(F)(F)C(Cl)Br;1;1;1;;;;;
+417;Hematoporphyrin;CC(O)c1c(C)c2cc3[nH]c(cc4nc(cc5[nH]c(cc1n2)c(C)c5C(C)O)c(C)c4CCC(O)=O)c(CCC(O)=O)c3C;0;1;0;;;;;
+418;Heparin;CC(=O)NC1C(O)OC(COS(O)(=O)=O)C(OC2OC(C(OC3OC(CO)C(OC4OC(C(O)C(O)C4OS(O)(=O)=O)C(O)=O)C(OS(O)(=O)=O)C3NS(O)(=O)=O)C(O)C2OS(O)(=O)=O)C(O)=O)C1O;1;1;1;;;;;
+419;Hesperidin;COc1ccc(cc1O)C1CC(=O)c2c(O)cc(OC3OC(COC4OC(C)C(O)C(O)C4O)C(O)C(O)C3O)cc2O1;1;1;0;;;;;
+420;Hexachlorobenzene;Clc1c(Cl)c(Cl)c(Cl)c(Cl)c1Cl;1;1;0;;;;;
+421;Hexachlorophene;Oc1c(Cl)cc(Cl)c(Cl)c1Cc1c(O)c(Cl)cc(Cl)c1Cl;0;1;0;;;;;
+422;Hexestrol;CCC(C(CC)c1ccc(O)cc1)c1ccc(O)cc1;0;1;0;;;;;
+423;Hexetidine;CCCCC(CC)CN1CN(CC(CC)CCCC)CC(C)(N)C1;0;1;0;;;;;
+424;Hexobarbital;CN1C(O)=NC(=O)C(C)(C1=O)C1=CCCCC1;0;1;0;;;;;
+425;Histamine Dihydrochloride;NCCc1cnc[nH]1;0;1;0;;;;;
+426;Homoharringtonine;COC(=O)CC(O)(CCCC(C)(C)O)C(=O)OC1C2c3cc4OCOc4cc3CCN3CCCC23C=C1OC;1;0;0;;;;;
+427;Hyaluronic acid;CC(=O)NC1C(O)OC(CO)C(O)C1OC1OC(C(OC2OC(CO)C(O)C(OC3OC(C(O)C(O)C3O)C(O)=O)C2NC(C)=O)C(O)C1O)C(O)=O;1;1;0;;;;;
+428;Hydralazine;NNc1nncc2ccccc12;1;1;1;;;;;
+429;Hydrochlorothiazide;NS(=O)(=O)c1cc2c(NCNS2(=O)=O)cc1Cl;1;0;0;;;;;
+430;Hydrocortisone;CC12CCC(=O)C=C1CCC1C3CCC(O)(C(=O)CO)C3(C)CC(O)C21;1;1;1;;;;;
+431;Hydrocortisone acetate;CC(=O)OCC(=O)C1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3C(O)CC12C;0;1;0;;;;;
+432;Hydroquinone;Oc1ccc(O)cc1;1;1;0;;;;;
+433;Hydroxyurea;NC(=O)NO;1;1;0;;;;;
+434;Hydroxyzine;OCCOCCN1CCN(CC1)C(c1ccccc1)c1ccc(Cl)cc1;1;0;0;;;;;
+435;Ibuprofen;CC(C)Cc1ccc(cc1)C(C)C(O)=O;1;1;0;;;;;
+436;Idebenone;COC1=C(OC)C(=O)C(CCCCCCCCCCO)=C(C)C1=O;0;1;0;;;;;
+437;IDN-6556;C[C@H](C(=O)ON[C@H](CCC(O)=O)C(=O)COc1c(F)c(F)cc(F)c1F)C(=O)C(=O)Nc1ccccc1C(C)(C)C;0;1;0;;;;;
+438;Idoxifene;CCC(c1ccccc1)=C(c1ccc(I)cc1)c1ccc(OCCN2CCCC2)cc1;0;1;0;;;;;
+439;Ifosfamide;ClCCNP1(=O)OCCCN1CCCl;1;0;0;;;;;
+440;ilomastat;CNC(=O)C(Cc1c[nH]c2ccccc12)NC(=O)C(CC(C)C)CC(=O)NO;0;1;0;;;;;
+441;Iloprost;CC#CCC(C)C(O)C=CC1C(O)CC2CC(CC12)=CCCCC(O)=O;0;1;0;;;;;
+442;Imatinib mesilate;CN1CCN(CC1)Cc1ccc(cc1)C(=O)Nc1ccc(C)c(Nc2nccc(n2)-c2cccnc2)c1;1;1;0;;;;;
+443;Imidapril;CCOC(=O)C(CCc1ccccc1)NC(C)C(=O)N1C(=O)N(C)C=C1C(O)=O;0;1;0;;;;;
+444;Imipramine;CN(C)CCCN1c2ccccc2CCc2ccccc12;1;1;0;;;;;
+445;Implanon;CCC12CC(=C)C3C(CCC4=CC(=O)CCC34)C1CCC2(O)C#C;1;0;0;;;;;
+446;Indapamide;CC1Cc2ccccc2N1NC(=O)c1ccc(Cl)c(c1)S(N)(=O)=O;0;1;0;;;;;
+447;Indinavir;CC(C)(C)NC(=O)C1CN(CCN1CC(O)CC(Cc1ccccc1)C(=O)NC1C(O)Cc2ccccc12)Cc1cccnc1;1;1;0;;;;;
+448;Indocyanine Green;CC1(C)C(C=CC=CC=CC=C2Cc3c(ccc4ccccc34)N2CCCCS(O)(=O)=O)=[N+](CCCCS(O)(=O)=O)c2ccc3ccccc3c12;0;1;0;;;;;
+449;Indomethacin;COc1ccc2n(C(=O)c3ccc(Cl)cc3)c(C)c(CC(O)=O)c2c1;1;1;1;;;;;
+450;Inosine;OCC1OC(C(O)C1O)n1cnc2C(=O)N=CNc12;0;1;0;;;;;
+451;Inositol;OC1C(O)C(O)C(O)C(O)C1O;1;0;0;;;;;
+452;intoplicine;CN(C)CCCNc1ncc(C)c2Nc3ccc4cc(O)ccc4c3-c12;1;0;0;;;;;
+453;Iodipamide;OC(=O)c1c(I)cc(I)c(NC(=O)CCCCC(=O)Nc2c(I)cc(I)c(C(O)=O)c2I)c1I;0;1;1;;;;;
+454;Iohexol;CC(=O)N(CC(O)CO)c1c(I)c(C(=O)NCC(O)CO)c(I)c(C(=O)NCC(O)CO)c1I;0;0;1;;;;;
+455;Iproclozide;CC(C)NNC(=O)COc1ccc(Cl)cc1;1;0;0;;;;;
+456;Iproniazid;CC(C)NNC(=O)c1ccncc1;1;1;0;;;;;
+457;Irbesartan;CCCCC1=NC2(CCCC2)C(=O)N1Cc1ccc(cc1)-c1ccccc1-c1nn[nH]n1;1;1;0;;;;;
+458;Irinotecan;CCc1c2CN3C(=O)C4=C(C=C3c2[nH]c2ccc(OC(=O)N3CCC(CC3)N3CCCCC3)cc12)C(O)(CC)C(=O)OC4;1;0;0;;;;;
+459;Isaxonine phosphate;CC(C)Nc1ncccn1;1;0;0;;;;;
+460;Isoflavone;O=C1C(=COc2ccccc12)c1ccccc1;0;0;1;;;;;
+461;Isoflurane;FC(F)OC(Cl)C(F)(F)F;1;1;1;;;;;
+462;Isoniazid;NNC(=O)c1ccncc1;1;1;1;;;;;
+463;Isoproterenol;CC(C)NCC(O)c1ccc(O)c(O)c1;1;1;1;;;;;
+464;Isosorbide dinitrate;O=N(=O)OC1COC2C(COC12)ON(=O)=O;1;1;0;;;;;
+465;Isosorbide mononitrate;OC1COC2C(COC12)ON(=O)=O;1;0;0;;;;;
+466;Isotretinoin;CC(\C=C\C1=C(C)CCCC1(C)C)=C/C=C/C(C)=C\C(O)=O;1;0;0;;;;;
+467;Isoxsuprine;CC(COc1ccccc1)NC(C)C(O)c1ccc(O)cc1;0;1;0;;;;;
+468;Isradipine;COC(=O)C1=C(C)NC(C)=C(C1c1cccc2nonc12)C(=O)OC(C)C;1;0;0;;;;;
+469;Itraconazole;CCC(C)N1N=CN(C1=O)c1ccc(cc1)N1CCN(CC1)c1ccc(OCC2COC(Cn3cncn3)(O2)c2ccc(Cl)cc2Cl)cc1;1;1;0;;;;;
+470;Ivermectin;CCC(C)C1OC2(CCC1C)CC1CC(CC=C(C)C(OC3CC(OC)C(OC4CC(OC)C(O)C(C)O4)C(C)O3)C(C)C=CC=C3COC4C(O)C(C)=CC(C(=O)O1)C34O)O2;1;0;0;;;;;
+471;ixabepilone;CC1CCCC2(C)OC2CC(NC(=O)CC(O)C(C)(C)C(=O)C(C)C1O)C(C)=Cc1csc(C)n1;1;0;0;;;;;
+472;JTT-501;Cc1oc(nc1CCOc1ccc(CC2C(=O)NOC2=O)cc1)-c1ccccc1;0;1;0;;;;;
+473;Kanamycin;NCC1OC(OC2C(N)CC(N)C(OC3OC(CO)C(O)C(N)C3O)C2O)C(N)C(O)C1O;0;0;1;;;;;
+474;Ketamine;CNC1(CCCCC1=O)c1ccccc1Cl;0;1;1;;;;;
+475;Ketanserin;Fc1ccc(cc1)C(=O)C1CCN(CC1)CCN1C(=O)Nc2ccccc2C1=O;1;1;0;;;;;
+476;Ketoconazole;CC(=O)N1CCN(CC1)c1ccc(OCC2COC(Cn3ccnc3)(O2)c2ccc(Cl)cc2Cl)cc1;1;1;1;;;;;
+477;Ketoprofen;CC(C(O)=O)c1cccc(c1)C(=O)c1ccccc1;1;0;0;;;;;
+478;Khellin;COc1c2OC(C)=CC(=O)c2c(OC)c2ccoc12;0;1;0;;;;;
+479;KRN 5500;CCCCCCCCCC=CC=CC(=O)NCC(=O)NC1C(O)C(O)C(Nc2ncnc3nc[nH]c23)OC1C(O)CO;0;1;0;;;;;
+480;L dopa;NC(Cc1ccc(O)c(O)c1)C(O)=O;1;1;0;;;;;
+481;Labetalol;CC(CCc1ccccc1)NCC(O)c1ccc(O)c(c1)C(N)=O;1;0;0;;;;;
+482;Lactic acid;CC(O)C(O)=O;1;1;1;;;;;
+483;Lactose;OCC1OC(OC2C(CO)OC(O)C(O)C2O)C(O)C(O)C1O;0;1;0;;;;;
+484;Lactulose;OCC1OC(OC2C(CO)OC(O)(CO)C2O)C(O)C(O)C1O;1;1;0;;;;;
+485;Lamivudine;NC1=NC(=O)N(C=C1)C1CSC(CO)O1;1;1;0;;;;;
+486;Lamotrigine;Nc1nnc(c(N)n1)-c1cccc(Cl)c1Cl;1;0;0;;;;;
+487;Lanreotide;CC(C)C1NC(=O)C(CCCCN)NC(=O)C(Cc2c[nH]c3ccccc23)NC(=O)C(Cc2ccc(O)cc2)NC(=O)C(CSSCC(NC1=O)C(=O)NC(C(C)O)C(N)=O)NC(=O)C(N)Cc1ccc2ccccc2c1;1;0;0;;;;;
+488;Lansoprazole;Cc1c(OCC(F)(F)F)cc[nH]c1CS(=O)c1nc2ccccc2[nH]1;1;0;0;;;;;
+489;L-Arginine;NC(CCC\N=C(\N)N)C(O)=O;1;1;1;;;;;
+490;Lecithin;C[N+](C)(C)CCOP(O)(=O)OCC(COC=O)OC=O;1;1;0;;;;;
+491;Leflunomide;Cc1oncc1C(=O)Nc1ccc(cc1)C(F)(F)F;1;1;0;;;;;
+492;Leucovorin;NC1=NC(=O)C2=C(NCC(CNc3ccc(cc3)C(=O)NC(CCC(O)=O)C(O)=O)N2C=O)N1;1;1;0;;;;;
+493;Levamisole;C1CN2CC(N=C2S1)c1ccccc1;1;1;0;;;;;
+494;Levetiracetam;CCC(N1CCCC1=O)C(N)=O;1;0;0;;;;;
+495;Levofloxacin;CC1COc2c(N3CCN(C)CC3)c(F)cc3C(=O)C(=CN1c23)C(O)=O;1;0;0;;;;;
+496;Levosimendan;CC1CC(=O)NN=C1c1ccc(N\N=C(\C#N)C#N)cc1;0;0;1;;;;;
+497;Lidocaine;CCN(CC)CC(=O)Nc1c(C)cccc1C;1;1;1;;;;;
+498;Lisinopril;NCCCCC(NC(CCc1ccccc1)C(O)=O)C(=O)N1CCCC1C(O)=O;1;1;0;;;;;
+499;Lodoxamide tromethamine;OC(=O)C(=O)Nc1cc(cc(NC(=O)C(O)=O)c1Cl)C#N;0;1;0;;;;;
+500;Lofepramine;CN(CCCN1c2ccccc2CCc2ccccc12)CC(=O)c1ccc(Cl)cc1;1;0;0;;;;;
+501;Lomustine;ClCCN(N=O)C(=O)NC1CCCCC1;1;1;0;;;;;
+502;Loperamide;CN(C)C(=O)C(CCN1CCC(O)(CC1)c1ccc(Cl)cc1)(c1ccccc1)c1ccccc1;0;1;0;;;;;
+503;Lopinavir;CC(C)C(N1CCCNC1=O)C(=O)NC(CC(O)C(Cc1ccccc1)NC(=O)COc1c(C)cccc1C)Cc1ccccc1;1;0;0;;;;;
+504;Lornoxicam;CN1C(C(=O)Nc2ccccn2)=C(O)c2sc(Cl)cc2S1(=O)=O;1;0;0;;;;;
+505;Losartan;CCCCc1nc(Cl)c(CO)n1Cc1ccc(cc1)-c1ccccc1-c1nn[nH]n1;1;1;0;;;;;
+506;Lovastatin;CCC(C)C(=O)OC1CC(C)C=C2C=CC(C)C(CCC3CC(O)CC(=O)O3)C12;1;1;1;;;;;
+507;Lysine acetylsalicylate;CC(=O)Oc1ccccc1C(O)=O;1;1;0;;;;;
+508;Malathion;CCOC(=O)CC(SP(=S)(OC)OC)C(=O)OCC;0;1;0;;;;;
+509;Malotilate;CC(C)OC(=O)C(\C(=O)OC(C)C)=C1\SC=CS1;1;1;0;;;;;
+510;Mangafodipir trisodium;Cc1ncc(COP(O)(O)=O)c(CN(CCN(CC(O)=O)Cc2c(COP(O)(O)=O)cnc(C)c2O)CC(O)=O)c1O;1;0;0;;;;;
+511;Mannitol;O[C@H]([C@H](O)CO)[C@H](O)[C@H](O)CO;1;1;1;;;;;
+512;Marimastat;CNC(=O)C(NC(=O)C(CC(C)C)C(O)C(=O)NO)C(C)(C)C;0;1;0;;;;;
+513;Marvelon;CCC12CC(=C)C3C(CCC4=CCCCC34)C1CCC2(O)C#C;1;0;0;;;;;
+514;Mazindol;OC1(N2CCN=C2c2ccccc12)c1ccc(Cl)cc1;0;1;0;;;;;
+515;Mebendazole;COC(=O)Nc1nc2ccc(cc2[nH]1)C(=O)c1ccccc1;1;0;1;;;;;
+516;Mebutamate;CCC(C)C(C)(COC(N)=O)COC(N)=O;0;1;0;;;;;
+517;Meclizine;Cc1cccc(CN2CCN(CC2)C(c2ccccc2)c2ccc(Cl)cc2)c1;0;1;0;;;;;
+518;Meclofenoxate;CN(C)CCOC(=O)COc1ccc(Cl)cc1;0;1;0;;;;;
+519;Medetomidine;CC(c1c[nH]c[nH]1)c1cccc(C)c1C;1;0;1;;;;;
+520;Medroxyprogesterone acetate;CC1CC2C(CCC3(C)C2CCC3(OC(C)=O)C(C)=O)C2(C)CCC(=O)C=C12;0;1;0;;;;;
+521;Mefenamic acid;Cc1cccc(Nc2ccccc2C(O)=O)c1C;1;1;0;;;;;
+522;Megestrol acetate;CC(=O)OC1(CCC2C3C=C(C)C4=CC(=O)CCC4(C)C3CCC12C)C(C)=O;1;1;0;;;;;
+523;Meloxicam;CN1C(C(=O)c2ccccc2S1(=O)=O)=C(O)Nc1ncc(C)s1;1;0;0;;;;;
+524;Melphalan;NC(Cc1ccc(cc1)N(CCCl)CCCl)C(O)=O;1;1;0;;;;;
+525;Menadione;CC1=CC(=O)c2ccccc2C1=O;1;1;1;;;;;
+526;Menthol;CC(C)c1ccc(C)cc1O;1;0;0;;;;;
+527;Meperidine;CCOC(=O)C1(CCN(C)CC1)c1ccccc1;1;0;0;;;;;
+528;Mephenytoin;CCC1(NC(=O)N(C)C1=O)c1ccccc1;0;1;0;;;;;
+529;Mephobarbital;CCC1(C(=O)NC(=O)N(C)C1=O)c1ccccc1;0;1;0;;;;;
+530;Mepivacaine;CC1CCCC(N1C)C(=O)N(C)c1ccccc1C;0;1;0;;;;;
+531;Meprobamate;CCCC(C)(COC(N)=O)COC(N)=O;0;1;0;;;;;
+532;Mequinol;COc1ccc(O)cc1;0;1;0;;;;;
+533;Meropenem;CC(O)C1C2C(C)C(SC3CCC(N3)C(=O)N(C)C)=C(N2C1=O)C(O)=O;1;0;0;;;;;
+534;Mersalyl;COC(CNC(=O)c1ccccc1OCC(O)=O)C[Hg]O;0;1;0;;;;;
+535;Mesalamine;Nc1ccc(O)c(c1)C(O)=O;1;1;0;;;;;
+536;Mesna;OS(=O)(=O)CCS;0;1;0;;;;;
+537;Metformin;CN(C)C(=N)\N=C(\N)N;1;1;1;;;;;
+538;Methadone;CCC(=O)C(CC(C)N(C)C)(c1ccccc1)c1ccccc1;1;1;0;;;;;
+539;Methamphetamine;CNC(C)Cc1ccccc1;1;1;0;;;;;
+540;Methaqualone;CC1=Nc2ccccc2C(=O)N1c1ccccc1C;0;1;0;;;;;
+541;Methimazole;Cn1ccnc1S;1;1;0;;;;;
+542;Methionine;CSCCC(N)C(O)=O;1;1;1;;;;;
+543;Methohexital sodium;CCC=CC(C)C1(CC=C)C(=O)NC(=O)N(C)C1=O;0;1;0;;;;;
+544;Methotrexate;CN(Cc1cnc2nc(N)nc(N)c2n1)c1ccc(cc1)C(=O)NC(CCC(O)=O)C(O)=O;1;1;1;;;;;
+545;Methoxsalen;COc1c2OC(=O)C=Cc2cc2ccoc12;1;1;0;;;;;
+546;Methoxyflurane;COC(F)(F)C(Cl)Cl;1;1;0;;;;;
+547;Methyl salicylate;COC(=O)c1ccccc1O;0;0;1;;;;;
+548;Methylcobalamin;CC(CNC(=O)CCC1(C)C(CC(N)=O)C2NC1=C(C)C1=NC(=CC3=NC(=C(C)C4=NC2(C)C(C)(CC(N)=O)C4CCC(N)=O)C(C)(CC(N)=O)C3CCC(N)=O)C(C)(C)C1CCC(N)=O)OP(O)(=O)OC1C(CO)OC(C1O)n1cnc2cc(C)c(C)cc12;0;1;0;;;;;
+549;Methyldopa;CC(N)(Cc1ccc(O)c(O)c1)C(O)=O;1;0;0;;;;;
+550;Methylphenidate;COC(=O)C(C1CCCCN1)c1ccccc1;1;1;0;;;;;
+551;Methylprednisolone;CC1CC2C3CCC(O)(C(=O)CO)C3(C)CC(O)C2C2(C)C=CC(=O)C=C12;1;1;0;;;;;
+552;Methyltestosterone;CC1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;1;1;0;;;;;
+553;Methyprylon;CCC1(CC)C(=O)NCC(C)C1=O;0;1;0;;;;;
+554;Metoclopramide;CCN(CC)CCNC(=O)c1cc(Cl)c(N)cc1OC;1;1;0;;;;;
+555;Metoprolol;COCCc1ccc(OCC(O)CNC(C)C)cc1;1;1;0;;;;;
+556;Metronidazole;Cc1ncc(n1CCO)N(=O)=O;1;1;0;;;;;
+557;Metyrapone;CC(C)(C(=O)c1cccnc1)c1cccnc1;1;1;0;;;;;
+558;Mianserin;CN1CCN2C(C1)c1ccccc1Cc1ccccc21;1;1;0;;;;;
+559;Micafungin;CCCCCOc1ccc(cc1)-c1cc(no1)-c1ccc(cc1)C(=O)NC1CC(O)C(O)NC(=O)C2C(O)C(C)CN2C(=O)C(NC(=O)C(NC(=O)C2CC(O)CN2C(=O)C(NC1=O)C(C)O)C(O)C(O)c1ccc(O)c(OS(O)(=O)=O)c1)C(O)CC(N)=O;1;0;0;;;;;
+560;Miconazole;Clc1ccc(COC(Cn2ccnc2)c2ccc(Cl)cc2Cl)c(Cl)c1;1;1;0;;;;;
+561;Midazolam;Cc1ncc2CN=C(c3ccccc3F)c3cc(Cl)ccc3-n12;0;1;0;;;;;
+562;Mifepristone;CC#CC1(O)CCC2C3CCC4=CC(=O)CCC4=C3C(CC12C)c1ccc(cc1)N(C)C;0;1;0;;;;;
+563;Miglustat;CCCCN1CC(O)C(O)C(O)C1CO;0;1;0;;;;;
+564;Miltefosine;CCCCCCCCCCCCCCCCOP(O)(=O)OCC[N+](C)(C)C;1;0;0;;;;;
+565;Minocycline;CN(C)C1C2CC3Cc4c(ccc(O)c4C(=O)C3=C(O)C2(O)C(=O)C(C(N)=O)=C1O)N(C)C;1;1;0;;;;;
+566;Minoxidil;NC1=CC(=NC(=N)N1O)N1CCCCC1;1;1;0;;;;;
+567;Mirtazapine;CN1CCN2C(C1)c1ccccc1Cc1cccnc21;1;0;0;;;;;
+568;Misoprostol;CCCCC(C)(O)CC=CC1C(O)CC(=O)C1CCCCCCC(=O)OC;0;1;1;;;;;
+569;Mithramycin;COC(C1Cc2cc3cc(OC4CC(OC5CC(O)C(O)C(C)O5)C(O)C(C)O4)c(C)c(O)c3c(O)c2C(=O)C1OC1CC(OC2CC(OC3CC(C)(O)C(O)C(C)O3)C(O)C(C)O2)C(O)C(C)O1)C(=O)C(O)C(C)O;1;1;0;;;;;
+570;mitiglinide;OC(=O)C(CC(=O)N1CC2CCCCC2C1)Cc1ccccc1;0;1;0;;;;;
+571;Mitomycin;COC12C3NC3CN1C1=C(C2COC(N)=O)C(=O)C(N)=C(C)C1=O;1;1;0;;;;;
+572;Mitotane;ClC(Cl)C(c1ccc(Cl)cc1)c1ccccc1Cl;1;0;1;;;;;
+573;Mitoxantrone;OCCNCCNc1ccc(NCCNCCO)c2C(=O)c3c(O)ccc(O)c3C(=O)c12;1;1;0;;;;;
+574;Mizolastine;CN(C1CCN(CC1)c1nc2ccccc2n1Cc1ccc(F)cc1)C1=NC=CC(=O)N1;1;0;0;;;;;
+575;Modafinil;NC(=O)CS(=O)C(c1ccccc1)c1ccccc1;1;0;0;;;;;
+576;Molsidomine;CCOC(=O)Nc1c[n+](no1)N1CCOCC1;1;1;0;;;;;
+577;Monensin sodium;CCC1(CCC(O1)C1(C)CCC2(CC(O)C(C)C(O2)C(C)C(OC)C(C)C(O)=O)O1)C1OC(CC1C)C1OC(O)(CO)C(C)CC1C;0;1;0;;;;;
+578;Montelukast;CC(C)(O)c1ccccc1CCC(SCC1(CC1)CC(O)=O)c1cccc(C=Cc2ccc3ccc(Cl)cc3n2)c1;1;0;0;;;;;
+579;Morphazinamide;O=C(NCN1CCOCC1)c1cnccn1;1;0;0;;;;;
+580;Morphine;CN1CCC23C4Oc5c(O)ccc(CC1C2C=CC4O)c35;1;1;0;;;;;
+581;Moxalactam;COC1(NC(=O)C(C(O)=O)c2ccc(O)cc2)C2OCC(CSc3nnnn3C)=C(N2C1=O)C(O)=O;1;0;0;;;;;
+582;Moxifloxacin;COc1c(N2CC3CCCNC3C2)c(F)cc2C(=O)C(=CN(C3CC3)c12)C(O)=O;1;0;0;;;;;
+583;Moxisylyte;CC(C)c1cc(OC(C)=O)c(C)cc1OCCN(C)C;1;0;0;;;;;
+584;Moxonidine;COc1nc(C)nc(Cl)c1\N=C1/NCCN1;0;1;0;;;;;
+585;Muzolimine;CC(N1N=C(N)CC1=O)c1ccc(Cl)c(Cl)c1;1;0;0;;;;;
+586;Mycophenolate;COc1c(C)c2COC(=O)c2c(O)c1CC=C(C)CCC(O)=O;1;1;0;;;;;
+587;Mycophenolate mofetil;COc1c(C)c2COC(=O)c2c(O)c1CC=C(C)CCC(=O)OCCN1CCOCC1;1;0;0;;;;;
+588;Nabumetone;COc1ccc2cc(CCC(C)=O)ccc2c1;0;1;0;;;;;
+589;Nadolol;CC(C)(C)NCC(O)COc1cccc2CC(O)C(O)Cc12;1;0;0;;;;;
+590;Nafarelin;CC(C)CC(NC(=O)C(N)Cc1ccc2ccccc2c1)C(=O)NC(CCCNC(N)=N)C(=O)N1CCCC1C(=O)NCC(=O)NC(=O)C(Cc1ccc(O)cc1)NC(=O)C(CO)NC(=O)C(Cc1c[nH]c2ccccc12)NC(=O)C(CC1C=NC=N1)NC(=O)C1CCC(=O)N1;1;0;0;;;;;
+591;Nalidixic acid;CCN1C=C(C(O)=O)C(=O)c2ccc(C)nc12;1;1;0;;;;;
+592;Nalorphine;OC1C=CC2C3Cc4ccc(O)c5OC1C2(CCCN3CC=C)c45;0;1;0;;;;;
+593;Naloxone;Oc1ccc2CC3N(CCC45C(Oc1c24)C(=O)CCC35O)CC=C;0;1;0;;;;;
+594;Naltrexone;Oc1ccc2CC3N(CCC45C(Oc1c24)C(=O)CCC35O)CC1CC1;1;1;0;;;;;
+595;Nandrolone Decanoate;CCCCCCCCCC(=O)OC1CCC2C3CCC4=CC(=O)CCC4C3CCC12C;0;1;0;;;;;
+596;Naproxen;COc1ccc2cc(ccc2c1)C(C)C(O)=O;1;1;0;;;;;
+597;N-aspartyl chlorin e6;CCc1c(C)c2cc3[nH]c(cc4nc(C(CCC(O)=O)C4C)c(CC(=O)NC(CC(O)=O)C(O)=O)c4nc(cc1[nH]2)c(C)c4C(O)=O)c(C)c3C=C;0;1;0;;;;;
+598;Natamycin;CC1CC=CC=CC=CC=CC(CC2OC(O)(CC(O)CC3OC3C=CC(=O)O1)CC(O)C2C(O)=O)OC1OC(C)C(O)C(N)C1O;0;0;1;;;;;
+599;Nefazodone;CCC1=NN(CCCN2CCN(CC2)c2cccc(Cl)c2)C(=O)N1CCOc1ccccc1;1;0;0;;;;;
+600;Nelfinavir;Cc1c(O)cccc1C(=O)NC(CSc1ccccc1)C(O)CN1CC2CCCCC2CC1C(=O)NC(C)(C)C;1;0;0;;;;;
+601;Neomycin;NCC1CC(OC2C(N)CC(N)C(O)C2OC2OC(CO)C(OC3OC(CN)C(O)C(O)C3N)C2O)C(N)C(O)C1O;0;1;0;;;;;
+602;Neomycin sulfate;NCC1OC(OC2C(N)CC(N)C(O)C2OC2OC(CO)C(OC3OC(CN)C(O)C(O)C3N)C2O)C(N)C(O)C1O;0;1;0;;;;;
+603;Neostigmine;CN(C)C(=O)Oc1cccc(c1)[N+](C)(C)C;0;1;0;;;;;
+604;Netilmicin;CCNC1CC(N)C(OC2OC(CN)=CCC2N)C(O)C1OC1OCC(C)(O)C(NC)C1O;1;1;0;;;;;
+605;Nevirapine;Cc1ccnc2N(C3CC3)c3ncccc3C(=O)Nc12;1;0;0;;;;;
+606;Niacin;OC(=O)c1cccnc1;1;1;0;;;;;
+607;Nialamide;O=C(CCNNC(=O)c1ccncc1)NCc1ccccc1;0;1;0;;;;;
+608;Nicergoline;COC12CC(COC(=O)c3cncc(Br)c3)CN(C)C1Cc1cn(C)c3cccc2c13;0;1;0;;;;;
+609;Niceritrol;O=C(OCC(COC(=O)c1cccnc1)(COC(=O)c1cccnc1)COC(=O)c1cccnc1)c1cccnc1;0;0;1;;;;;
+610;Nicorandil;O=C(NCCON(=O)=O)c1cccnc1;0;1;0;;;;;
+611;Nicotinamide;NC(=O)c1cccnc1;1;1;0;;;;;
+612;Nicotinamide adenine dinucleotide;NC(=O)c1ccc[n+](c1)C1OC(COP(O)(=O)OP(O)(=O)OCC2OC(C(O)C2O)n2cnc3c(N)ncnc23)C(O)C1O;0;1;0;;;;;
+613;Nicotine;CN1CCCC1c1cccnc1;1;1;0;;;;;
+614;Nicotinic acid;OC(=O)c1cccnc1;1;1;0;;;;;
+615;Nifedipine;COC(=O)C1=C(C)NC(C)=C(C1c1ccccc1N(=O)=O)C(=O)OC;1;1;1;;;;;
+616;Niflumic acid;OC(=O)c1cccnc1Nc1cccc(c1)C(F)(F)F;0;1;0;;;;;
+617;Nifurtimox;CC1CS(=O)(=O)CCN1N=Cc1ccc(o1)N(=O)=O;0;1;0;;;;;
+618;Nifurtoinol;OCN1C(=O)CN(CCc2ccc(o2)N(=O)=O)C1=O;1;0;0;;;;;
+619;Nilutamide;CC1(C)NC(=O)N(c2ccc(c(c2)C(F)(F)F)N(=O)=O)C1=O;1;1;0;;;;;
+620;Nimesulide;CS(=O)(=O)Nc1ccc(cc1Oc1ccccc1)N(=O)=O;1;1;0;;;;;
+621;Nimodipine;COCCOC(=O)C1=C(C)NC(C)=C(C1c1cccc(c1)N(=O)=O)C(=O)OC(C)C;1;1;0;;;;;
+622;Niridazole;O=C1NCCN1c1ncc(s1)N(=O)=O;0;1;0;;;;;
+623;Nisoldipine;COC(=O)C1=C(C)NC(C)=C(C1c1ccccc1N(=O)=O)C(=O)OCC(C)C;1;1;1;;;;;
+624;Nitisinone;FC(F)(F)c1ccc(C(=O)C2C(=O)CCCC2=O)c(c1)N(=O)=O;0;1;1;;;;;
+625;Nitrazepam;O=C1CN=C(c2ccccc2)c2cc(ccc2N1)N(=O)=O;0;1;0;;;;;
+626;Nitrendipine;CCOC(=O)C1=C(C)NC(C)=C(C1c1cccc(c1)N(=O)=O)C(=O)OC;1;1;0;;;;;
+627;Nitrofurantoin;O=C1CN(N=Cc2ccc(o2)N(=O)=O)C(=O)N1;1;1;1;;;;;
+628;Nitrofurazone;NC(=O)NN=Cc1ccc(o1)N(=O)=O;0;1;0;;;;;
+629;Nitroglycerin;O=N(=O)OCC(CON(=O)=O)ON(=O)=O;1;1;0;;;;;
+630;Nizatidine;CNC(NCCSCc1csc(CN(C)C)n1)=CN(=O)=O;1;1;0;;;;;
+631;N-methylglucamine;CNCC(O)C(O)C(O)C(O)CO;1;0;0;;;;;
+632;Nordihydroguaiaretic acid;CC(Cc1ccc(O)c(O)c1)C(C)Cc1ccc(O)c(O)c1;1;1;0;;;;;
+633;Norepinephrine;NCC(O)c1ccc(O)c(O)c1;1;1;1;;;;;
+634;Norethindrone;CC12CCC3C(CCC4=CC(=O)CCC34)C1CCC2(O)C#C;1;1;0;;;;;
+635;Norethindrone acetate;CC(=O)OC1(CCC2C3CCC4=CC(=O)CCC4C3CCC12C)C#C;0;1;1;;;;;
+636;Norethynodrel;CC12CCC3C(CCC4=C3CCC(=O)C4)C1CCC2(O)C#C;0;1;0;;;;;
+637;Norfloxacin;CCN1C=C(C(O)=O)C(=O)c2cc(F)c(cc12)N1CCNCC1;1;1;0;;;;;
+638;Norgestrel;CCC12CCC3C(CCC4=CC(=O)CCC34)C1CCC2(O)C#C;1;1;0;;;;;
+639;Nortriptyline;CNCC\C=C1\c2ccccc2CCc2ccccc12;1;0;0;;;;;
+640;Noscapine;COc1ccc2C(OC(=O)c2c1OC)C1N(C)CCc2cc3OCOc3c(OC)c12;0;1;0;;;;;
+641;Novobiocin;COC1C(OC(N)=O)C(O)C(Oc2ccc3C(O)=C(NC(=O)c4ccc(O)c(C\C=C(\C)C)c4)C(=O)Oc3c2C)OC1(C)C;1;1;0;;;;;
+642;Octapressin;NCCCCC(NC(=O)C1CCCN1C(=O)C1CSSCC(N)C(=O)NC(Cc2ccccc2)C(=O)NC(Cc2ccccc2)C(=O)NC(CCC(N)=O)C(=O)NC(CC(N)=O)C(=O)N1)C(=O)NCC(N)=O;0;0;1;;;;;
+643;Octreotide;CC(O)C(CO)NC(=O)C1CSSCC(NC(=O)C(N)Cc2ccccc2)C(=O)NC(Cc2ccccc2)C(=O)NC(Cc2c[nH]c3ccccc23)C(=O)NC(CCCCN)C(=O)NC(C(C)O)C(=O)N1;1;1;0;;;;;
+644;Olanzapine;CN1CCN(CC1)C1=Nc2ccccc2Nc2sc(C)cc12;1;0;0;;;;;
+645;Olmesartan medoxomil;O=C1O/C(=C(\O1)C)COC(=O)c2c(nc(n2Cc5ccc(c4ccccc4c3nnnn3)cc5)CCC)CC;0;1;0;;;;;
+646;Omeprazole;O=S(c2nc1ccc(OC)cc1n2)Cc3ncc(c(OC)c3C)C;1;1;0;;;;;
+647;Ondansetron;Cc1nccn1CC1CCc2c(C1=O)c1ccccc1n2C;1;0;0;;;;;
+648;Orlistat;CCCCCCCCCCCC(CC1OC(=O)C1CCCCCC)OC(=O)C(CC(C)C)NC=O;1;0;0;;;;;
+649;Orotic acid;OC(=O)C1=CC(=O)NC(=O)N1;0;1;0;;;;;
+650;Orphenadrine;CN(C)CCOC(c1ccccc1)c1ccccc1C;0;1;0;;;;;
+651;OSI-461;CC1=C(CC(=O)NCc2ccccc2)c2cc(F)ccc2C1=Cc1ccncc1;1;0;0;;;;;
+652;Oxacillin sodium;Cc1onc(-c2ccccc2)c1C(=O)NC1C2SC(C)(C)C(N2C1=O)C(O)=O;1;0;0;;;;;
+653;Oxaliplatin;NC1CCCCC1N;1;0;0;;;;;
+654;Oxamniquine;CC(C)NCC1CCc2cc(CO)c(cc2N1)N(=O)=O;0;1;0;;;;;
+655;Oxandrolone;CC1(O)CCC2C3CCC4CC(=O)OCC4(C)C3CCC12C;0;1;0;;;;;
+656;Oxaprozin;OC(=O)CCc1nc(-c2ccccc2)c(o1)-c1ccccc1;1;0;0;;;;;
+657;Oxazepam;OC1N=C(c2ccccc2)c2cc(Cl)ccc2NC1=O;1;1;0;;;;;
+658;Oxcarbazepine;NC(=O)N1c2ccccc2CC(=O)c2ccccc12;1;0;0;;;;;
+659;Oxethazaine;CN(C(=O)CN(CCO)CC(=O)N(C)C(C)(C)Cc1ccccc1)C(C)(C)Cc1ccccc1;0;1;0;;;;;
+660;Oxprenolol;CC(C)NCC(O)COc1ccccc1OCC=C;0;0;1;;;;;
+661;Oxybenzone;COc1ccc(C(=O)c2ccccc2)c(O)c1;0;1;0;;;;;
+662;Oxycodone hydrochloride;COc1ccc2CC3N(C)CCC45C(Oc1c24)C(=O)CCC35O;0;1;0;;;;;
+663;Oxymetholone;CC1(O)CCC2C3CCC4CC(=O)C(CC4(C)C3CCC12C)=CO;1;0;0;;;;;
+664;Oxymorphone;CN1CCC23C4Oc5c(O)ccc(CC1C2(O)CCC4=O)c35;1;0;0;;;;;
+665;Oxyphenbutazone;CCCCC1C(=O)N(N(C1=O)c1ccc(O)cc1)c1ccccc1;0;1;0;;;;;
+666;Oxyphenisatin;CC(=O)Oc1ccc(cc1)C1(C(=O)Nc2ccccc12)c1ccc(OC(C)=O)cc1;1;0;0;;;;;
+667;Oxypurinol;Oc1nc(O)c2cn[nH]c2n1;0;1;0;;;;;
+668;Oxytetracycline;CN(C)C1C2C(O)C3C(C(=O)c4c(O)cccc4C3(C)O)=C(O)C2(O)C(=O)C(C(N)=O)=C1O;0;1;0;;;;;
+669;Ozagrel;OC(=O)C=Cc1ccc(Cn2ccnc2)cc1;0;1;0;;;;;
+670;Paclitaxel;CC(=O)OC1C(=O)C2(C)C(O)CC3OCC3(OC(C)=O)C2C(OC(=O)c2ccccc2)C2(O)CC(OC(=O)C(O)C(NC(=O)c3ccccc3)c3ccccc3)C(C)=C1C2(C)C;1;1;0;;;;;
+671;Pantoprazole;COc1cc[nH]c(CS(=O)c2nc3ccc(OC(F)F)cc3[nH]2)c1OC;1;0;0;;;;;
+672;para-Aminosalicylic acid;Nc1ccc(C(O)=O)c(O)c1;1;0;0;;;;;
+673;Paromomycin;NCC1OC(OC2C(CO)OC(OC3C(O)C(N)CC(N)C3OC3OC(CO)C(O)C(O)C3N)C2O)C(N)C(O)C1O;0;1;0;;;;;
+674;Paroxetine;Fc1ccc(cc1)C1CCNCC1COc1ccc2OCOc2c1;1;0;0;;;;;
+675;Pemoline;NC1=NC(=O)C(O1)c1ccccc1;1;0;0;;;;;
+676;Penciclovir;NC1=NC(=O)c2ncn(CCC(CO)CO)c2N1;1;0;0;;;;;
+677;Penicillamine;CC(C)(S)C(N)C(O)=O;1;1;1;;;;;
+678;Penicillin;CC1(C)SC2C(NC(=O)Cc3ccccc3)C(=O)N2C1C(O)=O;1;1;1;;;;;
+679;Pentagastrin;CSCCC(NC(=O)C(Cc1c[nH]c2ccccc12)NC(=O)CCNC(=O)OCC(C)C)C(=O)NC(CC(O)=O)C(=O)NC(Cc1ccccc1)C(N)=O;1;1;0;;;;;
+680;Pentazocine;CC1C2Cc3ccc(O)cc3C1(C)CCN2C\C=C(\C)C;0;1;0;;;;;
+681;Pentobarbital;CCCC(C)C1(CC)C(=O)NC(=O)NC1=O;1;1;1;;;;;
+682;Pentoxifylline;CN1C(=O)N(CCCCC(C)=O)C(=O)c2c1ncn2C;1;1;1;;;;;
+683;Perazine;CN1CCN(CCCN2c3ccccc3Sc3ccccc23)CC1;1;0;0;;;;;
+684;Perfluorooctylbromide;FC(F)(F)C(F)(F)C(F)(F)C(F)(F)C(F)(F)C(F)(F)C(F)(F)C(F)(F)Br;1;0;1;;;;;
+685;Perindopril;CCCC(NC(C)C(=O)N1C2CCCCC2CC1C(O)=O)C(=O)OCC;1;1;0;;;;;
+686;Permethrin;CC1(C)C(\C=C(\Cl)Cl)C1C(=O)OCc1cccc(Oc2ccccc2)c1;0;1;0;;;;;
+687;Perphenazine;OCCN1CCN(CCCN2c3ccccc3Sc3ccc(Cl)cc23)CC1;0;1;0;;;;;
+688;Phenacetin;CCOc1ccc(NC(C)=O)cc1;1;1;0;;;;;
+689;Phenelzine;NNCCc1ccccc1;1;1;0;;;;;
+690;Phenformin;NC(=N)NC(=N)NCCc1ccccc1;0;1;0;;;;;
+691;Phenobarbital;CCC1(C(=O)NC(=O)NC1=O)c1ccccc1;1;1;1;;;;;
+692;Phenol;Oc1ccccc1;1;1;0;;;;;
+693;Phenolphthalein;Oc1ccc(cc1)C1(OC(=O)c2ccccc12)c1ccc(O)cc1;0;1;0;;;;;
+694;Phenoxybenzamine;CC(COc1ccccc1)N(CCCl)Cc1ccccc1;0;1;0;;;;;
+695;Phenprocoumon;CCC(c1ccccc1)C1=C(O)Oc2ccccc2C1=O;1;0;0;;;;;
+696;Phentolamine;Cc1ccc(cc1)N(CC1=NCCN1)c1cccc(O)c1;0;1;1;;;;;
+697;phenylacetate;OC(=O)Cc1ccccc1;0;1;0;;;;;
+698;Phenylbutazone;CCCCC1C(=O)N(N(C1=O)c1ccccc1)c1ccccc1;1;1;0;;;;;
+699;Phenylephrine;CNCC(O)c1cccc(O)c1;1;1;1;;;;;
+700;Phenylpropanolamine;CC(N)C(O)c1ccccc1;0;1;0;;;;;
+701;Phenyramidol;OC(CNc1ccccn1)c1ccccc1;1;0;0;;;;;
+702;Phenytoin;O=C1NC(=O)C(N1)(c1ccccc1)c1ccccc1;1;1;1;;;;;
+703;Physostigmine;CNC(=O)Oc1ccc2N(C)C3N(C)CCC3(C)c2c1;0;1;0;;;;;
+704;Picosulfate sodium;OS(=O)(=O)Oc1ccc(cc1)C(c1ccc(OS(O)(=O)=O)cc1)c1ccccn1;1;0;0;;;;;
+705;Pilocarpine;CCC1C(COC1=O)Cc1cncn1C;0;1;0;;;;;
+706;Pimozide;Fc1ccc(cc1)C(CCCN1CCC(CC1)N1C(=O)Nc2ccccc12)c1ccc(F)cc1;1;0;0;;;;;
+707;Pinacidil;CC(NC(Nc1cc[nH]cc1)=NC#N)C(C)(C)C;1;1;0;;;;;
+708;Pipemidic Acid;CCN1C=C(C(O)=O)C(=O)c2cnc(nc12)N1CCNCC1;1;0;0;;;;;
+709;Piperacillin;CCN1CCN(C(=O)NC(C(=O)NC2C3SC(C)(C)C(N3C2=O)C(O)=O)c2ccccc2)C(=O)C1=O;0;1;1;;;;;
+710;Piperazine;C1CNCCN1;0;1;0;;;;;
+711;Piperonyl butoxide;CCCCOCCOCCOCc1cc2OCOc2cc1CCC;0;1;0;;;;;
+712;Pirfenidone;CC1=CN(C(=O)C=C1)c1ccccc1;0;1;1;;;;;
+713;Piritrexim;COc1ccc(OC)c(Cc2cnc3[nH]c(N)nc(N)c3c2C)c1;1;0;0;;;;;
+714;Piroxicam;CN1C(C(=O)c2ccccc2S1(=O)=O)=C(O)Nc1ccccn1;1;1;0;;;;;
+715;Pirprofen;CC(C(O)=O)c1ccc(N2CC=CC2)c(Cl)c1;1;1;0;;;;;
+716;Pitavastatin;OC(CC(O)C=Cc1c(nc2ccccc2c1-c1ccc(F)cc1)C1CC1)CC(O)=O;1;1;0;;;;;
+717;Pizotyline;CN1CC\C(CC1)=C1/c2ccccc2CCc2sccc12;1;0;0;;;;;
+718;Podophyllotoxin;COc1cc(cc(OC)c1OC)C1C2C(COC2=O)C(O)c2cc3OCOc3cc12;0;1;0;;;;;
+719;Polyethylene glycol;OCCOCCOCCOCCOCCOCCOCCOCCOCCO;1;1;0;;;;;
+720;Polymyxin B;CCC(C)CCCCC(=O)NC(CCN)C(=O)NC(C(C)O)C(=O)NC(CCN)C(=O)NC1CCNC(=O)C(NC(=O)C(CCN)NC(=O)C(CCN)NC(=O)C(CC(C)C)NC(=O)C(Cc2ccccc2)NC(=O)C(CCN)NC1=O)C(C)O;1;1;0;;;;;
+721;Polyvinyl alcohol;OC=C;0;0;1;;;;;
+722;Porfimer sodium;CCc1c(C)c2cc3nc(cc4nc(cc5[nH]c(cc1[nH]2)c(C)c5C(C)O)c(C)c4CCC(O)=O)c(CCC(O)=O)c3C;0;1;0;;;;;
+723;Prasterone;CC12CCC3C(CC=C4CC(O)CCC34C)C1CCC2=O;1;1;0;;;;;
+724;Pravastatin;CCC(C)C(=O)OC1CC(O)C=C2C=CC(C)C(CCC(O)CC(O)CC(O)=O)C12;1;1;1;;;;;
+725;Praziquantel;O=C1CN(CC2N1CCc1ccccc21)C(=O)C1CCCCC1;0;1;1;;;;;
+726;Prazosin;COc1cc2nc(nc(N)c2cc1OC)N1CCN(CC1)C(=O)c1ccco1;1;1;0;;;;;
+727;Prednisolone;CC12CC(O)C3C(CCC4=CC(=O)C=CC34C)C1CCC2(O)C(=O)CO;1;1;0;;;;;
+728;Prednisone;CC12CC(=O)C3C(CCC4=CC(=O)C=CC34C)C1CCC2(O)C(=O)CO;1;0;1;;;;;
+729;Primaquine;COc1cc(NC(C)CCCN)c2ncccc2c1;1;1;0;;;;;
+730;Primidone;CCC1(C(=O)NCNC1=O)c1ccccc1;1;0;1;;;;;
+731;Probenecid;CCCN(CCC)S(=O)(=O)c1ccc(cc1)C(O)=O;1;1;0;;;;;
+732;Probucol;CC(C)(Sc1cc(c(O)c(c1)C(C)(C)C)C(C)(C)C)Sc1cc(c(O)c(c1)C(C)(C)C)C(C)(C)C;1;1;1;;;;;
+733;Procainamide;CCN(CC)CCNC(=O)c1ccc(N)cc1;1;1;0;;;;;
+734;Procarbazine;CNNCc1ccc(cc1)C(=O)NC(C)C;0;1;0;;;;;
+735;Prochlorperazine;CN1CCN(CCCN2c3ccccc3Sc3ccc(Cl)cc23)CC1;1;0;0;;;;;
+736;Progesterone;CC(=O)C1CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;1;1;1;;;;;
+737;Progestin;CC(=O)C1CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;1;0;0;;;;;
+738;Promethazine;CC(CN1c2ccccc2Sc2ccccc12)N(C)C;0;1;1;;;;;
+739;Propan-2-ol;CC(C)O;0;1;0;;;;;
+740;Propofol;CC(C)c1cccc(C(C)C)c1O;1;1;1;;;;;
+741;Propoxyphene;CCC(=O)OC(Cc1ccccc1)(C(C)CN(C)C)c1ccccc1;1;1;0;;;;;
+742;Propranolol;CC(C)NCC(O)COc1cccc2ccccc12;1;1;1;;;;;
+743;Propylthiouracil;CCCC1=CC(=O)NC(=S)N1;1;1;0;;;;;
+744;Prostaglandin E1;CCCCCC(O)C=CC1C(O)CC(=O)C1CCCCCCC(O)=O;1;1;1;;;;;
+745;Prostaglandin E2;CCCCCC(O)C=CC1C(O)CC(=O)C1CC=CCCCC(O)=O;1;1;0;;;;;
+746;Prostaglandin F2a;CCCCCC(O)C=CC1C(O)CC(O)C1CC=CCCCC(O)=O;1;1;1;;;;;
+747;Protionamide;CCCc1cc(ccn1)C(N)=S;1;0;0;;;;;
+748;Pyrazinamide;NC(=O)c1cnccn1;1;1;0;;;;;
+749;Pyridinol Carbamate;CNC(=O)OCc1cccc(COC(=O)NC)n1;1;1;0;;;;;
+750;Pyridoxal 5-Phosphate;Cc1ncc(COP(O)(O)=O)c(C=O)c1O;0;1;0;;;;;
+751;Pyridoxine;Cc1ncc(CO)c(CO)c1O;0;1;0;;;;;
+752;Pyrimethamine;CCc1nc(N)nc(N)c1-c1ccc(Cl)cc1;1;1;0;;;;;
+753;Quinacrine;CCN(CC)CCCC(C)Nc1c2ccc(Cl)cc2[nH]c2ccc(OC)cc12;1;1;1;;;;;
+754;Quinapril;CCOC(=O)C(CCc1ccccc1)NC(C)C(=O)N1Cc2ccccc2CC1C(O)=O;1;0;0;;;;;
+755;Quinestrol;CC12CCC3C(CCc4cc(OC5CCCC5)ccc34)C1CCC2(O)C#C;0;1;0;;;;;
+756;Quinidine;COc1ccc2nccc(C(O)C3CC4CCN3CC4C=C)c2c1;1;1;0;;;;;
+757;ragaglitazar;CCOC(Cc1ccc(OCCN2c3ccccc3Oc3ccccc23)cc1)C(O)=O;0;1;0;;;;;
+758;Raltitrexed;CN(Cc1ccc2NC(C)=NC(=O)c2c1)c1ccc(s1)C(=O)NC(CCC(O)=O)C(O)=O;1;0;0;;;;;
+759;Ramipril;CCOC(=O)C(CCc1ccccc1)NC(C)C(=O)N1C2CCCC2CC1C(O)=O;1;1;0;;;;;
+760;Ranitidine;CNC(NCCSCc1ccc(CN(C)C)o1)=CN(=O)=O;1;1;1;;;;;
+761;Rapamycin;COC1CC(CCC1O)CC(C)C1CC(=O)C(C)C=C(C)C(O)C(OC)C(=O)C(C)CC(C)C=CC=CC=C(C)C(CC2CCC(C)C(O)(O2)C(=O)C(=O)N2CCCCC2C(=O)O1)OC;1;1;0;;;;;
+762;Rebamipide;OC(=O)C(CC1=CC(=O)Nc2ccccc12)NC(=O)c1ccc(Cl)cc1;0;1;0;;;;;
+763;Reserpine;COC1C(CC2CN3CCc4c([nH]c5cc(OC)ccc45)C3CC2C1C(=O)OC)OC(=O)c1cc(OC)c(OC)c(OC)c1;0;1;0;;;;;
+764;Retinoic acid;CC(C=CC1=C(C)CCCC1(C)C)=CC=CC(C)=CC(O)=O;1;1;0;;;;;
+765;Ribavirin;NC(=O)c1ncn(n1)C1OC(CO)C(O)C1O;1;1;0;;;;;
+766;Riboflavin;Cc1cc2N=C3C(=O)NC(=O)N=C3N(CC(O)C(O)C(O)CO)c2cc1C;1;0;0;;;;;
+767;Rifabutin;COC1C=COC2(C)Oc3c(C)c(O)c4C(=O)C(NC(=O)C(C)=CC=CC(C)C(O)C(C)C(O)C(C)C(OC(C)=O)C1C)=C1NC5(CCN(CC5)CC(C)C)N=C1c4c3C2=O;1;1;0;;;;;
+768;Rifampicin;COC1C=COC2(C)OC3=C(C2=O)C2=C(O)C(=CNN4CCN(C)CC4)C(=NC(=O)C(C)=CC=CC(C)C(O)C(C)C(O)C(C)C(OC(C)=O)C1C)C(O)=C2C(O)=C3C;1;1;1;;;;;
+769;Rifamycin SV;COC1C=COC2(C)Oc3c(C)c(O)c4c(O)c(NC(=O)C(C)=CC=CC(C)C(O)C(C)C(O)C(C)C(OC(C)=O)C1C)cc(O)c4c3C2=O;1;1;0;;;;;
+770;Riluzole;Nc1nc2ccc(OC(F)(F)F)cc2s1;1;0;0;;;;;
+771;rimonabant;Cc1c(nn(-c2ccc(Cl)cc2Cl)c1-c1ccc(Cl)cc1)C(=O)NN1CCCCC1;0;1;0;;;;;
+772;Risperidone;CC1=C(CCN2CCC(CC2)c2noc3cc(F)ccc23)C(=O)N2CCCCC2=N1;1;0;0;;;;;
+773;Ritodrine hydrochloride;CC(NCCc1ccc(O)cc1)C(O)c1ccc(O)cc1;0;0;1;;;;;
+774;Ritonavir;CC(C)C(NC(=O)N(C)Cc1csc(n1)C(C)C)C(=O)NC(CC(O)C(Cc1ccccc1)NC(=O)OCc1cncs1)Cc1ccccc1;1;1;0;;;;;
+775;Rofecoxib;CS(=O)(=O)c1ccc(cc1)C1=C(C(=O)OC1)c1ccccc1;1;1;0;;;;;
+776;Rolitetracycline;CN(C)C1C2CC3C(C(=O)c4c(O)cccc4C3(C)O)=C(O)C2(O)C(=O)C(C(=O)NCN2CCCC2)=C1O;0;1;0;;;;;
+777;Rose Bengal;OC(=O)c1c(Cl)c(Cl)c(Cl)c(Cl)c1C1=C2C=C(I)C(=O)C(I)=C2Oc2c(I)c(O)c(I)cc12;0;1;0;;;;;
+778;Rosiglitazone;CN(CCOc1ccc(CC2SC(=O)NC2=O)cc1)c1ccccn1;1;1;0;;;;;
+779;Rosuvastatin;CC(C)c1[nH]c(nc(-c2ccc(F)cc2)c1C=CC(O)CC(O)CC(O)=O)N(C)S(C)(=O)=O;0;1;0;;;;;
+780;Roxithromycin;CCC1OC(=O)C(C)C(OC2CC(C)(OC)C(O)C(C)O2)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(O)CC(C)C(=NOCOCCOC)C(C)C(O)C1(C)O;1;1;0;;;;;
+781;Rubitecan;CCC1(O)C(=O)OCC2=C1C=C1N(Cc3cc4c(cccc4nc13)N(=O)=O)C2=O;1;0;0;;;;;
+782;Salbutamol;CC(C)(C)NCC(O)c1ccc(O)c(CO)c1;0;1;0;;;;;
+783;Salicylamide;NC(=O)c1ccccc1O;1;1;0;;;;;
+784;Salicylic acid;OC(=O)c1ccccc1O;0;1;0;;;;;
+785;Saquinavir;CC(C)(C)NC(=O)C1CC2CCCCC2CN1CC(O)C(Cc1ccccc1)NC(=O)C(CC(N)=O)NC(=O)c1ccc2ccccc2n1;1;0;0;;;;;
+786;Scillaren;CC1OC(OC2CCC3(C)C4CCC5(C)C(CCC5(O)C4CCC3=C2)C2=CC(=O)OC=C2)C(O)C(O)C1OC1OC(CO)C(O)C(O)C1O;0;1;0;;;;;
+787;Selegiline;CC(Cc1ccccc1)N(C)CC#C;1;0;0;;;;;
+788;Selenomethionine;C[Se]CCC(N)C(O)=O;0;1;1;;;;;
+789;Semaxanib;Cc1cc(C)c(C=C2C(=O)Nc3ccccc23)[nH]1;1;1;0;;;;;
+790;Seocalcitol;CCC(O)(CC)C=CC=CC(C)C1CCC2C(CCCC12C)=CC=C1CC(O)CC(O)C1=C;1;0;0;;;;;
+791;Sertraline;CNC1CCC(c2ccc(Cl)c(Cl)c2)c2ccccc12;1;1;0;;;;;
+792;Sevoflurane;FCOC(C(F)(F)F)C(F)(F)F;1;1;0;;;;;
+793;Sibutramine;CC(C)CC(N(C)C)C1(CCC1)c1ccc(Cl)cc1;1;0;0;;;;;
+794;Sildenafil;CCCc1nn(C)c2C(=O)NC(=Nc12)c1cc(ccc1OCC)S(=O)(=O)N1CCN(C)CC1;1;0;0;;;;;
+795;silybin;COc1cc(ccc1O)C1Oc2cc(ccc2OC1CO)C1Oc2cc(O)cc(O)c2C(=O)C1O;1;1;0;;;;;
+796;Silymarin;COc1cc(ccc1O)C1Oc2cc(ccc2OC1CO)C1Oc2cc(O)cc(O)c2C(=O)C1O;1;1;0;;;;;
+797;Simvastatin;CCC(C)(C)C(=O)OC1CC(C)C=C2C=CC(C)C(CCC3CC(O)CC(=O)O3)C12;1;1;1;;;;;
+798;Sitaxsentan;Cc1cc2OCOc2cc1CC(=O)c1sccc1S(=O)(=O)Nc1onc(C)c1Cl;1;0;0;;;;;
+799;Sodium acetate;CC(O)=O;0;1;0;;;;;
+800;Sodium benzoate;OC(=O)c1ccccc1;0;1;0;;;;;
+801;Sodium bicarbonate;OC(O)=O;0;1;1;;;;;
+802;SODIUM PHENYLBUTYRATE;OC(=O)CCCc1ccccc1;1;0;0;;;;;
+803;Sodium propionate;CCC(O)=O;0;1;0;;;;;
+804;Sorafenib tosylate;CNC(=O)c1cc(Oc2ccc(NC(=O)Nc3ccc(Cl)c(c3)C(F)(F)F)cc2)ccn1;1;0;0;;;;;
+805;Sorbitol;O=C([C@H](O)[C@@H](O)[C@H](O)CO)CO;1;1;0;;;;;
+806;Sotalol;CC(C)NCC(O)c1ccc(NS(C)(=O)=O)cc1;0;1;0;;;;;
+807;SP 600125;O=C1c2ccccc2-c2n[nH]c3cccc1c23;1;1;0;;;;;
+808;Sparfloxacin;CC1CN(CC(C)N1)c1c(F)c(N)c2C(=O)C(=CN(C3CC3)c2c1F)C(O)=O;1;0;0;;;;;
+809;Spiramycin;COC1C(CC(=O)OC(C)CCCCCC(OC2CCC(C(C)O2)N(C)C)C(C)CC(CC=O)C1OC1OC(C)C(OC2CC(C)(O)C(O)C(O)O2)C(C1O)N(C)C)OC(C)=O;1;0;0;;;;;
+810;Spironolactone;CC(=O)SC1CC2=CC(=O)CCC2(C)C2CCC3(C)C(CCC33CCC(=O)O3)C12;1;1;0;;;;;
+811;Stanozolol;CC1(O)CCC2C3CCC4Cc5[nH]ncc5CC4(C)C3CCC12C;1;1;0;;;;;
+812;Stavudine;CC1=CN(C2OC(CO)C=C2)C(=O)NC1=O;1;1;0;;;;;
+813;Streptomycin;CNC1C(O)C(O)C(CO)OC1OC1C(OC(C)C1(O)C=O)OC1C(O)C(O)C(NC(N)=N)C(O)C1NC(N)=N;1;1;1;;;;;
+814;Streptozotocin;CN(N=O)C(=O)NC1C(O)OC(CO)C(O)C1O;0;1;1;;;;;
+815;SU6668;Cc1[nH]c(C=C2C(=O)Nc3ccccc23)c(C)c1CCC(O)=O;1;0;0;;;;;
+816;Suberoylanilide hydroxamic acid;ONC(=O)CCCCCCC(=O)Nc1ccccc1;1;1;0;;;;;
+817;Sucrose;OCC1OC(OC2(CO)OC(CO)C(O)C2O)C(O)C(O)C1O;0;1;1;;;;;
+818;Sulbactam;CC1(C)C(N2C(CC2=O)S1(=O)=O)C(O)=O;1;0;0;;;;;
+819;Sulfadiazine;Nc1ccc(cc1)S(=O)(=O)Nc1ncccn1;1;0;0;;;;;
+820;Sulfadimethoxine;COc1cc(NS(=O)(=O)c2ccc(N)cc2)nc(OC)n1;1;0;0;;;;;
+821;Sulfadoxine;COc1ncnc(NS(=O)(=O)c2ccc(N)cc2)c1OC;0;0;1;;;;;
+822;Sulfamerazine;Cc1ccnc(NS(=O)(=O)c2ccc(N)cc2)n1;0;1;0;;;;;
+823;Sulfameter;COc1cnc(NS(=O)(=O)c2ccc(N)cc2)nc1;1;0;0;;;;;
+824;Sulfamethazine;Cc1cc(C)nc(NS(=O)(=O)c2ccc(N)cc2)n1;1;1;0;;;;;
+825;Sulfamethizole;Cc1nnc(NS(=O)(=O)c2ccc(N)cc2)s1;1;0;0;;;;;
+826;Sulfamethoxazole;Cc1cc(NS(=O)(=O)c2ccc(N)cc2)no1;1;0;0;;;;;
+827;Sulfanilamide;Nc1ccc(cc1)S(N)(=O)=O;1;0;0;;;;;
+828;Sulfaphenazole;Nc1ccc(cc1)S(=O)(=O)Nc1ccnn1-c1ccccc1;1;1;0;;;;;
+829;Sulfapyridine;Nc1ccc(cc1)S(=O)(=O)Nc1ccccn1;0;1;0;;;;;
+830;Sulfasalazine;OC(=O)c1cc(ccc1O)N=Nc1ccc(cc1)S(=O)(=O)Nc1ccccn1;1;1;0;;;;;
+831;Sulfinpyrazone;O=C1C(CCS(=O)c2ccccc2)C(=O)N(N1c1ccccc1)c1ccccc1;1;0;0;;;;;
+832;Sulindac;CC1=C(CC(O)=O)c2cc(F)ccc2C1=Cc1ccc(cc1)S(C)=O;1;1;0;;;;;
+833;Sulindac sulfone;CC1=C(CC(O)=O)c2cc(F)ccc2C1=Cc1ccc(cc1)S(C)(=O)=O;1;0;0;;;;;
+834;Suloctidil;CCCCCCCCNC(C)C(O)c1ccc(SC(C)C)cc1;1;0;0;;;;;
+835;Sulpiride;CCN1CCCC1CNC(=O)c1cc(ccc1OC)S(N)(=O)=O;1;0;1;;;;;
+836;Suprofen;CC(C(O)=O)c1ccc(cc1)C(=O)c1cccs1;1;0;0;;;;;
+837;Suramin;Cc1ccc(cc1NC(=O)c1cccc(NC(=O)Nc2cccc(c2)C(=O)Nc2cc(ccc2C)C(=O)Nc2ccc(c3cc(cc(c23)S(O)(=O)=O)S(O)(=O)=O)S(O)(=O)=O)c1)C(=O)Nc1ccc(c2cc(cc(c12)S(O)(=O)=O)S(O)(=O)=O)S(O)(=O)=O;1;1;1;;;;;
+838;Synephrine;CNCC(O)c1ccc(O)cc1;0;1;0;;;;;
+839;Tacrine hydrochloride;Nc1c2ccccc2nc2ccccc12;0;1;0;;;;;
+840;Tamoxifen;CCC(c1ccccc1)=C(c1ccccc1)c1ccc(OCCN(C)C)cc1;1;1;1;;;;;
+841;Tannic acid;OC1C(COC(=O)c2cc(O)c(O)c(O)c2)OC(OC(=O)c2cc(O)c(O)c(O)c2)C(O)C1OC(=O)c1cc(O)c(O)c(O)c1;0;1;1;;;;;
+842;Tegafur;FC1=CN(C2CCCO2)C(=O)NC1=O;1;0;0;;;;;
+843;Telithromycin;CCC1OC(=O)C(C)C(=O)C(C)C(OC2OC(C)CC(C2O)N(C)C)C(C)(CC(C)C(=O)C(C)C2N(CCCCn3cnc(c3)-c3cccnc3)C(=O)OC12C)OC;1;1;0;;;;;
+844;Telmisartan;CCCc1[nH]c2c(C)cc(cc2n1Cc1ccc(cc1)-c1ccccc1C(O)=O)-c1nc2ccccc2n1C;1;1;0;;;;;
+845;Temazepam;CN1C(=O)C(O)N=C(c2ccccc2)c2cc(Cl)ccc12;0;1;0;;;;;
+846;Temozolomide;CN1N=Nc2c(ncn2C1=O)C(N)=O;1;0;0;;;;;
+847;Teniposide;COc1cc(cc(OC)c1O)C1C2C(COC2=O)C(OC2OC(C)C(OC(O)c3cccs3)C(O)C2O)c2cc3OCOc3cc12;0;1;0;;;;;
+848;Tenofovir;CC(Cn1cnc2c(N)ncnc12)OCP(O)(O)=O;1;0;0;;;;;
+849;Tenoxicam;CN1C(C(=O)Nc2ccccn2)=C(O)c2sccc2S1(=O)=O;1;0;0;;;;;
+850;Terbinafine;CN(CC=CC#CC(C)(C)C)Cc1cccc2ccccc12;1;1;0;;;;;
+851;Terbutaline;CC(C)(C)NCC(O)c1cc(O)cc(O)c1;0;1;1;;;;;
+852;Terfenadine;CC(C)(C)c1ccc(cc1)C(O)CCCN1CCC(CC1)C(O)(c1ccccc1)c1ccccc1;1;0;0;;;;;
+853;Terlipressin;NCCCCC(NC(=O)C1CCCN1C(=O)C1CSSCC(NC(=O)CNC(=O)CNC(=O)CN)C(=O)NC(Cc2ccc(O)cc2)C(=O)NC(Cc2ccccc2)C(=O)NC(CCC(N)=O)C(=O)NC(CC(N)=O)C(=O)N1)C(=O)NCC(N)=O;1;0;1;;;;;
+854;Testosterone;CC12CCC3C(CCC4=CC(=O)CCC34C)C1CCC2O;1;1;1;;;;;
+855;Testosterone enanthate;CCCCCCC(=O)OC1CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;0;1;0;;;;;
+856;Testosterone propionate;CCC(=O)OC1CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C;0;1;0;;;;;
+857;Tetrabenazine;COc1cc2CCN3CC(CC(C)C)C(=O)CC3c2cc1OC;1;0;0;;;;;
+858;Tetracaine;CCCCNc1ccc(cc1)C(=O)OCCN(C)C;0;1;0;;;;;
+859;Tetrachloroethylene;Cl\C(Cl)=C(\Cl)Cl;1;1;0;;;;;
+860;Tetracycline;CN(C)C1C2CC3C(C(=O)c4c(O)cccc4C3(C)O)=C(O)C2(O)C(=O)C(C(N)=O)=C1O;1;1;1;;;;;
+861;Tetrandrine;COc1ccc2CC3N(C)CCc4cc(OC)c(OC)c(Oc5cc6C(Cc7ccc(Oc1c2)cc7)N(C)CCc6cc5OC)c34;1;1;0;;;;;
+862;Tezosentan;COc1ccccc1Oc1c(NS(=O)(=O)c2ccc(cn2)C(C)C)nc(nc1OCCO)-c1ccnc(c1)-c1nn[nH]n1;0;1;0;;;;;
+863;Thalidomide;O=C1CCC(N2C(=O)c3ccccc3C2=O)C(=O)N1;1;1;0;;;;;
+864;Theophylline;CN1C(=O)N(C)c2[nH]c[nH]c2C1=O;1;1;1;;;;;
+865;Thiabendazole;c1ccc2[nH]c(nc2c1)-c1cscn1;1;1;1;;;;;
+866;Thiamine;Cc1ncc(C[n+]2csc(CCO)c2C)c(N)n1;1;1;0;;;;;
+867;Thiamphenicol;CS(=O)(=O)c1ccc(cc1)C(O)C(CO)NC(=O)C(Cl)Cl;1;1;0;;;;;
+868;Thiocoraline;CSCC1N(C)C(=O)C2CSSCC(N(C)C(=O)CNC(=O)C(CSC1=O)NC(=O)c1nc3ccccc3cc1O)C(=O)N(C)C(CSC)C(=O)SCC(NC(=O)c1nc3ccccc3cc1O)C(=O)NCC(=O)N2C;1;0;0;;;;;
+869;Thioguanine;NC1=Nc2nc[nH]c2C(=S)N1;1;1;0;;;;;
+870;Thioridazine;CSc1ccc2Sc3ccccc3N(CCC3CCCCN3C)c2c1;1;1;0;;;;;
+871;Thyroxine;NC(Cc1cc(I)c(Oc2cc(I)c(O)c(I)c2)c(I)c1)C(O)=O;1;1;1;;;;;
+872;Tiadenol;OCCSCCCCCCCCCCSCCO;0;1;0;;;;;
+873;Tiaprofenic acid;CC(C(O)=O)c1ccc(s1)C(=O)c1ccccc1;0;1;0;;;;;
+874;Tiazofurin;NC(=O)c1csc(n1)C1OC(CO)C(O)C1O;0;1;0;;;;;
+875;Ticlopidine;Clc1ccccc1CN1CCc2sccc2C1;1;1;0;;;;;
+876;Ticrynafen;OC(=O)COc1ccc(C(=O)c2cccs2)c(Cl)c1Cl;1;1;0;;;;;
+877;Timolol;CC(C)(C)NCC(O)COc1nsnc1N1CCOCC1;1;0;0;;;;;
+878;Tiopronin;CC(S)C(=O)NCC(O)=O;1;1;0;;;;;
+879;Tirapazamine;Nc1n[n+](O)c2ccccc2[n+]1O;0;1;0;;;;;
+880;Tizanidine;Clc1ccc2nsnc2c1NC1=NCCN1;1;0;0;;;;;
+881;TNP-470;COC1C(CCC2(CO2)C1C1(C)OC1C\C=C(\C)C)OC(=O)NC(=O)CCl;0;1;1;;;;;
+882;Tobramycin;NCC1OC(OC2C(N)CC(N)C(OC3OC(CO)C(O)C(N)C3O)C2O)C(N)CC1O;0;1;0;;;;;
+883;Tocopherol acetate;CC(C)CCCC(C)CCCC(C)CCCC1(C)CCc2c(C)c(OC(C)=O)c(C)c(C)c2O1;0;1;0;;;;;
+884;Tolazamide;Cc1ccc(cc1)S(=O)(=O)NC(=O)NN1CCCCCC1;1;1;0;;;;;
+885;Tolbutamide;CCCCNC(=O)NS(=O)(=O)c1ccc(C)cc1;1;1;0;;;;;
+886;Tolcapone;Cc1ccc(cc1)C(=O)c1cc(O)c(O)c(c1)N(=O)=O;1;1;0;;;;;
+887;Tolperisone;CC(CN1CCCCC1)C(=O)c1ccc(C)cc1;1;0;0;;;;;
+888;Tolterodine;CC(C)N(CCC(c1ccccc1)c1cc(C)ccc1O)C(C)C;1;0;0;;;;;
+889;Topiramate;CC1(C)OC2COC3(COS(N)(=O)=O)OC(C)(C)OC3C2O1;1;0;0;;;;;
+890;Toremifene;CN(C)CCOc1ccc(cc1)C(c1ccccc1)=C(CCCl)c1ccccc1;1;1;0;;;;;
+891;Tramadol;COc1cccc(c1)C1(O)CCCCC1CN(C)C;1;1;0;;;;;
+892;Trandolapril;CCOC(=O)C(CCc1ccccc1)NC(C)C(=O)N1C2CCCCC2CC1C(O)=O;1;0;0;;;;;
+893;Tranexamic acid;NCC1CCC(CC1)C(O)=O;1;0;0;;;;;
+894;Tranilast;COc1ccc(C=CC(=O)Nc2ccccc2C(O)=O)cc1OC;0;1;0;;;;;
+895;Tranylcypromine;NC1CC1c1ccccc1;1;1;1;;;;;
+896;Trazodone;Clc1cccc(c1)N1CCN(CCCN2N=C3C=CC=CN3C2=O)CC1;1;0;0;;;;;
+897;Triac;OC(=O)Cc1cc(I)c(Oc2ccc(O)c(I)c2)c(I)c1;0;1;0;;;;;
+898;Triamcinolone;CC12CC(O)C3(F)C(CCC4=CC(=O)C=CC34C)C1CC(O)C2(O)C(=O)CO;0;1;0;;;;;
+899;Triamcinolone acetonide;CC1(C)OC2CC3C4CCC5=CC(=O)C=CC5(C)C4(F)C(O)CC3(C)C2(O1)C(=O)CO;0;1;0;;;;;
+900;Trichlorfon;COP(=O)(OC)C(O)C(Cl)(Cl)Cl;0;1;0;;;;;
+901;Trichloroacetic acid;OC(=O)C(Cl)(Cl)Cl;0;1;0;;;;;
+902;Trichloroethylene;Cl\C=C(\Cl)Cl;1;1;0;;;;;
+903;Triclosan;Oc1cc(Cl)ccc1Oc1ccc(Cl)cc1Cl;0;1;0;;;;;
+904;Triflupromazine;CN(C)CCCN1c2ccccc2Sc2ccc(cc12)C(F)(F)F;0;1;0;;;;;
+905;Triiodothyronine;NC(Cc1cc(I)c(Oc2ccc(O)c(I)c2)c(I)c1)C(O)=O;1;1;0;;;;;
+906;Trimetazidine;COc1ccc(CN2CCNCC2)c(OC)c1OC;1;1;0;;;;;
+907;Trimethadione;CN1C(=O)OC(C)(C)C1=O;1;1;0;;;;;
+908;Trimethaphan camsylate;O=C1N(Cc2ccccc2)C2C[S]3CCCC3C2N1Cc1ccccc1;0;0;1;;;;;
+909;Trimethoprim;COc1cc(Cc2cnc(N)nc2N)cc(OC)c1OC;1;1;0;;;;;
+910;Troglitazone;Cc1c(C)c2OC(C)(CCc2c(C)c1O)COc1ccc(CC2SC(=O)NC2=O)cc1;1;1;1;;;;;
+911;Troleandomycin;COC1CC(OC(C)C1OC(C)=O)OC1C(C)C(OC2OC(C)CC(C2OC(C)=O)N(C)C)C(C)CC2(CO2)C(=O)C(C)C(OC(C)=O)C(C)C(C)OC(=O)C1C;1;1;1;;;;;
+912;Trypan Blue;Cc1cc(ccc1N=Nc1c(O)c2c(N)cc(cc2cc1S(O)(=O)=O)S(O)(=O)=O)-c1ccc(N=Nc2c(O)c3c(N)cc(cc3cc2S(O)(=O)=O)S(O)(=O)=O)c(C)c1;0;1;0;;;;;
+913;Tryptophan;NC(Cc1c[nH]c2ccccc12)C(O)=O;0;1;0;;;;;
+914;Tyloxapol;CC(C)(C)CC(C)(C)c1ccc(O)cc1;0;1;0;;;;;
+915;Uracil;O=C1NC=CC(=O)N1;1;1;0;;;;;
+916;Urea;NC(N)=O;0;1;0;;;;;
+917;Uridine;OCC1OC(C(O)C1O)N1C=CC(=O)NC1=O;1;1;0;;;;;
+918;Ursodiol;O=C(O)CC[C@H]([C@H]1CC[C@@H]2[C@]1(C)CC[C@H]4[C@H]2[C@@H](O)C[C@@H]3C[C@H](O)CC[C@@]34C)C;1;1;1;;;;;
+919;Valproic acid;CCCC(CCC)C(O)=O;1;1;0;;;;;
+920;Valsartan;CCCCC(=O)N(Cc1ccc(cc1)-c1ccccc1-c1nn[nH]n1)C(C(C)C)C(O)=O;1;1;0;;;;;
+921;Valspodar;CC=CCC(C)C(=O)C1N(C)C(=O)C(C(C)C)N(C)C(=O)C(CC(C)C)N(C)C(=O)C(CC(C)C)N(C)C(=O)C(C)NC(=O)C(C)NC(=O)C(CC(C)C)N(C)C(=O)C(NC(=O)C(CC(C)C)N(C)C(=O)CN(C)C(=O)C(NC1=O)C(C)C)C(C)C;1;1;0;;;;;
+922;Vancomycin;CNC(CC(C)C)C(=O)NC1C(O)c2ccc(Oc3cc4cc(Oc5ccc(cc5Cl)C(O)C5NC(=O)C(NC(=O)C4NC(=O)C(CC(N)=O)NC1=O)c1ccc(O)c(c1)-c1c(O)cc(O)cc1C(NC5=O)C(O)=O)c3OC1OC(CO)C(O)C(O)C1OC1CC(C)(N)C(O)C(C)O1)c(Cl)c2;1;1;0;;;;;
+923;vandetanib;COc1cc2c(Nc3ccc(Br)cc3F)ncnc2cc1OCC1CCN(C)CC1;1;0;0;;;;;
+924;Venlafaxine;COc1ccc(cc1)C(CN(C)C)C1(O)CCCCC1;1;0;0;;;;;
+925;Verapamil;COc1ccc(CCN(C)CCCC(C#N)(C(C)C)c2ccc(OC)c(OC)c2)cc1OC;1;1;1;;;;;
+926;Vesnarinone;COc1ccc(cc1OC)C(=O)N1CCN(CC1)c1ccc2NC(=O)CCc2c1;1;1;0;;;;;
+927;Vidarabine;n2c1c(ncnc1n(c2)[C@@H]3O[C@@H]([C@@H](O)[C@@H]3O)CO)N;1;1;0;;;;;
+928;Vigabatrin;NC(CCC(O)=O)C=C;0;1;0;;;;;
+929;Vinblastine;CCC1(O)CC2CN(CCc3c([nH]c4ccccc34)C(C2)(C(=O)OC)c2cc3c(cc2OC)N(C)C2C(O)(C(OC(C)=O)C4(CC)C=CCN5CCC32C45)C(=O)OC)C1;1;1;0;;;;;
+930;Vincristine;CCC1(O)CC2CN(CCc3c([nH]c4ccccc34)C(C2)(C(=O)OC)c2cc3c(cc2OC)N(C=O)C2C(O)(C(OC(C)=O)C4(CC)C=CCN5CCC32C45)C(=O)OC)C1;1;1;1;;;;;
+931;vinflunine;CCC12C=CCN3CCC4(C13)C(N(C)c1cc(OC)c(cc41)C1(CC3CC(CN(C3)Cc3c1[nH]c1ccccc31)C(C)(F)F)C(=O)OC)C(O)(C2OC(C)=O)C(=O)OC;0;1;0;;;;;
+932;Vinorelbine;CCC1=CC2CN(C1)Cc1c([nH]c3ccccc13)C(C2)(C(=O)OC)c1cc2c(cc1OC)N(C)C1C(O)(C(OC(C)=O)C3(CC)C=CCN4CCC21C34)C(=O)OC;1;1;0;;;;;
+933;Vitamin A;CC(C=CC=C(C)C=CC1=C(C)CCCC1(C)C)=CCO;1;1;1;;;;;
+934;Vitamin B12;CC(CNC(=O)CCC1(C)C(CC(N)=O)C2=NC1=C(C)C1=NC(=CC3=NC(=C(C)C4=NC2(C)C(C)(CC(N)=O)C4CCC(N)=O)C(C)(CC(N)=O)C3CCC(N)=O)C(C)(C)C1CCC(N)=O)OP(O)(=O)OC1C(CO)OC(C1O)n1cnc2cc(C)c(C)cc12;1;0;0;;;;;
+935;Vitamin B6;Cc1ncc(CO)c(CO)c1O;1;1;0;;;;;
+936;Vitamin D3;CC(C)CCCC(C)C1CCC2C(CCCC12C)=CC=C1CC(O)CCC1=C;0;1;0;;;;;
+937;Vitamin E;CC(C)CCCC(C)CCCC(C)CCCC1(C)CCc2c(C)c(O)c(C)c(C)c2O1;1;1;1;;;;;
+938;Vitamin K;CC(C)CCCC(C)CCCC(C)CCCC(C)=CCC1=C(C)C(=O)c2ccccc2C1=O;1;1;0;;;;;
+939;Voglibose;OCC(CO)NC1CC(O)(CO)C(O)C(O)C1O;1;1;0;;;;;
+940;Voriconazole;CC(c1ncncc1F)C(O)(Cn1cncn1)c1ccc(F)cc1F;1;0;0;;;;;
+941;VX-950;CCCC(NC(=O)C1C2CCCC2CN1C(=O)C(NC(=O)C(NC(=O)c1cnccn1)C1CCCCC1)C(C)(C)C)C(=O)C(=O)NC1CC1;1;0;0;;;;;
+942;Warfarin;CC(=O)CC(c1ccccc1)C1=C(O)Oc2ccccc2C1=O;1;1;0;;;;;
+943;Ximelagatran;CCOC(=O)CNC(C1CCCCC1)C(=O)N1CCC1C(=O)NCc1ccc(cc1)C(N)=NO;1;0;0;;;;;
+944;Xipamide;Cc1cccc(C)c1NC(=O)c1cc(c(Cl)cc1O)S(N)(=O)=O;0;1;0;;;;;
+945;Yohimbine;COC(=O)C1C(O)CCC2CN3CCc4c([nH]c5ccccc45)C3CC12;0;1;0;;;;;
+946;Zafirlukast;COc1cc(ccc1Cc1cn(C)c2ccc(NC(=O)OC3CCCC3)cc12)C(=O)NS(=O)(=O)c1ccccc1C;1;0;0;;;;;
+947;Zalcitabine;NC1=NC(=O)N(C=C1)C1CCC(CO)O1;1;1;0;;;;;
+948;Zidovudine;CC1=CN(C2CC(NN=N)C(CO)O2)C(=O)NC1=O;1;1;0;;;;;
+949;Zileuton;CC(N(O)C(N)=O)c1cc2ccccc2s1;1;1;0;;;;;
+950;Zinc acetate;CC(O)=O;0;1;0;;;;;
+951;Zolpidem;CN(C)C(=O)Cc1c(nc2ccc(C)cn12)-c1ccc(C)cc1;1;0;0;;;;;
+952;zirconium;CCO[Zr](OCC)(OCC)OCC;0;0;0CC1=C(C2=CC3=NC(=
+953;hemoglobin;CC4=C(C(=C([N-]4)C=C5C(=C(C(=N5)C=C1N2)C=C)C)C)CCC(=O)[O-])C(=C3C)CCC(=O)O)C=C.[Fe+2];0;0;0
+954;test_salt;[Al].N.[Ba].[Bi].Br.[Ca].Cl.F.I.[K].[Li].[Mg].[Na].[Ag].[Sr].S.O.[Zn];0;0;0
+955;no_smiles_test;;0;0;0
+956;covalent_metal;CCC(=O)O[Na];0;0;0
+957;test_charge_recombination; CC([O-])=[N+](C)C; 0;0;0
+958;Chloroquine; CCN(CC)CCCC(C)NC1=C2C=CC(=CC2=NC=C1)Cl;0;0;0
+959;Water;O;0;0;0
+960;1,4-Dioxane;c1ccccc1O.O1CCOCC1;0;0;0
\ No newline at end of file
diff --git a/docs/tutorials/standardization.ipynb b/docs/tutorials/standardization.ipynb
new file mode 100644
index 00000000..59e4ae8a
--- /dev/null
+++ b/docs/tutorials/standardization.ipynb
@@ -0,0 +1,3785 @@
+{
+ "cells": [
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "# Implementation and evaluation of a computational standardization pipeline for chemical compounds\n",
+ "--------------------------------------------------------------\n",
+ "\n",
+ "> Based on [\"Trust, But Verify: On the Importance of Chemical Structure Curation in Cheminformatics and QSAR Modeling Research\" from 2010 (D. Fourches, ...)\"](https://pubmed.ncbi.nlm.nih.gov/20572635/)\n",
+ "\n",
+ "By Allen Dumler; reviewed by Jaime Rodríguez-Guerra, PhD."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Introduction \n",
+ "\n",
+ "This notebook serves to display the functionality of the `opencadd.compounds.standardization` subpackage. \n",
+ "\n",
+ "We are following the recommended standardization steps of [\"Trust, But Verify\" (Fourches et al., 2010)](https://pubmed.ncbi.nlm.nih.gov/20572635/), and using a modified¹ version of the dataset from the following paper: [Cheminformatics Analysis of Assertions Mined from Literature That Describe Drug-Induced Liver Injury in Different Species](https://pubs.acs.org/doi/10.1021/tx900326k).\n",
+ "\n",
+ "¹ We added some entries to trigger curation steps not covered by the original data."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Overview over the pipeline\n",
+ "------------------------------------------\n",
+ "\n",
+ "This pipeline has **five** main steps:\n",
+ "1. Structural Conversion\n",
+ "2. Filtering of Inorganics and Mixtures\n",
+ "3. Structural Cleaning \n",
+ "4. Normalization of Specific Chemotypes\n",
+ "5. Removal of Duplicates\n",
+ "\n",
+ "Each step consists of action performing tasks on the dataset. Actions are:\n",
+ "\n",
+ "- filtering\n",
+ "- cleaning\n",
+ "- normalizing\n",
+ "\n",
+ "**Filtering** actions will result in a score applied to the entries. The score is the number of the filtering task. You can use it to select subsets of the dataset sorting by the column `filtered_at`.\n",
+ "\n",
+ "**Cleaning** actions will result in a modification of the mol-representation of the entry, overwriting with the recent version calculated in the task. You can use it to select subsets of the dataset sorting by the column `cleaned_at`.\n",
+ "\n",
+ "**Normalizing** actions also will result in a modification of the mol-representation of the entry.You can use it to select subsets of the dataset sorting by the column `normalized_at`.\n",
+ "\n",
+ "At the end of the script, there is the possibility to export subsets of the dataset as a CSV. "
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 1,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Tutorial location: C:\\Users\\Allen.DESKTOP-O8FR8HB\\Documents\\DEV\\opencadd\\docs\\tutorials\n",
+ "Repo location: C:\\Users\\Allen.DESKTOP-O8FR8HB\\Documents\\DEV\\opencadd\n"
+ ]
+ }
+ ],
+ "source": [
+ "from pathlib import Path\n",
+ "\n",
+ "HERE = Path(_dh[-1])\n",
+ "REPO = HERE.parents[1]\n",
+ "\n",
+ "print(\"Tutorial location:\", HERE)\n",
+ "print(\"Repo location: \", REPO)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 2,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Import pandas and numpy\n",
+ "import pandas as pd\n",
+ "import numpy as np\n",
+ "\n",
+ "# Importing functions from the standardization API\n",
+ "from opencadd.compounds.standardization import (\n",
+ " convert_format,\n",
+ " detect_mixtures,\n",
+ " detect_metals,\n",
+ " detect_salts,\n",
+ " detect_inorganics,\n",
+ " handle_fragments,\n",
+ " handle_tautomers,\n",
+ " handle_charges,\n",
+ " handle_tautomers,\n",
+ " disconnect_metals,\n",
+ " remove_salts,\n",
+ " normalize_molecules,\n",
+ " validate_molecules,\n",
+ ")"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 3,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Utility function to compare SMILES\n",
+ "# JRG: Why is this function needed? You can use\n",
+ "# `operator.eq` builtin!\n",
+ "def smiles_string_changed(smiles_old, smiles_new):\n",
+ " \"\"\"\n",
+ " Compares SMILES strings. If they are identical, the value returned is False,\n",
+ " if they differ the value returned is True.\n",
+ "\n",
+ " Parameters\n",
+ " ----------\n",
+ " smiles_old: str\n",
+ " SMILES string\n",
+ " smiles_new: str\n",
+ " SMILES string\n",
+ "\n",
+ " Returns\n",
+ " -------\n",
+ " bool\n",
+ " True if changes were detectrd, false otherwise\n",
+ " \"\"\"\n",
+ " return smiles_old != smiles_new"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Initial dataset import and cleaning of empty entries\n",
+ "------------------------------------------------\n",
+ "Before any curation steps are can be applied, we need to import the dataset as a Pandas Dataframe.
\n",
+ "At this point you have the possibility to select the columns you need for the curation process. For our example dataset we will use columns IDs, Names and SMILEs.
\n",
+ "After that, we search for all entries which have empty strings saved under SMILES and remove them from the dataset.
\n",
+ "After the import, we add a Filtered_at column to track which standardization step filtered the entry. \n",
+ "The initial `task_number` will be 0, which leads to a default Filtered_at-value of 0 for all entries, where null stands for all the entries that passed without any filtering. "
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 4,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "
\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 1 | \n",
+ " (R)-Roscovitine | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ "
\n",
+ " \n",
+ " | 1 | \n",
+ " 2 | \n",
+ " 17-Methyltestosterone | \n",
+ " CC1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ "
\n",
+ " \n",
+ " | 2 | \n",
+ " 3 | \n",
+ " 1-alpha-Hydroxycholecalciferol | \n",
+ " CC(C)CCCC(C)C1CCC2C(CCCC12C)=CC=C1CC(O)CC(O)C1=C | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ "
\n",
+ " \n",
+ " | 3 | \n",
+ " 4 | \n",
+ " 2,3-Dimercaptosuccinic acid | \n",
+ " OC(=O)C(S)C(S)C(O)=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ "
\n",
+ " \n",
+ " | 4 | \n",
+ " 5 | \n",
+ " 2,4,6-Trinitrotoluene | \n",
+ " Cc1c(cc(cc1N(=O)=O)N(=O)=O)N(=O)=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "0 1 (R)-Roscovitine \n",
+ "1 2 17-Methyltestosterone \n",
+ "2 3 1-alpha-Hydroxycholecalciferol \n",
+ "3 4 2,3-Dimercaptosuccinic acid \n",
+ "4 5 2,4,6-Trinitrotoluene \n",
+ "\n",
+ " SMILES Filtered_at Cleaned_at \\\n",
+ "0 CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 0 0 \n",
+ "1 CC1(O)CCC2C3CCC4=CC(=O)CCC4(C)C3CCC12C 0 0 \n",
+ "2 CC(C)CCCC(C)C1CCC2C(CCCC12C)=CC=C1CC(O)CC(O)C1=C 0 0 \n",
+ "3 OC(=O)C(S)C(S)C(O)=O 0 0 \n",
+ "4 Cc1c(cc(cc1N(=O)=O)N(=O)=O)N(=O)=O 0 0 \n",
+ "\n",
+ " Normalized_at \n",
+ "0 0 \n",
+ "1 0 \n",
+ "2 0 \n",
+ "3 0 \n",
+ "4 0 "
+ ]
+ },
+ "execution_count": 4,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "task_number = 0\n",
+ "\n",
+ "# Import test-dataset\n",
+ "dataset = pd.read_csv(HERE / \"data\" / \"standardization_test_data.csv\", delimiter=\";\")\n",
+ "\n",
+ "# Filter columns\n",
+ "dataset = dataset[[\"IDs\", \"Names\", \"SMILEs\"]]\n",
+ "# Rename a column, due to an typo in the original dataset\n",
+ "dataset = dataset.rename(columns={\"SMILEs\": \"SMILES\"})\n",
+ "\n",
+ "# Delete empty entries from the main set.\n",
+ "# JRG: The usual thing here is to use df.dropna() function, possibly with a subset=XXX option\n",
+ "dataset = dataset[(dataset[\"SMILES\"].notna())]\n",
+ "\n",
+ "# Initializing the score to null at the 'Filtered_at'-column\n",
+ "# JRG: You can use a constant here, I believe: dataset[X] = task_number\n",
+ "dataset[\"Filtered_at\"] = dataset[\"SMILES\"].apply(\n",
+ " lambda x, task_number=task_number: task_number\n",
+ ")\n",
+ "\n",
+ "# Initializing the score to null at the 'Cleaned_at'-column\n",
+ "# JRG: Same as above\n",
+ "dataset[\"Cleaned_at\"] = dataset[\"SMILES\"].apply(\n",
+ " lambda x, task_number=task_number: task_number\n",
+ ")\n",
+ "\n",
+ "# Initializing the score to null at the 'Normalized_at'-column\n",
+ "# JRG: Same as above\n",
+ "dataset[\"Normalized_at\"] = dataset[\"SMILES\"].apply(\n",
+ " lambda x, task_number=task_number: task_number\n",
+ ")\n",
+ "\n",
+ "\n",
+ "# Reset the index to correct the deletion of the empty entries\n",
+ "dataset = dataset.reset_index(drop=True)\n",
+ "\n",
+ "# [Optional] Display empty entries for manual inspection.\n",
+ "# dataset[(dataset[\"SMILES\"].isnull())]\n",
+ "dataset.head()"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Step 1: Encoding Converison\n",
+ "------------------------------------------\n",
+ "\n",
+ "__Convert the SMILES representation format of the compounds into Mol-files__\n",
+ "\n",
+ "RDKit performs a sanitization of molecules converted to mol by default.
\n",
+ "In addition to some Nitro and Perchlorate transformations the following steps are taken²:\n",
+ "\n",
+ "\n",
+ "- Calculate explicit and implicit valence of all atoms. Fails when atoms have illegal valence.\n",
+ "- Calculate symmetrized SSSR. The slowest step fails in rare cases.\n",
+ "- Kekulize. Fails if a Kekule form cannot be found or non-ring bonds are marked as aromatic.\n",
+ "- Assign radicals if hydrogens set and bonds+hydrogens+charge < valence.\n",
+ "- Set aromaticity, if none set in input. Go round rings, Huckel rule to set atoms+bonds as aromatic.\n",
+ "- Set a conjugated property on bonds where applicable.\n",
+ "- Set hybridization property on atoms.\n",
+ "- Remove chirality markers from sp and sp2 hybridized centers.\n",
+ "\n",
+ "If the conversion from SMILES to mol fails, then those SMILES will get a **Filtered_at** marker added. \n",
+ "\n",
+ "> JRG: SMILES contains the end S already. It is not a plural form!\n",
+ "\n",
+ "To avoid molecule sanitization `convert_smiles_to_mol` can be called with the argument `sanitize=False`. Keep in mind that the generation of different Lewis structures serves to find alternative representation formats of the same molecule. \n",
+ "\n",
+ "__Overwrite the SMILES representation with ones compiled from our generated Mol-files__
\n",
+ "In order to register the changes we make to the entries, we have to calculate canonical SMILES with our function `convert_format`, which by default returns a canonical representation. The conversion back to SMILES has to happen since SMILES encodings vary depending on the algorithm used to calculate them. The newly calculated SMILES will be used as a validation parameter to determine any changes made to our entries further down the curation pipeline. \n",
+ "\n",
+ "References:\n",
+ "\n",
+ "² https://molvs.readthedocs.io/en/latest/guide/standardize.html?highlight=sanitize#rdkit-sanitize\n",
+ "* https://chemistry.stackexchange.com/questions/116498/what-is-kekulization-in-rdkit\n",
+ "* https://rdkit-discuss.narkive.com/QwnqcKcM/another-can-t-kekulize-mol-observation\n",
+ "* https://www.rdkit.org/docs/Cookbook.html\n",
+ "* https://www.rdkit.org/docs/source/rdkit.Chem.rdmolfiles.html"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "#### Task 1: Convert to RDKit Molecule Objects"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 5,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 943 | \n",
+ " 944 | \n",
+ " Xipamide | \n",
+ " Cc1cccc(C)c1NC(=O)c1cc(S(N)(=O)=O)c(Cl)cc1O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 944 | \n",
+ " 945 | \n",
+ " Yohimbine | \n",
+ " COC(=O)C1C(O)CCC2CN3CCc4c([nH]c5ccccc45)C3CC21 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 945 | \n",
+ " 946 | \n",
+ " Zafirlukast | \n",
+ " COc1cc(C(=O)NS(=O)(=O)c2ccccc2C)ccc1Cc1cn(C)c2... | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 946 | \n",
+ " 947 | \n",
+ " Zalcitabine | \n",
+ " Nc1ccn(C2CCC(CO)O2)c(=O)n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 947 | \n",
+ " 948 | \n",
+ " Zidovudine | \n",
+ " Cc1cn(C2CC(NN=N)C(CO)O2)c(=O)[nH]c1=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 948 | \n",
+ " 949 | \n",
+ " Zileuton | \n",
+ " CC(c1cc2ccccc2s1)N(O)C(N)=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 949 | \n",
+ " 950 | \n",
+ " Zinc acetate | \n",
+ " CC(=O)O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 950 | \n",
+ " 951 | \n",
+ " Zolpidem | \n",
+ " Cc1ccc(-c2nc3ccc(C)cn3c2CC(=O)N(C)C)cc1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 951 | \n",
+ " 952 | \n",
+ " zirconium | \n",
+ " CCO[Zr](OCC)(OCC)OCC | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 952 | \n",
+ " 953 | \n",
+ " hemoglobin | \n",
+ " C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 953 | \n",
+ " 954 | \n",
+ " test_salt | \n",
+ " Br.Cl.F.I.N.O.S.[Ag].[Al].[Ba].[Bi].[Ca].[K].[... | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 954 | \n",
+ " 956 | \n",
+ " covalent_metal | \n",
+ " CCC(=O)O[Na] | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 955 | \n",
+ " 957 | \n",
+ " test_charge_recombination | \n",
+ " CC([O-])=[N+](C)C | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 956 | \n",
+ " 958 | \n",
+ " Chloroquine | \n",
+ " CCN(CC)CCCC(C)Nc1ccnc2cc(Cl)ccc12 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 957 | \n",
+ " 959 | \n",
+ " Water | \n",
+ " O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ " | 958 | \n",
+ " 960 | \n",
+ " 1,4-Dioxane | \n",
+ " C1COCCO1.Oc1ccccc1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "943 944 Xipamide \n",
+ "944 945 Yohimbine \n",
+ "945 946 Zafirlukast \n",
+ "946 947 Zalcitabine \n",
+ "947 948 Zidovudine \n",
+ "948 949 Zileuton \n",
+ "949 950 Zinc acetate \n",
+ "950 951 Zolpidem \n",
+ "951 952 zirconium \n",
+ "952 953 hemoglobin \n",
+ "953 954 test_salt \n",
+ "954 956 covalent_metal \n",
+ "955 957 test_charge_recombination \n",
+ "956 958 Chloroquine \n",
+ "957 959 Water \n",
+ "958 960 1,4-Dioxane \n",
+ "\n",
+ " SMILES Filtered_at \\\n",
+ "943 Cc1cccc(C)c1NC(=O)c1cc(S(N)(=O)=O)c(Cl)cc1O 0 \n",
+ "944 COC(=O)C1C(O)CCC2CN3CCc4c([nH]c5ccccc45)C3CC21 0 \n",
+ "945 COc1cc(C(=O)NS(=O)(=O)c2ccccc2C)ccc1Cc1cn(C)c2... 0 \n",
+ "946 Nc1ccn(C2CCC(CO)O2)c(=O)n1 0 \n",
+ "947 Cc1cn(C2CC(NN=N)C(CO)O2)c(=O)[nH]c1=O 0 \n",
+ "948 CC(c1cc2ccccc2s1)N(O)C(N)=O 0 \n",
+ "949 CC(=O)O 0 \n",
+ "950 Cc1ccc(-c2nc3ccc(C)cn3c2CC(=O)N(C)C)cc1 0 \n",
+ "951 CCO[Zr](OCC)(OCC)OCC 0 \n",
+ "952 C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... 0 \n",
+ "953 Br.Cl.F.I.N.O.S.[Ag].[Al].[Ba].[Bi].[Ca].[K].[... 0 \n",
+ "954 CCC(=O)O[Na] 0 \n",
+ "955 CC([O-])=[N+](C)C 0 \n",
+ "956 CCN(CC)CCCC(C)Nc1ccnc2cc(Cl)ccc12 0 \n",
+ "957 O 0 \n",
+ "958 C1COCCO1.Oc1ccccc1 0 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "943 0 0 \n",
+ "944 0 0 \n",
+ "945 0 0 \n",
+ "946 0 0 \n",
+ "947 0 0 \n",
+ "948 0 0 \n",
+ "949 0 0 \n",
+ "950 0 0 \n",
+ "951 0 0 \n",
+ "952 0 0 \n",
+ "953 0 0 \n",
+ "954 0 0 \n",
+ "955 0 0 \n",
+ "956 0 0 \n",
+ "957 0 0 \n",
+ "958 0 0 \n",
+ "\n",
+ " mol \n",
+ "943 \n",
+ "Detecting inorganic structures is divided into two steps:
\n",
+ "First removing all entries not containing any Carbon at all, which are therefore not organic.
\n",
+ "Secondly, filtering out all compounds with inorganic substructures.
\n",
+ "\n",
+ "Similar problems occur for mixtures. Since most applications can not calculate descriptors for mixtures, filtering has to happen before processing.
\n",
+ "Additionally, since \"*inorganic compounds are known to have biological effects, like toxic effects*\" (Fourches 2010), we can often not distinguish if its organic or inorganic part causes the recorded activity of a mixed compound. Therefore the entry is useless and can be discarded. \n",
+ "\n",
+ "\n",
+ "Since the treatment is not as simple as it appears, the paper recommends deleting records containing mixtures.\n",
+ "Common and widely used practice is to retain molecules with the highest molecular weight or the largest number of atoms. Still, the paper states this might not be the best solution, and investigation in mixtures should only happen if there is a reason to believe the largest molecule and not the mixture itself is causing the biological activity."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "#### Task 2: Filter entries without Carbon\n",
+ "\n",
+ "The first task to determine if an entry is an organic molecule is to check for the presence of Carbon. `detect_carbon` is a function able to do this. It searches for the existence of carbon atoms. If the function finds at least one Carbon atom, it returns a boolean value of **TRUE**, if not **FALSE**. All entries that return **FALSE** will get the current task number (2) assigned into the *Filtered_at* column.\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 6,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Setting up the task_number\n",
+ "task_number = 2\n",
+ "\n",
+ "# Check for Carbon\n",
+ "dataset[\"Carbon_present\"] = dataset.apply(\n",
+ " lambda row: detect_inorganics.detect_carbon(row.mol)\n",
+ " if row.Filtered_at == 0\n",
+ " else None,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "# Add task_number to failed entries\n",
+ "dataset.loc[dataset[\"Carbon_present\"] == False, [\"Filtered_at\"]] = task_number"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries that do not contain any Carbon and thereby are inorganic molecules."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 7,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 953 | \n",
+ " 954 | \n",
+ " test_salt | \n",
+ " Br.Cl.F.I.N.O.S.[Ag].[Al].[Ba].[Bi].[Ca].[K].[... | \n",
+ " 2 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 957 | \n",
+ " 959 | \n",
+ " Water | \n",
+ " O | \n",
+ " 2 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names SMILES \\\n",
+ "953 954 test_salt Br.Cl.F.I.N.O.S.[Ag].[Al].[Ba].[Bi].[Ca].[K].[... \n",
+ "957 959 Water O \n",
+ "\n",
+ " Filtered_at Cleaned_at Normalized_at \\\n",
+ "953 2 0 0 \n",
+ "957 2 0 0 \n",
+ "\n",
+ " mol Carbon_present \n",
+ "953 \n",
+ "*While Astatine and Tennessine are also considered halogens, they are not included due to their radioactivity and rarity.*\n",
+ "
\n",
+ "\n",
+ "\n",
+ "###### An example of how to set up a custom set of elements and implement them in `detect_inorganic` \n",
+ "-----------------------------------------------------------------------------------------------------------\n",
+ "Defining a set:
\n",
+ "`elements = Chem.MolFromSmarts(\"[!#1&!#6&!#7&!#8&!#9&!#15&!#16&!#17&!#35&!#53]\")`
\n",
+ "(If you want to run this, import the following before: from rdkit import Chem)\n",
+ "\n",
+ "Pass the set as a parameter, where the `detect_inorganic` function is getting called:
\n",
+ "`lambda row: detect_inorganics.detect_inorganic(row.mol, elements)`"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 8,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Setting up the task_number\n",
+ "task_number = 3\n",
+ "\n",
+ "\n",
+ "# Check for inorganic structures\n",
+ "dataset[\"Inorganics\"] = dataset.apply(\n",
+ " lambda row: detect_inorganics.detect_inorganic(row.mol)\n",
+ " if row.Filtered_at == 0\n",
+ " else None,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "# Add task_number to failed entries\n",
+ "dataset.loc[dataset[\"Inorganics\"] == True, [\"Filtered_at\"]] = task_number"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries that contain other than our allowed elements. (Hydrogen, Carbon, Nitrogen, Oxygen, Fluorine, Phosphorus, Sulfur, Chlorine, Selenium, Bromine, Iodine)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 9,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 114 | \n",
+ " 115 | \n",
+ " Bortezomib | \n",
+ " CC(C)CC(NC(=O)C(Cc1ccccc1)NC(=O)c1cnccn1)B(O)O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 407 | \n",
+ " 408 | \n",
+ " Gold Sodium Thiomalate | \n",
+ " O=C(O)CC(S[Au])C(=O)O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 533 | \n",
+ " 534 | \n",
+ " Mersalyl | \n",
+ " COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 951 | \n",
+ " 952 | \n",
+ " zirconium | \n",
+ " CCO[Zr](OCC)(OCC)OCC | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 952 | \n",
+ " 953 | \n",
+ " hemoglobin | \n",
+ " C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 954 | \n",
+ " 956 | \n",
+ " covalent_metal | \n",
+ " CCC(=O)O[Na] | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "114 115 Bortezomib \n",
+ "407 408 Gold Sodium Thiomalate \n",
+ "533 534 Mersalyl \n",
+ "951 952 zirconium \n",
+ "952 953 hemoglobin \n",
+ "954 956 covalent_metal \n",
+ "\n",
+ " SMILES Filtered_at \\\n",
+ "114 CC(C)CC(NC(=O)C(Cc1ccccc1)NC(=O)c1cnccn1)B(O)O 3 \n",
+ "407 O=C(O)CC(S[Au])C(=O)O 3 \n",
+ "533 COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O 3 \n",
+ "951 CCO[Zr](OCC)(OCC)OCC 3 \n",
+ "952 C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... 3 \n",
+ "954 CCC(=O)O[Na] 3 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "114 0 0 \n",
+ "407 0 0 \n",
+ "533 0 0 \n",
+ "951 0 0 \n",
+ "952 0 0 \n",
+ "954 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "114 \n",
+ "\n",
+ "\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 958 | \n",
+ " 960 | \n",
+ " 1,4-Dioxane | \n",
+ " C1COCCO1.Oc1ccccc1 | \n",
+ " 4 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ ""
+ ],
+ "text/plain": [
+ " IDs Names SMILES Filtered_at Cleaned_at \\\n",
+ "958 960 1,4-Dioxane C1COCCO1.Oc1ccccc1 4 0 \n",
+ "\n",
+ " Normalized_at mol \\\n",
+ "958 0 __Note:__\n",
+ "While it is possible to clean and reuse entries containing metals or salts, those entries will not be curated here since it does not fit the scope of this tutorial. Nevertheless, we hint at the steps to do and which functions of the standardization API to use. \n",
+ "\n",
+ "³ (https://www.drugs.com/article/pharmaceutical-salts.html (03/12/21))"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "#### Task 5: Filter entries containing metals\n",
+ "\n",
+ "Entries can contain metals in different forms. Either as a regular compound in a mixture or as a counterion.
In the following steps, we search for those metals. When they are a counterion, we disconnect them from the non-metals they are bonding. \n",
+ "We might not find any metals due to previous filtering steps detecting mixtures and inorganics. Therefore we could search in the flagged entries and clean those entries later on. "
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 12,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Setting up the task_number\n",
+ "task_number = 5\n",
+ "\n",
+ "# Check for metals\n",
+ "dataset[\"metals\"] = dataset.apply(\n",
+ " lambda row: detect_metals(row.mol) if row.Filtered_at == 0 else None,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "# Add task_number to failed entries\n",
+ "dataset.loc[dataset[\"metals\"] == True, [\"Filtered_at\"]] = task_number"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries containing metals. We didn't find any entries in our filtered set, as already assumed."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 13,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ "Empty DataFrame\n",
+ "Columns: [IDs, Names, SMILES, Filtered_at, Cleaned_at, Normalized_at, mol, Carbon_present, Inorganics, mixture, metals]\n",
+ "Index: []"
+ ]
+ },
+ "execution_count": 13,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "dataset[dataset[\"Filtered_at\"] == 5]"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "So the next step would be to examine our filtered entries.
\n",
+ "For that, we make a copy of our current status of the dataset."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 14,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "score = [3, 4]\n",
+ "failed_entries_copy = dataset[dataset[\"Filtered_at\"].isin(score)].copy()"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "And we check for the presence of metals here"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 15,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Check for metals\n",
+ "failed_entries_copy[\"metals\"] = failed_entries_copy.apply(\n",
+ " lambda row: detect_metals(row.mol) if row.Filtered_at != 0 else None,\n",
+ " axis=1,\n",
+ ")"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 16,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 407 | \n",
+ " 408 | \n",
+ " Gold Sodium Thiomalate | \n",
+ " O=C(O)CC(S[Au])C(=O)O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 533 | \n",
+ " 534 | \n",
+ " Mersalyl | \n",
+ " COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 951 | \n",
+ " 952 | \n",
+ " zirconium | \n",
+ " CCO[Zr](OCC)(OCC)OCC | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 954 | \n",
+ " 956 | \n",
+ " covalent_metal | \n",
+ " CCC(=O)O[Na] | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names SMILES \\\n",
+ "407 408 Gold Sodium Thiomalate O=C(O)CC(S[Au])C(=O)O \n",
+ "533 534 Mersalyl COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O \n",
+ "951 952 zirconium CCO[Zr](OCC)(OCC)OCC \n",
+ "954 956 covalent_metal CCC(=O)O[Na] \n",
+ "\n",
+ " Filtered_at Cleaned_at Normalized_at \\\n",
+ "407 3 0 0 \n",
+ "533 3 0 0 \n",
+ "951 3 0 0 \n",
+ "954 3 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "407 But for this case, this does not make much sense since the resulting molecules after removing Zirconium would be not functional, and the covalent metal would be deleted entirely since all of its substructures are salts.\n",
+ "\n",
+ "However, what we could have done if it made sense:\n",
+ "1. disconnect_metals\n",
+ "2. handle_charges.uncharge\n",
+ "3. normalize_molecule.normalize\n",
+ "4. remove_salts\n",
+ "5. handle_charges.uncharge\n",
+ "6. normalize_molecule.normalize\n",
+ "7. handle_fragments.choose_largest_fragment\n",
+ "8. Apply a `cleaned_at` marker"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "#### Task 6: Removing salts \n",
+ "\n",
+ "This curation step can be applied to different subsets of the dataset.
\n",
+ "First, we will apply this to our entries that passed all steps before. \n",
+ "Since we filtered all mixtures out in previous steps, all salts found in this step are the only compound in the entry. Therefore they need to be deleted (filtered).\n",
+ "\n",
+ "More interesting might be the inspection of the *inorganics* **(Task 3)** or *mixtures* **(Task 4)**. We could check if any of those mixtures contain salts known in our dictionary. If so, we can delete those salts and reuse the entries if they are free of mixtures."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "First, we will search for salts in our dataset. "
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 17,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Setting up the task_number\n",
+ "task_number = 6\n",
+ "\n",
+ "# Check for salts\n",
+ "dataset[\"salts\"] = dataset.apply(\n",
+ " lambda row: detect_salts(row.mol) if row.Filtered_at == 0 else None,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "# Add task_number to failed entries\n",
+ "dataset.loc[dataset[\"salts\"] == True, [\"Filtered_at\"]] = task_number"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries containing salts."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 18,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 22 | \n",
+ " 23 | \n",
+ " Acetic acid | \n",
+ " CC(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 199 | \n",
+ " 200 | \n",
+ " Citric acid | \n",
+ " O=C(O)CC(O)(CC(=O)O)C(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 275 | \n",
+ " 276 | \n",
+ " Dimethyl sulfoxide | \n",
+ " CS(C)=O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 335 | \n",
+ " 336 | \n",
+ " Ethanol | \n",
+ " CCO | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 356 | \n",
+ " 357 | \n",
+ " Ferrous citrate | \n",
+ " O=C(O)CC(O)(CC(=O)O)C(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 400 | \n",
+ " 401 | \n",
+ " Glutamic acid | \n",
+ " NC(CCC(=O)O)C(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 405 | \n",
+ " 406 | \n",
+ " Glycerol | \n",
+ " OCC(O)CO | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 481 | \n",
+ " 482 | \n",
+ " Lactic acid | \n",
+ " CC(O)C(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 605 | \n",
+ " 606 | \n",
+ " Niacin | \n",
+ " O=C(O)c1cccnc1 | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 613 | \n",
+ " 614 | \n",
+ " Nicotinic acid | \n",
+ " O=C(O)c1cccnc1 | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 630 | \n",
+ " 631 | \n",
+ " N-methylglucamine | \n",
+ " CNCC(O)C(O)C(O)C(O)CO | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 696 | \n",
+ " 697 | \n",
+ " phenylacetate | \n",
+ " O=C(O)Cc1ccccc1 | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 709 | \n",
+ " 710 | \n",
+ " Piperazine | \n",
+ " C1CNCCN1 | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 783 | \n",
+ " 784 | \n",
+ " Salicylic acid | \n",
+ " O=C(O)c1ccccc1O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 798 | \n",
+ " 799 | \n",
+ " Sodium acetate | \n",
+ " CC(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 799 | \n",
+ " 800 | \n",
+ " Sodium benzoate | \n",
+ " O=C(O)c1ccccc1 | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 800 | \n",
+ " 801 | \n",
+ " Sodium bicarbonate | \n",
+ " O=C(O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 802 | \n",
+ " 803 | \n",
+ " Sodium propionate | \n",
+ " CCC(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 949 | \n",
+ " 950 | \n",
+ " Zinc acetate | \n",
+ " CC(=O)O | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names SMILES Filtered_at \\\n",
+ "22 23 Acetic acid CC(=O)O 6 \n",
+ "199 200 Citric acid O=C(O)CC(O)(CC(=O)O)C(=O)O 6 \n",
+ "275 276 Dimethyl sulfoxide CS(C)=O 6 \n",
+ "335 336 Ethanol CCO 6 \n",
+ "356 357 Ferrous citrate O=C(O)CC(O)(CC(=O)O)C(=O)O 6 \n",
+ "400 401 Glutamic acid NC(CCC(=O)O)C(=O)O 6 \n",
+ "405 406 Glycerol OCC(O)CO 6 \n",
+ "481 482 Lactic acid CC(O)C(=O)O 6 \n",
+ "605 606 Niacin O=C(O)c1cccnc1 6 \n",
+ "613 614 Nicotinic acid O=C(O)c1cccnc1 6 \n",
+ "630 631 N-methylglucamine CNCC(O)C(O)C(O)C(O)CO 6 \n",
+ "696 697 phenylacetate O=C(O)Cc1ccccc1 6 \n",
+ "709 710 Piperazine C1CNCCN1 6 \n",
+ "783 784 Salicylic acid O=C(O)c1ccccc1O 6 \n",
+ "798 799 Sodium acetate CC(=O)O 6 \n",
+ "799 800 Sodium benzoate O=C(O)c1ccccc1 6 \n",
+ "800 801 Sodium bicarbonate O=C(O)O 6 \n",
+ "802 803 Sodium propionate CCC(=O)O 6 \n",
+ "949 950 Zinc acetate CC(=O)O 6 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "22 0 0 \n",
+ "199 0 0 \n",
+ "275 0 0 \n",
+ "335 0 0 \n",
+ "356 0 0 \n",
+ "400 0 0 \n",
+ "405 0 0 \n",
+ "481 0 0 \n",
+ "605 0 0 \n",
+ "613 0 0 \n",
+ "630 0 0 \n",
+ "696 0 0 \n",
+ "709 0 0 \n",
+ "783 0 0 \n",
+ "798 0 0 \n",
+ "799 0 0 \n",
+ "800 0 0 \n",
+ "802 0 0 \n",
+ "949 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "22 \n",
+ "\n",
+ "\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 1 | \n",
+ " (R)-Roscovitine | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 1 | \n",
+ " 2 | \n",
+ " 17-Methyltestosterone | \n",
+ " CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 2 | \n",
+ " 3 | \n",
+ " 1-alpha-Hydroxycholecalciferol | \n",
+ " C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 3 | \n",
+ " 4 | \n",
+ " 2,3-Dimercaptosuccinic acid | \n",
+ " O=C(O)C(S)C(S)C(=O)O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 4 | \n",
+ " 5 | \n",
+ " 2,4,6-Trinitrotoluene | \n",
+ " Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA53... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ ""
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "0 1 (R)-Roscovitine \n",
+ "1 2 17-Methyltestosterone \n",
+ "2 3 1-alpha-Hydroxycholecalciferol \n",
+ "3 4 2,3-Dimercaptosuccinic acid \n",
+ "4 5 2,4,6-Trinitrotoluene \n",
+ "\n",
+ " SMILES Filtered_at Cleaned_at \\\n",
+ "0 CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 0 0 \n",
+ "1 CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O 0 0 \n",
+ "2 C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O 0 0 \n",
+ "3 O=C(O)C(S)C(S)C(=O)O 0 0 \n",
+ "4 Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] 0 0 \n",
+ "\n",
+ " Normalized_at mol \\\n",
+ "0 0 \n",
+ "\n",
+ "\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 22 | \n",
+ " 23 | \n",
+ " Acetic acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 199 | \n",
+ " 200 | \n",
+ " Citric acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 275 | \n",
+ " 276 | \n",
+ " Dimethyl sulfoxide | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 335 | \n",
+ " 336 | \n",
+ " Ethanol | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 356 | \n",
+ " 357 | \n",
+ " Ferrous citrate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 400 | \n",
+ " 401 | \n",
+ " Glutamic acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 405 | \n",
+ " 406 | \n",
+ " Glycerol | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 481 | \n",
+ " 482 | \n",
+ " Lactic acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 605 | \n",
+ " 606 | \n",
+ " Niacin | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 613 | \n",
+ " 614 | \n",
+ " Nicotinic acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 630 | \n",
+ " 631 | \n",
+ " N-methylglucamine | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 696 | \n",
+ " 697 | \n",
+ " phenylacetate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 709 | \n",
+ " 710 | \n",
+ " Piperazine | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 783 | \n",
+ " 784 | \n",
+ " Salicylic acid | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 798 | \n",
+ " 799 | \n",
+ " Sodium acetate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 799 | \n",
+ " 800 | \n",
+ " Sodium benzoate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 800 | \n",
+ " 801 | \n",
+ " Sodium bicarbonate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 802 | \n",
+ " 803 | \n",
+ " Sodium propionate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 949 | \n",
+ " 950 | \n",
+ " Zinc acetate | \n",
+ " | \n",
+ " 6 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ ""
+ ],
+ "text/plain": [
+ " IDs Names SMILES Filtered_at Cleaned_at Normalized_at \\\n",
+ "22 23 Acetic acid 6 0 0 \n",
+ "199 200 Citric acid 6 0 0 \n",
+ "275 276 Dimethyl sulfoxide 6 0 0 \n",
+ "335 336 Ethanol 6 0 0 \n",
+ "356 357 Ferrous citrate 6 0 0 \n",
+ "400 401 Glutamic acid 6 0 0 \n",
+ "405 406 Glycerol 6 0 0 \n",
+ "481 482 Lactic acid 6 0 0 \n",
+ "605 606 Niacin 6 0 0 \n",
+ "613 614 Nicotinic acid 6 0 0 \n",
+ "630 631 N-methylglucamine 6 0 0 \n",
+ "696 697 phenylacetate 6 0 0 \n",
+ "709 710 Piperazine 6 0 0 \n",
+ "783 784 Salicylic acid 6 0 0 \n",
+ "798 799 Sodium acetate 6 0 0 \n",
+ "799 800 Sodium benzoate 6 0 0 \n",
+ "800 801 Sodium bicarbonate 6 0 0 \n",
+ "802 803 Sodium propionate 6 0 0 \n",
+ "949 950 Zinc acetate 6 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "22 \n",
+ "\n",
+ "\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 114 | \n",
+ " 115 | \n",
+ " Bortezomib | \n",
+ " CC(C)CC(NC(=O)C(Cc1ccccc1)NC(=O)c1cnccn1)B(O)O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 407 | \n",
+ " 408 | \n",
+ " Gold Sodium Thiomalate | \n",
+ " O=C(O)CC(S[Au])C(=O)O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 533 | \n",
+ " 534 | \n",
+ " Mersalyl | \n",
+ " COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 951 | \n",
+ " 952 | \n",
+ " zirconium | \n",
+ " CCO[Zr](OCC)(OCC)OCC | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 952 | \n",
+ " 953 | \n",
+ " hemoglobin | \n",
+ " C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 954 | \n",
+ " 956 | \n",
+ " covalent_metal | \n",
+ " CCC(=O)O[Na] | \n",
+ " 3 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " True | \n",
+ " None | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 958 | \n",
+ " 960 | \n",
+ " 1,4-Dioxane | \n",
+ " C1COCCO1.Oc1ccccc1 | \n",
+ " 4 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA54... | \n",
+ " True | \n",
+ " False | \n",
+ " True | \n",
+ " None | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ ""
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "114 115 Bortezomib \n",
+ "407 408 Gold Sodium Thiomalate \n",
+ "533 534 Mersalyl \n",
+ "951 952 zirconium \n",
+ "952 953 hemoglobin \n",
+ "954 956 covalent_metal \n",
+ "958 960 1,4-Dioxane \n",
+ "\n",
+ " SMILES Filtered_at \\\n",
+ "114 CC(C)CC(NC(=O)C(Cc1ccccc1)NC(=O)c1cnccn1)B(O)O 3 \n",
+ "407 O=C(O)CC(S[Au])C(=O)O 3 \n",
+ "533 COC(CNC(=O)c1ccccc1OCC(=O)O)C[Hg]O 3 \n",
+ "951 CCO[Zr](OCC)(OCC)OCC 3 \n",
+ "952 C=CC1=C(C)c2cc3[n-]c(cc4nc(cc5[nH]c(cc1n2)c(C)... 3 \n",
+ "954 CCC(=O)O[Na] 3 \n",
+ "958 C1COCCO1.Oc1ccccc1 4 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "114 0 0 \n",
+ "407 0 0 \n",
+ "533 0 0 \n",
+ "951 0 0 \n",
+ "952 0 0 \n",
+ "954 0 0 \n",
+ "958 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "114 \n",
+ "\n",
+ "The steps would be:\n",
+ "\n",
+ "1. remove_salts\n",
+ "2. handle_charges.uncharge\n",
+ "3. normalize_molecule.normalize\n",
+ "4. handle_fragments.choose_largest_fragment"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Step 4: Normalization of Specific Chemotypes\n",
+ "\n",
+ "After we filtered all problematic entries in the previous steps and created subsets to curate entries containing metals and salts, the next task is to apply normalization transformations to the remaining entries to correct functional groups and recombine charges.
\n",
+ "The standardization API utilizes the Normalization transformations embedded in the rdMolStandardize-Package, which derives the rules described in the InChI technical manual.
\n",
+ "\n",
+ "*If available, custom conversions rules can be used and implemented but require modifying the `normalize_molecules.normalize` function to use them. (This might be covered in further development of this API.* "
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "#### Task 7: Normalization"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 25,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# Setting up the task_number\n",
+ "task_number = 7\n",
+ "\n",
+ "# Normalize the entries, overwrite the previous mol\n",
+ "dataset[\"mol\"] = dataset.apply(\n",
+ " lambda row: normalize_molecules.normalize(row.mol)\n",
+ " if row.Filtered_at == 0\n",
+ " else row.mol,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "# Calculate new SMILES for the entries to determine which entries needed to be normalized\n",
+ "dataset[\"SMILES_after_normalization\"] = dataset.apply(\n",
+ " lambda row: convert_format.convert_mol_to_smiles(row.mol)\n",
+ " if row.Filtered_at == 0\n",
+ " else row.SMILES,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "\n",
+ "# Compare the SMILES for changes after the normalization --> save as Boolean Value\n",
+ "dataset[\"normalized\"] = dataset.apply(\n",
+ " lambda row: smiles_string_changed(row.SMILES, row.SMILES_after_normalization)\n",
+ " if row.Filtered_at == 0\n",
+ " else None,\n",
+ " axis=1,\n",
+ ")\n",
+ "\n",
+ "\n",
+ "# Add task_number to normalized entries\n",
+ "dataset.loc[dataset[\"normalized\"] == True, [\"Normalized_at\"]] = task_number"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries where normalization steps took place."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 26,
+ "metadata": {
+ "scrolled": true
+ },
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ " SMILES_after_normalization | \n",
+ " normalized | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 574 | \n",
+ " 575 | \n",
+ " Modafinil | \n",
+ " NC(=O)CS(=O)C(c1ccccc1)c1ccccc1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 7 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " NC(=O)C[S+]([O-])C(c1ccccc1)c1ccccc1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 830 | \n",
+ " 831 | \n",
+ " Sulfinpyrazone | \n",
+ " O=C1C(CCS(=O)c2ccccc2)C(=O)N(c2ccccc2)N1c1ccccc1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 7 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " O=C1C(CC[S+]([O-])c2ccccc2)C(=O)N(c2ccccc2)N1c... | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 831 | \n",
+ " 832 | \n",
+ " Sulindac | \n",
+ " CC1=C(CC(=O)O)c2cc(F)ccc2C1=Cc1ccc(S(C)=O)cc1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 7 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC1=C(CC(=O)O)c2cc(F)ccc2C1=Cc1ccc([S+](C)[O-]... | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 955 | \n",
+ " 957 | \n",
+ " test_charge_recombination | \n",
+ " CC([O-])=[N+](C)C | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 7 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=O)N(C)C | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "574 575 Modafinil \n",
+ "830 831 Sulfinpyrazone \n",
+ "831 832 Sulindac \n",
+ "955 957 test_charge_recombination \n",
+ "\n",
+ " SMILES Filtered_at \\\n",
+ "574 NC(=O)CS(=O)C(c1ccccc1)c1ccccc1 0 \n",
+ "830 O=C1C(CCS(=O)c2ccccc2)C(=O)N(c2ccccc2)N1c1ccccc1 0 \n",
+ "831 CC1=C(CC(=O)O)c2cc(F)ccc2C1=Cc1ccc(S(C)=O)cc1 0 \n",
+ "955 CC([O-])=[N+](C)C 0 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "574 0 7 \n",
+ "830 0 7 \n",
+ "831 0 7 \n",
+ "955 0 7 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "574 save as Boolean Value\n",
+ "dataset[\"new_canonical_tautomer\"] = dataset.apply(\n",
+ " lambda row: smiles_string_changed(row.SMILES, row.canonicalized_tautomer_smiles)\n",
+ " if row.Filtered_at == 0\n",
+ " else None,\n",
+ " axis=1,\n",
+ ")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "Below you can see all entries where the canonicalized tautomer differs to the SMILES, that resulted from the curation process."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 29,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ " SMILES_after_normalization | \n",
+ " normalized | \n",
+ " canonicalized_tautomer_smiles | \n",
+ " new_canonical_tautomer | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 1 | \n",
+ " (R)-Roscovitine | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " False | \n",
+ " CCC(CO)N=c1[nH]c(=NCc2ccccc2)c2ncn(C(C)C)c2[nH]1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 11 | \n",
+ " 12 | \n",
+ " 5-Azacitidine | \n",
+ " Nc1ncn(C2OC(CO)C(O)C2O)c(=O)n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " Nc1ncn(C2OC(CO)C(O)C2O)c(=O)n1 | \n",
+ " False | \n",
+ " N=c1ncn(C2OC(CO)C(O)C2O)c(=O)[nH]1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 18 | \n",
+ " 19 | \n",
+ " Acenocoumarol | \n",
+ " CC(=O)CC(c1ccc([N+](=O)[O-])cc1)c1c(O)oc2ccccc... | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=O)CC(c1ccc([N+](=O)[O-])cc1)c1c(O)oc2ccccc... | \n",
+ " False | \n",
+ " CC(=O)CC(c1ccc([N+](=O)[O-])cc1)c1c(O)c2ccccc2... | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 21 | \n",
+ " 22 | \n",
+ " Acetazolamide | \n",
+ " CC(=O)Nc1nnc(S(N)(=O)=O)s1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=O)Nc1nnc(S(N)(=O)=O)s1 | \n",
+ " False | \n",
+ " CC(=O)N=c1[nH]nc(S(N)(=O)=O)s1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 24 | \n",
+ " 25 | \n",
+ " Acetohydroxamic acid | \n",
+ " CC(=O)NO | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=O)NO | \n",
+ " False | \n",
+ " CC(O)=NO | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ "
\n",
+ " \n",
+ " | 937 | \n",
+ " 938 | \n",
+ " Vitamin K | \n",
+ " CC(=CCC1=C(C)C(=O)c2ccccc2C1=O)CCCC(C)CCCC(C)C... | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=CCC1=C(C)C(=O)c2ccccc2C1=O)CCCC(C)CCCC(C)C... | \n",
+ " False | \n",
+ " CC(C=Cc1c(C)c(O)c2ccccc2c1O)=CCCC(C)CCCC(C)CCC... | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 941 | \n",
+ " 942 | \n",
+ " Warfarin | \n",
+ " CC(=O)CC(c1ccccc1)c1c(O)oc2ccccc2c1=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC(=O)CC(c1ccccc1)c1c(O)oc2ccccc2c1=O | \n",
+ " False | \n",
+ " CC(=O)CC(c1ccccc1)c1c(O)c2ccccc2oc1=O | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 946 | \n",
+ " 947 | \n",
+ " Zalcitabine | \n",
+ " Nc1ccn(C2CCC(CO)O2)c(=O)n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " Nc1ccn(C2CCC(CO)O2)c(=O)n1 | \n",
+ " False | \n",
+ " N=c1ccn(C2CCC(CO)O2)c(=O)[nH]1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 947 | \n",
+ " 948 | \n",
+ " Zidovudine | \n",
+ " Cc1cn(C2CC(NN=N)C(CO)O2)c(=O)[nH]c1=O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " Cc1cn(C2CC(NN=N)C(CO)O2)c(=O)[nH]c1=O | \n",
+ " False | \n",
+ " Cc1cn(C2CC(N=NN)C(CO)O2)c(=O)[nH]c1=O | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 956 | \n",
+ " 958 | \n",
+ " Chloroquine | \n",
+ " CCN(CC)CCCC(C)Nc1ccnc2cc(Cl)ccc12 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA65... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CCN(CC)CCCC(C)Nc1ccnc2cc(Cl)ccc12 | \n",
+ " False | \n",
+ " CCN(CC)CCCC(C)N=c1cc[nH]c2cc(Cl)ccc12 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
237 rows × 16 columns
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "0 1 (R)-Roscovitine \n",
+ "11 12 5-Azacitidine \n",
+ "18 19 Acenocoumarol \n",
+ "21 22 Acetazolamide \n",
+ "24 25 Acetohydroxamic acid \n",
+ ".. ... ... \n",
+ "937 938 Vitamin K \n",
+ "941 942 Warfarin \n",
+ "946 947 Zalcitabine \n",
+ "947 948 Zidovudine \n",
+ "956 958 Chloroquine \n",
+ "\n",
+ " SMILES Filtered_at \\\n",
+ "0 CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 0 \n",
+ "11 Nc1ncn(C2OC(CO)C(O)C2O)c(=O)n1 0 \n",
+ "18 CC(=O)CC(c1ccc([N+](=O)[O-])cc1)c1c(O)oc2ccccc... 0 \n",
+ "21 CC(=O)Nc1nnc(S(N)(=O)=O)s1 0 \n",
+ "24 CC(=O)NO 0 \n",
+ ".. ... ... \n",
+ "937 CC(=CCC1=C(C)C(=O)c2ccccc2C1=O)CCCC(C)CCCC(C)C... 0 \n",
+ "941 CC(=O)CC(c1ccccc1)c1c(O)oc2ccccc2c1=O 0 \n",
+ "946 Nc1ccn(C2CCC(CO)O2)c(=O)n1 0 \n",
+ "947 Cc1cn(C2CC(NN=N)C(CO)O2)c(=O)[nH]c1=O 0 \n",
+ "956 CCN(CC)CCCC(C)Nc1ccnc2cc(Cl)ccc12 0 \n",
+ "\n",
+ " Cleaned_at Normalized_at \\\n",
+ "0 0 0 \n",
+ "11 0 0 \n",
+ "18 0 0 \n",
+ "21 0 0 \n",
+ "24 0 0 \n",
+ ".. ... ... \n",
+ "937 0 0 \n",
+ "941 0 0 \n",
+ "946 0 0 \n",
+ "947 0 0 \n",
+ "956 0 0 \n",
+ "\n",
+ " mol Carbon_present \\\n",
+ "0 \n",
+ "An Example of how this can be done for the entry with the ID 12 is shown below:"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 30,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "{'N=c1ncn(C2OC(CO)C(O)C2O)c(=O)[nH]1',\n",
+ " 'N=c1ncn(C2OC(CO)C(O)C2O)c(O)n1',\n",
+ " 'Nc1ncn(C2OC(CO)C(O)C2O)c(=O)n1'}"
+ ]
+ },
+ "execution_count": 30,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "# Extract the SMILES for the entry with the ID you want to enumerate the tautomers for\n",
+ "smiles_to_enumerate_tautomer = \"\".join(\n",
+ " dataset.loc[dataset[\"IDs\"] == 12, [\"SMILES\"]].values[0]\n",
+ ")\n",
+ "\n",
+ "# Apply the enumerate_tautomer function on that SMILES\n",
+ "handle_tautomers.enumerate_tautomer(smiles_to_enumerate_tautomer)"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "### Step 5: Removal of duplicates\n",
+ "\n",
+ "Since RDKit can calculate the canonical version of SMILES, we can try to find all duplicate entries in our data frame through a SMILES string comparison.\n",
+ "\n",
+ "> JRG: Are you actually comparing canonical smiles here? I think they are just the raw values present in the dataset, aren't they? You might need to do a round trip MolFromSmiles->MolToSmiles to get the canonical version. Molecule comparison is tricky! Best way is to resort to graph homology, but we are not doing that now. Just ensure you are indeed using canonical smiles.\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 31,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Following IDs have identical SMILES 2 and 552\n",
+ "Following IDs have identical SMILES 23 and 799\n",
+ "Following IDs have identical SMILES 23 and 950\n",
+ "Following IDs have identical SMILES 30 and 507\n",
+ "Following IDs have identical SMILES 43 and 764\n",
+ "Following IDs have identical SMILES 47 and 883\n",
+ "Following IDs have identical SMILES 55 and 407\n",
+ "Following IDs have identical SMILES 58 and 864\n",
+ "Following IDs have identical SMILES 65 and 218\n",
+ "Following IDs have identical SMILES 168 and 169\n",
+ "Following IDs have identical SMILES 200 and 357\n",
+ "Following IDs have identical SMILES 211 and 212\n",
+ "Following IDs have identical SMILES 253 and 513\n",
+ "Following IDs have identical SMILES 322 and 326\n",
+ "Following IDs have identical SMILES 327 and 646\n",
+ "Following IDs have identical SMILES 606 and 614\n",
+ "Following IDs have identical SMILES 736 and 737\n",
+ "Following IDs have identical SMILES 751 and 935\n",
+ "Following IDs have identical SMILES 795 and 796\n",
+ "Following IDs have identical SMILES 799 and 950\n"
+ ]
+ }
+ ],
+ "source": [
+ "i = 0\n",
+ "\n",
+ "all_SMILES = dataset.loc[:, \"SMILES\"]\n",
+ "IDs = dataset.loc[:, \"IDs\"]\n",
+ "list_len = len(all_SMILES)\n",
+ "list_len\n",
+ "\n",
+ "while i < list_len:\n",
+ " SMILES = all_SMILES[i]\n",
+ " ID = IDs[i]\n",
+ " j = i + 1\n",
+ " while j < list_len:\n",
+ " compared_SMILES = all_SMILES[j]\n",
+ " compared_ID = IDs[j]\n",
+ " j += 1\n",
+ " if SMILES == compared_SMILES and ID != compared_ID:\n",
+ " print(\"Following IDs have identical SMILES\", ID, \"and\", compared_ID)\n",
+ " continue # JRG: What's this continue for? It's the last statement... do you mean `break`?\n",
+ " i += 1"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "# Export of the Dataset\n",
+ "\n",
+ "Various exporting possibilities are now open.
\n",
+ "You can:
\n",
+ "- export the whole dataset, with its scores in an dedicated row\n",
+ "- filter for only the entries that passed all filtering step (ergo have a Score of null in the *Filtered_at*-column)\n",
+ "- filter for the entries that where filtered at a step to perform some transformations on them (e.g. remove salts)\n",
+ "- export all the mol files you need into a SDF with the function `convert_mol_to_sdf` by firstly generating an array of all mol-entries (`mol_array`) and pass them to the function as a parameter, togheter with a filename `fn` (e.g. `convert_mol_to_sdf(mol_array, fn=\"my_sdf_export\")`)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 32,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " IDs | \n",
+ " Names | \n",
+ " SMILES | \n",
+ " Filtered_at | \n",
+ " Cleaned_at | \n",
+ " Normalized_at | \n",
+ " mol | \n",
+ " Carbon_present | \n",
+ " Inorganics | \n",
+ " mixture | \n",
+ " metals | \n",
+ " salts | \n",
+ " SMILES_after_normalization | \n",
+ " normalized | \n",
+ " canonicalized_tautomer_smiles | \n",
+ " new_canonical_tautomer | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 1 | \n",
+ " (R)-Roscovitine | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 | \n",
+ " False | \n",
+ " CCC(CO)N=c1[nH]c(=NCc2ccccc2)c2ncn(C(C)C)c2[nH]1 | \n",
+ " True | \n",
+ "
\n",
+ " \n",
+ " | 1 | \n",
+ " 2 | \n",
+ " 17-Methyltestosterone | \n",
+ " CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O | \n",
+ " False | \n",
+ " CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 2 | \n",
+ " 3 | \n",
+ " 1-alpha-Hydroxycholecalciferol | \n",
+ " C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O | \n",
+ " False | \n",
+ " C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 3 | \n",
+ " 4 | \n",
+ " 2,3-Dimercaptosuccinic acid | \n",
+ " O=C(O)C(S)C(S)C(=O)O | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " O=C(O)C(S)C(S)C(=O)O | \n",
+ " False | \n",
+ " O=C(O)C(S)C(S)C(=O)O | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ " | 4 | \n",
+ " 5 | \n",
+ " 2,4,6-Trinitrotoluene | \n",
+ " Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " 0 | \n",
+ " <rdkit.Chem.rdchem.Mol object at 0x0000016DA64... | \n",
+ " True | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " False | \n",
+ " Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] | \n",
+ " False | \n",
+ " Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] | \n",
+ " False | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " IDs Names \\\n",
+ "0 1 (R)-Roscovitine \n",
+ "1 2 17-Methyltestosterone \n",
+ "2 3 1-alpha-Hydroxycholecalciferol \n",
+ "3 4 2,3-Dimercaptosuccinic acid \n",
+ "4 5 2,4,6-Trinitrotoluene \n",
+ "\n",
+ " SMILES Filtered_at Cleaned_at \\\n",
+ "0 CCC(CO)Nc1nc(NCc2ccccc2)c2ncn(C(C)C)c2n1 0 0 \n",
+ "1 CC12CCC(=O)C=C1CCC1C2CCC2(C)C1CCC2(C)O 0 0 \n",
+ "2 C=C1C(=CC=C2CCCC3(C)C2CCC3C(C)CCCC(C)C)CC(O)CC1O 0 0 \n",
+ "3 O=C(O)C(S)C(S)C(=O)O 0 0 \n",
+ "4 Cc1c([N+](=O)[O-])cc([N+](=O)[O-])cc1[N+](=O)[O-] 0 0 \n",
+ "\n",
+ " Normalized_at mol \\\n",
+ "0 0 nglview index mapping based on file format
# (this is important to know how nglview will index residues)
self._map_residue_ids_names_nglixs(pocket)
+ self.pockets_residue_ngl_ixs[pocket.name] = (
+ pocket.residues.dropna()
+ .merge(self._residue_ids_to_ngl_ixs[pocket.name], how="left", on="residue.id")[
+ "residue.ngl_ix"
+ ]
+ .to_list()
+ )
# Load structure from text in nglview
self._add_structure(pocket, ligand_expo_id)
# Show regions
if show_regions:
- self._add_regions(pocket)
+ self._add_regions(pocket, show_only_pocket_residues)
# Show pocket center
if show_pocket_center:
@@ -281,7 +291,7 @@ def _add_ligand(self, pocket, ligand_expo_id):
f"hetero and not water and not ions"
)
- def _add_regions(self, pocket):
+ def _add_regions(self, pocket, show_only_pocket_residues=False):
"""
Color residues by regions.
@@ -308,9 +318,14 @@ def _add_regions(self, pocket):
residue_ngl_ix = residue_id2ix.loc[residue_id]
scheme_regions_list.append([color, residue_ngl_ix])
scheme_regions = nglview.color._ColorScheme(scheme_regions_list, label="scheme_regions")
+ if show_only_pocket_residues:
+ selection = " or ".join(self.pockets_residue_ngl_ixs[pocket.name])
+ else:
+ selection = "protein"
+ self.viewer.clear_representations(self._components_structures[pocket.name])
self.viewer.add_representation(
"cartoon",
- selection="protein",
+ selection=selection,
component=self._components_structures[pocket.name],
color=scheme_regions,
)
@@ -436,7 +451,9 @@ def _components_by_structure_name(self, structure_name):
components = []
components.append(self._components_structures[structure_name])
components.append(self._components_pocket_center[structure_name])
- components.extend(list(self._components_subpockets[structure_name].values()))
- components.extend(list(self._components_anchor_residues[structure_name].values()))
+ if len(self._components_subpockets) > 0:
+ components.extend(list(self._components_subpockets[structure_name].values()))
+ if len(self._components_anchor_residues) > 0:
+ components.extend(list(self._components_anchor_residues[structure_name].values()))
return components
diff --git a/opencadd/tests/compounds/standardization/data/new_mol.sdf b/opencadd/tests/compounds/standardization/data/new_mol.sdf
new file mode 100644
index 00000000..0c0242a2
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/data/new_mol.sdf
@@ -0,0 +1,130 @@
+NCGC00261900-01
+ RDKit 2D
+
+ 53 59 0 0 1 0 0 0 0 0999 V2000
+ 0.4125 0.7145 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 0.8250 -0.0000 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 1.6500 -0.0000 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 0.4125 -0.7145 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.4125 -0.7145 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.8250 0.0000 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.4125 0.7145 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ -1.6500 0.0000 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.3627 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5781 3.1622 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.0932 2.4947 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5781 1.8273 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.3232 1.0427 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.8752 0.4296 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.5162 0.8712 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.2613 0.0865 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.4543 -0.0850 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.1994 -0.8696 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.7514 -1.4827 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.5584 -1.3112 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.8133 -0.5266 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.4964 -2.2673 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.6895 -2.4389 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.4345 -3.2235 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.9866 -3.8366 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.7935 -3.6650 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.0485 -2.8804 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.3456 -4.2781 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.1740 -5.0851 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.8885 -5.4976 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5016 -4.9456 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.1661 -4.1919 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.6276 -3.3950 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.0145 -2.8430 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 3.3000 -3.2555 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 3.4715 -4.0624 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.2920 -4.1487 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.3627 2.0822 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.4215 1.2593 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 1.6697 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 2.0822 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 4.5572 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.2206 4.5572 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.9351 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 13.6496 4.5572 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.9351 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.2206 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 2.4947 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 1 2 1 0
+ 2 3 2 0
+ 2 4 1 0
+ 4 5 2 0
+ 5 6 1 0
+ 6 7 1 0
+ 6 8 2 0
+ 9 10 1 1
+ 10 11 1 0
+ 11 12 1 0
+ 12 13 1 1
+ 13 14 2 0
+ 13 15 1 0
+ 15 16 1 0
+ 16 17 1 0
+ 17 18 1 0
+ 18 19 1 0
+ 19 20 1 0
+ 20 21 1 0
+ 16 21 1 0
+ 19 22 1 0
+ 22 23 2 0
+ 23 24 1 0
+ 24 25 2 0
+ 25 26 1 0
+ 26 27 2 0
+ 22 27 1 0
+ 26 28 1 0
+ 28 29 1 0
+ 29 30 1 0
+ 30 31 1 0
+ 31 32 1 0
+ 28 32 1 0
+ 24 33 1 0
+ 33 34 1 0
+ 34 35 1 0
+ 35 36 1 0
+ 36 37 1 0
+ 33 37 1 0
+ 12 38 1 0
+ 9 38 1 0
+ 38 39 1 1
+ 38 40 1 0
+ 40 41 1 0
+ 41 42 2 0
+ 43 42 1 0
+ 9 43 1 0
+ 43 44 1 6
+ 44 45 1 0
+ 45 46 1 0
+ 46 47 2 0
+ 47 48 1 0
+ 48 49 2 0
+ 48 50 1 0
+ 50 51 2 0
+ 51 52 1 0
+ 42 52 1 0
+ 46 52 1 0
+ 52 53 1 1
+M END
+> (1)
+U 5882
+
+> (1)
+NCGC00261900
+
+> (1)
+NCGC00261900-01
+
+> (1)
+0
+
+$$$$
diff --git a/opencadd/tests/compounds/standardization/data/result_of_test_convert_mol_to_sdf.sdf b/opencadd/tests/compounds/standardization/data/result_of_test_convert_mol_to_sdf.sdf
new file mode 100644
index 00000000..0c0242a2
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/data/result_of_test_convert_mol_to_sdf.sdf
@@ -0,0 +1,130 @@
+NCGC00261900-01
+ RDKit 2D
+
+ 53 59 0 0 1 0 0 0 0 0999 V2000
+ 0.4125 0.7145 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 0.8250 -0.0000 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 1.6500 -0.0000 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 0.4125 -0.7145 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.4125 -0.7145 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.8250 0.0000 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ -0.4125 0.7145 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ -1.6500 0.0000 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.3627 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5781 3.1622 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.0932 2.4947 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5781 1.8273 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.3232 1.0427 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.8752 0.4296 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.5162 0.8712 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.2613 0.0865 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.4543 -0.0850 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.1994 -0.8696 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.7514 -1.4827 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.5584 -1.3112 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.8133 -0.5266 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.4964 -2.2673 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.6895 -2.4389 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.4345 -3.2235 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 5.9866 -3.8366 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 6.7935 -3.6650 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.0485 -2.8804 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.3456 -4.2781 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.1740 -5.0851 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 7.8885 -5.4976 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.5016 -4.9456 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 8.1661 -4.1919 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.6276 -3.3950 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.0145 -2.8430 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 3.3000 -3.2555 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 3.4715 -4.0624 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 4.2920 -4.1487 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.3627 2.0822 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 9.4215 1.2593 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 1.6697 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 2.0822 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.0772 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 10.7917 4.5572 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.2206 4.5572 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.9351 4.1447 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 13.6496 4.5572 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.9351 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 12.2206 2.9072 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 3.3197 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 11.5061 2.4947 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0
+ 1 2 1 0
+ 2 3 2 0
+ 2 4 1 0
+ 4 5 2 0
+ 5 6 1 0
+ 6 7 1 0
+ 6 8 2 0
+ 9 10 1 1
+ 10 11 1 0
+ 11 12 1 0
+ 12 13 1 1
+ 13 14 2 0
+ 13 15 1 0
+ 15 16 1 0
+ 16 17 1 0
+ 17 18 1 0
+ 18 19 1 0
+ 19 20 1 0
+ 20 21 1 0
+ 16 21 1 0
+ 19 22 1 0
+ 22 23 2 0
+ 23 24 1 0
+ 24 25 2 0
+ 25 26 1 0
+ 26 27 2 0
+ 22 27 1 0
+ 26 28 1 0
+ 28 29 1 0
+ 29 30 1 0
+ 30 31 1 0
+ 31 32 1 0
+ 28 32 1 0
+ 24 33 1 0
+ 33 34 1 0
+ 34 35 1 0
+ 35 36 1 0
+ 36 37 1 0
+ 33 37 1 0
+ 12 38 1 0
+ 9 38 1 0
+ 38 39 1 1
+ 38 40 1 0
+ 40 41 1 0
+ 41 42 2 0
+ 43 42 1 0
+ 9 43 1 0
+ 43 44 1 6
+ 44 45 1 0
+ 45 46 1 0
+ 46 47 2 0
+ 47 48 1 0
+ 48 49 2 0
+ 48 50 1 0
+ 50 51 2 0
+ 51 52 1 0
+ 42 52 1 0
+ 46 52 1 0
+ 52 53 1 1
+M END
+> (1)
+U 5882
+
+> (1)
+NCGC00261900
+
+> (1)
+NCGC00261900-01
+
+> (1)
+0
+
+$$$$
diff --git a/opencadd/tests/compounds/standardization/test_convert_format.py b/opencadd/tests/compounds/standardization/test_convert_format.py
new file mode 100644
index 00000000..6c2ce7bd
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_convert_format.py
@@ -0,0 +1,93 @@
+"""
+test for the module `convert_format`
+"""
+import pytest
+import sys
+import rdkit
+import os
+from pathlib import Path
+from rdkit import Chem
+
+from opencadd.compounds.standardization import convert_format
+
+
+def _evaluation_mol_generator(test_smiles=None, test_inchi=None):
+ """Creates mol files directly with rdkits functions for evaluation."""
+ if test_smiles is not None:
+ return Chem.MolFromSmiles(test_smiles)
+ if test_inchi is not None:
+ return Chem.MolFromInchi(test_inchi)
+
+
+def _evaluation_inchi(test_result):
+ test_result = Chem.MolFromSmiles(test_result)
+ test_result = Chem.MolToInchi(test_result)
+ return test_result
+
+
+def _test_path(fn):
+ """Leads to files saved in the data folder
+
+ Parameters
+ ----------
+ fn: str
+ The whole filename.
+
+ Returns
+ -------
+ The path of the file in the current working system.
+ """
+ return Path(__file__).parent / "data" / fn
+
+
+def test_convert_smiles_to_mol(test_smiles="C(C1C(C(C(C(O1)O)O)O)O)O"):
+ """Tests if the created file is a mol file."""
+ test_result = convert_format.convert_smiles_to_mol(test_smiles)
+ assert isinstance(test_result, rdkit.Chem.rdchem.Mol) == True
+
+
+def test_convert_inchi_to_mol(
+ test_inchi="InChI=1S/C6H12O6/c7-1-2-3(8)4(9)5(10)6(11)12-2/h2-11H,1H2/t2-,3-,4+,5-,6?/m1/s1",
+):
+ """Tests if the created file is a mol file."""
+ test_result = convert_format.convert_inchi_to_mol(test_inchi)
+ assert isinstance(test_result, rdkit.Chem.rdchem.Mol) == True
+
+
+def test_convert_mol_to_smiles(
+ test_inchi="InChI=1S/C8H7O4S.Na/c9-8(10)7-3-1-6(2-4-7)5-13(11)12;/h1-4H,5H2,(H,9,10);/q;+1/p-1",
+):
+ """Tests if the created file matches the file it originated from."""
+ test_result = convert_format.convert_mol_to_smiles(
+ _evaluation_mol_generator(test_inchi=test_inchi), canonical=False
+ )
+ test_result = _evaluation_inchi(test_result)
+ assert test_result == test_inchi
+
+
+def test_convert_mol_to_inchi(
+ test_inchi="InChI=1S/C6H12O6/c7-1-2-3(8)4(9)5(10)6(11)12-2/h2-11H,1H2/t2-,3-,4+,5-,6?/m1/s1",
+):
+ """Tests if the created file matches the file it originated from."""
+ test_result = convert_format.convert_mol_to_inchi(
+ _evaluation_mol_generator(test_inchi=test_inchi)
+ )
+ assert test_result == test_inchi
+
+
+def test_convert_sdf_to_mol_array():
+ fn = "new_mol"
+ mol_array = convert_format.convert_sdf_to_mol_array(str(_test_path(fn)))
+ mol = mol_array[0]
+ assert isinstance(mol, rdkit.Chem.rdchem.Mol) == True
+
+
+def test_convert_mol_to_sdf():
+ fn = "new_mol"
+ mol_array = convert_format.convert_sdf_to_mol_array(str(_test_path(fn)))
+ convert_format.convert_mol_to_sdf(
+ mol_array, fn=str(_test_path("result_of_test_convert_mol_to_sdf.sdf"))
+ )
+ assert (
+ os.path.isfile(str(_test_path("result_of_test_convert_mol_to_sdf.sdf"))) == True
+ )
diff --git a/opencadd/tests/compounds/standardization/test_detect_inorganic.py b/opencadd/tests/compounds/standardization/test_detect_inorganic.py
new file mode 100644
index 00000000..8bbea11b
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_detect_inorganic.py
@@ -0,0 +1,67 @@
+"""
+test for the module `detect_inorganics`
+"""
+import pytest
+import sys
+import rdkit
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import detect_inorganics
+
+
+def _evaluation_mol_generator(test_smiles=None, test_inchi=None):
+ """Creates mol files directly with rdkits functions for evaluation."""
+ if test_smiles is not None:
+ return Chem.MolFromSmiles(test_smiles)
+ if test_inchi is not None:
+ return Chem.MolFromInchi(test_inchi)
+
+
+def _atom_test(test_inchi=None, test_smiles=None):
+ mol = _evaluation_mol_generator(test_inchi=test_inchi, test_smiles=test_smiles)
+ return detect_inorganic(mol)
+
+
+def test_organic():
+ """Tests if organic structures are not detected as inorganic.
+ Organic atoms:
+ hydrogen, carbon, nitrogen, oxygen,
+ fluorine, phosphorus, sulfur, chlorine, bromine, iodine
+ """
+ assert _atom_test(test_inchi="InChI=1S/H") == False
+ assert _atom_test(test_inchi="InChI=1S/C") == False
+ assert _atom_test(test_inchi="InChI=1S/N") == False
+ assert _atom_test(test_inchi="InChI=1S/O") == False
+ assert _atom_test(test_inchi="InChI=1S/F") == False
+ assert _atom_test(test_inchi="InChI=1S/P") == False
+ assert _atom_test(test_inchi="InChI=1S/S") == False
+ assert _atom_test(test_inchi="InChI=1S/Cl") == False
+ assert _atom_test(test_inchi="InChI=1S/Br") == False
+ assert _atom_test(test_inchi="InChI=1S/I") == False
+
+
+def test_inorganic():
+ """Tests if inorganic structures are detected as inorganic.
+ Example atoms:
+ aluminum, selenium, sodium, magnesium
+ """
+ assert _atom_test(test_inchi="InChI=1S/Al") == True
+ assert _atom_test(test_inchi="InChI=1S/Se") == True
+ assert _atom_test(test_inchi="InChI=1S/Na") == True
+ assert _atom_test(test_inchi="InChI=1S/Mg") == True
+
+
+def test_organic_inorganic_combination():
+ """Tests if inorganic atoms are detected in combination with organic
+ atoms.
+ """
+ assert _atom_test(test_smiles="c1ccccc1C(=O)O[Ca]OC(=O)c1ccccc1") == True
+
+
+def test_organic_organic_combination():
+ """Tests if a combination of organic atoms doesn't resolve in a
+ detection of a inorganic structure.
+ Tested here: nitro group
+ """
+ assert _atom_test(test_smiles="[N](=O)(=O)O") == False
diff --git a/opencadd/tests/compounds/standardization/test_disconnect_metals.py b/opencadd/tests/compounds/standardization/test_disconnect_metals.py
new file mode 100644
index 00000000..adab99be
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_disconnect_metals.py
@@ -0,0 +1,49 @@
+"""
+test for the module `disconnect_metals`
+
+derived from MolVS's tests:
+https://github.com/mcs07/MolVS/blob/master/tests/test_metal.py
+"""
+import pytest
+import sys
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import disconnect_metals
+
+
+def _disconnect_metals_smiles(smiles):
+ mol = Chem.MolFromSmiles(smiles)
+ mol = disconnect_metals(mol)
+ if mol:
+ return Chem.MolToSmiles(mol)
+
+
+def test_disconnect_metals1():
+ assert (
+ _disconnect_metals_smiles("NC(CC(=O)O)C(=O)[O-].O.O.[Na+]")
+ == "NC(CC(=O)O)C(=O)[O-].O.O.[Na+]"
+ )
+
+
+def test_covalent_metal():
+ """Test if covalent metal is disconnected."""
+ assert _disconnect_metals_smiles("CCC(=O)O[Na]") == "CCC(=O)[O-].[Na+]"
+
+
+def test_no_accidental_deletion():
+ """Test metal ion is untouched."""
+ assert _disconnect_metals_smiles("CCC(=O)[O-].[Na+]") == "CCC(=O)[O-].[Na+]"
+
+
+def test_dimethylmercury():
+ """Test dimethylmercury is not disconnected."""
+ assert _disconnect_metals_smiles("C[Hg]C") == "C[Hg]C"
+
+
+def test_zirconium():
+ """Test zirconium (IV) ethoxide."""
+ assert (
+ _disconnect_metals_smiles("CCO[Zr](OCC)(OCC)OCC")
+ == "CC[O-].CC[O-].CC[O-].CC[O-].[Zr+4]"
+ )
diff --git a/opencadd/tests/compounds/standardization/test_handle_charges.py b/opencadd/tests/compounds/standardization/test_handle_charges.py
new file mode 100644
index 00000000..a87e80a5
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_handle_charges.py
@@ -0,0 +1,125 @@
+"""
+test for the module `handle_charges`
+
+derived from MolVS's tests: https://github.com/mcs07/MolVS/blob/master/tests/test_charge.py
+"""
+import pytest
+import sys
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import handle_charges
+
+
+def _uncharge_smiles(smiles):
+ """Utility function that returns the uncharged SMILES for a given
+ SMILES string.
+ """
+ mol = Chem.MolFromSmiles(smiles)
+ mol = handle_charges.uncharge(mol)
+ if mol:
+ return Chem.MolToSmiles(mol, isomericSmiles=True)
+
+
+def test_neutralization():
+ """Test neutralization of ionized acids and bases."""
+ assert (
+ _uncharge_smiles("C(C(=O)[O-])(Cc1n[n-]nn1)(C[NH3+])(C[N+](=O)[O-])")
+ == "NCC(Cc1nn[nH]n1)(C[N+](=O)[O-])C(=O)O"
+ )
+
+
+def test_zwitterion():
+ """Test preservation of zwitterion."""
+ assert _uncharge_smiles("n(C)1cc[n+]2cccc([O-])c12") == "Cn1cc[n+]2cccc([O-])c12"
+
+
+def test_choline():
+ """Choline should be left with a positive charge."""
+ assert _uncharge_smiles("C[N+](C)(C)CCO") == "C[N+](C)(C)CCO"
+
+
+def test_hydrogen():
+ """This should have the hydrogen removed to give deanol as a charge parent."""
+ assert _uncharge_smiles("C[NH+](C)CCO") == "CN(C)CCO"
+
+
+def test_neutrality():
+ """Overall system is already neutral."""
+ assert _uncharge_smiles("[Na+].O=C([O-])c1ccccc1") == "O=C([O-])c1ccccc1.[Na+]"
+
+
+def test_benzoate():
+ """Benzoate ion to benzoic acid."""
+ assert _uncharge_smiles("O=C([O-])c1ccccc1") == "O=C(O)c1ccccc1"
+
+
+def test_histidine():
+ """Charges in histidine should be neutralized."""
+ assert _uncharge_smiles("[NH3+]C(Cc1cnc[nH]1)C(=O)[O-]") == "NC(Cc1cnc[nH]1)C(=O)O"
+
+
+def test_fragment_neutralization():
+ """Neutralize both fragments."""
+ assert _uncharge_smiles("C[NH+](C)(C).[Cl-]") == "CN(C)C.Cl"
+
+
+def test_oxigen_neutralisation():
+ """Neutralise one oxygen."""
+ assert _uncharge_smiles("[N+](=O)([O-])[O-]") == "O=[N+]([O-])[O-]"
+
+
+def test_prefer_organic_fragments():
+ """Smaller organic fragment should be chosen over larger inorganic fragment."""
+ assert _uncharge_smiles("[N+](=O)([O-])[O-].[CH2]") == "O=[N+]([O-])[O-].[CH2]"
+
+
+def test_oxygen_balancing():
+ """Single oxygen should be protonated, the other left to balance the positive nitrogen."""
+ assert _uncharge_smiles("C[N+](C)(C)CC([O-])C[O-]") == "C[N+](C)(C)CC([O-])CO"
+
+
+def test_strongest_acid():
+ """Strongest acid should be left ionized."""
+ assert (
+ _uncharge_smiles("[O-]C(=O)C[n+]1ccn2cccc([O-])c12")
+ == "O=C([O-])C[n+]1ccn2cccc(O)c21"
+ )
+
+
+def test_charge_neutralization():
+ """All charges should be neutralized."""
+ assert _uncharge_smiles("[NH+](C)(C)CC([O-])C[O-]") == "CN(C)CC(O)CO"
+
+
+def test_uncharge():
+ """All charges should be neutralized."""
+ assert _uncharge_smiles("CNCC([O-])C[O-]") == "CNCC(O)CO"
+
+
+# Tests for Reionize
+
+
+def _reionize_smiles(smiles):
+ """Utility function that returns the uncharged SMILES for a given
+ SMILES string.
+ """
+ mol = Chem.MolFromSmiles(smiles)
+ mol = handle_charges.reionize(mol)
+ if mol:
+ return Chem.MolToSmiles(mol)
+
+
+def test_proton_to_weak_acid():
+ """Test reionizer moves proton to weaker acid."""
+ assert (
+ _reionize_smiles("C1=C(C=CC(=C1)[S]([O-])=O)[S](O)(=O)=O")
+ == "O=S(O)c1ccc(S(=O)(=O)[O-])cc1"
+ )
+
+
+def test_charged_carbon():
+ """Test charged carbon doesn't get recognised as
+ alpha-carbon-hydrogen-keto.
+ """
+ assert _reionize_smiles("CCOC(=O)C(=O)[CH-]C#N") == "CCOC(=O)C(=O)[CH-]C#N"
diff --git a/opencadd/tests/compounds/standardization/test_handle_fragments.py b/opencadd/tests/compounds/standardization/test_handle_fragments.py
new file mode 100644
index 00000000..8a57be61
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_handle_fragments.py
@@ -0,0 +1,80 @@
+"""
+test for the module `handle fragments`
+
+derived from MolVS's tests: https://github.com/mcs07/MolVS/blob/master/tests/test_fragment.py
+"""
+import pytest
+import sys
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import handle_fragments
+
+
+def _remove_fragment_smiles(smiles):
+ """Utility function that returns the result SMILES after
+ remove_fragments is applied to given a SMILES string."""
+ mol = Chem.MolFromSmiles(smiles)
+ mol = handle_fragments.remove_fragments(mol)
+ return Chem.MolToSmiles(mol)
+
+
+def test_remove_single_salt():
+ """Single Salt removal."""
+ assert _remove_fragment_smiles("CN(C)C.Cl") == "CN(C)C"
+
+
+def test_remove_multiple_salts():
+ """Multiple salt removal."""
+ assert _remove_fragment_smiles("CN(C)C.Cl.Cl.Br") == "CN(C)C"
+
+
+def test_fragment_patterns():
+ """FragmentPatterns should match entire fragments only, matches
+ within larger fragments should be left.
+ """
+ assert _remove_fragment_smiles("CN(Br)Cl") == "CN(Cl)Br"
+ assert _remove_fragment_smiles("CN(Br)Cl.Cl") == "CN(Cl)Br"
+
+
+def test_charged_salts():
+ """Charged salts."""
+ assert _remove_fragment_smiles("C[NH+](C)(C).[Cl-]") == "C[NH+](C)C"
+
+
+def test_last_match():
+ """Last match should be left."""
+ assert _remove_fragment_smiles("CC(=O)O.[Na]") == "CC(=O)O"
+
+
+def test_left_identical():
+ """Multiple identical last fragments should all be left."""
+ assert _remove_fragment_smiles("Br.Br") == "Br.Br"
+
+
+def test_remove_multiple_fragment():
+ """Test multiple fragment removal."""
+ assert (
+ _remove_fragment_smiles("[Na+].OC(=O)Cc1ccc(CN)cc1.OS(=O)(=O)C(F)(F)F")
+ == "NCc1ccc(CC(=O)O)cc1"
+ )
+
+
+def test_1_4_Dioxiane():
+ """1,4-Dioxane should be removed."""
+ assert _remove_fragment_smiles("c1ccccc1O.O1CCOCC1") == "Oc1ccccc1"
+
+
+def test_benzene():
+ """Benzene should be removed."""
+ assert _remove_fragment_smiles("c1ccccc1.CCCBr") == "CCCBr"
+
+
+def test_remove_various_fragments():
+ """Various fragments should be removed."""
+ assert (
+ _remove_fragment_smiles(
+ "CC(NC1=CC=C(O)C=C1)=O.CCCCC.O.CCO.CCCO.C1CCCCC1.C1CCCCCC1"
+ )
+ == "CC(=O)Nc1ccc(O)cc1"
+ )
diff --git a/opencadd/tests/compounds/standardization/test_handle_hydrogens.py b/opencadd/tests/compounds/standardization/test_handle_hydrogens.py
new file mode 100644
index 00000000..240187e9
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_handle_hydrogens.py
@@ -0,0 +1,29 @@
+"""
+test for the module `handle_hydrogens`
+"""
+import pytest
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import handle_hydrogens
+
+
+def _evaluation_mol_generator(test_smiles=None, test_inchi=None):
+ """Creates mol files directly with rdkits functions for evaluation."""
+ if test_smiles is not None:
+ return Chem.MolFromSmiles(test_smiles, sanitize=False)
+ if test_inchi is not None:
+ return Chem.MolFromInchi(test_inchi)
+
+
+def test_remove_hydrogen():
+ assert (
+ Chem.MolToInchi(
+ handle_hydrogens.remove_hydrogens(
+ _evaluation_mol_generator(
+ test_inchi="InChI=1S/2C7H6O2.Ca/c2*8-7(9)6-4-2-1-3-5-6;/h2*1-5H,(H,8,9);/q;;+2/p-2"
+ )
+ )
+ )
+ == "InChI=1S/2C7H6O2.Ca/c2*8-7(9)6-4-2-1-3-5-6;/h2*1-5H,(H,8,9);/q;;+2/p-2"
+ )
diff --git a/opencadd/tests/compounds/standardization/test_handle_tautomers.py b/opencadd/tests/compounds/standardization/test_handle_tautomers.py
new file mode 100644
index 00000000..e6985037
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_handle_tautomers.py
@@ -0,0 +1,543 @@
+"""
+test for the module `handle_tautomers`
+
+Partially derived from MolVS's tests and RDKIT MolStandardize tutorial:
+https://github.com/mcs07/MolVS/blob/master/tests/test_tautomer.py
+"""
+import pytest
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization.handle_tautomers import enumerate_tautomer, canonicalize_tautomer
+
+
+def test_1_3_keto_enol_enumeration():
+ """Enumerate 1,3 keto/enol tautomer."""
+ assert enumerate_tautomer("C1(=CCCCC1)O") == {"OC1=CCCCC1", "O=C1CCCCC1"}
+ assert enumerate_tautomer("C1(CCCCC1)=O") == {"OC1=CCCCC1", "O=C1CCCCC1"}
+
+
+def test_acetophenone_keto_enol_enumeration():
+ """Enumerate acetophenone keto/enol tautomer."""
+ assert enumerate_tautomer("C(=C)(O)C1=CC=CC=C1") == {
+ "C=C(O)c1ccccc1",
+ "CC(=O)c1ccccc1",
+ }
+ assert enumerate_tautomer("CC(C)=O") == {"CC(C)=O", "C=C(C)O"}
+
+
+def test_1_5_keto_enol_enumeration3():
+ """1,5 keto/enol tautomer"""
+ assert enumerate_tautomer("C1(=CC=CCC1)O") == {
+ "O=C1C=CCCC1",
+ "OC1=CCC=CC1",
+ "OC1=CC=CCC1",
+ "O=C1CC=CCC1",
+ "OC1=CCCC=C1",
+ }
+
+
+def test_aliphatic_imine_enumeration():
+ """aliphatic imine tautomer"""
+ assert enumerate_tautomer("C1(CCCCC1)=N") == {"N=C1CCCCC1", "NC1=CCCCC1"}
+ assert enumerate_tautomer("C1(=CCCCC1)N") == {"N=C1CCCCC1", "NC1=CCCCC1"}
+
+
+def test_special_imine_enumeration():
+ """special imine tautomer"""
+ assert enumerate_tautomer("C1(C=CC=CN1)=CC") == {
+ "CC=C1C=CC=CN1",
+ "CCc1ccccn1",
+ "CC=C1C=CCC=N1",
+ }
+ assert enumerate_tautomer("C1(=NC=CC=C1)CC") == {
+ "CC=C1C=CC=CN1",
+ "CCc1ccccn1",
+ "CC=C1C=CCC=N1",
+ }
+
+
+def test_1_3_aromatic_heteroatom_enumeration():
+ """1,3 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("O=c1cccc[nH]1") == {"Oc1ccccn1", "O=c1cccc[nH]1"}
+ assert enumerate_tautomer("Oc1ccccn1") == {"Oc1ccccn1", "O=c1cccc[nH]1"}
+ assert enumerate_tautomer("Oc1ncc[nH]1") == {"Oc1ncc[nH]1", "O=c1[nH]cc[nH]1"}
+
+
+def test_1_3_heteroatom_enumeration():
+ """1,3 heteroatom H shift"""
+ assert enumerate_tautomer("OC(C)=NC") == {"CN=C(C)O", "CNC(C)=O", "C=C(O)NC"}
+ assert enumerate_tautomer("CNC(C)=O") == {"CN=C(C)O", "CNC(C)=O", "C=C(O)NC"}
+ assert enumerate_tautomer("S=C(N)N") == {"N=C(N)S", "NC(N)=S"}
+ assert enumerate_tautomer("SC(N)=N") == {"N=C(N)S", "NC(N)=S"}
+ assert enumerate_tautomer("N=c1[nH]ccn(C)1") == {"Cn1ccnc1N", "Cn1cc[nH]c1=N"}
+ assert enumerate_tautomer("CN=c1[nH]cncc1") == {
+ "CN=c1ccnc[nH]1",
+ "CNc1ccncn1",
+ "CN=c1cc[nH]cn1",
+ }
+
+
+def test_1_5_aromatic_heteroatom_enumeration():
+ """1,5 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("Oc1cccc2ccncc12") == {
+ "O=c1cccc2cc[nH]cc1-2",
+ "Oc1cccc2ccncc12",
+ }
+ assert enumerate_tautomer("O=c1cccc2cc[nH]cc1-2") == {
+ "O=c1cccc2cc[nH]cc1-2",
+ "Oc1cccc2ccncc12",
+ }
+ assert enumerate_tautomer("Cc1n[nH]c2ncnn12") == {
+ "C=C1NNc2ncnn21",
+ "Cc1n[nH]c2ncnn12",
+ "Cc1nnc2[nH]cnn12",
+ "C=C1NN=C2N=CNN12",
+ "Cc1nnc2nc[nH]n12",
+ "C=C1NN=C2NC=NN12",
+ }
+ assert enumerate_tautomer("Cc1nnc2nc[nH]n12") == {
+ "C=C1NNc2ncnn21",
+ "Cc1n[nH]c2ncnn12",
+ "Cc1nnc2[nH]cnn12",
+ "C=C1NN=C2N=CNN12",
+ "Cc1nnc2nc[nH]n12",
+ "C=C1NN=C2NC=NN12",
+ }
+ assert enumerate_tautomer("Oc1ccncc1") == {"Oc1ccncc1", "O=c1cc[nH]cc1"}
+ assert enumerate_tautomer("Oc1c(cccc3)c3nc2ccncc12") == {
+ "Oc1c2ccccc2nc2ccncc12",
+ "O=c1c2ccccc2[nH]c2ccncc12",
+ "O=c1c2c[nH]ccc-2nc2ccccc12",
+ }
+ assert enumerate_tautomer("C2(=C1C(=NC=N1)[NH]C(=N2)N)O") == {
+ "N=c1[nH]c2ncnc-2c(O)[nH]1",
+ "Nc1nc(O)c2ncnc-2[nH]1",
+ "N=c1nc(O)c2nc[nH]c2[nH]1",
+ "Nc1nc2ncnc-2c(O)[nH]1",
+ "N=c1nc2nc[nH]c2c(O)[nH]1",
+ "N=c1[nH]c(=O)c2nc[nH]c2[nH]1",
+ "N=c1nc(O)c2[nH]cnc2[nH]1",
+ "N=c1[nH]c(=O)c2[nH]cnc2[nH]1",
+ "Nc1nc(=O)c2nc[nH]c2[nH]1",
+ "Nc1nc(O)c2nc[nH]c2n1",
+ "Nc1nc(=O)c2[nH]cnc2[nH]1",
+ "N=c1nc2[nH]cnc2c(O)[nH]1",
+ "Nc1nc2[nH]cnc2c(=O)[nH]1",
+ "Nc1nc2nc[nH]c2c(=O)[nH]1",
+ "Nc1nc(O)c2[nH]cnc2n1",
+ }
+ assert enumerate_tautomer("C2(C1=C([NH]C=N1)[NH]C(=N2)N)=O") == {
+ "N=c1[nH]c2ncnc-2c(O)[nH]1",
+ "Nc1nc(O)c2ncnc-2[nH]1",
+ "N=c1nc(O)c2nc[nH]c2[nH]1",
+ "Nc1nc2ncnc-2c(O)[nH]1",
+ "N=c1nc2nc[nH]c2c(O)[nH]1",
+ "N=c1[nH]c(=O)c2nc[nH]c2[nH]1",
+ "N=c1nc(O)c2[nH]cnc2[nH]1",
+ "N=c1[nH]c(=O)c2[nH]cnc2[nH]1",
+ "Nc1nc(=O)c2nc[nH]c2[nH]1",
+ "Nc1nc(O)c2nc[nH]c2n1",
+ "Nc1nc(=O)c2[nH]cnc2[nH]1",
+ "N=c1nc2[nH]cnc2c(O)[nH]1",
+ "Nc1nc2[nH]cnc2c(=O)[nH]1",
+ "Nc1nc2nc[nH]c2c(=O)[nH]1",
+ "Nc1nc(O)c2[nH]cnc2n1",
+ }
+ assert enumerate_tautomer("Oc1n(C)ncc1") == {
+ "Cn1nccc1O",
+ "CN1N=CCC1=O",
+ "Cn1[nH]ccc1=O",
+ }
+ assert enumerate_tautomer("O=c1nc2[nH]ccn2cc1") == {
+ "O=c1ccn2cc[nH]c2n1",
+ "Oc1ccn2ccnc2n1",
+ "O=c1ccn2ccnc2[nH]1",
+ }
+ assert enumerate_tautomer("N=c1nc[nH]cc1") == {
+ "N=c1cc[nH]cn1",
+ "N=c1ccnc[nH]1",
+ "Nc1ccncn1",
+ }
+ assert enumerate_tautomer("N=c(c1)ccn2cc[nH]c12") == {
+ "N=c1ccn2cc[nH]c2c1",
+ "Nc1ccn2ccnc2c1",
+ }
+ assert enumerate_tautomer("CN=c1nc[nH]cc1") == {
+ "CN=c1ccnc[nH]1",
+ "CNc1ccncn1",
+ "CN=c1cc[nH]cn1",
+ }
+
+
+def test_1_3_1_5_aromatic_heteroatom_enumeration():
+ """1,3 and 1,5 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("Oc1ncncc1") == {
+ "Oc1ccncn1",
+ "O=c1ccnc[nH]1",
+ "O=c1cc[nH]cn1",
+ }
+
+
+def test_1_7_aromatic_heteroatom_enumeration():
+ """1,7 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1") == {
+ "c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1",
+ "c1ccc2c(c1)=NC(c1nc3ccccc3[nH]1)N=2",
+ "c1ccc2c(c1)NC(=C1N=c3ccccc3=N1)N2",
+ }
+ assert enumerate_tautomer("c1ccc2c(c1)NC(=C1N=c3ccccc3=N1)N2") == {
+ "c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1",
+ "c1ccc2c(c1)=NC(c1nc3ccccc3[nH]1)N=2",
+ "c1ccc2c(c1)NC(=C1N=c3ccccc3=N1)N2",
+ }
+
+
+def test_1_9_aromatic_heteroatom_enumeration():
+ """1,9 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("CNc1ccnc2ncnn21") == {
+ "CN=c1cc[nH]c2ncnn12",
+ "CN=c1ccnc2nc[nH]n12",
+ "CN=c1ccnc2[nH]cnn12",
+ "CNc1ccnc2ncnn12",
+ }
+ assert enumerate_tautomer("CN=c1ccnc2nc[nH]n21") == {
+ "CN=c1ccnc2nc[nH]n12",
+ "CN=c1cc[nH]c2ncnn12",
+ "CN=c1ccnc2[nH]cnn12",
+ "CNc1ccnc2ncnn12",
+ }
+
+
+def test_1_11_aromatic_heteroatom_enumeration():
+ """1,11 aromatic heteroatom H shift"""
+ assert enumerate_tautomer("Nc1ccc(C=C2C=CC(=O)C=C2)cc1") == {
+ "Nc1ccc(C=C2C=CC(=O)C=C2)cc1",
+ "N=C1C=CC(=CC2C=CC(=O)C=C2)C=C1",
+ "N=C1C=CC(=Cc2ccc(O)cc2)C=C1",
+ "N=C1C=CC(C=C2C=CC(=O)C=C2)C=C1",
+ }
+ assert enumerate_tautomer("N=C1C=CC(=Cc2ccc(O)cc2)C=C1") == {
+ "Nc1ccc(C=C2C=CC(=O)C=C2)cc1",
+ "N=C1C=CC(=CC2C=CC(=O)C=C2)C=C1",
+ "N=C1C=CC(=Cc2ccc(O)cc2)C=C1",
+ "N=C1C=CC(C=C2C=CC(=O)C=C2)C=C1",
+ }
+
+
+def test_heterocyclic_enumeration():
+ """heterocyclic tautomer"""
+ assert enumerate_tautomer("n1ccc2ccc[nH]c12") == {
+ "c1c[nH]c2nccc-2c1",
+ "c1cnc2[nH]ccc2c1",
+ }
+ assert enumerate_tautomer("c1cc(=O)[nH]c2nccn12") == {
+ "O=c1ccn2cc[nH]c2n1",
+ "Oc1ccn2ccnc2n1",
+ "O=c1ccn2ccnc2[nH]1",
+ }
+ assert enumerate_tautomer("c1cnc2c[nH]ccc12") == {
+ "c1cc2cc[nH]c2cn1",
+ "c1cc2cc[nH]cc-2n1",
+ }
+ assert enumerate_tautomer("n1ccc2c[nH]ccc12") == {
+ "c1cc2[nH]ccc2cn1",
+ "c1cc2c[nH]ccc-2n1",
+ }
+ assert enumerate_tautomer("c1cnc2ccc[nH]c12") == {
+ "c1c[nH]c2ccnc-2c1",
+ "c1cnc2cc[nH]c2c1",
+ }
+
+
+def test_furanone_enumeration():
+ """furanone tautomer"""
+ assert enumerate_tautomer("C1=CC=C(O1)O") == {"Oc1ccco1", "O=C1CC=CO1"}
+ assert enumerate_tautomer("O=C1CC=CO1") == {"Oc1ccco1", "O=C1CC=CO1"}
+
+
+def test_keten_ynol_enumeration():
+ """keten/ynol tautomer"""
+ assert enumerate_tautomer("CC=C=O") == {"CC=C=O", "CC#CO"}
+ assert enumerate_tautomer("CC#CO") == {"CC=C=O", "CC#CO"}
+
+
+def test_ionic_nitro_aci_nitro_enumeration():
+ """ionic nitro/aci-nitro tautomer"""
+ assert enumerate_tautomer("C([N+](=O)[O-])C") == {
+ "CC[N+](=O)[O-]",
+ "CC=[N+]([O-])O",
+ }
+ assert enumerate_tautomer("C(=[N+](O)[O-])C") == {
+ "CC[N+](=O)[O-]",
+ "CC=[N+]([O-])O",
+ }
+
+
+def test_oxim_nitroso_enumeration():
+ """oxim nitroso tautomer"""
+ assert enumerate_tautomer("CC(C)=NO") == {"CC(C)N=O", "CC(C)=NO", "C=C(C)NO"}
+ assert enumerate_tautomer("CC(C)N=O") == {"CC(C)N=O", "CC(C)=NO", "C=C(C)NO"}
+ assert enumerate_tautomer("O=Nc1ccc(O)cc1") == {
+ "O=NC1C=CC(=O)C=C1",
+ "O=C1C=CC(=NO)C=C1",
+ "O=Nc1ccc(O)cc1",
+ }
+ assert enumerate_tautomer("O=C1C=CC(=NO)C=C1") == {
+ "O=NC1C=CC(=O)C=C1",
+ "O=C1C=CC(=NO)C=C1",
+ "O=Nc1ccc(O)cc1",
+ }
+
+
+def test_cyano_iso_cyanic_acid_enumeration():
+ """cyano/iso-cyanic acid tautomer"""
+ assert enumerate_tautomer("C(#N)O") == {"N#CO", "N=C=O"}
+ assert enumerate_tautomer("C(=N)=O") == {"N#CO", "N=C=O"}
+
+
+def test_isocyanide_enumeration():
+ """isocyanide tautomer"""
+ assert enumerate_tautomer("C#N") == {"[C-]#[NH+]", "C#N"}
+ assert enumerate_tautomer("[C-]#[NH+]") == {"[C-]#[NH+]", "C#N"}
+
+
+def test_phosphonic_acid_enumeration():
+ """phosphonic acid tautomer"""
+ assert enumerate_tautomer("[PH](=O)(O)(O)") == {"OP(O)O", "O=[PH](O)O"}
+ assert enumerate_tautomer("P(O)(O)O") == {"OP(O)O", "O=[PH](O)O"}
+
+
+def test_mobile_double_stereochemistry_enumeration():
+ """Remove stereochemistry from mobile double bonds"""
+ assert enumerate_tautomer("c1(ccccc1)/C=C(/O)\\C") == {
+ "C=C(O)Cc1ccccc1",
+ "CC(O)=Cc1ccccc1",
+ "CC(=O)Cc1ccccc1",
+ }
+ assert enumerate_tautomer("C/C=C/C(C)=O") == {
+ "C=C(O)C=CC",
+ "C=CCC(=C)O",
+ "CC=CC(C)=O",
+ "C=CCC(C)=O",
+ "C=CC=C(C)O",
+ }
+ assert enumerate_tautomer("C/C=C\\C(C)=O") == {
+ "C=C(O)C=CC",
+ "C=CCC(=C)O",
+ "CC=CC(C)=O",
+ "C=CCC(C)=O",
+ "C=CC=C(C)O",
+ }
+
+
+def test_gaunine_enumeration():
+ """Gaunine tautomers"""
+ assert enumerate_tautomer("N1C(N)=NC=2N=CNC2C1=O") == {
+ "N=c1[nH]c(=O)c2[nH]cnc2[nH]1",
+ "N=c1[nH]c(=O)c2nc[nH]c2[nH]1",
+ "N=c1[nH]c2ncnc-2c(O)[nH]1",
+ "N=c1nc(O)c2[nH]cnc2[nH]1",
+ "N=c1nc(O)c2nc[nH]c2[nH]1",
+ "N=c1nc2[nH]cnc2c(O)[nH]1",
+ "N=c1nc2nc[nH]c2c(O)[nH]1",
+ "Nc1nc(=O)c2[nH]cnc2[nH]1",
+ "Nc1nc(=O)c2nc[nH]c2[nH]1",
+ "Nc1nc(O)c2[nH]cnc2n1",
+ "Nc1nc(O)c2nc[nH]c2n1",
+ "Nc1nc(O)c2ncnc-2[nH]1",
+ "Nc1nc2[nH]cnc2c(=O)[nH]1",
+ "Nc1nc2nc[nH]c2c(=O)[nH]1",
+ "Nc1nc2ncnc-2c(O)[nH]1",
+ }
+
+
+def test_many_enumeration():
+ """Test a structure with hundreds of tautomers."""
+ assert len(enumerate_tautomer("[H][C](CO)(NC(=O)C1=C(O)C(O)=CC=C1)C(O)=O")) == 375
+
+
+def test_1_3_keto_enol_canonicalization():
+ """1,3 keto/enol tautomer"""
+ assert canonicalize_tautomer("C1(=CCCCC1)O") == "O=C1CCCCC1"
+ assert canonicalize_tautomer("C1(CCCCC1)=O") == "O=C1CCCCC1"
+
+
+def test_acetophenone_keto_enol_canonicalization():
+ """Acetophenone keto/enol tautomer"""
+ assert canonicalize_tautomer("C(=C)(O)C1=CC=CC=C1") == "CC(=O)c1ccccc1"
+
+
+def test_acetone_keto_enol_canonicalization():
+ """Acetone keto/enol tautomer"""
+ assert canonicalize_tautomer("CC(C)=O") == "CC(C)=O"
+
+
+def test_keto_enol_canonicalization():
+ """keto/enol tautomer"""
+ assert canonicalize_tautomer("OC(C)=C(C)C") == "CC(=O)C(C)C"
+
+
+def test_phenylpropanone_keto_enol_canonicalization():
+ """1-phenyl-2-propanone enol/keto"""
+ assert canonicalize_tautomer("c1(ccccc1)CC(=O)C") == "CC(=O)Cc1ccccc1"
+
+
+def test_1_5_keto_enol_canonicalization():
+ """1,5 keto/enol tautomer"""
+ assert (
+ canonicalize_tautomer("Oc1nccc2cc[nH]c(=N)c12") == "N=c1[nH]ccc2cc[nH]c(=O)c12"
+ )
+ assert canonicalize_tautomer("C1(C=CCCC1)=O") == "O=C1C=CCCC1"
+ assert canonicalize_tautomer("C1(=CC=CCC1)O") == "O=C1C=CCCC1"
+
+
+def test_aliphatic_imine_canonicalization():
+ """aliphatic imine tautomer"""
+ assert canonicalize_tautomer("C1(CCCCC1)=N") == "N=C1CCCCC1"
+ assert canonicalize_tautomer("C1(=CCCCC1)N") == "N=C1CCCCC1"
+
+
+def test_special_imine_canonicalization():
+ """special imine tautomer"""
+ assert canonicalize_tautomer("C1(C=CC=CN1)=CC") == "CCc1ccccn1"
+ assert canonicalize_tautomer("C1(=NC=CC=C1)CC") == "CCc1ccccn1"
+
+
+def test_1_3_aromatic_heteroatom_canonicalization():
+ """1,3 aromatic heteroatom H shift"""
+ assert canonicalize_tautomer("O=c1cccc[nH]1") == "O=c1cccc[nH]1"
+ assert canonicalize_tautomer("Oc1ccccn1") == "O=c1cccc[nH]1"
+ assert canonicalize_tautomer("Oc1ncc[nH]1") == "O=c1[nH]cc[nH]1"
+
+
+def test_1_3_heteroatom_canonicalization():
+ """1,3 heteroatom H shift"""
+ assert canonicalize_tautomer("OC(C)=NC") == "CNC(C)=O"
+ assert canonicalize_tautomer("CNC(C)=O") == "CNC(C)=O"
+ assert canonicalize_tautomer("S=C(N)N") == "NC(N)=S"
+ assert canonicalize_tautomer("SC(N)=N") == "NC(N)=S"
+ assert canonicalize_tautomer("N=c1[nH]ccn(C)1") == "Cn1cc[nH]c1=N"
+ assert canonicalize_tautomer("CN=c1[nH]cncc1") == "CN=c1cc[nH]cn1"
+
+
+def test_1_5_aromatic_heteroatom_canonicalization():
+ """1,5 aromatic heteroatom H shift"""
+ assert canonicalize_tautomer("Oc1cccc2ccncc12") == "Oc1cccc2ccncc12"
+ assert canonicalize_tautomer("O=c1cccc2cc[nH]cc1-2") == "Oc1cccc2ccncc12"
+ assert canonicalize_tautomer("Cc1n[nH]c2ncnn12") == "Cc1n[nH]c2ncnn12"
+ assert canonicalize_tautomer("Cc1nnc2nc[nH]n12") == "Cc1n[nH]c2ncnn12"
+ assert canonicalize_tautomer("Oc1ccncc1") == "O=c1cc[nH]cc1"
+ assert (
+ canonicalize_tautomer("Oc1c(cccc3)c3nc2ccncc12") == "O=c1c2ccccc2[nH]c2ccncc12"
+ )
+ assert (
+ canonicalize_tautomer("C2(=C1C(=NC=N1)[NH]C(=N2)N)O")
+ == "N=c1[nH]c(=O)c2[nH]cnc2[nH]1"
+ )
+ assert (
+ canonicalize_tautomer("C2(C1=C([NH]C=N1)[NH]C(=N2)N)=O")
+ == "N=c1[nH]c(=O)c2[nH]cnc2[nH]1"
+ )
+ assert canonicalize_tautomer("Oc1n(C)ncc1") == "Cn1[nH]ccc1=O"
+ assert canonicalize_tautomer("O=c1nc2[nH]ccn2cc1") == "O=c1ccn2cc[nH]c2n1"
+ assert canonicalize_tautomer("N=c1nc[nH]cc1") == "N=c1cc[nH]cn1"
+ assert canonicalize_tautomer("N=c(c1)ccn2cc[nH]c12") == "N=c1ccn2cc[nH]c2c1"
+ assert canonicalize_tautomer("CN=c1nc[nH]cc1") == "CN=c1cc[nH]cn1"
+
+
+def test_1_3_1_5_aromatic_heteroatom_canonicalization():
+ """1,3 and 1,5 aromatic heteroatom H shift"""
+ assert canonicalize_tautomer("Oc1ncncc1") == "O=c1cc[nH]cn1"
+
+
+def test_1_7_aromatic_heteroatom_canonicalization():
+ """1,7 aromatic heteroatom H shift"""
+ assert (
+ canonicalize_tautomer("c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1")
+ == "c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1"
+ )
+ assert (
+ canonicalize_tautomer("c1ccc2c(c1)NC(=C1N=c3ccccc3=N1)N2")
+ == "c1ccc2[nH]c(-c3nc4ccccc4[nH]3)nc2c1"
+ )
+
+
+def test_1_9_aromatic_heteroatom_canonicalization():
+ """1,9 aromatic heteroatom H shift"""
+ assert canonicalize_tautomer("CNc1ccnc2ncnn21") == "CN=c1cc[nH]c2ncnn12"
+ assert canonicalize_tautomer("CN=c1ccnc2nc[nH]n21") == "CN=c1cc[nH]c2ncnn12"
+
+
+def test_1_11_aromatic_heteroatom_canonicalization():
+ """1,11 aromatic heteroatom H shift"""
+ assert (
+ canonicalize_tautomer("Nc1ccc(C=C2C=CC(=O)C=C2)cc1")
+ == "Nc1ccc(C=C2C=CC(=O)C=C2)cc1"
+ )
+ assert (
+ canonicalize_tautomer("N=C1C=CC(=Cc2ccc(O)cc2)C=C1")
+ == "Nc1ccc(C=C2C=CC(=O)C=C2)cc1"
+ )
+
+
+def test_heterocyclic_canonicalization():
+ """heterocyclic tautomer"""
+ assert canonicalize_tautomer("n1ccc2ccc[nH]c12") == "c1cnc2[nH]ccc2c1"
+ assert canonicalize_tautomer("c1cc(=O)[nH]c2nccn12") == "O=c1ccn2cc[nH]c2n1"
+ assert canonicalize_tautomer("c1cnc2c[nH]ccc12") == "c1cc2cc[nH]c2cn1"
+ assert canonicalize_tautomer("n1ccc2c[nH]ccc12") == "c1cc2[nH]ccc2cn1"
+ assert canonicalize_tautomer("c1cnc2ccc[nH]c12") == "c1cnc2cc[nH]c2c1"
+
+
+def test_furanone_canonicalization():
+ """furanone tautomer"""
+ assert canonicalize_tautomer("C1=CC=C(O1)O") == "Oc1ccco1"
+ assert canonicalize_tautomer("O=C1CC=CO1") == "Oc1ccco1"
+
+
+def test_keten_ynol_canonicalization():
+ """keten/ynol tautomer"""
+ assert canonicalize_tautomer("CC=C=O") == "CC=C=O"
+ assert canonicalize_tautomer("CC#CO") == "CC=C=O"
+
+
+def test_ionic_nitro_aci_nitro_canonicalization():
+ """ionic nitro/aci-nitro tautomer"""
+ assert canonicalize_tautomer("C([N+](=O)[O-])C") == "CC[N+](=O)[O-]"
+ assert canonicalize_tautomer("C(=[N+](O)[O-])C") == "CC[N+](=O)[O-]"
+
+
+def test_oxim_nitroso_canonicalization():
+ """oxim nitroso tautomer"""
+ assert canonicalize_tautomer("CC(C)=NO") == "CC(C)=NO"
+ assert canonicalize_tautomer("CC(C)N=O") == "CC(C)=NO"
+
+
+def test_oxim_nitroso_phenol_canonicalization():
+ """oxim/nitroso tautomer via phenol"""
+ assert canonicalize_tautomer("O=Nc1ccc(O)cc1") == "O=Nc1ccc(O)cc1"
+ assert canonicalize_tautomer("O=C1C=CC(=NO)C=C1") == "O=Nc1ccc(O)cc1"
+
+
+def test_cyano_iso_cyanic_acid_canonicalization():
+ """cyano/iso-cyanic acid tautomer"""
+ assert canonicalize_tautomer("C(#N)O") == "N=C=O"
+ assert canonicalize_tautomer("C(=N)=O") == "N=C=O"
+
+
+def test_formamidinesulfinic_acid_canonicalization():
+ """formamidinesulfinic acid tautomer"""
+ assert canonicalize_tautomer("N=C(N)S(=O)O") == "N=C(N)S(=O)O"
+
+
+def test_isocyanide_canonicalization():
+ """isocyanide tautomer"""
+ assert canonicalize_tautomer("C#N") == "C#N"
+ assert canonicalize_tautomer("[C-]#[NH+]") == "C#N"
+
+
+def test_phosphonic_acid_canonicalization():
+ """phosphonic acid tautomer"""
+ assert canonicalize_tautomer("[PH](=O)(O)(O)") == "O=[PH](O)O"
+ assert canonicalize_tautomer("P(O)(O)O") == "O=[PH](O)O"
diff --git a/opencadd/tests/compounds/standardization/test_normalize_molecules.py b/opencadd/tests/compounds/standardization/test_normalize_molecules.py
new file mode 100644
index 00000000..e9bfc867
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_normalize_molecules.py
@@ -0,0 +1,64 @@
+"""
+test for the module `normalize_molecules`
+
+derived from MolVS's tests: https://github.com/mcs07/MolVS/blob/master/tests/test_normalize.py
+"""
+import pytest
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import normalize_molecules
+
+
+def normalization_for_smiles(smiles):
+ """Does normalization after converting a SMILES string into a mol."""
+ mol = Chem.MolFromSmiles(smiles, sanitize=False)
+ mol = normalize_molecules.normalize(mol)
+ if mol:
+ return Chem.MolToSmiles(mol, isomericSmiles=True)
+
+
+def test_nitro():
+ """Test nitro group normalozation."""
+ assert normalization_for_smiles("CN(=O)=O") == "C[N+](=O)[O-]"
+
+
+def test_sulfoxide():
+ """Test sulfoxide normalization."""
+ assert normalization_for_smiles("CS(C)=O") == "C[S+](C)[O-]"
+
+
+def test_sulfone():
+ """Test sulfone normalization."""
+ assert normalization_for_smiles("C[S+2]([O-])([O-])O") == "CS(=O)(=O)O"
+
+
+def test_1_3_charge_recombination():
+ """Test 1,3-seperated charges are recombined"""
+ assert normalization_for_smiles("CC([O-])=[N+](C)C") == "CC(=O)N(C)C"
+
+
+def test_1_3_charge_recombination_aromatic():
+ """Test 1,3-separated charges are recombined."""
+ assert normalization_for_smiles("C[n+]1ccccc1[O-]") == "Cn1ccccc1=O"
+
+
+def test_1_3_charge_recombination_exception():
+ """Test a case where 1,3-separated charges should not be recombined."""
+ assert (
+ normalization_for_smiles("CC12CCCCC1(Cl)[N+]([O-])=[N+]2[O-]")
+ == "CC12CCCCC1(Cl)[N+]([O-])=[N+]2[O-]"
+ )
+
+
+def test_1_5_charge_recombination():
+ """Test 1,5-separated charges are recombined."""
+ assert normalization_for_smiles("C[N+](C)=C\\C=C\\[O-]") == "CN(C)C=CC=O"
+
+
+def test_1_5_charge_recombination_exception():
+ """Test a case where 1,5-separated charges should not be recombined."""
+ assert (
+ normalization_for_smiles("C[N+]1=C2C=[N+]([O-])C=CN2CCC1")
+ == "C[N+]1=C2C=[N+]([O-])C=CN2CCC1"
+ )
diff --git a/opencadd/tests/compounds/standardization/test_remove_salts.py b/opencadd/tests/compounds/standardization/test_remove_salts.py
new file mode 100644
index 00000000..f7fd4d57
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_remove_salts.py
@@ -0,0 +1,71 @@
+"""
+test for the module `remove_salts`
+"""
+import pytest
+import sys
+import rdkit
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization.remove_salts import remove_salts
+
+
+def _evaluation_mol_generator(test_smiles=None, test_inchi=None):
+ """Creates mol files directly with rdkits functions for evaluation."""
+ if test_smiles is not None:
+ return Chem.MolFromSmiles(test_smiles)
+ if test_inchi is not None:
+ return Chem.MolFromInchi(test_inchi)
+
+
+def _molecule_test(test_inchi=None, test_smiles=None):
+ return Chem.MolToInchi(
+ remove_salts(
+ _evaluation_mol_generator(test_inchi=test_inchi, test_smiles=test_smiles)
+ )
+ )
+
+
+def test_structure():
+ """Only C(C(=O)[O-])(Cc1n[n-]nn1)(C[NH3+])(C[N+](=O)[O-] should be
+ left after stripping salts.
+ """
+ assert (
+ _molecule_test(
+ test_smiles="C(C(=O)[O-])(Cc1n[n-]nn1)(C[NH3+])(C[N+](=O)[O-].CCCCCCCCCCCCCCCCCC(=O)O.OCC(O)C1OC(=O)C(=C1O)O)"
+ )
+ == "InChI=1S/C6H10N6O4/c7-2-6(5(13)14,3-12(15)16)1-4-8-10-11-9-4/h1-3,7H2,(H2,8,9,10,11,13,14)/p-1"
+ )
+
+
+def test_single_salts():
+ """All salt fragments should be detected and stripped."""
+ assert (
+ _molecule_test(
+ test_smiles="[Al].N.[Ba].[Bi].Br.[Ca].Cl.F.I.[K].[Li].[Mg].[Na].[Ag].[Sr].S.O.[Zn]"
+ )
+ == ""
+ )
+
+
+def test_complex_salts():
+ """Complex salts, contained in salts.tsv should be detected."""
+ assert (
+ _molecule_test(test_smiles="OC(C(O)C(=O)O)C(=O)O.O=C1NS(=O)(=O)c2ccccc12") == ""
+ )
+
+
+def test_custom_dictionary():
+ """Configuration of a custom dictionary, by defining one, should
+ work.
+ """
+ assert (
+ Chem.MolToInchi(
+ remove_salts(
+ _evaluation_mol_generator(test_smiles="[Al].N.[Ba].[Bi]"),
+ dictionary=False,
+ defnData="[Al]",
+ )
+ )
+ == "InChI=1S/Ba.Bi.H3N.2H/h;;1H3;;"
+ )
diff --git a/opencadd/tests/compounds/standardization/test_standardizer.py b/opencadd/tests/compounds/standardization/test_standardizer.py
new file mode 100644
index 00000000..eeb6310a
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_standardizer.py
@@ -0,0 +1,14 @@
+"""
+Unit and regression test for the standardization package.
+"""
+
+# Import package, test suite, and other packages as needed
+
+import sys
+import pytest
+import opencadd.compounds.standardization
+
+
+def test_standardization_imported():
+ """Sample test, will always pass so long as import statement worked"""
+ assert "opencadd" in sys.modules
diff --git a/opencadd/tests/compounds/standardization/test_validate_molecules.py b/opencadd/tests/compounds/standardization/test_validate_molecules.py
new file mode 100644
index 00000000..0b6b5fcc
--- /dev/null
+++ b/opencadd/tests/compounds/standardization/test_validate_molecules.py
@@ -0,0 +1,93 @@
+"""
+test for the module `validate_molecules`
+
+Partially derived from MolVS's tests and RDKIT MolStandardize tutorial:
+https://github.com/mcs07/MolVS/blob/master/tests/test_validate.py
+https://github.com/susanhleung/rdkit/blob/dev/GSOC2018_MolVS_Integration/rdkit/Chem/MolStandardize/tutorial/MolStandardize.ipynb
+"""
+import pytest
+
+from rdkit import Chem
+
+from opencadd.compounds.standardization import validate_molecules
+
+
+def _evaluation_mol_generator(test_smiles=None, test_inchi=None):
+ """Creates mol files directly with rdkits functions for evaluation."""
+ if test_smiles is not None:
+ return Chem.MolFromSmiles(test_smiles, sanitize=False)
+ if test_inchi is not None:
+ return Chem.MolFromInchi(test_inchi)
+
+
+def test_no_atom():
+ """NoAtomValidation should log due to the lack of any atoms."""
+ assert validate_molecules.validate_default(
+ _evaluation_mol_generator(test_smiles="")
+ ) == ["ERROR: [NoAtomValidation] Molecule has no atoms"]
+
+
+def test_fragment_dichloroethane():
+ """FragmentValidation should identify 1,2-dichloroethane."""
+ assert validate_molecules.validate_fragment(
+ _evaluation_mol_generator(test_smiles="ClCCCl.c1ccccc1O")
+ ) == ["INFO: [FragmentValidation] 1,2-dichloroethane is present"]
+
+
+def test_fragment_dimethoxyethane():
+ """FragmentValidation should identify 1,2-dimethoxyethane."""
+ assert validate_molecules.validate_fragment(
+ _evaluation_mol_generator(test_smiles="COCCOC.CCCBr")
+ ) == ["INFO: [FragmentValidation] 1,2-dimethoxyethane is present"]
+
+
+def test_neutrality():
+ """NeutralValidation should identify net overall charge."""
+ assert validate_molecules.validate_neutrality(
+ _evaluation_mol_generator(test_smiles="O=C([O-])c1ccccc1")
+ ) == ["INFO: [NeutralValidation] Not an overall neutral system (-1)"]
+ assert validate_molecules.validate_neutrality(
+ _evaluation_mol_generator(test_smiles="CN=[NH+]CN=N")
+ ) == ["INFO: [NeutralValidation] Not an overall neutral system (+1)"]
+
+
+def test_isotope():
+ """IsotopeValidation should identify atoms with isotope labels."""
+ assert validate_molecules.validate_isotopes(
+ _evaluation_mol_generator(test_smiles="[13CH4]")
+ ) == ["INFO: [IsotopeValidation] Molecule contains isotope 13C"]
+ assert validate_molecules.validate_isotopes(
+ _evaluation_mol_generator(test_smiles="[2H]C(Cl)(Cl)Cl")
+ ) == ["INFO: [IsotopeValidation] Molecule contains isotope 2H"]
+ assert validate_molecules.validate_isotopes(
+ _evaluation_mol_generator(test_smiles="[2H]OC([2H])([2H])[2H]")
+ ) == ["INFO: [IsotopeValidation] Molecule contains isotope 2H"]
+
+
+def test_valency():
+ """check_valency should validate the valency of every atom in the
+ input molecule.
+ """
+ assert validate_molecules.check_valency(
+ _evaluation_mol_generator(test_smiles="CO(C)C")
+ ) == [
+ "INFO: [ValenceValidation] Explicit valence for atom # 1 O, 3, is greater than permitted"
+ ]
+
+
+def test_validate_allowed_atoms():
+ """validate_allowed_atoms should accept as input a list of atoms,
+ anything not on the list should throw an error.
+ """
+ assert validate_molecules.validate_allowed_atoms(
+ _evaluation_mol_generator(test_smiles="CC(=O)CF"), atomlist=[6, 7, 8]
+ ) == ["INFO: [AllowedAtomsValidation] Atom F is not in allowedAtoms list"]
+
+
+def test_validate_disallowed_atoms():
+ """validate_allowed_atoms should accept as input a list of atoms,
+ anything not on the list should throw an error.
+ """
+ assert validate_molecules.validate_disallowed_atoms(
+ _evaluation_mol_generator(test_smiles="CC(=O)CF"), atomlist=[9, 17, 35]
+ ) == ["INFO: [DisallowedAtomsValidation] Atom F is in disallowedAtoms list"]
diff --git a/opencadd/utils.py b/opencadd/utils.py
index d9eae239..88637d6b 100644
--- a/opencadd/utils.py
+++ b/opencadd/utils.py
@@ -7,6 +7,8 @@
import shutil
import tempfile
import contextlib
+from pathlib import Path
+
_logger = logging.getLogger(__name__)
@@ -66,3 +68,18 @@ class EmojiPerLevelFormatter(PerLevelFormatter):
101: "%(message)s",
25: "☑️ %(message)s",
}
+
+
+def data_path(fn):
+ """Leads to files saved in the data folder
+
+ Parameters
+ ----------
+ fn: str
+ The whole filename.
+
+ Returns
+ -------
+ The path of the file in the current working system.
+ """
+ return Path(__file__).parent / "data" / fn