OSCR

Design, preclinical evaluation, and multicenter phase 1 clinical study of HZ-A-018 for relapsed or refractory central nervous system lymphoma.

Code ↔ Paper

1 match between paragraphs of the paper and lines of its authors' code, computed by the harvester (lexical-v1). Click a colored paragraph or line to see its counterpart.

The 1 match
  1. [1] § Materials and methods › BTK inhibition activity and BBB permeability prediction model ↔ run.py, lines 343–432 · score 0.90 · search spaces, 0.05–0.1, attn_layers, output_dim, batch_size, Hyperopt

Paper

Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC

The paper is loaded when this pane is shown.

The authors' code

Python · 432 lines · 22 KB · MIT · 1 match

  1. import torch, argparse
  2. import torch.nn as nn
  3. import torch.optim as optim
  4. import numpy as np
  5. import pandas as pd
  6. from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, mean_absolute_error, mean_squared_error, r2_score
  7. from torch.nn import BCELoss
  8. from torch_geometric.data import DataLoader
  9. from TFM.Dataset import MolNet
  10. from TFM.model import Kno
  11. from TFM.utils import get_logger, metrics_c, metrics_r, set_seed, load_data
  12. from rdkit.Chem.SaltRemover import SaltRemover
  13. import hyperopt
  14. from hyperopt import fmin, hp, Trials
  15. from hyperopt.early_stop import no_progress_loss
  16. import warnings
  17. from datetime import datetime
  18. warnings.filterwarnings("ignore")
  19. remover = SaltRemover()
  20. bad = ['He', 'Be', 'Na', 'Mg', 'Al', 'K', 'Ca', 'Ti', 'V', 'Cr', 'Mn', 'Fe', 'Co', 'Ni', 'Cu', 'Zn', 'Ga', 'Ge', 'As', 'Se', 'Rb', 'Sr', 'Mo', 'Tc', 'Ru', 'Rh', 'Pd', 'Ag', 'Cd', 'In', 'Sn', 'Sb', 'Te', 'Gd', 'Tb', 'Ho', 'W', 'Ir', 'Pt', 'Au', 'Hg', 'Tl', 'Pb', 'Bi', 'Ac']
  21. _use_shared_memory = True
  22. torch.backends.cudnn.benchmark = True
  23. def training(model, train_loader, optimizer, loss_f, metric, task, device, mean, stds):
  24. loss_record, record_count = 0., 0.
  25. preds = torch.Tensor([]); tars = torch.Tensor([])
  26. model.train()
  27. if task == 'clas':
  28. for data in train_loader:
  29. if data.y.size()[0] > 1:
  30. y = data.y.to(device)
  31. logits = model(data)
  32. loss = loss_f(logits.squeeze(), y.squeeze())
  33. loss_record += float(loss.item())
  34. record_count += 1
  35. optimizer.zero_grad()
  36. loss.backward()
  37. nn.utils.clip_grad_value_(model.parameters(), clip_value=2)
  38. optimizer.step()
  39. pred = logits.detach().cpu()
  40. preds = torch.cat([preds, pred], 0); tars = torch.cat([tars, y.cpu()], 0)
  41. clas = preds > 0.5
  42. acc, f1, pre, rec, auc = metric(clas.squeeze().numpy(), preds.squeeze().numpy(), tars.squeeze().numpy())
  43. else:
  44. for data in train_loader:
  45. if data.y.size()[0] > 1:
  46. y = data.y.to(device)
  47. y_ = (y - mean) / (stds+1e-5)
  48. logits = model(data)
  49. loss = loss_f(logits.squeeze(), y_.squeeze())
  50. loss_record += float(loss.item())
  51. record_count += 1
  52. optimizer.zero_grad()
  53. loss.backward()
  54. nn.utils.clip_grad_value_(model.parameters(), clip_value=2)
  55. optimizer.step()
  56. pred = logits.detach()*stds+mean
  57. preds = torch.cat([preds, pred.cpu()], 0); tars = torch.cat([tars, y.cpu()], 0)
  58. acc, f1, pre, rec, auc = metric(preds.squeeze().numpy(), tars.squeeze().numpy())
  59. epoch_loss = loss_record / record_count
  60. return epoch_loss, acc, f1, pre, rec, auc
  61. def testing(model, test_loader, loss_f, metric, task, device, mean, stds, resu):
  62. loss_record, record_count = 0., 0.
  63. preds = torch.Tensor([]); tars = torch.Tensor([])
  64. model.eval()
  65. with torch.no_grad():
  66. if task == 'clas':
  67. for data in test_loader:
  68. if data.y.size()[0] > 1:
  69. y = data.y.to(device)
  70. logits = model(data)
  71. loss = loss_f(logits.squeeze(), y.squeeze())
  72. loss_record += float(loss.item())
  73. record_count += 1
  74. pred = logits.detach().cpu()
  75. preds = torch.cat([preds, pred], 0); tars = torch.cat([tars, y.cpu()], 0)
  76. preds, tars = preds.squeeze().numpy(), tars.squeeze().numpy()
  77. clas = preds > 0.5
  78. acc, f1, pre, rec, auc = metric(clas, preds, tars)
  79. else:
  80. for data in test_loader:
  81. if data.y.size()[0] > 1:
  82. y = data.y.to(device)
  83. y_ = (y - mean) / (stds+1e-5)
  84. logits = model(data)
  85. loss = loss_f(logits.squeeze(), y_.squeeze())
  86. loss_record += float(loss.item())
  87. record_count += 1
  88. pred = logits.detach()*stds+mean
  89. preds = torch.cat([preds, pred.cpu()], 0); tars = torch.cat([tars, y.cpu()], 0)
  90. preds, tars = preds.squeeze().numpy(), tars.squeeze().numpy()
  91. acc, f1, pre, rec, auc = metric(preds, tars)
  92. epoch_loss = loss_record / record_count
  93. if resu:
  94. return epoch_loss, acc, f1, pre, rec, auc, preds, tars
  95. else:
  96. return epoch_loss, acc, f1, pre, rec, auc
  97. def main(tasks, task, dataset, device, train_epoch, seed, fold, batch_size, rate, split, modelpath, logger, lr, attn_head, output_dim, attn_layers, dropout, mean, stds, D, useedge, met, savem):
  98. logger.info('Dataset: {} task: {} train_epoch: {}'.format(dataset, task, train_epoch))
  99. d_k, seed_ = round(output_dim/attn_head), seed
  100. fold_result = [[], []]
  101. if task == 'clas':
  102. loss_f = BCELoss().to(device)
  103. metric = metrics_c(accuracy_score, precision_score, recall_score, f1_score, roc_auc_score)
  104. for fol in range(1, fold+1):
  105. best_val_auc, best_test_auc = 0., 0.
  106. if seed is not None:
  107. seed_ = seed + fol-1
  108. set_seed(seed_)
  109. model = Kno(task, tasks, attn_head, output_dim, d_k, attn_layers, D, dropout, useedge, device).to(device)
  110. optimizer = optim.AdamW(model.parameters(), lr=lr, weight_decay=0.1)
  111. scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=50, eta_min=1e-6, last_epoch=-1)
  112. train_loader, valid_loader, test_loader = load_data(dataset, batch_size, rate[0], rate[1], 0, task, split, seed_)
  113. logger.info('Dataset: {} Fold: {:<4d}'.format(moldata, fol))
  114. for i in range(1,train_epoch+1):
  115. train_loss, train_acc, train_f1, train_pre, train_rec, train_auc = training(model, train_loader, optimizer, loss_f, metric, task, device, mean, stds)
  116. if sche:
  117. scheduler.step()
  118. valid_loss, valid_acc, valid_f1, valid_pre, valid_rec, valid_auc = testing(model, valid_loader, loss_f, metric, task, device, mean, stds, False)
  119. logger.info('Dataset: {} Epoch: {:<3d} train_loss: {:.4f} train_auc: {:.4f}'.format(dataset ,i, train_loss, train_auc))
  120. logger.info('Dataset: {} Epoch: {:<3d} valid_loss: {:.4f} valid_auc: {:.4f}'.format(dataset, i, valid_loss, valid_auc))
  121. if valid_auc > best_val_auc:
  122. best_val_auc = valid_auc
  123. if savem:
  124. model_save_path = modelpath + '{}_{}_{}.pkl'.format(dataset, i, round(valid_auc, 4))
  125. torch.save(model.state_dict(), model_save_path)
  126. test_loss, test_acc, test_f1, test_pre, test_rec, test_auc = testing(model, test_loader, loss_f, metric, task, device, mean, stds, False)
  127. logger.info('Dataset: {} Epoch: {:<3d} test__loss: {:.4f} test__auc: {:.4f}'.format(dataset, i, test_loss, test_auc))
  128. best_test_auc = test_auc
  129. fold_result[0].append(best_val_auc)
  130. fold_result[1].append(best_test_auc)
  131. logger.info('Dataset: {} Fold: {} best_val_auc: {:.4f} best_test_auc: {:.4f}'.format(dataset, fol, best_val_auc, best_test_auc))
  132. logger.info('Dataset: {} Fold result: {}'.format(dataset, fold_result))
  133. return fold_result
  134. else:
  135. if met == 'mae':
  136. loss_f = nn.L1Loss().to(device)
  137. else:
  138. loss_f = nn.MSELoss().to(device)
  139. metric = metrics_r(mean_absolute_error, mean_squared_error, r2_score)
  140. for fol in range(1, fold+1):
  141. best_val_rmse, best_test_rmse = 9999., 9999.
  142. if seed is not None:
  143. seed_ = seed + fol-1
  144. set_seed(seed_)
  145. model = Kno(task, tasks, attn_head, output_dim, d_k, attn_layers, D, dropout, useedge, device).to(device)
  146. optimizer = optim.AdamW(model.parameters(), lr=lr, weight_decay=0.1)
  147. scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=50, eta_min=1e-6, last_epoch=-1)
  148. train_loader, valid_loader, test_loader = load_data(dataset, batch_size, rate[0], rate[1], 0, task, split, seed_)
  149. logger.info('Dataset: {} Fold: {:<4d}'.format(moldata, fol))
  150. for i in range(1, train_epoch+1):
  151. train_loss, train_mae, train_rmse, train_r2, _, _ = training(model, train_loader, optimizer, loss_f, metric, task, device, mean, stds)
  152. if sche:
  153. scheduler.step()
  154. valid_loss, valid_mae, valid_rmse, valid_r2, _, _ = testing(model, valid_loader, loss_f, metric, task, device, mean, stds, False)
  155. logger.info('Dataset: {} Epoch: {:<3d} train_loss: {:.4f} train_mae: {:.4f} train_rmse: {:.4f}'.format(dataset, i, train_loss, train_mae, train_rmse))
  156. logger.info('Dataset: {} Epoch: {:<3d} valid_loss: {:.4f} valid_mae: {:.4f} valid_rmse: {:.4f}'.format(dataset, i, valid_loss, valid_mae, valid_rmse))
  157. if met == 'rmse':
  158. if valid_rmse < best_val_rmse:
  159. best_val_rmse = valid_rmse
  160. if savem:
  161. model_save_path = modelpath + '{}_{}_{}.pkl'.format(dataset, i, round(valid_rmse,4))
  162. torch.save(model.state_dict(), model_save_path)
  163. test_loss, test_mae, test_rmse, test_r2, _, _ = testing(model, test_loader, loss_f, metric, task, device, mean, stds, False)
  164. logger.info('Dataset: {} Epoch: {:<3d} test_loss: {:.4f} test_rmse: {:.4f}'.format(dataset, i, test_loss, test_rmse))
  165. best_test_rmse = test_rmse
  166. elif met == 'mae':
  167. if valid_mae < best_val_rmse:
  168. best_val_rmse = valid_mae
  169. if savem:
  170. model_save_path = modelpath + '{}_{}_{}.pkl'.format(dataset, i, round(valid_rmse,4))
  171. torch.save(model.state_dict(), model_save_path)
  172. test_loss, test_mae, test_rmse, test_r2, _, _ = testing(model, test_loader, loss_f, metric, task, device, mean, stds, False)
  173. logger.info('Dataset: {} Epoch: {:<3d} test_loss: {:.4f} test_mae: {:.4f}'.format(dataset, i, test_loss, test_mae))
  174. best_test_rmse = test_mae
  175. else:
  176. raise ValueError('regression metric must be rmse or mae')
  177. fold_result[0].append(best_val_rmse)
  178. fold_result[1].append(best_test_rmse)
  179. logger.info('Dataset: {} Fold: {} best_val_{}: {:.4f} best_test_{}: {:.4f}'.format(dataset, fol, met, best_val_rmse, met, best_test_rmse))
  180. logger.info('Dataset: {} Fold result: {}'.format(dataset, fold_result))
  181. return fold_result
  182. def test(tasks, task, dataset, device, seed, batch_size, logger, attn_head, output_dim, attn_layers, dropout, pretrain, mean, stds, D, useedge, met):
  183. logger.info('Dataset: {} task: {} testing:'.format(dataset, task))
  184. d_k = round(output_dim/attn_head)
  185. if seed is not None:
  186. set_seed(seed)
  187. model = Kno(task, tasks, attn_head, output_dim, d_k, attn_layers, D, dropout, useedge, device).to(device)
  188. state_dict = torch.load(pretrain)
  189. model.load_state_dict(state_dict)
  190. data = MolNet(root='./dataset', dataset=dataset)
  191. loader = DataLoader(data, batch_size=batch_size, shuffle=False, pin_memory=True, num_workers=0, drop_last=False)
  192. if task == 'clas':
  193. loss_f = BCELoss().to(device)
  194. metric = metrics_c(accuracy_score, precision_score, recall_score, f1_score, roc_auc_score)
  195. loss, acc, f1, pre, rec, auc, preds, tars = testing(model, loader, loss_f, metric, task, device, mean, stds, True)
  196. logger.info('Dataset: {} test_loss: {:.4f} test_acc: {:.4f} test_f1: {:.4f} test_auc: {:.4f} test_pre: {:.4f} test_rec: {:.4f}'.format(dataset, loss, acc, f1, auc, pre, rec))
  197. results = {
  198. 'test_loss': loss,
  199. 'test_acc': acc,
  200. 'test_f1': f1,
  201. 'test_pre': pre,
  202. 'test_rec': rec,
  203. 'test_auc': auc,
  204. }
  205. df_prediction = pd.DataFrame({'prediction': preds})
  206. df_target = pd.DataFrame({'target': tars})
  207. df_single_values = pd.DataFrame({k: [v] for k, v in results.items()})
  208. for col in df_single_values.columns:
  209. df_single_values[col] = df_single_values[col].reindex(df_prediction.index, method='ffill')
  210. df = pd.concat([df_single_values, df_prediction, df_target], axis=1)
  211. df.to_csv('log/Result'+moldata+'_test.csv', index=False)
  212. else:
  213. if met == 'mae':
  214. loss_f = nn.L1Loss().to(device)
  215. else:
  216. loss_f = nn.MSELoss().to(device)
  217. metric = metrics_r(mean_absolute_error, mean_squared_error, r2_score)
  218. loss, mae, rmse, r2, _, _, preds, tars= testing(model, loader, loss_f, metric, task, device, mean, stds, True)
  219. logger.info('Dataset: {} test_loss: {:.4f} test_mae: {:.4f} test_rmse: {:.4f} test_r2: {:.4f}'.format(dataset, loss, mae, rmse, r2))
  220. results = {
  221. 'test_loss': loss,
  222. 'test_mae': mae,
  223. 'test_rmse': rmse,
  224. 'test_r2': r2,
  225. }
  226. df_prediction = pd.DataFrame({'prediction': preds})
  227. df_target = pd.DataFrame({'target': tars})
  228. df_single_values = pd.DataFrame({k: [v] for k, v in results.items()})
  229. for col in df_single_values.columns:
  230. df_single_values[col] = df_single_values[col].reindex(df_prediction.index, method='ffill')
  231. df = pd.concat([df_single_values, df_prediction, df_target], axis=1)
  232. df.to_csv('log/Result'+moldata+'_test.csv', index=False)
  233. def psearch(params):
  234. logger.info('Optimizing Hyperparameters')
  235. fold_result = main(params['tasks'],params['task'],params['moldata'],params['device'],params['train_epoch'],params['seed'],params['fold'],params['batch_size'],params['rate'],params['split'],params['modelpath'],params['logger'],params['lr'],params['attn_head'],params['output_dim'],params['attn_layers'],params['dropout'],params['mean'], params['std'], params['D'], params['useedge'], params['metric'], False)
  236. if task == 'reg':
  237. valid_res = np.mean(fold_result[1])
  238. else:
  239. valid_res = -np.mean(fold_result[1])
  240. return valid_res
  241. if __name__ == '__main__':
  242. parser = argparse.ArgumentParser(description='TransFoxMol')
  243. parser.add_argument('mode', type=str, choices=['train', 'test', 'search'], help='train, test or hyperparameter_search')
  244. parser.add_argument('moldata', type=str, help='Dataset name')
  245. parser.add_argument('--task', type=str, choices=['clas', 'reg'], help='Classification or Regression')
  246. parser.add_argument('--numtasks', type=int, default=1, help='Number of tasks (default: 1).')
  247. parser.add_argument('--device', type=str, default='cuda:0', help='Which gpu to use if any (default: cuda:0)')
  248. parser.add_argument('--batch_size', type=int, default=32, help='Input batch size for training (default: 32)')
  249. parser.add_argument('--train_epoch', type=int, default=50, help='Number of epochs to train (default: 50)')
  250. parser.add_argument('--max_eval', type=int, default=100, help='Number hyperparameter settings to try (default: 100)')
  251. parser.add_argument('--lr', type=float, default=0.001, help='learning rate')
  252. parser.add_argument('--valrate', type=float, default=0.1, help='valid rate (default: 0.1)')
  253. parser.add_argument('--testrate', type=float, default=0.1, help='test rate (default: 0.1)')
  254. parser.add_argument('--fold', type=int, default=3, help='Number of folds for cross validation (default: 3)')
  255. parser.add_argument('--dropout', type=float, default=0.05, help='dropout ratio')
  256. parser.add_argument('--split', type =str, default='random_scaffold', help = 'random_scaffold/balan_scaffold/random (default: random_scaffold)')
  257. parser.add_argument('--attn_layers', type=int, default=2, help='Number of feature learning layers')
  258. parser.add_argument('--output_dim', type=int, default=256, help='Hidden size of embedding layer')
  259. parser.add_argument('--D', type=int, default=4, help='Hidden size of readout layer')
  260. parser.add_argument('--seed', type=int, help = "Seed for splitting the dataset")
  261. parser.add_argument('--pretrain', type=str, help = "Path of retrained weights")
  262. parser.add_argument('--metric', type=str, choices=['rmse', 'mae'], help='Metric to evaluate the regression performance')
  263. args = parser.parse_args()
  264. device = torch.device(args.device)
  265. moldata = args.moldata
  266. attn_head = 10
  267. max_eval = args.max_eval
  268. rate = [args.valrate, args.testrate]
  269. useedge = False
  270. sche = True
  271. if moldata in ['esol', 'freesolv', 'lipo', 'qm7', 'qm8', 'qm9']:
  272. task = 'reg'
  273. if moldata == 'qm8':
  274. numtasks = 12
  275. elif moldata == 'qm9':
  276. numtasks = 3
  277. else:
  278. numtasks = 1
  279. elif moldata in ['bbbp', 'sider', 'clintox', 'tox21', 'toxcast', 'bace', 'pcba', 'muv', 'hiv']:
  280. task = 'clas'
  281. if moldata == 'sider':
  282. numtasks = 27
  283. elif moldata == 'clintox':
  284. numtasks = 2
  285. useedge = True
  286. elif moldata == 'tox21':
  287. numtasks = 12
  288. elif moldata == 'toxcast':
  289. numtasks = 617
  290. elif moldata == 'pcba':
  291. numtasks = 128
  292. elif moldata == 'muv':
  293. numtasks = 17
  294. else:
  295. numtasks = 1
  296. else:
  297. task = args.task
  298. numtasks = args.numtasks
  299. logf = 'log/{}_{}_{}_{}.log'.format(moldata, args.task, args.split, args.mode)
  300. modelpath = 'log/checkpoint/'
  301. logger = get_logger(logf)
  302. logger.info("Arguments:")
  303. for arg in vars(args):
  304. logger.info(f"{arg}: {getattr(args, arg)}")
  305. moldata += task
  306. try:
  307. data = MolNet(root='./dataset', dataset=moldata)
  308. except:
  309. raise ValueError('Process the dataset first!')
  310. length = len(data)
  311. if task == 'clas':
  312. mean, std = None, None
  313. else:
  314. max_eval = 50
  315. if numtasks > 1:
  316. ys = np.asarray([d.y.numpy() for d in data])
  317. mean, std = np.mean(ys, 0), np.std(ys, 0)
  318. mean, stds = torch.FloatTensor(mean).to(device), torch.FloatTensor(std).to(device)
  319. else:
  320. ys = np.asarray([d.y.item() for d in data])
  321. mean, std = np.mean(ys, 0), np.std(ys, 0)
  322. dps = [0.05, 0.1]
  323. if args.mode == 'search':
  324. trials = Trials()
  325. if length < 500:
  326. batch_size = 8
  327. attn_head = 8
  328. elif length < 5000:
  329. batch_size = 32
  330. else:
  331. batch_size = 256
  332. max_eval = 50
  333. if task == 'clas':
  334. dps = [0.1, 0.2, 0.3]
  335. if args.moldata == 'bbbp':
  336. lrs = [1e-4, 5e-5, 1e-5]
  337. sche = False
  338. else:
  339. lrs = [1e-2, 5e-3, 1e-3]
  340. parm_space = { # search space of param
  341. 'tasks': numtasks,
  342. 'task': task,
  343. 'moldata': moldata,
  344. 'mean': mean,
  345. 'std': std,
  346. 'device': args.device,
  347. 'modelpath': modelpath,
  348. 'logger': logger,
  349. 'useedge': useedge,
  350. 'seed': args.seed,
  351. 'fold': args.fold,
  352. 'metric': args.metric,
  353. 'rate': rate,
  354. 'split': args.split,
  355. 'train_epoch': args.train_epoch,
  356. 'attn_head': attn_head,
  357. 'output_dim': hp.choice('output_dim', [128, 256]),
  358. 'attn_layers': hp.choice('attn_layers', [1, 2, 3, 4]),
  359. 'dropout': hp.choice('dropout', dps),
  360. 'lr': hp.choice('lr', lrs),
  361. 'D': hp.choice('D', [2, 4, 6, 8, 12, 16]),
  362. 'batch_size': batch_size
  363. }
  364. param_mappings = {
  365. 'output_dim': [128, 256],
  366. 'attn_layers': [1, 2, 3, 4],
  367. 'dropout': dps,
  368. 'lr': lrs,
  369. 'D': [2, 4, 6, 8, 12, 16]
  370. }
  371. best = fmin(fn=psearch, space=parm_space, algo=hyperopt.tpe.suggest, max_evals=max_eval, trials=trials, early_stop_fn=no_progress_loss(int(max_eval/2)))
  372. best_values = {k: param_mappings[k][v] if k in param_mappings else v for k, v in best.items()}
  373. ys = [t['result']['loss'] for t in trials.trials]
  374. logger.info('Dataset {} Hyperopt Results: {}'.format(moldata, ys))
  375. logger.info('Dataset {} Best Params: {}'.format(moldata, best_values))
  376. logger.info('Dataset {} Best Perform: {}'.format(moldata, np.min(ys)))
  377. elif args.mode == 'train':
  378. logger.info('Training')
  379. fold_result = main(numtasks, task, moldata, device, args.train_epoch, args.seed, args.fold, args.batch_size, rate, args.split, modelpath, logger, args.lr, attn_head, args.output_dim, args.attn_layers, args.dropout, mean, std, args.D, useedge, args.metric, True)
  380. elif args.mode == 'test':
  381. assert (args.pretrain is not None)
  382. fold_result = test(numtasks, task, moldata, device, args.seed, args.batch_size, logger, attn_head, args.output_dim, args.attn_layers, args.dropout, args.pretrain, mean, std, args.D, useedge, args.metric)
  383. else:pass

run.py at commit ce054fc, under MIT · at the source

Overview

Authors: Shenglan Li1, Jialiang Lu2,3, Zhuang Kang1, Haiyan Yang4, Wenbin Qian5, Xi Chen4, Xianggui Yuan5, Miao Hu6, Qiuqiu Shi2,3, Xinglu Zhou6, Feng Chen1, Xiaowu Dong2,3, Wenbin Li1
ORCID iDs: Xiaowu Dong
  1. Cancer Center, Beijing Tiantan Hospital, Capital Medical University, Beijing 100070, China
  2. College of Pharmaceutical Sciences, Zhejiang University, Hangzhou 310058, China
  3. ZJUCPS-HealZen Joint Laboratory of Drug Innovation and Transformation, Zhejiang University, Hangzhou 310058, China
  4. Department of Lymphoma, Zhejiang Cancer Hospital, Hangzhou 310022, China
  5. Department of Hematology, The Second Affiliated Hospital, Zhejiang University School of Medicine, Hangzhou 310009, China
  6. HealZen Therapeutics Co., Ltd., Hangzhou 310018, China
Journal: Acta pharmaceutica Sinica. B, volume 16, issue 8, pages 5351-5362
Dates: received 13 August 2025; accepted 14 April 2026; published online 23 May 2026; in print August 2026
Type: Research article · Language: English
License: CC BY-NC-ND
Identifiers: DOI 10.1016/j.apsb.2026.05.021 · PMID 42592300 · PMCID PMC13464085 · OpenAlex W7162195956
Open access: gold, a free copy (OpenAlex)
Status: code verified
Categories: human (organism), clinical / translational (subfield)
Methods: Statistics, Connectivity, Machine learning, Smoothing, state filtering, decompositions
Keywords: Bruton tyrosine kinase, Central nervous system lymphoma, BTK inhibitors, HZ-A-018, Blood‒brain barrier, BBB permeability, Phase 1 clinical trial, AI-driven drug discovery
Topic: CNS Lymphoma Diagnosis and Treatment (Neurology, Medicine), according to OpenAlex
Funding: HealZen Therapeutics Co
Citations: not cited yet (Europe PMC); 31 references in the paper

Abstract

The abstract is not reproduced here: the paper's license (CC BY-NC-ND) does not allow it. Read it in the paper, at the publisher or on Europe PMC.

Repository

Its files are read in the Code ↔ Paper reader above, with 1 match between paragraphs and lines of code.

gaojianl/KnoMol

License: MIT
State: the link answers, verified on 28 September 2026
Evidence: files inventoried
Commit: ce054fcc139afe2f68fe945be736a5c1bc45cfe8, 26 December 2024
Languages: Python (5)
Size: 14 files, 5 scripts
Software Heritage: not archived
Found in: the text, “BTK inhibition activity and BBB permeability pre”
Holds: README, license file, environment (requirements.txt)
Not found: CITATION.cff, tests, continuous integration, documentation
Tools: NumPy (4 files), PyTorch Geometric (4 files), PyTorch (4 files), pandas (3 files), RDKit (3 files), NetworkX (1 file), scikit-learn (1 file)
Availability: 1 check, the latest on 28 September 2026: the link answers
  • 28 September 2026: the link answers
7 files

Tracing map

Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.

What the map holds:

  • 1 repository of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
  • 5 scripts, each with its path and the digest of its content;
  • 1 match between paragraphs of the paper and lines of the code (method lexical-v1);
  • neither the text of the paper nor the code itself.

Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.

Data

Datasets cited

Versions

The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.

Version 2, 28 September 2026

  • Authors: added Xiaowu Dong (0000-0002-2178-4372); removed Xiaowu Dong

Version 1, 28 September 2026: the first record

Recorded: type, language, journal, volume, issue, pages, dates, 13 authors, 8 keywords, 1 funder, 31 references.

Cite

This paper

Li, S., Lu, J., Kang, Z., Yang, H., Qian, W., Chen, X., Yuan, X., Hu, M., Shi, Q., Zhou, X., Chen, F., Dong, X., & Li, W. (2026). Design, preclinical evaluation, and multicenter phase 1 clinical study of HZ-A-018 for relapsed or refractory central nervous system lymphoma. Acta pharmaceutica Sinica. B, 16(8), 5351-5362. https://doi.org/10.1016/j.apsb.2026.05.021

BibTeX

@article{li2026design,
author = {Li, Shenglan and Lu, Jialiang and Kang, Zhuang and Yang, Haiyan and Qian, Wenbin and Chen, Xi and Yuan, Xianggui and Hu, Miao and Shi, Qiuqiu and Zhou, Xinglu and Chen, Feng and Dong, Xiaowu and Li, Wenbin},
title = {{Design, preclinical evaluation, and multicenter phase 1 clinical study of HZ-A-018 for relapsed or refractory central nervous system lymphoma}},
journal = {Acta pharmaceutica Sinica. B},
year = {2026},
month = may,
volume = {16},
number = {8},
pages = {5351--5362},
publisher = {Elsevier},
issn = {2211-3835},
doi = {10.1016/j.apsb.2026.05.021},
url = {https://doi.org/10.1016/j.apsb.2026.05.021},
pmid = {42592300},
pmcid = {PMC13464085}
}

RIS

TY - JOUR
AU - Li, Shenglan
AU - Lu, Jialiang
AU - Kang, Zhuang
AU - Yang, Haiyan
AU - Qian, Wenbin
AU - Chen, Xi
AU - Yuan, Xianggui
AU - Hu, Miao
AU - Shi, Qiuqiu
AU - Zhou, Xinglu
AU - Chen, Feng
AU - Dong, Xiaowu
AU - Li, Wenbin
TI - Design, preclinical evaluation, and multicenter phase 1 clinical study of HZ-A-018 for relapsed or refractory central nervous system lymphoma
T2 - Acta pharmaceutica Sinica. B
J2 - Acta Pharm Sin B
PY - 2026
DA - 2026/05/23
VL - 16
IS - 8
SP - 5351
EP - 5362
SN - 2211-3835
PB - Elsevier
DO - 10.1016/j.apsb.2026.05.021
UR - https://doi.org/10.1016/j.apsb.2026.05.021
LA - en
ER -

CSL-JSON

{
"id": "10.1016/j.apsb.2026.05.021",
"type": "article-journal",
"title": "Design, preclinical evaluation, and multicenter phase 1 clinical study of HZ-A-018 for relapsed or refractory central nervous system lymphoma",
"container-title": "Acta pharmaceutica Sinica. B",
"author": [
{
"family": "Li",
"given": "Shenglan"
},
{
"family": "Lu",
"given": "Jialiang"
},
{
"family": "Kang",
"given": "Zhuang"
},
{
"family": "Yang",
"given": "Haiyan"
},
{
"family": "Qian",
"given": "Wenbin"
},
{
"family": "Chen",
"given": "Xi"
},
{
"family": "Yuan",
"given": "Xianggui"
},
{
"family": "Hu",
"given": "Miao"
},
{
"family": "Shi",
"given": "Qiuqiu"
},
{
"family": "Zhou",
"given": "Xinglu"
},
{
"family": "Chen",
"given": "Feng"
},
{
"family": "Dong",
"given": "Xiaowu"
},
{
"family": "Li",
"given": "Wenbin"
}
],
"container-title-short": "Acta Pharm Sin B",
"volume": "16",
"issue": "8",
"page": "5351-5362",
"DOI": "10.1016/j.apsb.2026.05.021",
"PMID": "42592300",
"PMCID": "PMC13464085",
"ISSN": "2211-3835",
"publisher": "Elsevier",
"URL": "https://doi.org/10.1016/j.apsb.2026.05.021",
"language": "en",
"issued": {
"date-parts": [
[
2026,
5,
23
]
]
}
}

The tracing map gets a citation of its own once an author has validated it and it has a DOI.

Similar papers

The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.

[1] doi:10.3390/ph19081319 [code]
DeepBBB: A Data-Composition-Aware Graph Screening Workflow for BBB-Focused CNS Library Construction and Prospective PAMPA-BBB Evaluation.
Journal: Pharmaceuticals (Basel, Switzerland)
In common: RDKit, PyTorch Geometric, NetworkX, 4 other tools, clinical / translational
[2] doi:10.1371/journal.pone.0345854 [code]
Shedding light on neural learning to rank models for anticancer drug prioritization.
Journal: PloS one
In common: RDKit, PyTorch Geometric, NetworkX, 4 other tools
[3] doi:10.1093/nar/gkag706 [code]
scDifformer: diffusion-based post-training for virtual cell modeling across large-scale single-cell data.
Journal: Nucleic acids research
In common: RDKit, PyTorch Geometric, NetworkX, 4 other tools
[4] doi:10.1093/bib/bbag118 [code]
Drug screening for α-synuclein aggregation inhibitors via multimodal graph neural network.
Journal: Briefings in bioinformatics
In common: RDKit, PyTorch Geometric, NetworkX, 4 other tools
[5] doi:10.1093/bioinformatics/btag153 [code]
MAISNet: a multi-species integrated graph neural network for acetylcholinesterase inhibitor screening.
Journal: Bioinformatics (Oxford, England)
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools, clinical / translational
[6] doi:10.1038/s41598-026-53415-5 [code]
Computational design and immunoinformatics validation of a T cell multi-epitope vaccine targeting glioblastoma stem cells.
Journal: Scientific reports
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools
[7] doi:10.1038/s41586-026-10391-0 [code]
Cell-type-targeted mitochondrial transplantation rescues cell degeneration.
Journal: Nature
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools
[8] doi:10.1021/acs.biochem.5c00596 [code]
Cargo Recognition of Nesprin-2 by the Dynein Adapter Bicaudal D2 for a Nuclear Positioning Pathway That Is Important for Brain Development.
Journal: Biochemistry
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools
[9] doi:10.34133/csbj.0184 [code]
Cross-Species Multitask Learning with Molecular and ADME Descriptors for Liver Microsomal Metabolic Stability.
Journal: Computational and structural biotechnology journal
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools
[10] doi:10.3390/ijms27156614 [code]
Candidalysin Inhibits &lt;i&gt;Porphyromonas gingivalis&lt;/i&gt; Lipoprotein-Induced IL-1β Production in BV-2 Microglia via Hydrophobic Microbial Interactions.
Journal: International journal of molecular sciences
In common: RDKit, PyTorch Geometric, PyTorch, 3 other tools

Contribute

The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.

Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.

Request its removal

To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).

Discussion, reproductions, activity

Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.

Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.

Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.