浏览代码

Modifications to analyses

Nicholas Schense 7 小时之前
父节点
当前提交
cf2d6b37f7

+ 227 - 0
alnn_rewrite.log

@@ -5734,3 +5734,230 @@ RuntimeError: App is not running
 2026-07-27 03:25:26,000 - INFO - [Analyze Evaluations] Analysis artifacts written to outputs/exp1/analysis/.
 2026-07-27 03:25:26,003 - INFO - [Analyze Evaluations] Task completed.
 2026-07-27 03:25:26,005 - INFO - [SYSTEM] All tasks finished successfully!
+2026-07-27 08:12:13,951 - INFO - [SYSTEM] Application initialized and ready.
+2026-07-27 08:12:13,953 - INFO - [SYSTEM] Successfully loaded config from ./outputs/exp1/config.toml
+2026-07-27 08:12:37,323 - INFO - [SYSTEM] Successfully loaded config from ./outputs/exp1/config.toml
+2026-07-27 08:14:01,970 - INFO - [SYSTEM] Initiating Scenario: 6. Extend Noise Sweep (extra sigmas + merge + analyze)
+2026-07-27 08:14:01,982 - INFO - [SYSTEM] Starting Task 1/7: Load Image and ADNIMERGE
+2026-07-27 08:14:02,135 - INFO - [Load Image and ADNIMERGE] Seeded RNGs with base seed 42
+2026-07-27 08:14:02,136 - INFO - [Load Image and ADNIMERGE] Loading Files
+2026-07-27 08:14:06,198 - INFO - [Load Image and ADNIMERGE] Dataloaders initalized
+2026-07-27 08:14:06,206 - INFO - [Load Image and ADNIMERGE] Task completed.
+2026-07-27 08:14:06,209 - INFO - [SYSTEM] Starting Task 2/7: Load Saved Models
+2026-07-27 08:14:06,210 - INFO - [Load Saved Models] Found 30 normal model(s) in normal_models/.
+2026-07-27 08:14:06,211 - INFO - [Load Saved Models] Found 30 bayesian model(s) in bayesian_models/.
+2026-07-27 08:14:06,212 - INFO - [Load Saved Models] Task completed.
+2026-07-27 08:14:06,213 - INFO - [SYSTEM] Starting Task 3/7: Evaluate Additional Noise Levels
+2026-07-27 08:14:06,486 - INFO - [Evaluate Additional Noise Levels] Evaluating 30 normal model(s) on 97 test samples, 4 noise level(s), 1 MC pass(es) each.
+2026-07-27 08:14:06,493 - INFO - [Evaluate Additional Noise Levels] Loading normal model 1/30: model_1.pt
+2026-07-27 08:14:10,789 - INFO - [Evaluate Additional Noise Levels] Loading normal model 2/30: model_2.pt
+2026-07-27 08:14:14,780 - INFO - [Evaluate Additional Noise Levels] Loading normal model 3/30: model_3.pt
+2026-07-27 08:14:18,785 - INFO - [Evaluate Additional Noise Levels] Loading normal model 4/30: model_4.pt
+2026-07-27 08:14:22,790 - INFO - [Evaluate Additional Noise Levels] Loading normal model 5/30: model_5.pt
+2026-07-27 08:14:26,818 - INFO - [Evaluate Additional Noise Levels] Loading normal model 6/30: model_6.pt
+2026-07-27 08:14:30,820 - INFO - [Evaluate Additional Noise Levels] Loading normal model 7/30: model_7.pt
+2026-07-27 08:14:34,828 - INFO - [Evaluate Additional Noise Levels] Loading normal model 8/30: model_8.pt
+2026-07-27 08:14:38,838 - INFO - [Evaluate Additional Noise Levels] Loading normal model 9/30: model_9.pt
+2026-07-27 08:14:42,869 - INFO - [Evaluate Additional Noise Levels] Loading normal model 10/30: model_10.pt
+2026-07-27 08:14:46,895 - INFO - [Evaluate Additional Noise Levels] Loading normal model 11/30: model_11.pt
+2026-07-27 08:14:50,935 - INFO - [Evaluate Additional Noise Levels] Loading normal model 12/30: model_12.pt
+2026-07-27 08:14:54,961 - INFO - [Evaluate Additional Noise Levels] Loading normal model 13/30: model_13.pt
+2026-07-27 08:14:58,993 - INFO - [Evaluate Additional Noise Levels] Loading normal model 14/30: model_14.pt
+2026-07-27 08:15:03,026 - INFO - [Evaluate Additional Noise Levels] Loading normal model 15/30: model_15.pt
+2026-07-27 08:15:07,044 - INFO - [Evaluate Additional Noise Levels] Loading normal model 16/30: model_16.pt
+2026-07-27 08:15:11,065 - INFO - [Evaluate Additional Noise Levels] Loading normal model 17/30: model_17.pt
+2026-07-27 08:15:15,105 - INFO - [Evaluate Additional Noise Levels] Loading normal model 18/30: model_18.pt
+2026-07-27 08:15:19,142 - INFO - [Evaluate Additional Noise Levels] Loading normal model 19/30: model_19.pt
+2026-07-27 08:15:23,185 - INFO - [Evaluate Additional Noise Levels] Loading normal model 20/30: model_20.pt
+2026-07-27 08:15:27,227 - INFO - [Evaluate Additional Noise Levels] Loading normal model 21/30: model_21.pt
+2026-07-27 08:15:31,281 - INFO - [Evaluate Additional Noise Levels] Loading normal model 22/30: model_22.pt
+2026-07-27 08:15:35,326 - INFO - [Evaluate Additional Noise Levels] Loading normal model 23/30: model_23.pt
+2026-07-27 08:15:39,367 - INFO - [Evaluate Additional Noise Levels] Loading normal model 24/30: model_24.pt
+2026-07-27 08:15:43,402 - INFO - [Evaluate Additional Noise Levels] Loading normal model 25/30: model_25.pt
+2026-07-27 08:15:47,438 - INFO - [Evaluate Additional Noise Levels] Loading normal model 26/30: model_26.pt
+2026-07-27 08:15:51,472 - INFO - [Evaluate Additional Noise Levels] Loading normal model 27/30: model_27.pt
+2026-07-27 08:15:55,509 - INFO - [Evaluate Additional Noise Levels] Loading normal model 28/30: model_28.pt
+2026-07-27 08:15:59,580 - INFO - [Evaluate Additional Noise Levels] Loading normal model 29/30: model_29.pt
+2026-07-27 08:16:03,606 - INFO - [Evaluate Additional Noise Levels] Loading normal model 30/30: model_30.pt
+2026-07-27 08:16:07,681 - INFO - [Evaluate Additional Noise Levels] Saved normal test evaluation to outputs/exp1/evaluations/extra_noisy_normal.nc (models=30, samples=97, noise_levels=4).
+2026-07-27 08:16:07,764 - INFO - [Evaluate Additional Noise Levels] Evaluating 30 bayesian model(s) on 97 test samples, 4 noise level(s), 20 MC pass(es) each.
+2026-07-27 08:16:07,771 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 1/30: model_1.pt
+2026-07-27 08:17:24,582 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 2/30: model_2.pt
+2026-07-27 08:18:41,597 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 3/30: model_3.pt
+2026-07-27 08:19:58,634 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 4/30: model_4.pt
+2026-07-27 08:21:15,849 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 5/30: model_5.pt
+2026-07-27 08:22:32,936 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 6/30: model_6.pt
+2026-07-27 08:23:50,134 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 7/30: model_7.pt
+2026-07-27 08:25:07,160 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 8/30: model_8.pt
+2026-07-27 08:26:24,151 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 9/30: model_9.pt
+2026-07-27 08:27:41,265 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 10/30: model_10.pt
+2026-07-27 08:28:58,399 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 11/30: model_11.pt
+2026-07-27 08:30:15,532 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 12/30: model_12.pt
+2026-07-27 08:31:32,684 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 13/30: model_13.pt
+2026-07-27 08:32:49,777 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 14/30: model_14.pt
+2026-07-27 08:34:06,928 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 15/30: model_15.pt
+2026-07-27 08:35:24,116 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 16/30: model_16.pt
+2026-07-27 08:36:41,202 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 17/30: model_17.pt
+2026-07-27 08:37:58,298 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 18/30: model_18.pt
+2026-07-27 08:39:15,667 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 19/30: model_19.pt
+2026-07-27 08:40:32,961 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 20/30: model_20.pt
+2026-07-27 08:41:50,069 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 21/30: model_21.pt
+2026-07-27 08:43:07,285 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 22/30: model_22.pt
+2026-07-27 08:44:28,540 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 23/30: model_23.pt
+2026-07-27 08:45:52,012 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 24/30: model_24.pt
+2026-07-27 08:47:15,354 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 25/30: model_25.pt
+2026-07-27 08:48:38,783 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 26/30: model_26.pt
+2026-07-27 08:50:02,316 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 27/30: model_27.pt
+2026-07-27 08:51:25,715 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 28/30: model_28.pt
+2026-07-27 08:52:49,349 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 29/30: model_29.pt
+2026-07-27 08:54:12,794 - INFO - [Evaluate Additional Noise Levels] Loading bayesian model 30/30: model_30.pt
+2026-07-27 08:55:36,334 - INFO - [Evaluate Additional Noise Levels] Saved bayesian test evaluation to outputs/exp1/evaluations/extra_noisy_bayesian.nc (models=30, samples=97, noise_levels=4).
+2026-07-27 08:55:36,335 - INFO - [Evaluate Additional Noise Levels] Task completed.
+2026-07-27 08:55:36,337 - INFO - [SYSTEM] Starting Task 4/7: Evaluate Additional Noise Levels (Validation)
+2026-07-27 08:55:36,455 - INFO - [Evaluate Additional Noise Levels (Validation)] Evaluating 30 normal model(s) on 149 val samples, 4 noise level(s), 1 MC pass(es) each.
+2026-07-27 08:55:36,462 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 1/30: model_1.pt
+2026-07-27 08:55:42,919 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 2/30: model_2.pt
+2026-07-27 08:55:49,396 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 3/30: model_3.pt
+2026-07-27 08:55:55,869 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 4/30: model_4.pt
+2026-07-27 08:56:02,388 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 5/30: model_5.pt
+2026-07-27 08:56:08,878 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 6/30: model_6.pt
+2026-07-27 08:56:15,353 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 7/30: model_7.pt
+2026-07-27 08:56:21,818 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 8/30: model_8.pt
+2026-07-27 08:56:28,318 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 9/30: model_9.pt
+2026-07-27 08:56:34,839 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 10/30: model_10.pt
+2026-07-27 08:56:41,318 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 11/30: model_11.pt
+2026-07-27 08:56:47,802 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 12/30: model_12.pt
+2026-07-27 08:56:54,283 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 13/30: model_13.pt
+2026-07-27 08:57:00,766 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 14/30: model_14.pt
+2026-07-27 08:57:07,220 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 15/30: model_15.pt
+2026-07-27 08:57:13,734 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 16/30: model_16.pt
+2026-07-27 08:57:20,244 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 17/30: model_17.pt
+2026-07-27 08:57:26,719 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 18/30: model_18.pt
+2026-07-27 08:57:33,186 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 19/30: model_19.pt
+2026-07-27 08:57:39,663 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 20/30: model_20.pt
+2026-07-27 08:57:46,171 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 21/30: model_21.pt
+2026-07-27 08:57:52,664 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 22/30: model_22.pt
+2026-07-27 08:57:59,122 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 23/30: model_23.pt
+2026-07-27 08:58:05,606 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 24/30: model_24.pt
+2026-07-27 08:58:12,104 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 25/30: model_25.pt
+2026-07-27 08:58:18,616 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 26/30: model_26.pt
+2026-07-27 08:58:25,143 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 27/30: model_27.pt
+2026-07-27 08:58:31,626 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 28/30: model_28.pt
+2026-07-27 08:58:38,130 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 29/30: model_29.pt
+2026-07-27 08:58:44,600 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading normal model 30/30: model_30.pt
+2026-07-27 08:58:51,096 - INFO - [Evaluate Additional Noise Levels (Validation)] Saved normal val evaluation to outputs/exp1/evaluations/val_extra_noisy_normal.nc (models=30, samples=149, noise_levels=4).
+2026-07-27 08:58:51,205 - INFO - [Evaluate Additional Noise Levels (Validation)] Evaluating 30 bayesian model(s) on 149 val samples, 4 noise level(s), 20 MC pass(es) each.
+2026-07-27 08:58:51,212 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 1/30: model_1.pt
+2026-07-27 09:00:57,995 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 2/30: model_2.pt
+2026-07-27 09:03:05,150 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 3/30: model_3.pt
+2026-07-27 09:05:12,096 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 4/30: model_4.pt
+2026-07-27 09:07:19,051 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 5/30: model_5.pt
+2026-07-27 09:09:26,087 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 6/30: model_6.pt
+2026-07-27 09:11:33,078 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 7/30: model_7.pt
+2026-07-27 09:13:40,080 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 8/30: model_8.pt
+2026-07-27 09:15:47,002 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 9/30: model_9.pt
+2026-07-27 09:17:54,161 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 10/30: model_10.pt
+2026-07-27 09:20:01,369 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 11/30: model_11.pt
+2026-07-27 09:22:08,239 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 12/30: model_12.pt
+2026-07-27 09:24:15,030 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 13/30: model_13.pt
+2026-07-27 09:26:21,764 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 14/30: model_14.pt
+2026-07-27 09:28:28,871 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 15/30: model_15.pt
+2026-07-27 09:30:35,940 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 16/30: model_16.pt
+2026-07-27 09:32:43,102 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 17/30: model_17.pt
+2026-07-27 09:34:50,185 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 18/30: model_18.pt
+2026-07-27 09:36:57,201 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 19/30: model_19.pt
+2026-07-27 09:39:04,485 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 20/30: model_20.pt
+2026-07-27 09:41:11,661 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 21/30: model_21.pt
+2026-07-27 09:43:18,639 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 22/30: model_22.pt
+2026-07-27 09:45:25,811 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 23/30: model_23.pt
+2026-07-27 09:47:33,027 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 24/30: model_24.pt
+2026-07-27 09:49:40,049 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 25/30: model_25.pt
+2026-07-27 09:51:47,128 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 26/30: model_26.pt
+2026-07-27 09:53:54,218 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 27/30: model_27.pt
+2026-07-27 09:56:01,246 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 28/30: model_28.pt
+2026-07-27 09:58:08,287 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 29/30: model_29.pt
+2026-07-27 10:00:15,224 - INFO - [Evaluate Additional Noise Levels (Validation)] Loading bayesian model 30/30: model_30.pt
+2026-07-27 10:02:22,505 - INFO - [Evaluate Additional Noise Levels (Validation)] Saved bayesian val evaluation to outputs/exp1/evaluations/val_extra_noisy_bayesian.nc (models=30, samples=149, noise_levels=4).
+2026-07-27 10:02:22,511 - INFO - [Evaluate Additional Noise Levels (Validation)] Task completed.
+2026-07-27 10:02:22,515 - INFO - [SYSTEM] Starting Task 5/7: Combine Split Evaluations
+2026-07-27 10:02:22,556 - INFO - [Combine Split Evaluations] combine_evaluations: wrote combined_normal.nc (246 samples x 1 noise level(s) from normal.nc, val_normal.nc).
+2026-07-27 10:02:22,614 - INFO - [Combine Split Evaluations] combine_evaluations: wrote combined_noisy_normal.nc (246 samples x 17 noise level(s) from noisy_normal.nc, extra_noisy_normal.nc, val_noisy_normal.nc, val_extra_noisy_normal.nc).
+2026-07-27 10:02:22,641 - INFO - [Combine Split Evaluations] combine_evaluations: wrote combined_bayesian.nc (246 samples x 1 noise level(s) from bayesian.nc, val_bayesian.nc).
+2026-07-27 10:02:22,700 - INFO - [Combine Split Evaluations] combine_evaluations: wrote combined_noisy_bayesian.nc (246 samples x 17 noise level(s) from noisy_bayesian.nc, extra_noisy_bayesian.nc, val_noisy_bayesian.nc, val_extra_noisy_bayesian.nc).
+2026-07-27 10:02:22,706 - INFO - [Combine Split Evaluations] Task completed.
+2026-07-27 10:02:22,708 - INFO - [SYSTEM] Starting Task 6/7: Generate Extra Information
+2026-07-27 10:02:22,708 - INFO - [Generate Extra Information] Extra info: dataset_summary
+2026-07-27 10:02:22,842 - INFO - [Generate Extra Information] dataset_summary: train=526 scans/301 patients, val=149 scans/86 patients, test=97 scans/43 patients. Wrote dataset_summary.png/.json.
+2026-07-27 10:02:22,849 - INFO - [Generate Extra Information] Extra info: noise_examples
+2026-07-27 10:02:23,339 - INFO - [Generate Extra Information] noise_examples: wrote noise_examples.png (3 images x 8 noise levels).
+2026-07-27 10:02:23,346 - INFO - [Generate Extra Information] Extra information written to outputs/exp1/analysis/.
+2026-07-27 10:02:23,347 - INFO - [Generate Extra Information] Task completed.
+2026-07-27 10:02:23,348 - INFO - [SYSTEM] Starting Task 7/7: Analyze Evaluations
+2026-07-27 10:02:23,388 - INFO - [Analyze Evaluations] Loaded evaluation 'bayesian' with dims {'model': 30, 'noise_level': 1, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,400 - INFO - [Analyze Evaluations] Loaded evaluation 'combined_bayesian' with dims {'model': 30, 'noise_level': 1, 'sample': 246, 'class_name': 2}.
+2026-07-27 10:02:23,404 - INFO - [Analyze Evaluations] Loaded evaluation 'combined_noisy_bayesian' with dims {'model': 30, 'noise_level': 17, 'sample': 246, 'class_name': 2}.
+2026-07-27 10:02:23,413 - INFO - [Analyze Evaluations] Loaded evaluation 'combined_noisy_normal' with dims {'model': 30, 'noise_level': 17, 'sample': 246, 'class_name': 2}.
+2026-07-27 10:02:23,417 - INFO - [Analyze Evaluations] Loaded evaluation 'combined_normal' with dims {'model': 30, 'noise_level': 1, 'sample': 246, 'class_name': 2}.
+2026-07-27 10:02:23,420 - INFO - [Analyze Evaluations] Loaded evaluation 'extra_noisy_bayesian' with dims {'model': 30, 'noise_level': 4, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,430 - INFO - [Analyze Evaluations] Loaded evaluation 'extra_noisy_normal' with dims {'model': 30, 'noise_level': 4, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,433 - INFO - [Analyze Evaluations] Loaded evaluation 'noisy_bayesian' with dims {'model': 30, 'noise_level': 13, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,437 - INFO - [Analyze Evaluations] Loaded evaluation 'noisy_normal' with dims {'model': 30, 'noise_level': 13, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,440 - INFO - [Analyze Evaluations] Loaded evaluation 'normal' with dims {'model': 30, 'noise_level': 1, 'sample': 97, 'class_name': 2}.
+2026-07-27 10:02:23,449 - INFO - [Analyze Evaluations] Loaded evaluation 'val_bayesian' with dims {'model': 30, 'noise_level': 1, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,453 - INFO - [Analyze Evaluations] Loaded evaluation 'val_extra_noisy_bayesian' with dims {'model': 30, 'noise_level': 4, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,456 - INFO - [Analyze Evaluations] Loaded evaluation 'val_extra_noisy_normal' with dims {'model': 30, 'noise_level': 4, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,465 - INFO - [Analyze Evaluations] Loaded evaluation 'val_noisy_bayesian' with dims {'model': 30, 'noise_level': 13, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,469 - INFO - [Analyze Evaluations] Loaded evaluation 'val_noisy_normal' with dims {'model': 30, 'noise_level': 13, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,472 - INFO - [Analyze Evaluations] Loaded evaluation 'val_normal' with dims {'model': 30, 'noise_level': 1, 'sample': 149, 'class_name': 2}.
+2026-07-27 10:02:23,478 - INFO - [Analyze Evaluations] Running analysis: model_report — Per-model accuracy and uncertainty summary (PDF)
+2026-07-27 10:02:23,711 - INFO - [Analyze Evaluations] model_report: wrote model_report.pdf (2 family/families, 60 model(s)).
+2026-07-27 10:02:23,712 - INFO - [Analyze Evaluations] Running analysis: ensemble_size_accuracy — Accuracy vs. number of ensemble members
+2026-07-27 10:02:23,735 - INFO - [Analyze Evaluations] ensemble_size_accuracy: normal - sizes 1..30, up to 30 configs each.
+2026-07-27 10:02:23,755 - INFO - [Analyze Evaluations] ensemble_size_accuracy: bayesian - sizes 1..30, up to 30 configs each.
+2026-07-27 10:02:23,963 - INFO - [Analyze Evaluations] ensemble_size_accuracy: wrote ensemble_size_accuracy.png (2 family/families).
+2026-07-27 10:02:23,970 - INFO - [Analyze Evaluations] Running analysis: bootstrap_ensemble_ci — Bootstrap confidence intervals for ensemble accuracy
+2026-07-27 10:02:24,011 - INFO - [Analyze Evaluations] bootstrap_ensemble_ci: normal full-ensemble acc = 71.5% (95% CI 62.4-79.9%, n=246 scans / 129 patients).
+2026-07-27 10:02:24,052 - INFO - [Analyze Evaluations] bootstrap_ensemble_ci: bayesian full-ensemble acc = 80.5% (95% CI 72.3-87.9%, n=246 scans / 129 patients).
+2026-07-27 10:02:24,171 - INFO - [Analyze Evaluations] bootstrap_ensemble_ci: wrote bootstrap_ci.png (2 family/families).
+2026-07-27 10:02:24,177 - INFO - [Analyze Evaluations] Running analysis: coverage — Coverage curves: accuracy and F1 vs. retained fraction
+2026-07-27 10:02:24,748 - INFO - [Analyze Evaluations] coverage: wrote coverage_accuracy.png (5 measure panels).
+2026-07-27 10:02:25,376 - INFO - [Analyze Evaluations] coverage: wrote coverage_f1.png (5 measure panels).
+2026-07-27 10:02:25,384 - INFO - [Analyze Evaluations] coverage: mean_output_entropy / normal_ensemble — AURC=0.1189 (accuracy at full coverage 0.715).
+2026-07-27 10:02:25,387 - INFO - [Analyze Evaluations] coverage: mean_output_entropy / normal_single — AURC=0.1335 (accuracy at full coverage 0.725).
+2026-07-27 10:02:25,389 - INFO - [Analyze Evaluations] coverage: mean_output_entropy / bayesian_ensemble — AURC=0.0763 (accuracy at full coverage 0.805).
+2026-07-27 10:02:25,389 - INFO - [Analyze Evaluations] coverage: mean_output_entropy / bayesian_single — AURC=0.1594 (accuracy at full coverage 0.699).
+2026-07-27 10:02:25,390 - INFO - [Analyze Evaluations] coverage: mean_predictive_entropy / normal_ensemble — AURC=0.1096 (accuracy at full coverage 0.715).
+2026-07-27 10:02:25,390 - INFO - [Analyze Evaluations] coverage: mean_predictive_entropy / bayesian_ensemble — AURC=0.0923 (accuracy at full coverage 0.805).
+2026-07-27 10:02:25,391 - INFO - [Analyze Evaluations] coverage: ensemble_std / normal_ensemble — AURC=0.2589 (accuracy at full coverage 0.715).
+2026-07-27 10:02:25,391 - INFO - [Analyze Evaluations] coverage: ensemble_std / bayesian_ensemble — AURC=0.0979 (accuracy at full coverage 0.805).
+2026-07-27 10:02:25,391 - INFO - [Analyze Evaluations] coverage: ensemble_mutual_information / normal_ensemble — AURC=0.2778 (accuracy at full coverage 0.715).
+2026-07-27 10:02:25,392 - INFO - [Analyze Evaluations] coverage: ensemble_mutual_information / bayesian_ensemble — AURC=0.1134 (accuracy at full coverage 0.805).
+2026-07-27 10:02:25,392 - INFO - [Analyze Evaluations] coverage: mean_mutual_information / bayesian_ensemble — AURC=0.1170 (accuracy at full coverage 0.805).
+2026-07-27 10:02:25,392 - INFO - [Analyze Evaluations] coverage: mean_mutual_information / bayesian_single — AURC=0.2047 (accuracy at full coverage 0.699).
+2026-07-27 10:02:25,394 - INFO - [Analyze Evaluations] Running analysis: noise_performance — Accuracy and uncertainty vs. input noise
+2026-07-27 10:02:26,274 - INFO - [Analyze Evaluations] noise_performance: normal_ensemble — accuracy 0.715 at σ=0 → 0.585 at σ=32.
+2026-07-27 10:02:26,280 - INFO - [Analyze Evaluations] noise_performance: normal_single — accuracy 0.725 at σ=0 → 0.569 at σ=32.
+2026-07-27 10:02:26,282 - INFO - [Analyze Evaluations] noise_performance: bayesian_ensemble — accuracy 0.809 at σ=0 → 0.472 at σ=32.
+2026-07-27 10:02:26,282 - INFO - [Analyze Evaluations] noise_performance: bayesian_single — accuracy 0.686 at σ=0 → 0.477 at σ=32.
+2026-07-27 10:02:26,283 - INFO - [Analyze Evaluations] noise_performance: wrote noise_performance.png (7 panels).
+2026-07-27 10:02:26,284 - INFO - [Analyze Evaluations] Running analysis: noise_correlation — Uncertainty vs. accuracy across noise levels (with fit)
+2026-07-27 10:02:27,395 - INFO - [Analyze Evaluations] noise_correlation: wrote noise_correlation_mean_output_entropy.png.
+2026-07-27 10:02:27,401 - INFO - [Analyze Evaluations] noise_correlation: mean_output_entropy / normal_ensemble — exponential fit R²=0.671, Spearman ρ=-0.823.
+2026-07-27 10:02:27,402 - INFO - [Analyze Evaluations] noise_correlation: mean_output_entropy / normal_single — exponential fit R²=0.799, Spearman ρ=-0.826.
+2026-07-27 10:02:27,403 - INFO - [Analyze Evaluations] noise_correlation: mean_output_entropy / bayesian_ensemble — exponential fit R²=0.012, Spearman ρ=-0.132.
+2026-07-27 10:02:27,403 - INFO - [Analyze Evaluations] noise_correlation: mean_output_entropy / bayesian_single — exponential fit R²=0.036, Spearman ρ=-0.206.
+2026-07-27 10:02:27,717 - INFO - [Analyze Evaluations] noise_correlation: wrote noise_correlation_mean_predictive_entropy.png.
+2026-07-27 10:02:27,718 - INFO - [Analyze Evaluations] noise_correlation: mean_predictive_entropy / normal_ensemble — exponential fit R²=0.740, Spearman ρ=-0.851.
+2026-07-27 10:02:27,726 - INFO - [Analyze Evaluations] noise_correlation: mean_predictive_entropy / bayesian_ensemble — exponential fit R²=0.002, Spearman ρ=-0.084.
+2026-07-27 10:02:28,065 - INFO - [Analyze Evaluations] noise_correlation: wrote noise_correlation_ensemble_std.png.
+2026-07-27 10:02:28,071 - INFO - [Analyze Evaluations] noise_correlation: ensemble_std / normal_ensemble — exponential fit R²=0.012, Spearman ρ=-0.091.
+2026-07-27 10:02:28,072 - INFO - [Analyze Evaluations] noise_correlation: ensemble_std / bayesian_ensemble — exponential fit R²=0.339, Spearman ρ=-0.520.
+2026-07-27 10:02:28,396 - INFO - [Analyze Evaluations] noise_correlation: wrote noise_correlation_ensemble_mutual_information.png.
+2026-07-27 10:02:28,403 - INFO - [Analyze Evaluations] noise_correlation: ensemble_mutual_information / normal_ensemble — exponential fit R²=0.014, Spearman ρ=0.144.
+2026-07-27 10:02:28,404 - INFO - [Analyze Evaluations] noise_correlation: ensemble_mutual_information / bayesian_ensemble — exponential fit R²=0.426, Spearman ρ=-0.659.
+2026-07-27 10:02:28,731 - INFO - [Analyze Evaluations] noise_correlation: wrote noise_correlation_mean_mutual_information.png.
+2026-07-27 10:02:28,737 - INFO - [Analyze Evaluations] noise_correlation: mean_mutual_information / bayesian_ensemble — exponential fit R²=0.259, Spearman ρ=-0.432.
+2026-07-27 10:02:28,737 - INFO - [Analyze Evaluations] noise_correlation: mean_mutual_information / bayesian_single — exponential fit R²=0.285, Spearman ρ=-0.498.
+2026-07-27 10:02:28,740 - INFO - [Analyze Evaluations] Analysis artifacts written to outputs/exp1/analysis/.
+2026-07-27 10:02:28,744 - INFO - [Analyze Evaluations] Task completed.
+2026-07-27 10:02:28,745 - INFO - [SYSTEM] All tasks finished successfully!

+ 2 - 2
analysis/context.py

@@ -38,8 +38,8 @@ CONFIG_LABELS = {
 
 #: Family -> hue (dataviz categorical slots, CVD-safe in this fixed order).
 FAMILY_COLORS = {
-    MODEL_KIND_NORMAL: "#003f5c",
-    MODEL_KIND_BAYESIAN: "#ffa600",
+    MODEL_KIND_NORMAL: "#2a78d6",     # blue
+    MODEL_KIND_BAYESIAN: "#e34948",   # red
 }
 
 #: Configuration -> line style. Identity is never carried by colour alone.

+ 5 - 5
analysis/measures.py

@@ -155,30 +155,30 @@ class Measure:
 MEASURES: Dict[str, Measure] = {
     "mean_output_entropy": Measure(
         name="mean_output_entropy",
-        label="Mean-output entropy (total)",
+        label="Mean output",
         fn=mean_output_entropy,
     ),
     "mean_predictive_entropy": Measure(
         name="mean_predictive_entropy",
-        label="Mean predictive entropy (aleatoric)",
+        label="Predictive uncertainty",
         fn=mean_predictive_entropy,
         duplicate_when_single=True,
     ),
     "ensemble_std": Measure(
         name="ensemble_std",
-        label="Ensemble std. dev. (disagreement)",
+        label="Standard deviation",
         fn=ensemble_std,
         requires_ensemble=True,
     ),
     "ensemble_mutual_information": Measure(
         name="ensemble_mutual_information",
-        label="Ensemble mutual information (epistemic)",
+        label="Ensemble model uncertainty",
         fn=ensemble_mutual_information,
         requires_ensemble=True,
     ),
     "mean_mutual_information": Measure(
         name="mean_mutual_information",
-        label="Mean MC mutual information (epistemic)",
+        label="Model uncertainty",
         fn=mean_mutual_information,
         families=(MODEL_KIND_BAYESIAN,),
         needs_member_mi=True,

+ 29 - 20
analysis/plots/coverage.py

@@ -89,22 +89,27 @@ def _series_curves(
         }
 
 
-def _plot(
+def _plot_measure(
     ctx: AnalysisContext,
-    results: Dict[str, Dict[str, Dict[str, Any]]],
-    metric: str,
-    std_key: str,
-    ylabel: str,
-    stem: str,
+    measure_name: str,
+    per_series: Dict[str, Dict[str, Any]],
 ) -> None:
-    """One panel per measure; one line per series."""
-    panels = [name for name, per_series in results.items() if per_series]
-    if not panels:
-        return
-
-    fig, axes = panel_grid(len(panels), n_cols=2)
-    for ax, measure_name in zip(axes, panels):
-        for series_key, payload in results[measure_name].items():
+    """One FIGURE per uncertainty measure: accuracy and F1 side by side.
+
+    One file per measure keeps each panel to a handful of lines; a single grid of
+    every measure was too dense to read.
+    """
+    label = ms.MEASURES[measure_name].label
+    fig, axes = panel_grid(2, n_cols=2, panel_size=(5.2, 3.8))
+
+    for ax, (metric, std_key, ylabel) in zip(
+        axes,
+        [
+            ("accuracy", "accuracy_std", "Accuracy"),
+            ("f1", "f1_std", f"F1 ({mt.POSITIVE_CLASS})"),
+        ],
+    ):
+        for payload in per_series.values():
             series = payload["_series"]
             x = np.asarray(payload["coverage"], dtype=float) * 100.0
             y = np.asarray(payload[metric], dtype=float)
@@ -119,15 +124,18 @@ def _plot(
                 color=series.color, linestyle=series.linestyle,
                 linewidth=2, marker="o", markersize=3, label=series.label,
             )
-        ax.set_title(ms.MEASURES[measure_name].label, fontsize=10)
+        ax.set_title(ylabel, fontsize=10)
         ax.set_xlabel("Coverage: most-confident samples retained (%)")
         ax.set_ylabel(ylabel)
         style_axes(ax)
+        # Full dataset on the left, progressively more restricted to the right,
+        # so the curve reads in the direction of "discard more".
+        ax.invert_xaxis()
 
-    fig.suptitle(f"Coverage curves — {ylabel} vs. retained fraction")
+    fig.suptitle(f"Coverage curves — {label}")
     add_legend(fig, axes)
-    png = save_figure(fig, ctx.out_dir, stem)
-    ctx.log.info(f"coverage: wrote {png.name} ({len(panels)} measure panels).")
+    png = save_figure(fig, ctx.out_dir, f"coverage_{measure_name}")
+    ctx.log.info(f"coverage: wrote {png.name}.")
 
 
 @register(
@@ -152,8 +160,9 @@ def coverage(ctx: AnalysisContext) -> None:
         ctx.log.error("coverage: no usable evaluations.")
         return
 
-    _plot(ctx, results, "accuracy", "accuracy_std", "Accuracy", "coverage_accuracy")
-    _plot(ctx, results, "f1", "f1_std", "F1 (AD)", "coverage_f1")
+    for measure_name, per_series in results.items():
+        if per_series:
+            _plot_measure(ctx, measure_name, per_series)
 
     serializable = {
         measure: {

+ 2 - 1
analysis/plots/model_report.py

@@ -15,6 +15,7 @@ import typst
 
 from analysis.sources import FAMILY_LABELS, baseline_noise_index
 from analysis.context import AnalysisContext
+from analysis.plotting import data_dir
 from analysis.registry import register
 
 _TEMPLATE = pl.Path(__file__).resolve().parent.parent / "templates" / "model_report.typ"
@@ -99,7 +100,7 @@ def model_report(ctx: AnalysisContext) -> None:
         str(_TEMPLATE), sys_inputs={"data": payload_json}, format="pdf"
     )
     (ctx.out_dir / "model_report.pdf").write_bytes(pdf_bytes)
-    (ctx.out_dir / "model_report.json").write_text(payload_json)
+    (data_dir(ctx.out_dir) / "model_report.json").write_text(payload_json)
 
     total_models = sum(f["n_models"] for f in families)
     ctx.log.info(

+ 14 - 1
analysis/plots/noise_correlation.py

@@ -40,7 +40,20 @@ from analysis.plotting import panel_grid, save_figure, save_json, style_axes
 from analysis.registry import register
 
 #: Equal-count uncertainty bins per noise level.
-N_BINS = 10
+#
+# 1 = one point per noise level, at that level's mean uncertainty and accuracy.
+# This is the form the earlier codebase used. It cannot show *within*-condition
+# structure, but it does test something that is not guaranteed: accuracy and
+# uncertainty are each only assumed monotone in sigma, so a clean relationship
+# between them across levels is real evidence they are linked, and its shape
+# characterises that link.
+#
+# Raise it (3-5) to also resolve the uncertainty->error relationship *inside*
+# each noise level. Each bin's accuracy is a binomial proportion, so its standard
+# error is sqrt(p(1-p)/n): with ~291 pooled samples the ensemble series gets
+# ~97/bin at 3 bins (SE ~0.04) but only ~29 at 10 (SE ~0.07), which visibly
+# dominated the scatter.
+N_BINS = 1
 
 #: Minimum samples in a bin for its accuracy to be trustworthy.
 MIN_BIN_SAMPLES = 5

+ 29 - 23
analysis/plots/noise_performance.py

@@ -124,16 +124,10 @@ def noise_performance(ctx: AnalysisContext) -> None:
         ctx.log.error("noise_performance: no noise-sweep evaluations available.")
         return
 
-    # Panels: accuracy, F1, then every measure that any series produced.
-    measure_panels = [
-        name
-        for name in ms.MEASURES
-        if any(name in p["measures"] for p in responses.values())
-    ]
-    panel_keys = ["accuracy", "f1"] + measure_panels
-    fig, axes = panel_grid(len(panel_keys), n_cols=2)
-
-    for ax, key in zip(axes, panel_keys):
+    x_label = next(iter(responses.values()))["noise_axis"]
+
+    def _draw(ax, key: str) -> None:
+        """Plot every series' response for one quantity onto ``ax``."""
         is_performance = key in ("accuracy", "f1")
         for payload in responses.values():
             series: Series = payload["_series"]
@@ -158,23 +152,35 @@ def noise_performance(ctx: AnalysisContext) -> None:
                 color=series.color, linestyle=series.linestyle,
                 linewidth=2, marker="o", markersize=4, label=series.label,
             )
-
-        if key == "accuracy":
-            ax.set_title("Accuracy", fontsize=10)
-            ax.set_ylabel("Accuracy")
-        elif key == "f1":
-            ax.set_title(f"F1 ({mt.POSITIVE_CLASS})", fontsize=10)
-            ax.set_ylabel("F1")
-        else:
-            ax.set_title(ms.MEASURES[key].label, fontsize=10)
-            ax.set_ylabel("Uncertainty (nats)")
-        ax.set_xlabel(next(iter(responses.values()))["noise_axis"])
+        ax.set_xlabel(x_label)
         style_axes(ax)
 
-    fig.suptitle("Performance and uncertainty vs. input noise")
+    # One figure for performance, then one per uncertainty measure -- a single
+    # 7-panel grid was too dense to read.
+    fig, axes = panel_grid(2, n_cols=2, panel_size=(5.2, 3.8))
+    for ax, key, title in (
+        (axes[0], "accuracy", "Accuracy"),
+        (axes[1], "f1", f"F1 ({mt.POSITIVE_CLASS})"),
+    ):
+        _draw(ax, key)
+        ax.set_title(title, fontsize=10)
+        ax.set_ylabel(title)
+    fig.suptitle("Performance vs. input noise")
     add_legend(fig, axes)
     png = save_figure(fig, ctx.out_dir, "noise_performance")
 
+    for name in ms.MEASURES:
+        if not any(name in p["measures"] for p in responses.values()):
+            continue
+        label = ms.MEASURES[name].label
+        mfig, maxes = panel_grid(1, n_cols=1, panel_size=(6.4, 4.2))
+        _draw(maxes[0], name)
+        maxes[0].set_title(f"{label} vs. input noise", fontsize=11)
+        maxes[0].set_ylabel(label)
+        add_legend(mfig, maxes)
+        mpng = save_figure(mfig, ctx.out_dir, f"noise_uncertainty_{name}")
+        ctx.log.info(f"noise_performance: wrote {mpng.name}.")
+
     save_json(
         {
             "positive_class": mt.POSITIVE_CLASS,
@@ -194,4 +200,4 @@ def noise_performance(ctx: AnalysisContext) -> None:
             f"noise_performance: {key} — accuracy {acc[0]:.3f} at σ={levels[0]:g} "
             f"→ {acc[-1]:.3f} at σ={levels[-1]:g}."
         )
-    ctx.log.info(f"noise_performance: wrote {png.name} ({len(panel_keys)} panels).")
+    ctx.log.info(f"noise_performance: wrote {png.name}.")

+ 25 - 6
analysis/plotting.py

@@ -83,20 +83,39 @@ def add_legend(fig: Figure, axes: Sequence[Axes], *, ncol: int = 4) -> None:
     )
 
 
+#: Figures and the numbers behind them live in separate subdirectories of the
+#: analysis output dir, so the plots can be browsed without wading through JSON.
+PLOTS_SUBDIR = "plots"
+DATA_SUBDIR = "data"
+
+
+def plots_dir(out_dir: pl.Path) -> pl.Path:
+    """``<analysis>/plots`` -- every figure lands here."""
+    path = out_dir / PLOTS_SUBDIR
+    path.mkdir(parents=True, exist_ok=True)
+    return path
+
+
+def data_dir(out_dir: pl.Path) -> pl.Path:
+    """``<analysis>/data`` -- the numbers behind the figures."""
+    path = out_dir / DATA_SUBDIR
+    path.mkdir(parents=True, exist_ok=True)
+    return path
+
+
 def save_figure(fig: Figure, out_dir: pl.Path, stem: str) -> pl.Path:
-    """Write ``stem.png`` (raster) and ``stem.pdf`` (vector); close the figure."""
-    out_dir.mkdir(parents=True, exist_ok=True)
-    png_path = out_dir / f"{stem}.png"
+    """Write ``plots/stem.png`` (raster) and ``plots/stem.pdf``; close the figure."""
+    target = plots_dir(out_dir)
+    png_path = target / f"{stem}.png"
     fig.savefig(png_path, dpi=DPI, bbox_inches="tight")
-    fig.savefig(out_dir / f"{stem}.pdf", bbox_inches="tight")
+    fig.savefig(target / f"{stem}.pdf", bbox_inches="tight")
     plt.close(fig)
     return png_path
 
 
 def save_json(payload: Dict[str, Any], out_dir: pl.Path, stem: str) -> pl.Path:
     """Persist the numbers behind a figure so plots can be rebuilt or reused."""
-    out_dir.mkdir(parents=True, exist_ok=True)
-    path = out_dir / f"{stem}.json"
+    path = data_dir(out_dir) / f"{stem}.json"
     path.write_text(json.dumps(payload, indent=2, default=_json_default))
     return path
 

+ 6 - 1
config.toml

@@ -50,9 +50,14 @@ deterministic = false # force deterministic cuDNN kernels (slower, exact reprodu
 # so the curve is followed all the way into the unusable regime, giving the fit
 # both plateaus (clean accuracy and chance level) plus the transition between.
 # 0.0 is the clean baseline and must stay first; values must be unique.
-noise_levels = [0.0, 0.05, 0.1, 0.2, 0.3, 0.5, 0.7, 1.0, 1.5, 2.0, 3.0, 5.0, 8.0]
+noise_levels = [0.0, 0.3, 0.5, 0.7, 1.0, 1.5, 2.0, 3.0, 5.0, 7.0, 9.0, 11.0, 15.0, 20.0]
 # false => sigma is an absolute value in raw voxel units (not recommended here).
 noise_relative = true
+# ADDITIONAL sigmas evaluated by the "scen_extend_noise" scenario. Those runs
+# compute ONLY these levels and merge them onto the sweep already on disk, so the
+# noise range can be extended without recomputing anything. Levels already
+# present are ignored (the base sweep wins). Leave empty when not extending.
+extra_noise_levels = []
 # Monte-Carlo forward passes used to estimate Bayesian predictive/model
 # uncertainty (step 5/6). Ignored for the deterministic ensemble.
 mc_passes = 30

+ 2 - 0
evaluation/__init__.py

@@ -15,6 +15,7 @@ from evaluation.schema import (
     build_evaluation_dataset,
     compute_uncertainty,
     concat_evaluations,
+    concat_noise_levels,
     load_evaluation,
     save_evaluation,
 )
@@ -28,6 +29,7 @@ __all__ = [
     "build_evaluation_dataset",
     "compute_uncertainty",
     "concat_evaluations",
+    "concat_noise_levels",
     "load_evaluation",
     "save_evaluation",
 ]

+ 35 - 0
evaluation/schema.py

@@ -332,6 +332,41 @@ def concat_evaluations(
     return combined
 
 
+def concat_noise_levels(
+    datasets: Sequence[xr.Dataset], attrs: Optional[Mapping[str, Any]] = None
+) -> xr.Dataset:
+    """Union evaluations of the SAME samples that differ only in noise level.
+
+    This is what lets a noise sweep be extended without recomputing it: a later
+    run evaluates additional sigmas over the same split and models, and the
+    result is merged onto the existing ``noise_level`` axis here.
+
+    The parts must share the ``model``, ``sample`` and ``class_name`` axes; only
+    ``noise_level`` grows. The result is sorted ascending by sigma (plots and
+    curve fits assume a monotonic axis) and duplicate levels are dropped, keeping
+    the first occurrence -- so re-running an already-computed sigma is harmless.
+
+    Args:
+        datasets: Evaluations over the same samples at different noise levels.
+        attrs: Attributes merged over the result.
+    """
+    parts = list(datasets)
+    if not parts:
+        raise ValueError("concat_noise_levels requires at least one dataset.")
+
+    combined = xr.concat(parts, dim="noise_level")
+    levels = np.round(
+        np.asarray(combined["noise_level"].values, dtype=float), 12
+    )
+    # np.unique returns the index of the FIRST occurrence of each value.
+    _, first_occurrence = np.unique(levels, return_index=True)
+    combined = combined.isel(noise_level=np.sort(first_occurrence))
+    combined = combined.sortby("noise_level")
+    if attrs:
+        combined.attrs.update(dict(attrs))
+    return combined
+
+
 # ---------------------------------------------------------------------------
 # Incremental builder
 # ---------------------------------------------------------------------------

+ 47 - 0
tasks/__init__.py

@@ -24,6 +24,7 @@ from typing import Any, Callable, Dict, List, Tuple, TypedDict
 
 from . import analyze
 from . import evaluate
+from . import extra_info
 from . import load_data
 from . import load_models
 from . import train_bayesian
@@ -82,6 +83,18 @@ PIPELINE_TASKS: Dict[str, TaskEntry] = {
         "task_name": "Evaluate Models on Noised Data (Validation)",
         "task_func": evaluate.evaluate_noisy_val_task,
     },
+    "evaluate_noisy_extra": {
+        "task_name": "Evaluate Additional Noise Levels",
+        "task_func": evaluate.evaluate_noisy_extra_task,
+    },
+    "evaluate_noisy_extra_val": {
+        "task_name": "Evaluate Additional Noise Levels (Validation)",
+        "task_func": evaluate.evaluate_noisy_extra_val_task,
+    },
+    "extra_info": {
+        "task_name": "Generate Extra Information",
+        "task_func": extra_info.extra_info_task,
+    },
     "combine_evaluations": {
         "task_name": "Combine Split Evaluations",
         "task_func": evaluate.combine_evaluations_task,
@@ -143,6 +156,9 @@ TASK_WRITES = {
     "evaluate_regular_val": ("evaluations",),
     "evaluate_bayesian_val": ("evaluations",),
     "evaluate_noisy_val": ("evaluations",),
+    "evaluate_noisy_extra": ("evaluations",),
+    "evaluate_noisy_extra_val": ("evaluations",),
+    "extra_info": ("analysis",),
     "combine_evaluations": ("evaluations",),
     "run_analysis": ("analysis",),
     # k-fold writes per-fold models+evals under folds/ and the aggregated OOF
@@ -209,6 +225,7 @@ SCENARIOS: Dict[str, ScenarioEntry] = {
             *_EVAL_ALL_SPLITS,
             *_EVAL_ALL_SPLITS_NOISY,
             "combine_evaluations",
+            "extra_info",
             "run_analysis",
         ],
     },
@@ -220,6 +237,7 @@ SCENARIOS: Dict[str, ScenarioEntry] = {
             *_EVAL_ALL_SPLITS,
             *_EVAL_ALL_SPLITS_NOISY,
             "combine_evaluations",
+            "extra_info",
             "run_analysis",
         ],
     },
@@ -230,6 +248,35 @@ SCENARIOS: Dict[str, ScenarioEntry] = {
             "load_models",
             *_EVAL_ALL_SPLITS,
             "combine_evaluations",
+            "extra_info",
+            "run_analysis",
+        ],
+    },
+    # Extend an existing noise sweep: computes ONLY the extra sigmas, merges them
+    # onto the precomputed sweep, and re-analyses. No retraining, no recompute of
+    # the noise levels already on disk.
+    "scen_extend_noise": {
+        "label": "6. Extend Noise Sweep (extra sigmas + merge + analyze)",
+        "tasks": [
+            "load_data",
+            "load_models",
+            "evaluate_noisy_extra",
+            "evaluate_noisy_extra_val",
+            "combine_evaluations",
+            "extra_info",
+            "run_analysis",
+        ],
+    },
+    # Re-run the analysis over evaluations already on disk. No GPU work: it only
+    # re-merges the per-split / base+extra evaluation files, regenerates the
+    # extra-info figures (which need the dataset, hence load_data), and redraws
+    # every analysis. Use after changing an analysis parameter.
+    "scen_reanalyze": {
+        "label": "7. Re-analyze Existing Evaluations (merge + extras + analysis)",
+        "tasks": [
+            "load_data",
+            "combine_evaluations",
+            "extra_info",
             "run_analysis",
         ],
     },

+ 109 - 39
tasks/evaluate.py

@@ -35,10 +35,13 @@ import numpy as np
 import toml
 import torch
 
+import xarray as xr
+
 from evaluation.schema import (
     CLASS_NAMES,
     EvaluationAccumulator,
     concat_evaluations,
+    concat_noise_levels,
     load_evaluation,
     save_evaluation,
 )
@@ -286,14 +289,19 @@ def _out_path(config: Dict[str, Any], name: str) -> pl.Path:
     return pl.Path(config["work_dir"]) / _EVAL_SUBDIR / name
 
 
-def eval_filename(split: str, kind: str, noisy: bool = False) -> str:
-    """Canonical evaluation filename for a (split, kind, noisy) combination.
+def eval_filename(
+    split: str, kind: str, noisy: bool = False, extra: bool = False
+) -> str:
+    """Canonical evaluation filename for a (split, kind, noisy, extra) combination.
 
     The test split keeps its historical un-prefixed names (``normal.nc``,
     ``noisy_bayesian.nc``); other splits are prefixed (``val_normal.nc``).
+    ``extra`` marks an *additional* noise sweep computed in a later run
+    (``extra_noisy_normal.nc``), which is merged onto the base sweep's
+    ``noise_level`` axis by :func:`combine_evaluations_task`.
     """
     prefix = "" if split == SPLIT_TEST else f"{split}_"
-    return f"{prefix}{'noisy_' if noisy else ''}{kind}.nc"
+    return f"{prefix}{'extra_' if extra else ''}{'noisy_' if noisy else ''}{kind}.nc"
 
 
 def _evaluate_clean(
@@ -331,9 +339,25 @@ def _evaluate_noisy(
     config: Dict[str, Any],
     state: Dict[str, Any],
     split: str,
+    extra: bool = False,
 ) -> None:
-    """Evaluate both ensembles across the noise sweep for ``split`` (step 6)."""
+    """Evaluate both ensembles across a noise sweep for ``split`` (step 6).
+
+    With ``extra=True`` the sweep uses ``evaluation.extra_noise_levels`` and is
+    written to separate ``extra_*`` files, so additional sigmas can be computed
+    later and merged without recomputing the original sweep.
+    """
     noise_levels, mc_passes = _eval_cfg(config)
+    if extra:
+        noise_levels = list(
+            config.get("evaluation", {}).get("extra_noise_levels", [])
+        )
+        if not noise_levels:
+            log.error(
+                "No evaluation.extra_noise_levels configured; nothing to add. "
+                "Set them in config.toml to extend the sweep."
+            )
+            return
     any_models = False
     for kind, key, n_passes in (
         (KIND_NORMAL, "normal_model_paths", 1),
@@ -353,7 +377,9 @@ def _evaluate_noisy(
             track=track,
             noise_levels=noise_levels,
             n_passes=n_passes,
-            out_path=_out_path(config, eval_filename(split, kind, noisy=True)),
+            out_path=_out_path(
+                config, eval_filename(split, kind, noisy=True, extra=extra)
+            ),
             split=split,
         )
     if not any_models:
@@ -393,47 +419,85 @@ def evaluate_noisy_val_task(track, log, config, state) -> None:
     _evaluate_noisy(track, log, config, state, SPLIT_VAL)
 
 
+def evaluate_noisy_extra_task(track, log, config, state) -> None:
+    """ADDITIONAL noise levels on the TEST split (extends an existing sweep)."""
+    _evaluate_noisy(track, log, config, state, SPLIT_TEST, extra=True)
+
+
+def evaluate_noisy_extra_val_task(track, log, config, state) -> None:
+    """ADDITIONAL noise levels on the VALIDATION split."""
+    _evaluate_noisy(track, log, config, state, SPLIT_VAL, extra=True)
+
+
 def combine_evaluations_task(
     track: ProgressTracker,
     log: PipelineLogger,
     config: Dict[str, Any],
     state: Dict[str, Any],
 ) -> None:
-    """Pool per-split evaluations into ``combined_*.nc`` for analysis.
+    """Merge per-split evaluations into ``combined_*.nc`` for analysis.
+
+    Two independent merges happen here:
 
-    Concatenates the configured splits (default test + val) along the ``sample``
-    axis. The splits are patient-disjoint by construction, so pooling them yields
-    a larger evaluation set with no duplicated samples -- which is the point:
-    more evaluation samples means tighter accuracy/uncertainty estimates.
-    Analysis prefers these pooled files over the single-split ones.
+    1. **Noise levels** (``concat_noise_levels``) -- within one split, the base
+       sweep and any ``extra_*`` sweep are unioned along ``noise_level``. This is
+       what lets extra sigmas be computed later and folded in without recomputing
+       the original sweep.
+    2. **Splits** (``concat_evaluations``) -- the configured splits (default test
+       + val) are pooled along ``sample``. They are patient-disjoint by
+       construction, so pooling enlarges the evaluation set with no duplicated
+       samples, giving tighter accuracy/uncertainty estimates.
+
+    Analysis prefers the resulting ``combined_*`` files over single-split ones.
     """
     eval_dir = pl.Path(config["work_dir"]) / _EVAL_SUBDIR
     splits = list(
         config.get("evaluation", {}).get("combine_splits", [SPLIT_TEST, SPLIT_VAL])
     )
-    if len(splits) < 2:
-        log.info(
-            f"combine_evaluations: only {splits} configured; nothing to pool."
-        )
-        return
 
     written = 0
     for kind in (KIND_NORMAL, KIND_BAYESIAN):
         for noisy in (False, True):
-            paths = [
-                eval_dir / eval_filename(split, kind, noisy=noisy) for split in splits
-            ]
-            present = [p for p in paths if p.exists()]
-            if len(present) < 2:
-                continue
-
-            parts = [load_evaluation(str(p)) for p in present]
+            per_split: List[xr.Dataset] = []
+            sources: List[str] = []
+            merged_noise = False
             try:
-                # Guard against a mis-specified split: the same patient must not
-                # appear in two pooled parts (that would double-count them).
+                for split in splits:
+                    # Stage 1: base + extra sweeps for this split.
+                    candidates = [
+                        eval_dir / eval_filename(split, kind, noisy=noisy),
+                    ]
+                    if noisy:
+                        candidates.append(
+                            eval_dir
+                            / eval_filename(split, kind, noisy=True, extra=True)
+                        )
+                    present = [p for p in candidates if p.exists()]
+                    if not present:
+                        continue
+
+                    parts = [load_evaluation(str(p)) for p in present]
+                    if len(parts) > 1:
+                        merged = concat_noise_levels(parts)
+                        for part in parts:
+                            part.close()
+                        merged_noise = True
+                    else:
+                        merged = parts[0]
+                    per_split.append(merged)
+                    sources.extend(p.name for p in present)
+
+                if not per_split:
+                    continue
+                # Nothing gained from writing a copy of a single source.
+                if len(per_split) < 2 and not merged_noise:
+                    continue
+
+                # Guard: the same patient must never appear in two pooled splits
+                # (that would double-count them).
                 seen: set[str] = set()
                 overlap: set[str] = set()
-                for part in parts:
+                for part in per_split:
                     part_ptids = {str(p) for p in np.atleast_1d(part["ptid"].values)}
                     overlap |= seen & part_ptids
                     seen |= part_ptids
@@ -445,28 +509,34 @@ def combine_evaluations_task(
                     )
                     continue
 
-                combined = concat_evaluations(
-                    parts,
-                    attrs={
-                        "split": "+".join(splits),
-                        "combined_from": ", ".join(p.name for p in present),
-                        "n_sources": len(present),
-                    },
+                # Stage 2: pool the splits along `sample`.
+                combined = (
+                    concat_evaluations(per_split)
+                    if len(per_split) > 1
+                    else per_split[0]
+                )
+                combined.attrs.update(
+                    {
+                        "split": "+".join(splits[: len(per_split)]),
+                        "combined_from": ", ".join(sources),
+                        "n_sources": len(sources),
+                    }
                 )
                 out_name = f"combined_{'noisy_' if noisy else ''}{kind}.nc"
                 save_evaluation(combined, str(eval_dir / out_name))
                 written += 1
                 log.info(
                     f"combine_evaluations: wrote {out_name} "
-                    f"({combined.sizes['sample']} samples from "
-                    f"{', '.join(p.name for p in present)})."
+                    f"({combined.sizes['sample']} samples x "
+                    f"{combined.sizes['noise_level']} noise level(s) from "
+                    f"{', '.join(sources)})."
                 )
             finally:
-                for part in parts:
+                for part in per_split:
                     part.close()
 
     if written == 0:
         log.error(
-            "combine_evaluations: found no split pairs to pool. Run the test and "
-            "validation evaluation steps first."
+            "combine_evaluations: nothing to merge. Run the test/validation "
+            "evaluation steps (and any extra noise steps) first."
         )

+ 278 - 0
tasks/extra_info.py

@@ -0,0 +1,278 @@
+"""Extra information step -- figures and facts that are NOT in the evaluations.
+
+The analysis stage reads only the evaluation netCDF files, so anything that needs
+the *data itself* (split composition, what a noised image actually looks like)
+has to be produced here, while the dataloaders are still in ``state``. Runs after
+``load_data`` and before ``run_analysis``, writing into the same regeneratable
+``analysis/`` directory.
+
+Adding an extra: write a function taking an :class:`ExtraContext` and register it
+in :data:`EXTRAS`.
+"""
+
+import json
+import pathlib as pl
+from dataclasses import dataclass
+from typing import Any, Callable, Dict, List, Sequence
+
+import matplotlib
+
+matplotlib.use("Agg")  # headless; runs in a background worker thread
+
+import matplotlib.pyplot as plt  # noqa: E402
+import numpy as np  # noqa: E402
+import torch  # noqa: E402
+
+from analysis.plotting import data_dir, plots_dir  # noqa: E402
+from evaluation import CLASS_NAMES  # noqa: E402
+from util.progress import ProgressTracker  # noqa: E402
+from util.ui_logger import PipelineLogger  # noqa: E402
+
+ANALYSIS_SUBDIR = "analysis"
+
+#: Splits summarised, in pipeline order.
+_SPLITS = ("train", "val", "test")
+
+#: Example images shown in the noise-degradation grid.
+_N_EXAMPLES = 3
+
+#: Maximum noise columns in that grid (levels are subsampled evenly to fit).
+_MAX_NOISE_COLUMNS = 8
+
+
+@dataclass
+class ExtraContext:
+    """What an extra-info producer gets: pipeline state, config, output dir."""
+
+    config: Dict[str, Any]
+    state: Dict[str, Any]
+    out_dir: pl.Path
+    log: PipelineLogger
+
+
+ExtraFn = Callable[[ExtraContext], None]
+
+
+def _split_arrays(loader) -> tuple[np.ndarray, np.ndarray]:
+    """Return ``(image_ids, class_indices)`` for a loader WITHOUT loading images.
+
+    Reads straight from the underlying ``ADNIDataset`` through the ``Subset``
+    indices, so summarising the (large) training split costs nothing.
+    """
+    subset = loader.dataset
+    base = getattr(subset, "dataset", subset)
+    indices = list(getattr(subset, "indices", range(len(base))))
+    image_ids = np.array([int(base.filename_ids[i]) for i in indices], dtype=int)
+    classes = base.expected_classes[indices].argmax(dim=1).cpu().numpy()
+    return image_ids, classes
+
+
+def dataset_summary(ctx: ExtraContext) -> None:
+    """Split sizes, class balance and patient counts for every split."""
+    image_to_ptid: Dict[int, str] = ctx.state.get("image_to_ptid", {})
+
+    summary: Dict[str, Any] = {
+        "seed": ctx.config["data"].get("seed"),
+        "data_splits": ctx.config["data"].get("data_splits"),
+        "class_names": list(CLASS_NAMES),
+        "splits": {},
+    }
+    all_patients: Dict[str, set[str]] = {}
+
+    for split in _SPLITS:
+        loader = ctx.state.get(f"{split}_loader")
+        if loader is None:
+            continue
+        image_ids, classes = _split_arrays(loader)
+        patients = {image_to_ptid.get(int(i), "unknown") for i in image_ids}
+        all_patients[split] = patients
+        summary["splits"][split] = {
+            "n_scans": int(image_ids.size),
+            "n_patients": len(patients),
+            "scans_per_patient": round(image_ids.size / max(len(patients), 1), 2),
+            "class_counts": {
+                name: int(np.sum(classes == index))
+                for index, name in enumerate(CLASS_NAMES)
+            },
+            "class_fractions": {
+                name: round(float(np.mean(classes == index)), 4)
+                for index, name in enumerate(CLASS_NAMES)
+            },
+        }
+
+    if not summary["splits"]:
+        ctx.log.error("dataset_summary: no dataloaders in state; skipping.")
+        return
+
+    # Leakage check: the patient-grouped split must keep splits disjoint.
+    leaks: Dict[str, int] = {}
+    names = list(all_patients)
+    for i, a in enumerate(names):
+        for b in names[i + 1 :]:
+            shared = all_patients[a] & all_patients[b] - {"unknown"}
+            if shared:
+                leaks[f"{a}&{b}"] = len(shared)
+    summary["patient_overlap_between_splits"] = leaks
+    if leaks:
+        ctx.log.error(f"dataset_summary: PATIENT LEAKAGE between splits: {leaks}")
+
+    (data_dir(ctx.out_dir) / "dataset_summary.json").write_text(
+        json.dumps(summary, indent=2)
+    )
+
+    # Figure: scans per split, stacked by class.
+    splits = list(summary["splits"])
+    fig, axes = plt.subplots(1, 2, figsize=(10.0, 3.6), layout="constrained")
+    bottom = np.zeros(len(splits))
+    for index, name in enumerate(CLASS_NAMES):
+        heights = np.array(
+            [summary["splits"][s]["class_counts"][name] for s in splits], dtype=float
+        )
+        axes[0].bar(
+            splits, heights, bottom=bottom, label=name,
+            color=["#2a78d6", "#e34948"][index % 2],
+        )
+        bottom += heights
+    axes[0].set_ylabel("Scans")
+    axes[0].set_title("Split size and class balance", fontsize=10)
+    axes[0].legend(frameon=False)
+
+    axes[1].bar(
+        splits,
+        [summary["splits"][s]["n_patients"] for s in splits],
+        color="#7a7a7a",
+    )
+    axes[1].set_ylabel("Patients")
+    axes[1].set_title("Distinct patients per split", fontsize=10)
+
+    for ax in axes:
+        ax.grid(True, axis="y", color="#000000", alpha=0.08, linewidth=0.7)
+        for spine in ("top", "right"):
+            ax.spines[spine].set_visible(False)
+    fig.savefig(
+        plots_dir(ctx.out_dir) / "dataset_summary.png", dpi=150, bbox_inches="tight"
+    )
+    plt.close(fig)
+
+    parts = ", ".join(
+        f"{s}={summary['splits'][s]['n_scans']} scans/"
+        f"{summary['splits'][s]['n_patients']} patients"
+        for s in splits
+    )
+    ctx.log.info(f"dataset_summary: {parts}. Wrote dataset_summary.png/.json.")
+
+
+def _noise_columns(config: Dict[str, Any]) -> List[float]:
+    """Noise levels to show, evenly subsampled to at most _MAX_NOISE_COLUMNS."""
+    eval_cfg = config.get("evaluation", {})
+    levels = sorted(
+        {float(v) for v in eval_cfg.get("noise_levels", [0.0])}
+        | {float(v) for v in eval_cfg.get("extra_noise_levels", [])}
+    )
+    if len(levels) <= _MAX_NOISE_COLUMNS:
+        return levels
+    picks = np.linspace(0, len(levels) - 1, _MAX_NOISE_COLUMNS).round().astype(int)
+    return [levels[i] for i in sorted(set(picks.tolist()))]
+
+
+def noise_examples(ctx: ExtraContext) -> None:
+    """Grid of example images degrading across the configured noise levels.
+
+    Applies exactly the same perturbation the evaluation uses (relative to each
+    image's own intensity std when ``noise_relative``), so the figure shows what
+    the model actually saw at each sigma.
+    """
+    loader = ctx.state.get("test_loader")
+    if loader is None:
+        ctx.log.error("noise_examples: no test_loader in state; skipping.")
+        return
+
+    subset = loader.dataset
+    base = getattr(subset, "dataset", subset)
+    indices = list(getattr(subset, "indices", range(len(base))))[:_N_EXAMPLES]
+    if not indices:
+        ctx.log.error("noise_examples: split is empty; skipping.")
+        return
+
+    levels = _noise_columns(ctx.config)
+    relative = bool(ctx.config.get("evaluation", {}).get("noise_relative", True))
+    seed = int(ctx.config["data"].get("seed", 0) or 0)
+
+    fig, axes = plt.subplots(
+        len(indices),
+        len(levels),
+        figsize=(1.55 * len(levels), 1.75 * len(indices)),
+        squeeze=False,
+        layout="constrained",
+    )
+
+    for row, dataset_index in enumerate(indices):
+        volume = base.mri_data[dataset_index].float()  # (C, D, H, W)
+        image_id = int(base.filename_ids[dataset_index])
+        label = CLASS_NAMES[int(base.expected_classes[dataset_index].argmax())]
+        std = float(volume.std())
+        # A mid-axial slice, chosen once so every column shows the same anatomy.
+        slice_index = volume.shape[-1] // 2
+        generator = torch.Generator().manual_seed(seed + dataset_index)
+        clean_slice = volume[0, :, :, slice_index]
+        # Fix the display window on the CLEAN image so added noise visibly
+        # washes the image out instead of being renormalised away.
+        vmin, vmax = float(clean_slice.min()), float(clean_slice.max())
+
+        for col, sigma in enumerate(levels):
+            noisy = volume
+            if sigma > 0:
+                noise = torch.randn(volume.shape, generator=generator)
+                noisy = volume + noise * (sigma * (std if relative else 1.0))
+            ax = axes[row][col]
+            ax.imshow(
+                noisy[0, :, :, slice_index].numpy().T,
+                cmap="gray", vmin=vmin, vmax=vmax, origin="lower",
+            )
+            ax.set_xticks([])
+            ax.set_yticks([])
+            if row == 0:
+                ax.set_title(f"σ = {sigma:g}", fontsize=9)
+            if col == 0:
+                ax.set_ylabel(f"{image_id}\n({label})", fontsize=8)
+
+    unit = "× image std" if relative else "raw voxel units"
+    fig.suptitle(f"Image degradation with added Gaussian noise (σ in {unit})")
+    fig.savefig(
+        plots_dir(ctx.out_dir) / "noise_examples.png", dpi=150, bbox_inches="tight"
+    )
+    plt.close(fig)
+    ctx.log.info(
+        f"noise_examples: wrote noise_examples.png "
+        f"({len(indices)} images x {len(levels)} noise levels)."
+    )
+
+
+#: Registered extra-info producers, in run order.
+EXTRAS: Dict[str, ExtraFn] = {
+    "dataset_summary": dataset_summary,
+    "noise_examples": noise_examples,
+}
+
+
+def extra_info_task(
+    track: ProgressTracker,
+    log: PipelineLogger,
+    config: Dict[str, Any],
+    state: Dict[str, Any],
+) -> None:
+    """Run every registered extra-info producer."""
+    out_dir = pl.Path(config["work_dir"]) / ANALYSIS_SUBDIR
+    out_dir.mkdir(parents=True, exist_ok=True)
+    ctx = ExtraContext(config=config, state=state, out_dir=out_dir, log=log)
+
+    track.update(total=len(EXTRAS), advance=0)
+    for name, fn in EXTRAS.items():
+        log.info(f"Extra info: {name}")
+        try:
+            fn(ctx)
+        except Exception as exc:  # noqa: BLE001 - one failure must not stop the rest
+            log.error(f"Extra info '{name}' failed: {exc}")
+            log.file_logger.error(f"Extra info '{name}' failed", exc_info=True)
+        track.update(advance=1)
+    log.info(f"Extra information written to {out_dir}/.")