File indexing completed on 2026-10-08 08:36:03
0001
0002 """Check Yallfile graph structure without ROOT, raw data, or a scheduler.
0003
0004 Run with a yall-run checkout containing yall-run PR #33's partial ``@each``
0005 binding support (merged as 1081e9dd39418262588248272618130ce0503b8a).
0006 """
0007 from pathlib import Path
0008 import os
0009 import re
0010 import tempfile
0011 import unittest
0012 from unittest.mock import patch
0013
0014 from yall_run.model import load_spec
0015
0016 EXAMPLES = Path(__file__).resolve().parent
0017 EXPECTED = {'scan-set-1': 55, 'scan-set-2': 187, 'lfhcal-simple': 9}
0018 FULLSET_EXPECTED = {
0019 'fullset-b1-repro': 25,
0020 'fullset-b2-repro': 19,
0021 'fullset-c1-repro': 24,
0022 'fullset-c2-repro': 18,
0023 'fullset-c3-repro': 17,
0024 'fullset-d1-repro': 29,
0025 'fullset-d2-repro': 17,
0026 'fullset-e1-repro': 18,
0027 'fullset-e2-repro': 17,
0028 'fullset-e3-repro': 18,
0029 'fullset-f1-repro': 17,
0030 'fullset-f2-repro': 18,
0031 'fullset-g1-repro': 19,
0032 'fullset-g2-repro': 20,
0033 }
0034 FULLSET_SUFFIX = {
0035 'fullset-b1-repro': 'b1',
0036 'fullset-b2-repro': 'b2',
0037 'fullset-c1-repro': 'c1',
0038 'fullset-c2-repro': 'c2',
0039 'fullset-c3-repro': 'c3',
0040 'fullset-d1-repro': 'd1',
0041 'fullset-d2-repro': 'd2',
0042 'fullset-e1-repro': 'e1',
0043 'fullset-e2-repro': 'e2',
0044 'fullset-e3-repro': 'e3',
0045 'fullset-f1-repro': 'f1',
0046 'fullset-f2-repro': 'f2',
0047 'fullset-g1-repro': 'g1',
0048 'fullset-g2-repro': 'g2',
0049 }
0050 RUNDB_NAME = 'DataTakingDB_TBSPSH2_202605_HGCROC.csv'
0051 HVSCAN_MUONS = ('194', '195', '196', '197', '198', '199', '200', '201', '202')
0052 HVSCAN_EXPECTED = 92
0053
0054
0055 def pair_rows(text):
0056 match = re.search(r'^@table pairs ped (run|mip):\n((?:[ \t]+[^\n]*\n)+)', text, re.M)
0057 if match is None:
0058 raise AssertionError('missing top-level pairs table')
0059 return match, [tuple(line.split()) for line in match[2].splitlines()]
0060
0061
0062 def run_rows(text):
0063 match = re.search(r'^@table runs type run:\n((?:[ \t]+[^\n]*\n)+)', text, re.M)
0064 if match is None:
0065 raise AssertionError('missing top-level typed runs table')
0066 return match, [tuple(line.split()) for line in match[1].splitlines()]
0067
0068
0069 class SharedConversionTests(unittest.TestCase):
0070 def setUp(self):
0071 self.temp = tempfile.TemporaryDirectory()
0072 self.addCleanup(self.temp.cleanup)
0073 self.env = patch.dict(os.environ, {
0074 'LFHCAL_DATA': str(Path(self.temp.name) / 'data'),
0075 'LFHCAL_WORK': str(Path(self.temp.name) / 'work'),
0076 'EIC_SHELL': '/not-executed/eic-shell',
0077 })
0078 self.env.start()
0079 self.addCleanup(self.env.stop)
0080
0081 def load_text(self, text):
0082 source = Path(self.temp.name) / 'Yallfile'
0083 source.write_text(text)
0084 return {t.name: t for t in load_spec(source).tasks}
0085
0086 def check_graph(self, which, text):
0087 _, pairs = pair_rows(text)
0088 tasks = self.load_text(text)
0089 runs = list(dict.fromkeys([p for p, _ in pairs] + [m for _, m in pairs]))
0090 pedestals = list(dict.fromkeys(p for p, _ in pairs))
0091 self.assertEqual([n for n in tasks if n.startswith('convert-')],
0092 ['convert-' + r for r in runs])
0093 self.assertEqual([n for n in tasks if n.startswith('pedestal-')],
0094 ['pedestal-' + p for p in pedestals])
0095 self.assertNotIn('converted', tasks)
0096 for run in runs:
0097 self.assertEqual(tasks['convert-' + run].parents, ())
0098 for ped in pedestals:
0099 self.assertEqual(tasks['pedestal-' + ped].parents, ('convert-' + ped,))
0100
0101 if which == 'lfhcal-simple':
0102 self.assertFalse(any(n.startswith('calibration-') for n in tasks))
0103 self.assertFalse(any(n.startswith('summary-') for n in tasks))
0104 else:
0105 stage = 'transfer'
0106 self.assertEqual([n for n in tasks if n.startswith(stage + '-')],
0107 [f'{stage}-{p}-{m}' for p, m in pairs])
0108 for ped, muon in pairs:
0109 self.assertEqual(tasks[f'{stage}-{ped}-{muon}'].parents,
0110 (f'pedestal-{ped}', f'convert-{muon}'))
0111 self.assertEqual(tasks[f'mip-{ped}-{muon}'].parents,
0112 (f'transfer-{ped}-{muon}',))
0113
0114
0115 owners = {ref.path: t.name for t in tasks.values() for ref in t.outputs}
0116 def ancestors(name):
0117 result = set()
0118 todo = list(tasks[name].parents)
0119 while todo:
0120 parent = todo.pop()
0121 if parent not in result:
0122 result.add(parent)
0123 todo.extend(tasks[parent].parents)
0124 return result
0125 for task in tasks.values():
0126 upstream = ancestors(task.name)
0127 for ref in task.inputs:
0128 if ref.path in owners:
0129 self.assertIn(owners[ref.path], upstream, (task.name, ref.path))
0130 return tasks
0131
0132 def check_fullset_graph(self, which, text):
0133 _, rows = run_rows(text)
0134 tasks = self.load_text(text)
0135 pedestal_runs = [run for run_type, run in rows if run_type == 'pedestal']
0136 muon_runs = [run for run_type, run in rows if run_type == 'muon']
0137 self.assertEqual(len(pedestal_runs), 1, (which, rows))
0138 self.assertTrue(muon_runs, (which, rows))
0139
0140 conversions = [f'convert-{run_type}-{run}' for run_type, run in rows]
0141 self.assertEqual([name for name in tasks if name.startswith('convert-')],
0142 conversions)
0143 for name in conversions:
0144 self.assertEqual(tasks[name].parents, (), (which, name))
0145
0146 muon_parents = tuple(f'convert-muon-{run}' for run in muon_runs)
0147 self.assertEqual(tasks['merge-muon'].parents, muon_parents)
0148 merge_inputs = tuple(Path(ref.path).name for ref in tasks['merge-muon'].inputs)
0149 self.assertEqual(merge_inputs,
0150 tuple(f'rawHGCROC_{run}.root' for run in muon_runs))
0151 self.assertNotIn(
0152 f'rawHGCROC_{pedestal_runs[0]}.root',
0153 merge_inputs,
0154 (which, 'pedestal conversion leaked into merge-muon'),
0155 )
0156
0157 pedestal_run = pedestal_runs[0]
0158 pedestal_name = f'pedestal-{pedestal_run}'
0159 self.assertEqual(tasks[pedestal_name].parents,
0160 (f'convert-pedestal-{pedestal_run}',))
0161 self.assertTrue(any(
0162 Path(ref.path).name == f'rawHGCROC_{pedestal_run}.root'
0163 for ref in tasks[pedestal_name].inputs
0164 ))
0165
0166 transfer = tasks[f'transfer-{FULLSET_SUFFIX[which]}']
0167 self.assertEqual(transfer.parents,
0168 ('merge-muon', pedestal_name))
0169 pedestal_outputs = {ref.path for ref in tasks[pedestal_name].outputs}
0170 transfer_inputs = {ref.path for ref in transfer.inputs}
0171 self.assertEqual(len(pedestal_outputs & transfer_inputs), 1,
0172 (which, pedestal_outputs, transfer_inputs))
0173 return tasks
0174
0175 def test_default_graphs(self):
0176 for which, count in EXPECTED.items():
0177 with self.subTest(example=which):
0178 text = (EXAMPLES / which / 'Yallfile').read_text()
0179 self.assertEqual(len(self.check_graph(which, text)), count)
0180
0181 def test_fullset_graphs(self):
0182 for which, count in FULLSET_EXPECTED.items():
0183 with self.subTest(example=which):
0184 text = (EXAMPLES / which / 'Yallfile').read_text()
0185 self.assertEqual(len(self.check_fullset_graph(which, text)), count)
0186
0187 def test_fullset_pedestal_dependencies_follow_run_table(self):
0188 for which, count in FULLSET_EXPECTED.items():
0189 with self.subTest(example=which):
0190 text = (EXAMPLES / which / 'Yallfile').read_text()
0191 match, rows = run_rows(text)
0192 old_run = next(run for run_type, run in rows
0193 if run_type == 'pedestal')
0194 new_run = str(int(old_run) + 1000)
0195 changed_rows = [
0196 (run_type, new_run if run_type == 'pedestal' else run)
0197 for run_type, run in rows
0198 ]
0199 replacement = ''.join(
0200 f' {run_type:<8} {run}\n'
0201 for run_type, run in changed_rows
0202 )
0203 text = text[:match.start(1)] + replacement + text[match.end(1):]
0204 tasks = self.check_fullset_graph(which, text)
0205 self.assertEqual(len(tasks), count)
0206 self.assertIn(f'convert-pedestal-{new_run}', tasks)
0207 self.assertIn(f'pedestal-{new_run}', tasks)
0208 self.assertNotIn(f'convert-pedestal-{old_run}', tasks)
0209 self.assertNotIn(f'pedestal-{old_run}', tasks)
0210
0211 def test_hvscan_graph(self):
0212 text = (EXAMPLES / 'hvscan-repro' / 'Yallfile').read_text()
0213 tasks = self.load_text(text)
0214 self.assertEqual(len(tasks), HVSCAN_EXPECTED)
0215 self.assertEqual(tasks['pedestal-188'].parents, ('convert-pedestal-188',))
0216 self.assertEqual(
0217 [name for name in tasks if name.startswith('convert-')],
0218 ['convert-pedestal-188'] + [f'convert-muon-{run}' for run in HVSCAN_MUONS],
0219 )
0220 for run in HVSCAN_MUONS:
0221 self.assertEqual(
0222 tasks[f'transfer-muon-{run}'].parents,
0223 (f'convert-muon-{run}', 'pedestal-188'),
0224 )
0225 self.assertEqual(
0226 tasks[f'final-muon-{run}'].parents,
0227 (f'refine5-muon-{run}',),
0228 )
0229
0230 def test_shared_pedestal_is_converted_and_fitted_once(self):
0231 for which, count in EXPECTED.items():
0232 with self.subTest(example=which):
0233 text = (EXAMPLES / which / 'Yallfile').read_text()
0234 match, rows = pair_rows(text)
0235 rows[1] = (rows[0][0], rows[1][1])
0236 text = text[:match.start(2)] + ''.join(f' {p} {m}\n' for p, m in rows) + text[match.end(2):]
0237 self.assertEqual(len(self.check_graph(which, text)), count - 2)
0238
0239 def test_run_in_both_columns_is_converted_once(self):
0240 for which, count in EXPECTED.items():
0241 with self.subTest(example=which):
0242 text = (EXAMPLES / which / 'Yallfile').read_text()
0243 match, rows = pair_rows(text)
0244 rows[1] = (rows[0][1], rows[1][1])
0245 text = text[:match.start(2)] + ''.join(f' {p} {m}\n' for p, m in rows) + text[match.end(2):]
0246 self.assertEqual(len(self.check_graph(which, text)), count - 1)
0247
0248 def test_run_based_output_names_still_reject_multiple_calibrations_of_one_muon(self):
0249 for which in ('scan-set-1', 'scan-set-2'):
0250 with self.subTest(example=which):
0251 text = (EXAMPLES / which / 'Yallfile').read_text()
0252 match, rows = pair_rows(text)
0253 rows[1] = (rows[1][0], rows[0][1])
0254 text = text[:match.start(2)] + ''.join(f' {p} {m}\n' for p, m in rows) + text[match.end(2):]
0255 with self.assertRaisesRegex(ValueError, 'owned by both'):
0256 self.load_text(text)
0257
0258 def test_dataprep_stages_pass_run_database(self):
0259 cases = {
0260 'lfhcal-simple': ('pedestal-296',),
0261 'calibration-pair': ('pedestal', 'mip'),
0262 }
0263 for which, names in cases.items():
0264 with self.subTest(example=which):
0265 text = (EXAMPLES / which / 'Yallfile').read_text()
0266 tasks = self.load_text(text)
0267 for name in names:
0268 task = tasks[name]
0269 command = task.command if isinstance(task.command, str) else ' '.join(task.command)
0270 self.assertRegex(command, r'(?:^|\s)-r\s+', (which, name, command))
0271 self.assertTrue(
0272 any(Path(ref.path).name == RUNDB_NAME for ref in task.inputs),
0273 (which, name),
0274 )
0275
0276
0277 if __name__ == '__main__':
0278 unittest.main()