Add split miner implementation

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-03-12 07:03:52 +08:00
co-authored by Claude Opus 4.6
parent 94c97629de
commit 446e6f90fe
27 changed files with 6281 additions and 0 deletions
+3
View File
@@ -0,0 +1,3 @@
[
[["C", "F", "D", "G", "I", "E", "B", "J", "A", "H", "C", "F", "D", "G", "I", "E", "B", "J", "A", "H", "C", "F", "D", "G", "I", "E", "B", "J", "A", "H", "C", "F", "D", "C", "F", "D"], 1]
]
+5
View File
@@ -0,0 +1,5 @@
[
[["C", "F", "D", "G", "J", "E", "B", "K", "A", "I"], 10],
[["C", "F", "D"], 5],
[["C", "F", "D", "G", "J", "E", "B", "K", "H"], 3]
]
+216
View File
@@ -0,0 +1,216 @@
[
[["D", "B", "I", "F", "H", "J", "C"], 1774],
[["D", "B", "I", "F", "H", "J", "A", "C"], 736],
[["J", "C", "H", "D", "B", "I", "F"], 252],
[["D", "B", "I", "H", "F", "J", "A", "C"], 86],
[["J", "C", "H"], 432],
[["J", "D", "B", "I", "F", "H", "A", "C"], 210],
[["J", "C", "D", "B", "I", "F", "H"], 1150],
[["H", "D", "B", "I", "F", "J", "A", "C"], 80],
[["D", "B", "H", "I", "F", "J", "A", "C"], 50],
[["J", "A", "C", "H", "D", "B", "I", "F"], 61],
[["C", "H", "J"], 194],
[["J", "A", "C", "D", "B", "I", "H", "F"], 30],
[["C", "D", "B", "I", "F", "H", "J"], 684],
[["D", "B", "I", "F", "H", "J", "E", "K", "C"], 23],
[["J", "C", "D", "B", "H", "I", "F"], 101],
[["J", "A", "C", "D", "B", "I", "F", "H"], 187],
[["H", "J", "C"], 606],
[["H", "J", "G", "A", "C"], 6],
[["J", "A", "C", "H"], 80],
[["C", "D", "B", "I", "H", "F", "J"], 53],
[["J", "H", "A", "C"], 87],
[["H", "D", "B", "I", "F", "J", "C"], 219],
[["J", "D", "B", "I", "F", "H", "C"], 16],
[["J", "E", "K", "C", "H"], 1],
[["D", "B", "H", "I", "F", "J", "C"], 96],
[["H", "J", "A", "C"], 258],
[["J", "G", "D", "B", "I", "F", "H", "A", "C"], 9],
[["J", "G", "H", "A", "C"], 2],
[["D", "H", "B", "I", "F", "J", "C"], 14],
[["A", "C", "D", "B", "I", "F", "H", "J"], 93],
[["D", "H", "B", "J", "I", "F", "A", "C"], 1],
[["H", "D", "B", "I", "F", "J", "E", "K", "C"], 1],
[["J", "C", "D", "B", "I", "H", "F"], 127],
[["D", "B", "I", "H", "F", "J", "C"], 194],
[["J", "E", "K", "A", "C", "D", "B", "I", "F", "H"], 6],
[["D", "B", "I", "F", "H", "J", "G", "A", "C"], 15],
[["D", "B", "I", "H", "J", "C"], 27],
[["J", "A", "C", "D", "B", "H", "I", "F"], 20],
[["D", "B", "I", "F", "C", "H", "J"], 8],
[["J", "A", "G", "D", "B", "I", "F", "H"], 1],
[["J", "C", "D", "B", "I", "H"], 15],
[["J", "D", "B", "H", "I", "F", "A", "C"], 16],
[["J", "C", "D", "H", "B", "I", "F"], 19],
[["D", "B", "I", "F", "H", "E", "J", "K", "C"], 6],
[["E", "K", "C", "D", "B", "I", "F", "H", "J"], 7],
[["H", "G", "J", "A", "C"], 1],
[["D", "B", "I", "F", "J", "C", "H"], 8],
[["H", "J", "C", "D", "B", "I", "F"], 13],
[["D", "B", "I", "H", "F", "J", "K", "E", "A", "C"], 1],
[["A", "C", "D", "H", "J", "B", "I", "F"], 1],
[["J", "E", "K", "C", "H", "D", "B", "I", "F"], 3],
[["D", "B", "I", "H", "J", "F", "C"], 6],
[["D", "B", "H", "J", "C"], 1],
[["J", "E", "K", "C", "D", "B", "I", "H", "F"], 1],
[["H", "J", "D", "B", "I", "F", "C"], 2],
[["J", "H", "D", "B", "I", "F", "A", "C"], 44],
[["C", "D", "B", "I", "H", "J"], 21],
[["J", "D", "H", "B", "I", "F", "A", "C"], 1],
[["D", "B", "I", "F", "H", "J", "A"], 18],
[["D", "B", "I", "F", "H", "J", "E", "C", "K"], 1],
[["J", "H", "C"], 6],
[["J", "G", "H", "D", "B", "I", "F", "A", "C"], 1],
[["A", "C", "H", "J"], 28],
[["D", "B", "I", "F", "J", "H", "A", "C"], 3],
[["J", "E", "K", "A", "C", "H", "D", "B", "I", "F"], 3],
[["J", "A", "H"], 2],
[["J", "E", "K", "H", "A", "C"], 1],
[["F", "H", "J", "C"], 6],
[["J", "E", "K", "C", "H", "F"], 1],
[["D", "B", "I", "H", "F", "J", "K", "E", "C"], 1],
[["H", "J", "G", "C"], 2],
[["D", "B", "I", "F", "H", "J", "E", "K", "A", "C"], 17],
[["H", "J", "A"], 8],
[["D", "B", "I", "F", "H", "J", "K", "E", "A", "C"], 2],
[["J", "E", "K", "D", "B", "I", "F", "H", "A", "C"], 6],
[["I", "F", "H", "J", "A", "C"], 2],
[["G", "J", "H", "A", "C"], 1],
[["H", "J", "C", "G"], 1],
[["J", "G", "A", "C", "H"], 4],
[["J", "G", "A", "C", "H", "D", "B", "I", "F"], 1],
[["J", "D", "B", "I", "H", "F", "A", "C"], 21],
[["J", "E", "K", "C", "D", "B", "I", "F", "H"], 5],
[["E", "K", "D", "B", "I", "F", "H", "J", "A", "C"], 8],
[["D", "B", "I", "F", "J", "A", "C", "H"], 1],
[["H", "D", "B", "I", "F", "K", "J", "E", "A", "C"], 1],
[["C", "H", "J", "D", "B", "I", "F"], 5],
[["K", "E", "D", "B", "I", "H", "F", "J", "A", "C"], 2],
[["E", "J", "K", "C", "H"], 3],
[["J", "E", "K", "A", "C", "H"], 2],
[["A", "C", "D", "B", "I", "H", "F", "J"], 7],
[["H", "J", "E", "K", "C"], 5],
[["J", "C", "H", "D", "B", "I"], 5],
[["G", "D", "B", "I", "F", "H", "J", "A", "C"], 5],
[["J", "H", "D", "B", "I", "A", "C"], 3],
[["J", "C", "F", "H"], 4],
[["J", "C", "H", "F"], 5],
[["G", "J", "D", "B", "I", "F", "H", "A", "C"], 2],
[["D", "B", "I", "H", "J", "A", "C"], 15],
[["K", "J", "E", "C", "D", "B", "I", "F", "H"], 2],
[["H", "J", "C", "I", "F"], 1],
[["I", "F", "H", "J", "C"], 2],
[["J", "G", "A", "C", "D", "B", "I", "F", "H"], 4],
[["E", "K", "C", "H", "J"], 2],
[["D", "B", "I", "H", "F", "J", "E", "K", "A", "C"], 5],
[["J", "G", "D", "B", "I", "H", "F", "A", "C"], 1],
[["K", "J", "E", "A", "C", "H", "D", "B", "I", "F"], 2],
[["H", "D", "B", "I", "F", "J", "E", "K", "A", "C"], 2],
[["J", "E", "K", "C", "D", "B", "H", "I", "F"], 2],
[["H", "J", "C", "F"], 2],
[["C", "F", "H", "J"], 1],
[["C", "I", "F", "H", "J"], 1],
[["J", "H", "A", "C", "I", "F"], 1],
[["E", "K", "A", "C", "D", "B", "I", "F", "H", "J"], 5],
[["C", "D", "B", "I", "H", "J", "F"], 2],
[["H", "D", "J", "B", "I", "F", "C"], 1],
[["F", "J", "C", "H"], 3],
[["J", "A", "D", "B", "I", "F", "H"], 5],
[["D", "H", "B", "I", "F", "J", "A", "C"], 2],
[["H", "D", "B", "I", "G", "J", "A", "C"], 1],
[["D", "B", "J", "I", "H", "F", "A", "C"], 1],
[["H", "D", "B", "I", "J", "C"], 4],
[["I", "H", "J", "C"], 1],
[["D", "H", "J", "B", "I", "C"], 1],
[["D", "H", "B", "I", "J", "C"], 1],
[["B", "D", "I", "F", "H", "J", "A", "C"], 1],
[["D", "B", "I", "F", "H", "K", "J", "E", "A"], 1],
[["D", "B", "I", "F", "H", "J", "G", "C"], 6],
[["H", "D", "B", "I", "F", "G", "J", "A", "C"], 2],
[["J", "A", "C", "H", "D", "B", "I"], 1],
[["F", "C", "H", "J"], 1],
[["E", "J", "K", "C", "D", "B", "I", "F", "H"], 7],
[["K", "E", "J", "C", "H", "D", "B", "I", "F"], 1],
[["C", "H", "J", "F"], 1],
[["H", "D", "B", "I", "F", "K", "J", "E", "C"], 1],
[["H", "J", "E", "K", "A", "C"], 8],
[["K", "H", "D", "B", "I", "F", "J", "E", "A", "C"], 1],
[["D", "B", "I", "F", "H", "G", "J", "A", "C"], 1],
[["K", "J", "E", "D", "B", "I", "F", "H", "A", "C"], 1],
[["D", "B", "I", "F", "K", "J", "E", "H", "A", "C"], 1],
[["D", "B", "I", "H", "J", "A"], 1],
[["E", "K", "H", "J", "A", "C"], 2],
[["D", "B", "I", "H", "F", "J", "A"], 6],
[["D", "B", "I", "F", "H", "K", "J", "E", "A", "C"], 4],
[["D", "B", "I", "F", "H", "J", "G", "E", "K", "A"], 1],
[["J", "A", "C", "D", "H", "B", "I", "F"], 2],
[["D", "B", "I", "F", "H", "J", "G", "A"], 2],
[["B", "C", "D", "I", "F", "H", "J"], 1],
[["E", "D", "B", "I", "F", "H", "J", "K", "C"], 2],
[["J", "A", "D", "B", "I", "H", "F"], 1],
[["J", "D", "B", "I", "H", "F", "A"], 1],
[["H", "J", "A", "C", "D", "B", "I", "F"], 3],
[["K", "J", "E", "A", "C", "D", "B", "I", "F", "H"], 2],
[["J", "G", "C", "D", "B", "I", "F", "H"], 4],
[["D", "B", "I", "H", "F", "J", "G", "A", "C"], 3],
[["D", "B", "I", "H", "J", "E", "K", "A", "C"], 1],
[["H", "J", "E", "K", "A"], 1],
[["J", "A", "C", "H", "F"], 1],
[["H", "F", "J", "A", "C"], 1],
[["J", "H", "A", "C", "D", "B", "I", "F"], 6],
[["H", "E", "J", "K", "C"], 1],
[["H", "J", "A", "C", "D", "B", "I"], 1],
[["D", "B", "H", "I", "F", "E", "J", "K", "C"], 1],
[["J", "G", "C", "D", "B", "I", "H"], 1],
[["H", "D", "B", "I", "J", "E", "K", "C"], 1],
[["D", "B", "I", "H", "F", "J", "E", "K", "C"], 2],
[["K", "J", "E", "A", "C", "D", "B", "I", "H", "F"], 1],
[["B", "J", "A", "C", "D", "I", "F", "H"], 1],
[["J", "D", "B", "I", "H", "F", "C"], 1],
[["K", "E", "A", "C", "D", "B", "I", "H", "F", "J"], 1],
[["K", "E", "C", "D", "B", "I", "F", "H", "J"], 1],
[["K", "E", "J", "C", "H"], 1],
[["A", "C", "D", "B", "H", "J"], 1],
[["K", "J", "E", "D", "B", "I", "H", "F", "A", "C"], 1],
[["J", "A", "C", "D", "B", "F", "I", "H"], 1],
[["J", "G", "D", "H", "B", "I", "F", "A", "C"], 1],
[["J", "C", "D", "B", "H"], 1],
[["D", "B", "H", "I", "J", "A", "C"], 1],
[["D", "B", "H", "I", "F", "J", "G", "A", "C"], 2],
[["H", "G", "J", "E", "K", "A", "C"], 2],
[["G", "J", "A", "C", "H", "D", "B", "I", "F"], 1],
[["J", "D", "B", "I", "H", "A", "C"], 1],
[["J", "E", "K", "A", "C", "D", "B", "I", "H"], 1],
[["H", "J", "K", "E", "C"], 2],
[["D", "B", "J", "I", "F", "H", "C"], 2],
[["J", "H", "C", "D", "B", "I", "F"], 1],
[["G", "J", "A", "C", "D", "B", "I", "F", "H"], 2],
[["J", "C", "G", "D", "B", "I", "F", "H"], 1],
[["J", "D", "B", "I", "F", "C", "H"], 1],
[["F", "A", "C", "H", "J"], 1],
[["D", "B", "I", "F", "H", "J", "G", "E", "K", "A", "C"], 1],
[["E", "J", "K", "C", "H", "D", "B", "I", "F"], 1],
[["D", "B", "I", "H", "J", "E", "K", "C"], 1],
[["H", "J", "E", "K", "C", "F"], 1],
[["J", "F", "H", "A", "C"], 1],
[["G", "A", "C", "D", "B", "I", "F", "H", "J"], 1],
[["F", "H", "J", "A", "C"], 1],
[["J", "H", "F", "A", "C"], 1],
[["J", "A", "D", "B", "I", "H"], 1],
[["J", "D", "B", "A", "C", "I", "F", "H"], 1],
[["J", "D", "B", "C", "I", "F", "H"], 1],
[["J", "A", "C", "D", "B", "I", "H"], 1],
[["D", "B", "I", "H", "F", "K", "J", "E", "C"], 1],
[["J", "A", "C", "D", "B", "H"], 1],
[["C", "D", "B", "H", "J"], 1],
[["D", "B", "I", "H", "J", "G", "E", "K", "A", "C"], 1],
[["A", "C", "D", "B", "I", "H", "J"], 2],
[["D", "B", "I", "F", "J", "H", "C"], 1],
[["E", "C", "K", "D", "B", "I", "F", "H", "J"], 1],
[["G", "C", "D", "B", "I", "F", "H", "J"], 1],
[["B", "C", "D", "I", "H", "F", "J"], 1],
[["D", "B", "I", "F", "H", "C", "J"], 1],
[["E", "J", "K", "C", "D", "B", "H", "I", "F"], 1],
[["J", "C", "I", "H"], 1],
[["J", "A", "C", "I", "H"], 1]
]
+8
View File
@@ -0,0 +1,8 @@
[
[["C", "L", "A", "D", "E", "H"], 1],
[["C", "L", "A", "D", "E", "J", "I", "K", "F"], 1],
[["C", "L", "G", "C", "G", "A", "C", "A", "D", "E", "J", "I", "K", "F"], 1],
[["C", "L", "G", "A", "D", "A", "C", "A", "D", "E", "J", "K", "I", "K", "I", "K", "F"], 1],
[["C", "L", "G", "A", "D", "A", "C", "A", "D", "E", "J", "I", "K", "B", "F"], 1],
[["C", "L", "A", "D", "E", "J", "K", "B", "F"], 1]
]
+129
View File
@@ -0,0 +1,129 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/10
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for the DFG construction with a 5-task example.
Uses a small event log with 5 tasks to verify DFG construction,
edge frequencies, and source/sink detection.
"""
from __future__ import annotations
import unittest
from split_miner.bpmn import Node, Task
from split_miner.dfg import DirectlyFollowsGraph
def _make_tasks(
labels: str,
) -> dict[str, Task]:
"""Create a Task for each single-character label.
:param labels: The labels as a string.
:return: A dict mapping label to Task.
"""
return {ch: Task(ch, ch) for ch in labels}
def _make_five_task_log() -> tuple[
dict[tuple[Node, ...], int], dict[str, Task]
]:
"""Build a 5-task event log.
L = {<a,b,c,d>^3, <a,c,b,d>^2, <a,e,d>^1}
:return: The event log and the task map.
"""
t: dict[str, Task] = _make_tasks("abcde")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["c"], t["d"]): 3,
(t["a"], t["c"], t["b"], t["d"]): 2,
(t["a"], t["e"], t["d"]): 1,
}
return traces, t
class TestFiveTaskDFG(unittest.TestCase):
"""Tests for DFG with 5 tasks (a, b, c, d, e)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_five_task_log()
self.__dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
def test_nodes(self) -> None:
"""DFG has the correct 5 nodes."""
self.assertEqual(
self.__dfg.nodes,
set(self.__t.values()),
)
def test_edge_frequencies(self) -> None:
"""All 8 edge frequencies are correct."""
t: dict[str, Task] = self.__t
expected: dict[tuple[Node, Node], int] = {
(t["a"], t["b"]): 3,
(t["a"], t["c"]): 2,
(t["a"], t["e"]): 1,
(t["b"], t["c"]): 3,
(t["b"], t["d"]): 2,
(t["c"], t["b"]): 2,
(t["c"], t["d"]): 3,
(t["e"], t["d"]): 1,
}
for (src, tgt), freq in expected.items():
self.assertEqual(
self.__dfg.df_frequency(src, tgt),
freq,
f"|{src.node_id} -> {tgt.node_id}|"
f" should be {freq}"
)
def test_edges(self) -> None:
"""DFG has the correct 8 edges."""
t: dict[str, Task] = self.__t
expected: set[tuple[Node, Node]] = {
(t["a"], t["b"]), (t["a"], t["c"]),
(t["a"], t["e"]),
(t["b"], t["c"]), (t["b"], t["d"]),
(t["c"], t["b"]), (t["c"], t["d"]),
(t["e"], t["d"]),
}
self.assertEqual(self.__dfg.edges, expected)
def test_sources(self) -> None:
"""Sources include a (first task of traces)."""
self.assertIn(
self.__t["a"], self.__dfg.sources
)
def test_sinks(self) -> None:
"""Sinks include d (last task of traces)."""
self.assertIn(
self.__t["d"], self.__dfg.sinks
)
if __name__ == "__main__":
unittest.main()
+67
View File
@@ -0,0 +1,67 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/10
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for empty and minimal event logs.
Verifies proper error handling when input data is
insufficient for the Split Miner pipeline.
"""
from __future__ import annotations
import unittest
from split_miner import split_miner
from split_miner.bpmn import Node, Task
from split_miner.dfg import DirectlyFollowsGraph
class TestEmptyLog(unittest.TestCase):
"""Tests for empty event log handling."""
def test_empty_log_dfg(self) -> None:
"""Empty log produces a DFG with no nodes."""
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph({})
)
self.assertEqual(dfg.nodes, set())
self.assertEqual(dfg.edges, set())
def test_empty_log_no_sources(self) -> None:
"""Empty DFG has no sources."""
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph({})
)
self.assertEqual(dfg.sources, set())
def test_empty_log_split_miner(self) -> None:
"""Split Miner handles empty log gracefully."""
model = split_miner({})
self.assertEqual(len(model.edges), 0)
def test_single_event_trace(self) -> None:
"""Single-event trace produces no edges."""
a: Task = Task("a", "a")
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph({(a,): 1})
)
self.assertEqual(dfg.nodes, {a})
self.assertEqual(dfg.edges, set())
if __name__ == "__main__":
unittest.main()
+196
View File
@@ -0,0 +1,196 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/11
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for the epsilon parameter.
Ported from the bpmn project's test_epsilon.py.
High epsilon values may break the graph.
"""
from __future__ import annotations
import unittest
from split_miner import split_miner
TRACES_1: dict[tuple[str, ...], int] = {
("a", "b", "f", "g", "i", "j", "k"): 1150,
("b", "f", "g", "i", "j", "k", "a"): 684,
("a", "b", "k"): 432,
("a", "b", "k", "f", "g", "i", "j"): 252,
("b", "k", "a"): 194,
("a", "b", "f", "g", "i", "k", "j"): 192,
("a", "h", "f", "g", "i", "j", "k", "b"): 190,
("a", "h", "b", "f", "g", "i", "j", "k"): 188,
("a", "h", "k", "b"): 80,
("a", "h", "b", "k"): 79,
("a", "h", "b", "k", "f", "g", "i", "j"): 61,
("b", "f", "g", "i", "k", "j", "a"): 53,
("a", "h", "b", "f", "g", "i", "k", "j"): 41,
("a", "h", "k", "f", "g", "i", "j", "b"): 40,
("a", "b", "f", "g", "k", "i", "j"): 36,
("a", "h", "f", "g", "i", "k", "j", "b"): 28,
("b", "f", "g", "i", "k", "a"): 21,
("a", "b", "f", "k", "g", "i", "j"): 19,
("a", "f", "g", "i", "j", "k", "h", "b"): 19,
("a", "f", "g", "i", "j", "k", "b"): 16,
("a", "b", "f", "g", "i", "k"): 15,
("a", "h", "b", "f", "g", "k", "i", "j"): 9,
("a", "k", "h", "b"): 7,
("a", "k", "b"): 6,
("a", "b", "k", "f", "g", "i"): 5,
("a", "b", "k", "j"): 5,
("a", "c", "e", "b", "f", "g", "i", "j",
"k"): 5,
("a", "h", "f", "g", "i", "j", "k"): 5,
("a", "h", "k", "b", "f", "g", "i", "j"): 5,
("b", "k", "a", "f", "g", "i", "j"): 5,
("a", "b", "j", "k"): 4,
("a", "d", "h", "f", "g", "i", "j", "k",
"b"): 4,
("a", "f", "g", "i", "k", "j", "h", "b"): 4,
("a", "h", "f", "g", "k", "i", "j", "b"): 4,
("a", "k", "f", "g", "i", "j", "h", "b"): 4,
("a", "c", "e", "b", "f", "g", "i", "k",
"j"): 3,
("a", "c", "e", "b", "k", "f", "g", "i",
"j"): 3,
("a", "c", "e", "h", "b", "f", "g", "i",
"j", "k"): 3,
("a", "c", "h", "e", "b", "f", "g", "i",
"j", "k"): 3,
("a", "d", "h", "b", "k"): 3,
("a", "h", "k", "f", "g", "i", "b"): 3,
("a", "c", "e", "f", "g", "i", "j", "k",
"h", "b"): 2,
("a", "c", "e", "h", "f", "g", "i", "j",
"k", "b"): 2,
("a", "c", "h", "e", "b", "k", "f", "g",
"i", "j"): 2,
("a", "c", "h", "e", "f", "g", "i", "j",
"k", "b"): 2,
("a", "d", "b", "f", "g", "i", "j", "k"): 2,
("a", "h", "b", "f", "k", "g", "i", "j"): 2,
("a", "h", "k"): 2,
("b", "f", "g", "i", "k", "a", "j"): 2,
("a", "b", "f", "g", "k"): 1,
("a", "b", "i", "k"): 1,
("a", "c", "e", "b", "k"): 1,
("a", "c", "e", "b", "k", "j"): 1,
("a", "c", "e", "h", "b", "f", "g", "i",
"k"): 1,
("a", "c", "e", "h", "b", "k"): 1,
("a", "c", "e", "h", "b", "k", "f", "g",
"i", "j"): 1,
("a", "c", "e", "h", "k", "b"): 1,
("a", "c", "h", "e", "b", "k"): 1,
("a", "d", "h", "b", "f", "g", "i", "j",
"k"): 1,
("a", "f", "g", "h", "i", "j", "k", "b"): 1,
("a", "f", "g", "i", "b", "j", "k"): 1,
("a", "f", "g", "i", "j", "b", "k"): 1,
("a", "f", "g", "i", "k", "j", "b"): 1,
("a", "f", "g", "i", "k", "j", "h"): 1,
("a", "f", "g", "k", "i", "j", "h", "b"): 1,
("a", "h", "b", "f", "g", "i", "k"): 1,
("a", "h", "b", "f", "g", "k"): 1,
("a", "h", "b", "i", "k"): 1,
("a", "h", "b", "k", "f", "g", "i"): 1,
("a", "h", "b", "k", "j"): 1,
("a", "h", "f", "g", "i", "b", "j", "k"): 1,
("a", "h", "f", "g", "i", "k"): 1,
("a", "h", "f", "g", "i", "k", "b"): 1,
("a", "h", "f", "g", "i", "k", "j"): 1,
("a", "h", "f", "k", "g", "i", "j", "b"): 1,
("a", "h", "j", "k", "b"): 1,
("a", "h", "k", "b", "i", "j"): 1,
("a", "h", "k", "j", "b"): 1,
("a", "k", "b", "f", "g", "i", "j"): 1,
("a", "k", "h", "b", "f", "g", "i", "j"): 1,
("b", "f", "g", "k", "a"): 1,
("b", "i", "j", "k", "a"): 1,
("b", "j", "k", "a"): 1,
("b", "k", "a", "j"): 1,
}
"""Traces from the KMU log, filtered and anonymized."""
TRACES_2: dict[tuple[str, ...], int] = {
("a", "g", "e", "c", "d", "f", "h"): 252,
("a", "g", "c", "d", "f", "e", "h"): 192,
("a", "b", "c", "d", "f", "h", "e", "g"): 190,
("a", "g", "c", "d", "e", "f", "h"): 36,
("a", "g", "c", "e", "d", "f", "h"): 19,
}
"""Simplified traces from TRACES_1."""
TRACES_3: dict[tuple[str, ...], int] = {
("a", "b", "d", "b", "e", "d", "c", "e",
"d", "e"): 1,
("a", "b", "d", "e", "c", "d", "e", "a",
"b", "d", "e", "a", "b", "d", "e", "c",
"d", "a", "c", "e", "d", "e", "e"): 1,
}
"""Traces from the rent data."""
class TestEpsilon(unittest.TestCase):
"""Tests for the epsilon parameter edge cases."""
def test_traces_1(self) -> None:
"""Tests TRACES_1.
:return: None.
"""
self.__test_traces(TRACES_1)
def test_traces_2(self) -> None:
"""Tests TRACES_2.
:return: None.
"""
self.__test_traces(TRACES_2)
def test_traces_3(self) -> None:
"""Tests TRACES_3.
:return: None.
"""
self.__test_traces(TRACES_3)
def __test_traces(
self,
traces: dict[tuple[str, ...], int],
) -> None:
"""Tests a trace set.
A high epsilon (0.8) prunes many edges, which can
produce a graph with cut vertices. The SPQR-tree
construction raises ValueError for non-biconnected
graphs, but build_rpst() catches this and falls
back to a single fragment. All three epsilon
values should succeed without raising.
:param traces: The traces.
:return: None.
"""
split_miner(traces, epsilon=0.8, eta=0.8)
split_miner(traces, epsilon=0.33, eta=0.8)
split_miner(traces)
if __name__ == "__main__":
unittest.main()
+430
View File
@@ -0,0 +1,430 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/11
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for real-world event logs.
Ported from the bpmn project's test_event_log.py.
Task labels are anonymized per log-anonymization.md.
"""
from __future__ import annotations
import json
import unittest
from pathlib import Path
from split_miner import (
BPMNModel,
Gateway,
GatewayType,
Task,
split_miner,
)
def _load_traces(
log_file: str,
) -> dict[tuple[str, ...], int]:
"""Load traces from a JSON log file.
:param log_file: The path to the log file.
:return: The traces as a dict mapping trace tuples
to frequencies.
"""
with open(log_file) as f:
raw: list[list] = json.loads(f.read())
return {tuple(x[0]): x[1] for x in raw}
def _run_split_miner(
log_file: str,
epsilon: float = 0.8,
eta: float = 0.8,
) -> BPMNModel:
"""Run split miner on a log file.
:param log_file: The path to the log file.
:param epsilon: The concurrency threshold.
:param eta: The filtering percentile.
:return: The discovered BPMN model.
"""
traces: dict[tuple[str, ...], int] = (
_load_traces(log_file)
)
return split_miner(
traces, epsilon=epsilon, eta=eta
)
def _count_gateways(
model: BPMNModel,
gw_type: GatewayType,
is_split: bool,
) -> int:
"""Count gateways of a given type and role.
:param model: The BPMN model.
:param gw_type: The gateway type.
:param is_split: True for splits, False for joins.
:return: The count.
"""
result: int = 0
for node in model.all_nodes:
if not isinstance(node, Gateway):
continue
if node.gateway_type != gw_type:
continue
if is_split:
if len(model.outgoing_edges(node)) > 1:
result += 1
else:
if len(model.incoming_edges(node)) > 1:
result += 1
return result
_LOGS_DIR: str = str(
Path(__file__).parent / "logs"
)
class TestMultiSourceSinkEventLog(unittest.TestCase):
"""Tests for multi_source_sink.json (kmu.json)."""
__LOG_FILE: str = str(
Path(_LOGS_DIR) / "multi_source_sink.json"
)
__TASKS: set[str] = {
"A", "B", "C", "D", "E",
"F", "G", "H", "I", "J", "K",
}
def test_event_log(self) -> None:
"""Tests the event log with default parameters.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE
)
self.assertEqual(len(model.all_nodes), 21)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 4
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 2
)
self.assertEqual(len(model.edges), 30)
def test_event_log_eta_0(self) -> None:
"""Tests the event log with eta=0.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE, eta=0
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(len(model.all_nodes), 42)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 16
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 2
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 9
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 1
)
self.assertEqual(len(model.edges), 78)
def test_event_log_eta_1(self) -> None:
"""Tests the event log with eta=1.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE, eta=1
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(len(model.all_nodes), 21)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 4
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 2
)
self.assertEqual(len(model.edges), 30)
class TestCyclicTraceEventLog(unittest.TestCase):
"""Tests for cyclic_trace.json (ntphrf.json)."""
__LOG_FILE: str = str(
Path(_LOGS_DIR) / "cyclic_trace.json"
)
__TASKS: set[str] = {
"A", "B", "C", "D", "E",
"F", "G", "H", "I", "J",
}
def test_event_log(self) -> None:
"""Tests the event log with default parameters.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(len(model.all_nodes), 14)
self.assertEqual(len(model.edges), 14)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 1
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 0
)
class TestMultiSinkEventLog(unittest.TestCase):
"""Tests for multi_sink.json (ntp_job403.json)."""
__LOG_FILE: str = str(
Path(_LOGS_DIR) / "multi_sink.json"
)
__TASKS: set[str] = {
"A", "B", "C", "D", "E",
"F", "G", "H", "I", "J", "K",
}
def test_event_log(self) -> None:
"""Tests the event log with default parameters.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(len(model.all_nodes), 15)
self.assertEqual(len(model.edges), 16)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 2
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 0
)
class TestShortLoopsEventLog(unittest.TestCase):
"""Tests for short_loops.json (lottery.json)."""
__LOG_FILE: str = str(
Path(_LOGS_DIR) / "short_loops.json"
)
__TASKS: set[str] = {
"A", "B", "C", "D", "E", "F",
"G", "H", "I", "J", "K", "L",
}
def test_event_log(self) -> None:
"""Tests the event log with default parameters.
:return: None.
"""
model: BPMNModel = _run_split_miner(
self.__LOG_FILE
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(task_labels, self.__TASKS)
self.assertEqual(len(model.all_nodes), 21)
self.assertEqual(len(model.edges), 24)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, True
), 4
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, True
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.XOR, False
), 3
)
self.assertEqual(
_count_gateways(
model, GatewayType.AND, False
), 0
)
self.assertEqual(
_count_gateways(
model, GatewayType.OR, False
), 0
)
if __name__ == "__main__":
unittest.main()
+263
View File
@@ -0,0 +1,263 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/10
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for Fig. 5(b) joins and Fig. 6/7 OR minimization.
Tests join gateway discovery and OR-joins minimization using
manually constructed BPMN models from the SM 1.0 paper
figures.
"""
from __future__ import annotations
import unittest
from split_miner.bpmn import (
BPMNModel,
EndEvent,
Gateway,
GatewayType,
Node,
StartEvent,
Task,
)
from split_miner.joins import discover_joins
from split_miner.or_minimization import replace_or_joins
def _make_fig5b_model() -> BPMNModel:
"""Build the model from Fig. 5(b) of the paper.
Graph structure (after splits, before joins):
- start -> gx1 (XOR split)
- gx1 -> {a, b}
- a -> gx2 (XOR split)
- b -> gx3 (XOR split)
- gx2 -> {j, c}
- gx3 -> {j, d}
- j -> i
- c -> i
- d -> k
- i -> k
- k -> end
:return: The BPMN model.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
tasks: dict[str, Task] = {}
for label in ["a", "b", "c", "d", "i", "j", "k"]:
t: Task = Task(label, label)
model.add_task(t)
tasks[label] = t
gx1: Gateway = Gateway("gx1", GatewayType.XOR)
gx2: Gateway = Gateway("gx2", GatewayType.XOR)
gx3: Gateway = Gateway("gx3", GatewayType.XOR)
model.add_gateway(gx1)
model.add_gateway(gx2)
model.add_gateway(gx3)
nodes: dict[str, Node] = {
"start": start, "end": end,
"gx1": gx1, "gx2": gx2, "gx3": gx3,
}
nodes.update(tasks)
for src, tgt in [
("start", "gx1"),
("gx1", "a"), ("gx1", "b"),
("a", "gx2"), ("b", "gx3"),
("gx2", "j"), ("gx3", "j"),
("gx2", "c"), ("gx3", "d"),
("j", "i"), ("c", "i"),
("d", "k"), ("i", "k"),
("k", "end"),
]:
model.add_edge(nodes[src], nodes[tgt])
return model
def _make_fig6_model() -> tuple[
BPMNModel, Gateway, Gateway, Gateway
]:
"""Build the model from Fig. 6 of the paper.
Graph structure (after joins, before OR minimization):
- start -> a -> gx1 (XOR split)
- gx1 -> {b, c}
- b -> ga1 (AND split)
- c -> ga2 (AND split)
- ga1 -> {d, go2}
- ga2 -> {go1, e}
- d -> go1
- e -> go2
- go1 (OR join) -> f
- go2 (OR join) -> g
- f -> go3 (OR join)
- g -> go3
- go3 -> h -> end
:return: The model and the three OR-join gateways.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
tasks: dict[str, Task] = {}
for label in [
"a", "b", "c", "d", "e", "f", "g", "h"
]:
t: Task = Task(label, label)
model.add_task(t)
tasks[label] = t
gx1: Gateway = Gateway("gx1", GatewayType.XOR)
ga1: Gateway = Gateway("ga1", GatewayType.AND)
ga2: Gateway = Gateway("ga2", GatewayType.AND)
go1: Gateway = Gateway("go1", GatewayType.OR)
go2: Gateway = Gateway("go2", GatewayType.OR)
go3: Gateway = Gateway("go3", GatewayType.OR)
for gw in [gx1, ga1, ga2, go1, go2, go3]:
model.add_gateway(gw)
nodes: dict[str, Node] = {
"start": start, "end": end,
"gx1": gx1, "ga1": ga1, "ga2": ga2,
"go1": go1, "go2": go2, "go3": go3,
}
nodes.update(tasks)
for src, tgt in [
("start", "a"), ("a", "gx1"),
("gx1", "b"), ("gx1", "c"),
("b", "ga1"), ("c", "ga2"),
("ga1", "d"), ("ga1", "go2"),
("ga2", "go1"), ("ga2", "e"),
("d", "go1"), ("e", "go2"),
("go1", "f"), ("go2", "g"),
("f", "go3"), ("g", "go3"),
("go3", "h"), ("h", "end"),
]:
model.add_edge(nodes[src], nodes[tgt])
return model, go1, go2, go3
class TestFig5bJoins(unittest.TestCase):
"""Tests joins discovery from Fig. 5(b)."""
def test_joins_discovery(self) -> None:
"""All three joins are XOR (all splits are XOR).
After discover_joins:
- j gets an XOR join (from gx2 and gx3)
- i gets an XOR join (from j and c)
- k gets an XOR join (from d and i)
"""
model: BPMNModel = _make_fig5b_model()
discover_joins(model)
# j should have a join gateway predecessor
j: Task = model.get_task("j")
j_preds: set[Node] = model.predecessors(j)
self.assertEqual(len(j_preds), 1)
j_join: Node = next(iter(j_preds))
self.assertIsInstance(j_join, Gateway)
assert isinstance(j_join, Gateway)
self.assertEqual(
j_join.gateway_type, GatewayType.XOR,
"Join for j should be XOR (all splits "
"are XOR)"
)
# i should have a join gateway predecessor
i: Task = model.get_task("i")
i_preds: set[Node] = model.predecessors(i)
self.assertEqual(len(i_preds), 1)
i_join: Node = next(iter(i_preds))
self.assertIsInstance(i_join, Gateway)
assert isinstance(i_join, Gateway)
self.assertEqual(
i_join.gateway_type, GatewayType.XOR,
"Join for i should be XOR"
)
# k should have a join gateway predecessor
k: Task = model.get_task("k")
k_preds: set[Node] = model.predecessors(k)
self.assertEqual(len(k_preds), 1)
k_join: Node = next(iter(k_preds))
self.assertIsInstance(k_join, Gateway)
assert isinstance(k_join, Gateway)
self.assertEqual(
k_join.gateway_type, GatewayType.XOR,
"Join for k should be XOR"
)
def test_gateway_count(self) -> None:
"""6 gateways after joins (3 splits + 3 joins)."""
model: BPMNModel = _make_fig5b_model()
discover_joins(model)
self.assertEqual(len(model.gateways), 6)
class TestFig6Fig7OrMinimization(unittest.TestCase):
"""Tests OR-joins minimization (Fig. 6 -> Fig. 7)."""
def test_or_joins_minimization(self) -> None:
"""OR-joins are minimized to correct types.
After OR-joins minimization:
- go1 becomes XOR (fed by XOR split gx1)
- go2 becomes XOR (fed by XOR split gx1)
- go3 becomes AND (fed by AND splits ga1, ga2)
"""
model: BPMNModel
go1: Gateway
go2: Gateway
go3: Gateway
model, go1, go2, go3 = _make_fig6_model()
replace_or_joins(model)
self.assertEqual(
go1.gateway_type, GatewayType.XOR,
"go1 should become XOR"
)
self.assertEqual(
go2.gateway_type, GatewayType.XOR,
"go2 should become XOR"
)
self.assertEqual(
go3.gateway_type, GatewayType.AND,
"go3 should become AND"
)
def test_edge_count_unchanged(self) -> None:
"""OR minimization doesn't change edges."""
model: BPMNModel
model, _, _, _ = _make_fig6_model()
edges_before: int = len(model.edges)
replace_or_joins(model)
self.assertEqual(len(model.edges), edges_before)
if __name__ == "__main__":
unittest.main()
+246
View File
@@ -0,0 +1,246 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/11
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for node-splitting normalization in build_rpst().
When aggressive edge filtering produces a graph with cut
vertices, the completed version C(G) is not biconnected.
Polyvyanyy et al. (2011), Section 4 describes node-splitting
as the correct fix: split each node with >1 incoming AND >1
outgoing edges into two nodes, making C(G) biconnected.
These tests verify that build_rpst() correctly applies
node-splitting normalization instead of falling back to a
single fragment.
Reference:
Polyvyanyy, A., Vanhatalo, J., & Volzer, H. (2011).
Simplified Computation and Generalization of the
Refined Process Structure Tree. Section 4.
"""
from __future__ import annotations
import unittest
from split_miner.bpmn import (
BPMNModel,
EndEvent,
Gateway,
GatewayType,
Node,
StartEvent,
Task,
)
from split_miner.joins import SESEFragment, build_rpst
from split_miner.joins import discover_joins
def _make_cut_vertex_model() -> tuple[
BPMNModel, dict[str, Node],
]:
"""Build a model with a cut vertex at node c.
Graph structure:
- start -> a -> c -> d -> end
- c -> e -> f -> c (loop)
Node c has 2 incoming edges (from a and f) and
2 outgoing edges (to d and e). In C(G), removing c
disconnects {e, f} from the rest, making c a cut
vertex (separation point).
:return: The model and its named nodes.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
tasks: dict[str, Task] = {}
for label in ["a", "c", "d", "e", "f"]:
t: Task = Task(label, label)
model.add_task(t)
tasks[label] = t
nodes: dict[str, Node] = {
"start": start, "end": end,
}
nodes.update(tasks)
for src, tgt in [
("start", "a"), ("a", "c"),
("c", "d"), ("d", "end"),
("c", "e"), ("e", "f"), ("f", "c"),
]:
model.add_edge(nodes[src], nodes[tgt])
return model, nodes
class TestBuildRpstCutVertex(unittest.TestCase):
"""Tests build_rpst with a cut vertex graph.
Verifies that node-splitting normalization produces
proper RPST fragments instead of a single fallback
fragment.
"""
def setUp(self) -> None:
"""Set up the cut vertex model.
:return: None.
"""
self.__model: BPMNModel
self.__nodes: dict[str, Node]
self.__model, self.__nodes = (
_make_cut_vertex_model()
)
self.__fragments: list[SESEFragment] = (
build_rpst(self.__model)
)
def test_multiple_fragments(self) -> None:
"""Produces multiple fragments, not single fallback.
With node-splitting normalization, the SPQR-tree
should decompose the graph into multiple SESE
fragments instead of falling back to a single
R-type fragment.
:return: None.
"""
self.assertGreater(len(self.__fragments), 1)
def test_all_edges_covered(self) -> None:
"""Union of fragment edges covers all model edges.
:return: None.
"""
all_frag_edges: set[tuple[Node, Node]] = set()
for f in self.__fragments:
all_frag_edges |= f.edges
self.assertEqual(
all_frag_edges, self.__model.edges
)
def test_entry_exit_are_model_nodes(self) -> None:
"""Entry and exit are original model nodes.
No split proxy nodes should appear as fragment
entry or exit.
:return: None.
"""
all_nodes: set[Node] = self.__model.all_nodes
for f in self.__fragments:
self.assertIn(f.entry, all_nodes)
self.assertIn(f.exit_node, all_nodes)
def test_fragment_nodes_are_model_nodes(self) -> None:
"""All fragment nodes are original model nodes.
No split proxy nodes should leak into fragment
node sets.
:return: None.
"""
all_nodes: set[Node] = self.__model.all_nodes
for f in self.__fragments:
for node in f.nodes:
self.assertIn(
node, all_nodes,
f"Proxy node {node!r} leaked "
f"into fragment",
)
def test_bottom_up_order(self) -> None:
"""Fragments are ordered bottom-up (small first).
:return: None.
"""
sizes: list[int] = [
len(f.edges) for f in self.__fragments
]
self.assertEqual(sizes, sorted(sizes))
class TestDiscoverJoinsCutVertex(unittest.TestCase):
"""Tests discover_joins on a graph with a cut vertex.
Verifies that join gateway discovery works correctly
when the graph requires node-splitting normalization.
"""
def setUp(self) -> None:
"""Set up and run discover_joins.
:return: None.
"""
self.__model: BPMNModel
self.__nodes: dict[str, Node]
self.__model, self.__nodes = (
_make_cut_vertex_model()
)
discover_joins(self.__model)
def test_completes_without_raising(self) -> None:
"""discover_joins completes without exception.
:return: None.
"""
# If we get here, it didn't raise.
self.assertTrue(True)
def test_join_for_c(self) -> None:
"""Task c gets a join gateway predecessor.
Task c has 2 incoming edges (from a and f),
so it should get a join gateway.
:return: None.
"""
c: Task = self.__model.get_task("c")
preds: set[Node] = (
self.__model.predecessors(c)
)
self.assertEqual(len(preds), 1)
join: Node = next(iter(preds))
self.assertIsInstance(join, Gateway)
def test_join_for_c_is_xor(self) -> None:
"""Task c's join is XOR (loop-join).
The f -> c edge creates a cycle (c -> e -> f
-> c), making this a loop-join which should be
XOR per Definition 12 of the SM 1.0 paper.
:return: None.
"""
c: Task = self.__model.get_task("c")
preds: set[Node] = (
self.__model.predecessors(c)
)
join: Node = next(iter(preds))
assert isinstance(join, Gateway)
self.assertEqual(
join.gateway_type, GatewayType.XOR,
"Loop-join for c should be XOR",
)
if __name__ == "__main__":
unittest.main()
+482
View File
@@ -0,0 +1,482 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/10
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Detailed pipeline stage tests for the paper example.
Verifies each stage of the Split Miner pipeline using the
running example from Section 3 of the SM 1.0 paper.
"""
from __future__ import annotations
import unittest
from split_miner import (
BPMNModel,
Gateway,
GatewayType,
Node,
Task,
split_miner,
)
from split_miner.concurrency import PrunedDFG
from split_miner.dfg import DirectlyFollowsGraph
from split_miner.filtering import FilteredDFG
def _make_tasks(
labels: str,
) -> dict[str, Task]:
"""Create a Task for each single-character label.
:param labels: The labels as a string.
:return: A dict mapping label to Task.
"""
return {ch: Task(ch, ch) for ch in labels}
def _make_paper_node_log() -> tuple[
dict[tuple[Node, ...], int], dict[str, Task]
]:
"""Build the paper example log with Node objects.
:return: The Node-based traces and the task map.
"""
t: dict[str, Task] = _make_tasks("abcdefgh")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["c"], t["g"],
t["e"], t["h"]): 10,
(t["a"], t["b"], t["c"], t["f"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["d"], t["g"],
t["e"], t["h"]): 10,
(t["a"], t["b"], t["d"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["e"], t["c"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["e"], t["d"],
t["g"], t["h"]): 10,
(t["a"], t["c"], t["b"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["c"], t["b"], t["f"],
t["g"], t["h"]): 10,
(t["a"], t["d"], t["b"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["d"], t["b"], t["f"],
t["g"], t["h"]): 10,
}
return traces, t
def _make_paper_str_log() -> dict[
tuple[str, ...], int
]:
"""Build the paper example log with string labels.
:return: The string-based traces.
"""
return {
("a", "b", "c", "g", "e", "h"): 10,
("a", "b", "c", "f", "g", "h"): 10,
("a", "b", "d", "g", "e", "h"): 10,
("a", "b", "d", "e", "g", "h"): 10,
("a", "b", "e", "c", "g", "h"): 10,
("a", "b", "e", "d", "g", "h"): 10,
("a", "c", "b", "e", "g", "h"): 10,
("a", "c", "b", "f", "g", "h"): 10,
("a", "d", "b", "e", "g", "h"): 10,
("a", "d", "b", "f", "g", "h"): 10,
}
class TestDFGAllEdges(unittest.TestCase):
"""Tests for all DFG edge frequencies (Table 1)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
self.__dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
def test_all_edge_frequencies(self) -> None:
"""All 20 DFG edge frequencies match Table 1.
Verifies every directly-follows frequency from the
paper's example event log.
"""
t: dict[str, Task] = self.__t
expected: dict[tuple[Node, Node], int] = {
(t["a"], t["b"]): 60,
(t["a"], t["c"]): 20,
(t["a"], t["d"]): 20,
(t["b"], t["c"]): 20,
(t["b"], t["d"]): 20,
(t["b"], t["e"]): 40,
(t["b"], t["f"]): 20,
(t["c"], t["b"]): 20,
(t["c"], t["f"]): 10,
(t["c"], t["g"]): 20,
(t["d"], t["b"]): 20,
(t["d"], t["e"]): 10,
(t["d"], t["g"]): 20,
(t["e"], t["c"]): 10,
(t["e"], t["d"]): 10,
(t["e"], t["g"]): 30,
(t["e"], t["h"]): 20,
(t["f"], t["g"]): 30,
(t["g"], t["e"]): 20,
(t["g"], t["h"]): 80,
}
for (src, tgt), freq in expected.items():
self.assertEqual(
self.__dfg.df_frequency(src, tgt),
freq,
f"|{src.node_id} -> {tgt.node_id}|"
f" should be {freq}"
)
def test_all_edges(self) -> None:
"""The DFG has exactly 20 edges."""
t: dict[str, Task] = self.__t
expected: set[tuple[Node, Node]] = {
(t["a"], t["b"]), (t["a"], t["c"]),
(t["a"], t["d"]),
(t["b"], t["c"]), (t["b"], t["d"]),
(t["b"], t["e"]), (t["b"], t["f"]),
(t["c"], t["b"]), (t["c"], t["f"]),
(t["c"], t["g"]),
(t["d"], t["b"]), (t["d"], t["e"]),
(t["d"], t["g"]),
(t["e"], t["c"]), (t["e"], t["d"]),
(t["e"], t["g"]), (t["e"], t["h"]),
(t["f"], t["g"]),
(t["g"], t["e"]), (t["g"], t["h"]),
}
self.assertEqual(self.__dfg.edges, expected)
def test_edge_count(self) -> None:
"""The DFG has 20 edges."""
self.assertEqual(len(self.__dfg.edges), 20)
class TestPrunedDFGEdges(unittest.TestCase):
"""Tests for PDFG edge set (Section 3.2)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
self.__pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=0.2
)
def test_pdfg_edges(self) -> None:
"""PDFG has 12 edges after concurrent pruning.
Concurrent pairs b||c, b||d, d||e, e||g are
removed along with their reverse edges.
"""
t: dict[str, Task] = self.__t
expected: set[tuple[Node, Node]] = {
(t["a"], t["b"]), (t["a"], t["c"]),
(t["a"], t["d"]),
(t["b"], t["e"]), (t["b"], t["f"]),
(t["c"], t["f"]), (t["c"], t["g"]),
(t["d"], t["g"]),
(t["e"], t["c"]), (t["e"], t["h"]),
(t["f"], t["g"]),
(t["g"], t["h"]),
}
self.assertEqual(self.__pdfg.edges, expected)
def test_pdfg_edge_count(self) -> None:
"""PDFG has 12 edges."""
self.assertEqual(len(self.__pdfg.edges), 12)
def test_concurrent_pairs(self) -> None:
"""All four concurrent pairs are detected."""
t: dict[str, Task] = self.__t
self.assertTrue(
self.__pdfg.is_concurrent(t["b"], t["c"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["b"], t["d"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["d"], t["e"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["e"], t["g"])
)
def test_not_concurrent(self) -> None:
"""Non-concurrent pairs."""
t: dict[str, Task] = self.__t
self.assertFalse(
self.__pdfg.is_concurrent(t["a"], t["b"])
)
self.assertFalse(
self.__pdfg.is_concurrent(t["c"], t["d"])
)
self.assertFalse(
self.__pdfg.is_concurrent(t["c"], t["f"])
)
class TestFilteredDFGEdges(unittest.TestCase):
"""Tests for filtered PDFG edge set (Section 3.3)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
pdfg: PrunedDFG = PrunedDFG(dfg, epsilon=0.2)
self.__fdfg: FilteredDFG = FilteredDFG(
pdfg, eta=0.4
)
def test_filtered_edges(self) -> None:
"""Filtered PDFG has 10 edges.
Edges c->f and e->c are filtered out.
"""
t: dict[str, Task] = self.__t
expected: set[tuple[Node, Node]] = {
(t["a"], t["b"]), (t["a"], t["c"]),
(t["a"], t["d"]),
(t["b"], t["e"]), (t["b"], t["f"]),
(t["c"], t["g"]),
(t["d"], t["g"]),
(t["e"], t["h"]),
(t["f"], t["g"]),
(t["g"], t["h"]),
}
self.assertEqual(self.__fdfg.edges, expected)
def test_filtered_edge_count(self) -> None:
"""Filtered PDFG has 10 edges."""
self.assertEqual(len(self.__fdfg.edges), 10)
def test_removed_edges(self) -> None:
"""Edges c->f and e->c are not in filtered PDFG."""
t: dict[str, Task] = self.__t
self.assertNotIn(
(t["c"], t["f"]), self.__fdfg.edges
)
self.assertNotIn(
(t["e"], t["c"]), self.__fdfg.edges
)
class TestPaperExampleStructure(unittest.TestCase):
"""Tests for the final paper example structure."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[str, ...], int] = (
_make_paper_str_log()
)
self.__model: BPMNModel = split_miner(
traces, epsilon=0.2, eta=0.4
)
def test_node_count(self) -> None:
"""The final model has 16 nodes."""
self.assertEqual(
len(self.__model.all_nodes), 16
)
def test_edge_count(self) -> None:
"""The final model has 18 edges."""
self.assertEqual(len(self.__model.edges), 18)
def test_gateway_counts(self) -> None:
"""6 gateways: AND=1, XOR=4, OR=1."""
gw_types: list[GatewayType] = [
gw.gateway_type
for gw in self.__model.gateways.values()
]
self.assertEqual(len(gw_types), 6)
self.assertEqual(
gw_types.count(GatewayType.AND), 1
)
self.assertEqual(
gw_types.count(GatewayType.XOR), 4
)
self.assertEqual(
gw_types.count(GatewayType.OR), 1
)
def test_and_split_after_a(self) -> None:
"""Task a leads to an AND split gateway."""
a: Task = self.__model.get_task("a")
a_succs: set[Node] = (
self.__model.successors(a)
)
self.assertEqual(len(a_succs), 1)
and_gw: Node = next(iter(a_succs))
self.assertIsInstance(and_gw, Gateway)
assert isinstance(and_gw, Gateway)
self.assertEqual(
and_gw.gateway_type, GatewayType.AND
)
def test_and_split_successors(self) -> None:
"""AND split has successors b and XOR split."""
a: Task = self.__model.get_task("a")
and_gw: Node = next(
iter(self.__model.successors(a))
)
and_succs: set[Node] = (
self.__model.successors(and_gw)
)
self.assertEqual(len(and_succs), 2)
b: Task = self.__model.get_task("b")
self.assertIn(b, and_succs)
def test_xor_split_cd(self) -> None:
"""XOR split for c and d (successor of AND)."""
a: Task = self.__model.get_task("a")
and_gw: Node = next(
iter(self.__model.successors(a))
)
and_succs: set[Node] = (
self.__model.successors(and_gw)
)
b: Task = self.__model.get_task("b")
xor1: Node = (and_succs - {b}).pop()
self.assertIsInstance(xor1, Gateway)
assert isinstance(xor1, Gateway)
self.assertEqual(
xor1.gateway_type, GatewayType.XOR
)
xor1_succs: set[Node] = (
self.__model.successors(xor1)
)
c: Task = self.__model.get_task("c")
d: Task = self.__model.get_task("d")
self.assertEqual(xor1_succs, {c, d})
def test_xor_split_ef(self) -> None:
"""XOR split for e and f (after b)."""
b: Task = self.__model.get_task("b")
b_succs: set[Node] = (
self.__model.successors(b)
)
self.assertEqual(len(b_succs), 1)
xor2: Node = next(iter(b_succs))
self.assertIsInstance(xor2, Gateway)
assert isinstance(xor2, Gateway)
self.assertEqual(
xor2.gateway_type, GatewayType.XOR
)
xor2_succs: set[Node] = (
self.__model.successors(xor2)
)
e: Task = self.__model.get_task("e")
f: Task = self.__model.get_task("f")
self.assertEqual(xor2_succs, {e, f})
def test_xor_join_cd(self) -> None:
"""XOR join for c and d."""
c: Task = self.__model.get_task("c")
d: Task = self.__model.get_task("d")
c_succs: set[Node] = (
self.__model.successors(c)
)
d_succs: set[Node] = (
self.__model.successors(d)
)
self.assertEqual(len(c_succs), 1)
self.assertEqual(len(d_succs), 1)
self.assertEqual(c_succs, d_succs)
join: Node = next(iter(c_succs))
self.assertIsInstance(join, Gateway)
assert isinstance(join, Gateway)
self.assertEqual(
join.gateway_type, GatewayType.XOR
)
def test_or_join_to_g(self) -> None:
"""OR join for {XOR-join, f} leading to g."""
g: Task = self.__model.get_task("g")
g_preds: set[Node] = (
self.__model.predecessors(g)
)
self.assertEqual(len(g_preds), 1)
or_gw: Node = next(iter(g_preds))
self.assertIsInstance(or_gw, Gateway)
assert isinstance(or_gw, Gateway)
self.assertEqual(
or_gw.gateway_type, GatewayType.OR
)
or_preds: set[Node] = (
self.__model.predecessors(or_gw)
)
self.assertEqual(len(or_preds), 2)
f: Task = self.__model.get_task("f")
self.assertIn(f, or_preds)
def test_xor_join_to_h(self) -> None:
"""XOR join for {e, g} leading to h.
This was an OR-join that became XOR after
OR-joins minimization (Algorithm 9).
"""
h: Task = self.__model.get_task("h")
h_preds: set[Node] = (
self.__model.predecessors(h)
)
self.assertEqual(len(h_preds), 1)
join: Node = next(iter(h_preds))
self.assertIsInstance(join, Gateway)
assert isinstance(join, Gateway)
self.assertEqual(
join.gateway_type, GatewayType.XOR
)
join_preds: set[Node] = (
self.__model.predecessors(join)
)
self.assertEqual(len(join_preds), 2)
e: Task = self.__model.get_task("e")
g: Task = self.__model.get_task("g")
self.assertIn(e, join_preds)
self.assertIn(g, join_preds)
if __name__ == "__main__":
unittest.main()
+645
View File
@@ -0,0 +1,645 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/10
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for the Split Miner algorithm.
Uses the running example from Section 3 of the SM 1.0 journal
paper (Augusto et al., 2018).
"""
from __future__ import annotations
import unittest
from collections import deque
from split_miner import (
BPMNModel,
Gateway,
GatewayType,
Node,
Task,
split_miner,
)
from split_miner.concurrency import PrunedDFG
from split_miner.dfg import DirectlyFollowsGraph
from split_miner.filtering import FilteredDFG
def _make_tasks(
labels: str,
) -> dict[str, Task]:
"""Create a Task for each single-character label.
:param labels: The labels as a string.
:return: A dict mapping label to Task.
"""
return {ch: Task(ch, ch) for ch in labels}
def _make_paper_node_log() -> tuple[
dict[tuple[Node, ...], int], dict[str, Task]
]:
"""Build the paper example log with Node objects.
:return: The Node-based traces and the task map.
"""
t: dict[str, Task] = _make_tasks("abcdefgh")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["c"], t["g"],
t["e"], t["h"]): 10,
(t["a"], t["b"], t["c"], t["f"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["d"], t["g"],
t["e"], t["h"]): 10,
(t["a"], t["b"], t["d"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["e"], t["c"],
t["g"], t["h"]): 10,
(t["a"], t["b"], t["e"], t["d"],
t["g"], t["h"]): 10,
(t["a"], t["c"], t["b"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["c"], t["b"], t["f"],
t["g"], t["h"]): 10,
(t["a"], t["d"], t["b"], t["e"],
t["g"], t["h"]): 10,
(t["a"], t["d"], t["b"], t["f"],
t["g"], t["h"]): 10,
}
return traces, t
def _make_paper_str_log() -> dict[
tuple[str, ...], int
]:
"""Build the paper example log with string labels.
:return: The string-based traces.
"""
return {
("a", "b", "c", "g", "e", "h"): 10,
("a", "b", "c", "f", "g", "h"): 10,
("a", "b", "d", "g", "e", "h"): 10,
("a", "b", "d", "e", "g", "h"): 10,
("a", "b", "e", "c", "g", "h"): 10,
("a", "b", "e", "d", "g", "h"): 10,
("a", "c", "b", "e", "g", "h"): 10,
("a", "c", "b", "f", "g", "h"): 10,
("a", "d", "b", "e", "g", "h"): 10,
("a", "d", "b", "f", "g", "h"): 10,
}
class TestDFGConstruction(unittest.TestCase):
"""Tests for DFG construction (Section 3.1)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
self.__dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
def test_nodes(self) -> None:
"""The DFG has the correct set of nodes."""
self.assertEqual(
self.__dfg.nodes,
set(self.__t.values()),
)
def test_sources_and_sinks(self) -> None:
"""The DFG has correct sources and sinks."""
self.assertIn(
self.__t["a"], self.__dfg.sources
)
self.assertIn(
self.__t["h"], self.__dfg.sinks
)
def test_df_frequencies(self) -> None:
"""Selected directly-follows frequencies match."""
t: dict[str, Task] = self.__t
# a -> b: appears in 6 trace types * 10 = 60
self.assertEqual(
self.__dfg.df_frequency(
t["a"], t["b"]
), 60
)
# a -> c: 2 trace types * 10 = 20
self.assertEqual(
self.__dfg.df_frequency(
t["a"], t["c"]
), 20
)
# a -> d: 2 trace types * 10 = 20
self.assertEqual(
self.__dfg.df_frequency(
t["a"], t["d"]
), 20
)
def test_no_self_loops(self) -> None:
"""The paper example has no self-loops."""
self.assertEqual(self.__dfg.self_loops, set())
def test_no_short_loops(self) -> None:
"""The paper example has no short-loops."""
self.assertEqual(
self.__dfg.short_loops, set()
)
def test_self_loop_detection(self) -> None:
"""Self-loops are correctly detected."""
t: dict[str, Task] = _make_tasks("abc")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["b"], t["c"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
self.assertIn(t["b"], dfg.self_loops)
self.assertNotIn(t["a"], dfg.self_loops)
def test_self_loop_edges_excluded(self) -> None:
"""Self-loop edges are excluded from edges set."""
t: dict[str, Task] = _make_tasks("abc")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["b"], t["c"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
self.assertNotIn(
(t["b"], t["b"]), dfg.edges
)
self.assertIn(
(t["a"], t["b"]), dfg.edges
)
self.assertIn(
(t["b"], t["c"]), dfg.edges
)
def test_short_loop_detection(self) -> None:
"""Short-loops are correctly detected."""
t: dict[str, Task] = _make_tasks("abcd")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["c"],
t["b"], t["d"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
self.assertIn(
(t["b"], t["c"]), dfg.short_loops
)
self.assertIn(
(t["c"], t["b"]), dfg.short_loops
)
class TestConcurrencyDiscovery(unittest.TestCase):
"""Tests for concurrency discovery (Section 3.2)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
self.__pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=0.2
)
def test_concurrent_pairs(self) -> None:
"""Correct concurrency relations with epsilon=0.2.
The paper identifies: b||c, b||d, d||e, e||g.
"""
t: dict[str, Task] = self.__t
# Check expected concurrent pairs
self.assertTrue(
self.__pdfg.is_concurrent(t["b"], t["c"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["b"], t["d"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["d"], t["e"])
)
self.assertTrue(
self.__pdfg.is_concurrent(t["e"], t["g"])
)
# Non-concurrent pairs
self.assertFalse(
self.__pdfg.is_concurrent(t["a"], t["b"])
)
self.assertFalse(
self.__pdfg.is_concurrent(t["c"], t["d"])
)
def test_self_loop_skipped_in_concurrency(
self,
) -> None:
"""Self-loop nodes are never concurrent."""
# b has a self-loop; a and b could look
# concurrent but b should be skipped.
t: dict[str, Task] = _make_tasks("abc")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["b"],
t["a"], t["c"]): 10,
(t["a"], t["b"],
t["a"], t["c"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=1.0
)
self.assertFalse(
pdfg.is_concurrent(t["a"], t["b"])
)
def test_short_loop_not_concurrent(self) -> None:
"""Short-loop pairs are not concurrent.
Condition 4 prevents short-loop pairs from being
declared concurrent.
"""
t: dict[str, Task] = _make_tasks("abcd")
traces: dict[tuple[Node, ...], int] = {
(t["a"], t["b"], t["c"],
t["b"], t["d"]): 10,
(t["a"], t["c"],
t["b"], t["d"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=1.0
)
self.assertFalse(
pdfg.is_concurrent(t["b"], t["c"])
)
class TestFiltering(unittest.TestCase):
"""Tests for edge filtering (Section 3.3)."""
def setUp(self) -> None:
"""Set up the test.
:return: None.
"""
traces: dict[tuple[Node, ...], int]
traces, self.__t = _make_paper_node_log()
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=0.2
)
self.__fdfg: FilteredDFG = FilteredDFG(
pdfg, eta=1.0
)
def test_filtered_edges_retain_best(self) -> None:
"""Filtered DFG retains best incoming/outgoing
edges.
Per Table 1 in the paper, edges (e,c) and (c,f)
should be dropped.
"""
t: dict[str, Task] = self.__t
edges: set[tuple[Node, Node]] = (
self.__fdfg.edges
)
# These should be retained (best edges)
self.assertIn((t["a"], t["b"]), edges)
self.assertIn((t["b"], t["e"]), edges)
self.assertIn((t["f"], t["g"]), edges)
self.assertIn((t["g"], t["h"]), edges)
def test_sources_and_sinks_preserved(self) -> None:
"""Filtering preserves sources and sinks."""
t: dict[str, Task] = self.__t
self.assertIn(
t["a"], self.__fdfg.sources
)
self.assertIn(
t["h"], self.__fdfg.sinks
)
class TestEndToEnd(unittest.TestCase):
"""End-to-end test for Split Miner."""
def test_paper_example_basic(self) -> None:
"""Split Miner produces a valid BPMN model.
The discovered model should have start/end events,
all 8 tasks, gateways, and proper connectivity.
"""
traces: dict[tuple[str, ...], int] = (
_make_paper_str_log()
)
model: BPMNModel = split_miner(
traces, epsilon=0.2, eta=0.4
)
# Has start and end
self.assertIsNotNone(model.start)
self.assertIsNotNone(model.end)
# Has all 8 tasks
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(
task_labels,
{"a", "b", "c", "d", "e", "f", "g", "h"},
)
# Has edges
self.assertGreater(len(model.edges), 0)
# Start has outgoing edge
self.assertGreater(
len(model.outgoing_edges(model.start)), 0
)
# End has incoming edge
self.assertGreater(
len(model.incoming_edges(model.end)), 0
)
def test_paper_example_has_gateways(self) -> None:
"""The paper example produces split and join
gateways.
Per Fig. 3c, the model should have both XOR and
AND gateways (or OR gateways that get minimized).
"""
traces: dict[tuple[str, ...], int] = (
_make_paper_str_log()
)
model: BPMNModel = split_miner(
traces, epsilon=0.2, eta=0.4
)
self.assertGreater(len(model.gateways), 0)
gw_types: set[GatewayType] = {
gw.gateway_type
for gw in model.gateways.values()
}
# Should have at least XOR or AND gateways
self.assertTrue(
GatewayType.XOR in gw_types
or GatewayType.AND in gw_types,
f"Expected XOR or AND gateways, "
f"got {gw_types}"
)
def test_paper_example_all_tasks_connected(
self,
) -> None:
"""Every task is reachable from start.
Verifies syntactic correctness: all tasks on a
path from start to end.
"""
traces: dict[tuple[str, ...], int] = (
_make_paper_str_log()
)
model: BPMNModel = split_miner(
traces, epsilon=0.2, eta=0.4
)
# BFS from start
reachable: set[Node] = set()
queue: deque[Node] = deque([model.start])
while queue:
node: Node = queue.popleft()
if node in reachable:
continue
reachable.add(node)
for _, succ in model.outgoing_edges(node):
queue.append(succ)
# All tasks should be reachable
for task in model.tasks.values():
self.assertIn(
task, reachable,
f"Task {task.label!r} not reachable "
f"from start"
)
# End should be reachable
self.assertIn(model.end, reachable)
def test_paper_example_all_tasks_reach_end(
self,
) -> None:
"""Every task can reach the end event.
Verifies syntactic correctness by backward BFS.
"""
traces: dict[tuple[str, ...], int] = (
_make_paper_str_log()
)
model: BPMNModel = split_miner(
traces, epsilon=0.2, eta=0.4
)
# Backward BFS from end
can_reach_end: set[Node] = set()
queue: deque[Node] = deque([model.end])
while queue:
node: Node = queue.popleft()
if node in can_reach_end:
continue
can_reach_end.add(node)
for pred, _ in model.incoming_edges(node):
queue.append(pred)
# All tasks should reach end
for task in model.tasks.values():
self.assertIn(
task, can_reach_end,
f"Task {task.label!r} cannot reach end"
)
def test_simple_sequence(self) -> None:
"""A simple sequential log produces no gateways."""
traces: dict[tuple[str, ...], int] = {
("a", "b", "c"): 10,
}
model: BPMNModel = split_miner(traces)
self.assertEqual(len(model.gateways), 0)
self.assertEqual(len(model.tasks), 3)
def test_simple_xor_choice(self) -> None:
"""A log with exclusive choice produces XOR
gateways.
Log: {<a,b,d>^10, <a,c,d>^10}
Expected: a -> XOR-split -> {b, c} ->
XOR-join -> d
"""
traces: dict[tuple[str, ...], int] = {
("a", "b", "d"): 10,
("a", "c", "d"): 10,
}
model: BPMNModel = split_miner(
traces, epsilon=0.1, eta=0.4
)
# Should have tasks a, b, c, d
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(
task_labels, {"a", "b", "c", "d"}
)
# Should have gateways
self.assertGreater(len(model.gateways), 0)
# All tasks reachable from start
reachable: set[Node] = set()
queue: deque[Node] = deque([model.start])
while queue:
node: Node = queue.popleft()
if node in reachable:
continue
reachable.add(node)
for _, s in model.outgoing_edges(node):
queue.append(s)
for task in model.tasks.values():
self.assertIn(task, reachable)
def test_simple_concurrency(self) -> None:
"""A log with concurrency produces AND gateways.
Log: {<a,b,c,d>^10, <a,c,b,d>^10}
b and c are concurrent.
"""
traces: dict[tuple[str, ...], int] = {
("a", "b", "c", "d"): 10,
("a", "c", "b", "d"): 10,
}
model: BPMNModel = split_miner(
traces, epsilon=1.0, eta=0.4
)
task_labels: set[str] = {
t.label for t in model.tasks.values()
if t.label is not None
}
self.assertEqual(
task_labels, {"a", "b", "c", "d"}
)
# Should have AND gateways for b||c
and_gws: list[Gateway] = [
gw for gw in model.gateways.values()
if gw.gateway_type == GatewayType.AND
]
self.assertGreater(
len(and_gws), 0,
"Expected AND gateways for concurrent "
"b and c"
)
class TestSelfLoopHandling(unittest.TestCase):
"""Tests for self-loop handling."""
def test_self_loop_restored(self) -> None:
"""Self-loops are restored in the final BPMN model.
A self-loop on task b should produce XOR-join and
XOR-split gateways around b with a back-edge.
"""
traces: dict[tuple[str, ...], int] = {
("a", "b", "c"): 10,
("a", "b", "b", "c"): 10,
("a", "b", "b", "b", "c"): 10,
}
model: BPMNModel = split_miner(traces)
# Task b should have a gateway predecessor
# and a gateway successor (the self-loop
# XOR-join and XOR-split)
b: Node = model.get_task("b")
b_preds: set[Node] = model.predecessors(b)
b_succs: set[Node] = model.successors(b)
# b should have exactly 1 predecessor (XOR-join)
# and 1 successor (XOR-split)
self.assertEqual(len(b_preds), 1)
self.assertEqual(len(b_succs), 1)
join_node: Node = next(iter(b_preds))
split_node: Node = next(iter(b_succs))
self.assertIsInstance(join_node, Gateway)
self.assertIsInstance(split_node, Gateway)
assert isinstance(join_node, Gateway)
assert isinstance(split_node, Gateway)
self.assertEqual(
join_node.gateway_type, GatewayType.XOR
)
self.assertEqual(
split_node.gateway_type, GatewayType.XOR
)
# Back-edge: split -> join
self.assertIn(
split_node,
model.predecessors(join_node)
)
class TestShortLoopHandling(unittest.TestCase):
"""Tests for short-loop handling."""
def test_short_loop_not_concurrent(self) -> None:
"""Short-loop pairs are excluded from concurrency.
If a and b form a short-loop, they must not be
declared concurrent even if they appear in both
orders.
"""
# a,b,a pattern = short-loop
t: dict[str, Task] = _make_tasks("abxy")
traces: dict[tuple[Node, ...], int] = {
(t["x"], t["a"], t["b"],
t["a"], t["y"]): 10,
(t["x"], t["b"], t["a"],
t["b"], t["y"]): 10,
(t["x"], t["a"], t["y"]): 10,
(t["x"], t["b"], t["y"]): 10,
}
dfg: DirectlyFollowsGraph = (
DirectlyFollowsGraph(traces)
)
# Should detect short-loop
self.assertIn(
(t["a"], t["b"]), dfg.short_loops
)
# Should NOT be concurrent
pdfg: PrunedDFG = PrunedDFG(
dfg, epsilon=1.0
)
self.assertFalse(
pdfg.is_concurrent(t["a"], t["b"])
)
if __name__ == "__main__":
unittest.main()
+651
View File
@@ -0,0 +1,651 @@
# Split Miner - BPMN process discovery from event logs.
# Authors:
# imacat@mail.imacat.idv.tw (imacat), 2026/3/11
# AI assistance: Claude Code (Anthropic)
# Copyright (c) 2026 imacat.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
# implied. See the License for the specific language governing
# permissions and limitations under the License.
"""Tests for SPQR-tree and RPST integration.
Tests that the spqrtree library is correctly integrated via
build_rpst(), using known graph decompositions from:
- Wikimedia SPQR tree example
- RPST paper (Polyvyanyy et al., 2011) Fig. 3(a)
- SM 1.0 paper Fig. 5(b)
References:
* https://commons.wikimedia.org/wiki/File:SPQR_tree_2.svg
* Polyvyanyy, A., Vanhatalo, J., & Voelzer, H. (2011).
Simplified computation and generalization of the refined
process structure tree. Lecture Notes in Computer
Science, 25-41.
"""
from __future__ import annotations
import unittest
from spqrtree import MultiGraph, NodeType, SPQRTree
from split_miner.bpmn import (
BPMNModel,
EndEvent,
Gateway,
GatewayType,
Node,
StartEvent,
Task,
)
from split_miner.joins import SESEFragment, build_rpst
def _make_serial_model() -> tuple[
BPMNModel, StartEvent, EndEvent,
Task, Task, Task,
]:
"""Build a serial chain: start -> a -> b -> c -> end.
:return: The model and its nodes.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
a: Task = Task("a", "a")
b: Task = Task("b", "b")
c: Task = Task("c", "c")
for t in [a, b, c]:
model.add_task(t)
model.add_edge(start, a)
model.add_edge(a, b)
model.add_edge(b, c)
model.add_edge(c, end)
return model, start, end, a, b, c
def _make_diamond_model() -> tuple[
BPMNModel, StartEvent, EndEvent,
Task, Task,
]:
"""Build a diamond: start -> {a, b} -> end.
:return: The model and its nodes.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
a: Task = Task("a", "a")
b: Task = Task("b", "b")
model.add_task(a)
model.add_task(b)
model.add_edge(start, a)
model.add_edge(start, b)
model.add_edge(a, end)
model.add_edge(b, end)
return model, start, end, a, b
def _make_fig5b_model() -> tuple[
BPMNModel, dict[str, Node],
]:
"""Build the model from Fig. 5(b) of the SM 1.0 paper.
Graph structure (after splits, before joins):
- start -> gx1 (XOR split)
- gx1 -> {a, b}
- a -> gx2 (XOR split), b -> gx3 (XOR split)
- gx2 -> {j, c}, gx3 -> {j, d}
- j -> i, c -> i, d -> k, i -> k
- k -> end
:return: The model and its named nodes.
"""
start: StartEvent = StartEvent("start")
end: EndEvent = EndEvent("end")
model: BPMNModel = BPMNModel(start, end)
tasks: dict[str, Task] = {}
for label in ["a", "b", "c", "d", "i", "j", "k"]:
t: Task = Task(label, label)
model.add_task(t)
tasks[label] = t
gx1: Gateway = Gateway("gx1", GatewayType.XOR)
gx2: Gateway = Gateway("gx2", GatewayType.XOR)
gx3: Gateway = Gateway("gx3", GatewayType.XOR)
model.add_gateway(gx1)
model.add_gateway(gx2)
model.add_gateway(gx3)
nodes: dict[str, Node] = {
"start": start, "end": end,
"gx1": gx1, "gx2": gx2, "gx3": gx3,
}
nodes.update(tasks)
for src, tgt in [
("start", "gx1"),
("gx1", "a"), ("gx1", "b"),
("a", "gx2"), ("b", "gx3"),
("gx2", "j"), ("gx3", "j"),
("gx2", "c"), ("gx3", "d"),
("j", "i"), ("c", "i"),
("d", "k"), ("i", "k"),
("k", "end"),
]:
model.add_edge(nodes[src], nodes[tgt])
return model, nodes
class TestSpqrTreeWikimedia(unittest.TestCase):
"""Tests SPQR-tree on the Wikimedia Commons example.
Tests the set of all SPQR-tree nodes (type + vertices)
regardless of root choice, since the unrooted tree
structure is unique but the rooting may vary.
Reference:
https://commons.wikimedia.org/wiki/File:SPQR_tree_2.svg
"""
def setUp(self) -> None:
"""Set up the Wikimedia example graph.
:return: None.
"""
mg: MultiGraph = MultiGraph()
for v in "abcdefghijklmnop":
mg.add_vertex(v)
for u, v in [
("a", "b"), ("a", "c"), ("a", "g"),
("b", "d"), ("b", "h"),
("c", "d"), ("c", "e"),
("d", "f"), ("e", "f"), ("e", "g"),
("f", "h"),
("h", "i"), ("h", "j"),
("i", "j"), ("i", "n"),
("j", "k"),
("k", "m"), ("k", "n"), ("m", "n"),
("l", "m"), ("l", "o"), ("l", "p"),
("m", "o"), ("m", "p"),
("o", "p"),
("g", "l"),
]:
mg.add_edge(u, v)
self.__tree: SPQRTree = SPQRTree(mg)
self.__all_nodes: list[
tuple[str, frozenset[str]]
] = []
_collect_all_nodes(
self.__tree.root, self.__all_nodes
)
def test_node_count(self) -> None:
"""The tree has 5 nodes (1 S, 1 P, 3 R).
:return: None.
"""
self.assertEqual(len(self.__all_nodes), 5)
def test_node_types(self) -> None:
"""Node types are P, R, R, R, S (sorted).
:return: None.
"""
types: list[str] = sorted(
t for t, _ in self.__all_nodes
)
self.assertEqual(
types, ["P", "R", "R", "R", "S"]
)
def test_s_node(self) -> None:
"""S-node has vertices {g, h, l, m}.
:return: None.
"""
s_nodes: list[frozenset[str]] = [
v for t, v in self.__all_nodes
if t == "S"
]
self.assertEqual(len(s_nodes), 1)
self.assertEqual(
s_nodes[0],
frozenset({"g", "h", "l", "m"}),
)
def test_p_node(self) -> None:
"""P-node has vertices {l, m}.
:return: None.
"""
p_nodes: list[frozenset[str]] = [
v for t, v in self.__all_nodes
if t == "P"
]
self.assertEqual(len(p_nodes), 1)
self.assertEqual(
p_nodes[0], frozenset({"l", "m"})
)
def test_r_node_1(self) -> None:
"""R-node {a,b,c,d,e,f,g,h} exists.
:return: None.
"""
r_verts: list[frozenset[str]] = [
v for t, v in self.__all_nodes
if t == "R"
]
self.assertIn(
frozenset({
"a", "b", "c", "d",
"e", "f", "g", "h",
}),
r_verts,
)
def test_r_node_2(self) -> None:
"""R-node {h,i,j,k,m,n} exists.
:return: None.
"""
r_verts: list[frozenset[str]] = [
v for t, v in self.__all_nodes
if t == "R"
]
self.assertIn(
frozenset({
"h", "i", "j", "k", "m", "n",
}),
r_verts,
)
def test_r_node_3(self) -> None:
"""R-node {l,m,o,p} exists.
:return: None.
"""
r_verts: list[frozenset[str]] = [
v for t, v in self.__all_nodes
if t == "R"
]
self.assertIn(
frozenset({"l", "m", "o", "p"}),
r_verts,
)
class TestSpqrTreeRpstFig3a(unittest.TestCase):
"""Tests SPQR-tree on RPST paper Fig. 3(a).
Reference: Polyvyanyy et al. (2011), Fig. 3(a).
Graph: s->u, u->{v,w}, v->{w,x}, w->x, x->y,
y->z (x2), z->y, z->t, plus back-edge t->s.
"""
def setUp(self) -> None:
"""Set up the RPST Fig 3a graph.
:return: None.
"""
mg: MultiGraph = MultiGraph()
for v in [
"s", "u", "v", "w", "x",
"y", "z", "t",
]:
mg.add_vertex(v)
for u, v in [
("s", "u"), ("u", "v"), ("u", "w"),
("v", "w"), ("v", "x"), ("w", "x"),
("x", "y"),
("y", "z"), ("y", "z"), ("z", "y"),
("z", "t"), ("t", "s"),
]:
mg.add_edge(u, v)
self.__tree: SPQRTree = SPQRTree(mg)
def test_root_type(self) -> None:
"""The root is an S-node.
:return: None.
"""
self.assertEqual(
self.__tree.root.type, NodeType.S
)
def test_root_vertices(self) -> None:
"""Root S-node contains {s,t,u,x,y,z}.
:return: None.
"""
verts: set[str] = _skeleton_vertices(
self.__tree.root
)
self.assertEqual(
verts, {"s", "t", "u", "x", "y", "z"}
)
def test_child_count(self) -> None:
"""The root has 2 children: R and P.
:return: None.
"""
self.assertEqual(
len(self.__tree.root.children), 2
)
def test_r_child(self) -> None:
"""R-node child has {u,v,w,x}.
:return: None.
"""
r1 = _find_child_by_vertices(
self.__tree.root, {"u", "v", "w", "x"}
)
self.assertIsNotNone(r1)
assert r1 is not None
self.assertEqual(r1.type, NodeType.R)
def test_p_child(self) -> None:
"""P-node child has {y,z}.
:return: None.
"""
p1 = _find_child_by_vertices(
self.__tree.root, {"y", "z"}
)
self.assertIsNotNone(p1)
assert p1 is not None
self.assertEqual(p1.type, NodeType.P)
def test_r_child_real_edges(self) -> None:
"""R-node has 5 real edges (the biconnected core).
:return: None.
"""
r1 = _find_child_by_vertices(
self.__tree.root, {"u", "v", "w", "x"}
)
assert r1 is not None
real: list[tuple[str, str]] = [
(e.u, e.v) for e in r1.skeleton.edges
if not e.virtual
]
self.assertEqual(len(real), 5)
def test_p_child_real_edges(self) -> None:
"""P-node has 3 real edges (y->z x2, z->y).
:return: None.
"""
p1 = _find_child_by_vertices(
self.__tree.root, {"y", "z"}
)
assert p1 is not None
real: list[tuple[str, str]] = [
(e.u, e.v) for e in p1.skeleton.edges
if not e.virtual
]
self.assertEqual(len(real), 3)
class TestBuildRpstSerial(unittest.TestCase):
"""Tests build_rpst on a serial chain."""
def setUp(self) -> None:
"""Set up a serial model: start->a->b->c->end.
:return: None.
"""
model: BPMNModel
model, _, _, _, _, _ = _make_serial_model()
self.__fragments: list[SESEFragment] = (
build_rpst(model)
)
def test_single_fragment(self) -> None:
"""A serial chain produces one S-type fragment.
:return: None.
"""
self.assertEqual(len(self.__fragments), 1)
def test_fragment_type(self) -> None:
"""The fragment is S-type (serial).
:return: None.
"""
self.assertEqual(
self.__fragments[0].fragment_type,
NodeType.S,
)
def test_fragment_edges(self) -> None:
"""The fragment contains all 4 edges.
:return: None.
"""
self.assertEqual(
len(self.__fragments[0].edges), 4
)
def test_fragment_nodes(self) -> None:
"""The fragment contains all 5 nodes.
:return: None.
"""
self.assertEqual(
len(self.__fragments[0].nodes), 5
)
class TestBuildRpstDiamond(unittest.TestCase):
"""Tests build_rpst on a diamond graph."""
def setUp(self) -> None:
"""Set up a diamond: start->{a,b}->end.
:return: None.
"""
model: BPMNModel
model, _, _, _, _ = _make_diamond_model()
self.__fragments: list[SESEFragment] = (
build_rpst(model)
)
def test_fragment_count(self) -> None:
"""Diamond produces 3 fragments (2 S + 1 P).
:return: None.
"""
self.assertEqual(len(self.__fragments), 3)
def test_has_p_fragment(self) -> None:
"""There is a P-type (parallel) fragment.
:return: None.
"""
p_frags: list[SESEFragment] = [
f for f in self.__fragments
if f.fragment_type == NodeType.P
]
self.assertEqual(len(p_frags), 1)
def test_p_fragment_covers_all(self) -> None:
"""The P-type fragment contains all 4 edges.
:return: None.
"""
p_frag: SESEFragment = [
f for f in self.__fragments
if f.fragment_type == NodeType.P
][0]
self.assertEqual(len(p_frag.edges), 4)
def test_p_fragment_nodes(self) -> None:
"""The P-type fragment contains all 4 nodes.
:return: None.
"""
p_frag: SESEFragment = [
f for f in self.__fragments
if f.fragment_type == NodeType.P
][0]
self.assertEqual(len(p_frag.nodes), 4)
def test_s_fragments(self) -> None:
"""Two S-type fragments (one per branch).
:return: None.
"""
s_frags: list[SESEFragment] = [
f for f in self.__fragments
if f.fragment_type == NodeType.S
]
self.assertEqual(len(s_frags), 2)
def test_s_fragment_edges(self) -> None:
"""Each S-type fragment has 2 edges.
:return: None.
"""
for f in self.__fragments:
if f.fragment_type == NodeType.S:
self.assertEqual(len(f.edges), 2)
def test_bottom_up_order(self) -> None:
"""Fragments are ordered bottom-up (small first).
:return: None.
"""
sizes: list[int] = [
len(f.edges) for f in self.__fragments
]
self.assertEqual(sizes, sorted(sizes))
class TestBuildRpstFig5b(unittest.TestCase):
"""Tests build_rpst on SM 1.0 paper Fig. 5(b)."""
def setUp(self) -> None:
"""Set up the Fig. 5(b) model.
:return: None.
"""
model: BPMNModel
model, _ = _make_fig5b_model()
self.__fragments: list[SESEFragment] = (
build_rpst(model)
)
def test_has_fragments(self) -> None:
"""At least one fragment is produced.
:return: None.
"""
self.assertGreater(len(self.__fragments), 0)
def test_all_edges_covered(self) -> None:
"""Union of fragment edges covers all model edges.
:return: None.
"""
model: BPMNModel
model, _ = _make_fig5b_model()
all_frag_edges: set[tuple[Node, Node]] = set()
for f in self.__fragments:
all_frag_edges |= f.edges
self.assertEqual(all_frag_edges, model.edges)
def test_has_r_fragment(self) -> None:
"""There is at least one R-type (rigid) fragment.
:return: None.
"""
r_frags: list[SESEFragment] = [
f for f in self.__fragments
if f.fragment_type == NodeType.R
]
self.assertGreater(len(r_frags), 0)
def test_bottom_up_order(self) -> None:
"""Fragments are ordered bottom-up (small first).
:return: None.
"""
sizes: list[int] = [
len(f.edges) for f in self.__fragments
]
self.assertEqual(sizes, sorted(sizes))
def test_entry_exit_are_nodes(self) -> None:
"""Entry and exit of each fragment are model nodes.
:return: None.
"""
model: BPMNModel
model, _ = _make_fig5b_model()
all_nodes: set[Node] = model.all_nodes
for f in self.__fragments:
self.assertIn(f.entry, all_nodes)
self.assertIn(f.exit_node, all_nodes)
def _collect_all_nodes(
spqr_node,
result: list[tuple[str, frozenset]],
) -> None:
"""Collect all SPQR-tree nodes as (type, vertices).
:param spqr_node: The SPQR-tree node.
:param result: The output list.
"""
verts: frozenset = frozenset(
_skeleton_vertices(spqr_node)
)
result.append((spqr_node.type.name, verts))
for child in spqr_node.children:
_collect_all_nodes(child, result)
def _skeleton_vertices(spqr_node) -> set:
"""Extract vertex set from an SPQR-tree node skeleton.
:param spqr_node: The SPQR-tree node.
:return: The set of vertices.
"""
verts: set = set()
for e in spqr_node.skeleton.edges:
verts.add(e.u)
verts.add(e.v)
return verts
def _find_child_by_vertices(
parent, target_verts: set
):
"""Find a child SPQR node by its vertex set.
:param parent: The parent SPQR-tree node.
:param target_verts: The expected vertex set.
:return: The matching child, or None.
"""
for child in parent.children:
if _skeleton_vertices(child) == target_verts:
return child
return None
if __name__ == "__main__":
unittest.main()