DEVX-62: feat: weighted LPT distribution, workflow fixes, decouple vikunja/sync-wiki from release
Post-merge / detect-type (push) Successful in 30s
Post-merge / validate-commit-msg (push) Successful in 40s
Post-merge / vikunja (push) Successful in 44s
Post-merge / release (push) Successful in 59s
Post-merge / badges (push) Successful in 58s
Post-merge / sync-wiki (push) Successful in 1m2s
Post-merge / configure-repo (push) Successful in 37s

This commit was merged in pull request #102.
This commit is contained in:
2026-06-26 17:57:03 +00:00
parent e271c79e93
commit 41c631d5f5
8 changed files with 301 additions and 95 deletions
+31 -6
View File
@@ -1,10 +1,14 @@
#!/usr/bin/env python3
"""Distribute a list of files across N parallel runners (round-robin).
"""Distribute a list of files across N parallel runners using LPT scheduling.
Generic file-based test distribution for CI matrix jobs. Discovers files
matching a glob pattern, sorts them for deterministic ordering, then
assigns them round-robin to *max_runners* groups. The assigned group for
*runner_index* is written to ``$GITHUB_ENV`` for use by subsequent steps.
assigns them to *max_runners* groups using LPT (Longest Processing Time
first) scheduling — files are weighted by size (as a proxy for test
runtime) and assigned to the runner with the least total weight.
The assigned group for *runner_index* is written to ``$GITHUB_ENV`` for
use by subsequent steps.
Usage::
@@ -32,11 +36,32 @@ def discover_files(pattern: str) -> list[str]:
return sorted(glob.glob(pattern))
def _file_weight(path: str) -> int:
"""Estimate a weight for a file based on its size in bytes.
Falls back to 1 if the file cannot be stat'd (e.g. in tests).
"""
try:
return max(1, os.path.getsize(path))
except OSError:
return 1
def distribute(files: list[str], max_runners: int) -> list[list[str]]:
"""Split *files* into *max_runners* balanced groups (round-robin)."""
"""Split *files* into *max_runners* balanced groups using LPT scheduling.
Files are weighted by size (as a proxy for runtime) and assigned to
the runner with the least total weight.
"""
weights = [_file_weight(f) for f in files]
groups: list[list[str]] = [[] for _ in range(max_runners)]
for i, f in enumerate(files):
groups[i % max_runners].append(f)
loads = [0] * max_runners
# Sort by weight descending, preserving original order for ties
indexed = sorted(enumerate(files), key=lambda x: (-weights[x[0]], x[0]))
for orig_idx, f in indexed:
min_runner = min(range(max_runners), key=lambda r: loads[r])
groups[min_runner].append(f)
loads[min_runner] += weights[orig_idx]
return groups
+62 -10
View File
@@ -132,14 +132,63 @@ def build_multi_role_pairs(
return [MultiRoleTestPair(r, s, p) for r, s in role_scenarios for p in platforms]
def distribute_multi_role(pairs: list[MultiRoleTestPair], max_runners: int) -> list[list[MultiRoleTestPair]]:
"""Split *pairs* into *max_runners* balanced groups (round-robin)."""
groups: list[list[MultiRoleTestPair]] = [[] for _ in range(max_runners)]
for i, pair in enumerate(pairs):
groups[i % max_runners].append(pair)
# Heuristic weights for known heavy molecule scenarios.
# These are estimated from CI run times — scenarios that pull large Docker
# images or run complex Ansible playbooks take longer.
_SCENARIO_WEIGHTS: dict[str, int] = {
"nextcloud": 10,
"gitea": 8,
"vaultwarden": 7,
"zitadel": 7,
"postgresql": 6,
"redis": 5,
"backup": 5,
"docker-base": 4,
"default": 3,
"binary": 2,
}
_DEFAULT_SCENARIO_WEIGHT = 3
def _scenario_weight(scenario: str) -> int:
"""Estimate a weight for a scenario based on its name."""
s = scenario.lower()
for key, weight in _SCENARIO_WEIGHTS.items():
if key in s:
return weight
return _DEFAULT_SCENARIO_WEIGHT
def _lpt_distribute[T](items: list[T], weights: list[int], max_runners: int) -> list[list[T]]:
"""Distribute *items* across *max_runners* using LPT (Longest Processing Time first).
Sorts items by weight (descending), then assigns each to the runner
with the least total weight. This produces a more balanced distribution
than naive round-robin when items have varying costs.
"""
groups: list[list[T]] = [[] for _ in range(max_runners)]
loads = [0] * max_runners
# Sort by weight descending, preserving original order for ties
indexed = sorted(enumerate(items), key=lambda x: (-weights[x[0]], x[0]))
for orig_idx, item in indexed:
# Find the runner with the minimum load
min_runner = min(range(max_runners), key=lambda r: loads[r])
groups[min_runner].append(item)
loads[min_runner] += weights[orig_idx]
return groups
def distribute_multi_role(pairs: list[MultiRoleTestPair], max_runners: int) -> list[list[MultiRoleTestPair]]:
"""Split *pairs* into *max_runners* balanced groups using LPT scheduling.
Each pair is weighted by scenario name heuristics (e.g. ``nextcloud`` is
heavier than ``binary``). Pairs are sorted by weight descending and
assigned to the runner with the least total weight.
"""
weights = [_scenario_weight(p.scenario) for p in pairs]
return _lpt_distribute(pairs, weights, max_runners)
def multi_role_pairs_for_runner(
pairs: list[MultiRoleTestPair], runner_index: int, max_runners: int
) -> list[MultiRoleTestPair]:
@@ -153,11 +202,14 @@ def multi_role_pairs_for_runner(
def distribute(pairs: list[TestPair], max_runners: int) -> list[list[TestPair]]:
"""Split *pairs* into *max_runners* balanced groups (round-robin)."""
groups: list[list[TestPair]] = [[] for _ in range(max_runners)]
for i, pair in enumerate(pairs):
groups[i % max_runners].append(pair)
return groups
"""Split *pairs* into *max_runners* balanced groups using LPT scheduling.
Each pair is weighted by scenario name heuristics (e.g. ``nextcloud`` is
heavier than ``binary``). Pairs are sorted by weight descending and
assigned to the runner with the least total weight.
"""
weights = [_scenario_weight(p.scenario) for p in pairs]
return _lpt_distribute(pairs, weights, max_runners)
def pairs_for_runner(pairs: list[TestPair], runner_index: int, max_runners: int) -> list[TestPair]: