Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions graphify/extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -4320,6 +4320,15 @@ def _extract_parallel(
max_workers = min(max_workers, 61)
max_workers = max(max_workers, 1)

# A one-worker pool buys no parallelism: it still pays process spawn plus an
# IPC round trip per file, and it is the one residual case where the parent's
# rebuild watchdog (os._exit) can orphan a worker that is mid-task. The
# Windows post-commit hook exports GRAPHIFY_MAX_WORKERS=1, so this is the
# default there. Hand the work back so the caller extracts sequentially in
# this process instead (#2173).
if max_workers == 1:
return False

# root anchors hash keys / node ids / XAML boundary; cache_location is where
# the cache dir is written (defaults to root when not decoupled) (#1774).
root_str = str(root)
Expand Down
57 changes: 57 additions & 0 deletions tests/test_extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -1488,6 +1488,63 @@ def submit(self, *a, **kw):
assert "__main__" in out, "warning must hint at the Windows __main__ guard idiom"


def test_extract_parallel_skips_pool_when_max_workers_is_one(tmp_path, monkeypatch):
"""#2173: a resolved worker count of 1 must not spawn a ProcessPoolExecutor.

The Windows post-commit hook exports GRAPHIFY_MAX_WORKERS=1, so before this the
rebuild spawned a one-worker pool for >= _PARALLEL_THRESHOLD files: no
parallelism, one process spawn plus an IPC round trip per file, and the only
window where the parent's rebuild watchdog (os._exit) can orphan a worker
mid-task. _extract_parallel must decline (return False) so the caller extracts
sequentially in-process.
"""
import concurrent.futures
from graphify import extract as extract_mod

spawned = {"count": 0}

def fake_pool(*args, **kwargs):
spawned["count"] += 1
raise AssertionError("ProcessPoolExecutor must not be constructed for 1 worker")

monkeypatch.setattr(concurrent.futures, "ProcessPoolExecutor", fake_pool)
monkeypatch.setenv("GRAPHIFY_MAX_WORKERS", "1")

uncached = [(i, FIXTURES / "sample.py") for i in range(25)] # >= _PARALLEL_THRESHOLD
per_file: list = [None] * len(uncached)

ok = extract_mod._extract_parallel(uncached, per_file, tmp_path, None, len(uncached))
assert ok is False, "must hand the work back for sequential extraction"
assert spawned["count"] == 0, "no pool may be spawned when max_workers resolves to 1"


def test_extract_parallel_still_spawns_pool_for_multiple_workers(tmp_path, monkeypatch):
"""Guard the #2173 skip: >1 worker must still take the pool path."""
import concurrent.futures
from graphify import extract as extract_mod

spawned = {"count": 0}

class FakePool:
def __init__(self, *a, **kw):
spawned["count"] += 1
def __enter__(self):
return self
def __exit__(self, *a):
return False
def submit(self, *a, **kw):
raise concurrent.futures.process.BrokenProcessPool("stop here")

monkeypatch.setattr(concurrent.futures, "ProcessPoolExecutor", FakePool)
monkeypatch.setenv("GRAPHIFY_MAX_WORKERS", "4")

uncached = [(i, FIXTURES / "sample.py") for i in range(25)]
per_file: list = [None] * len(uncached)

extract_mod._extract_parallel(uncached, per_file, tmp_path, None, len(uncached))
assert spawned["count"] == 1, "multi-worker runs must still use the pool"


# ---------------------------------------------------------------------------
# Bash extractor tests (#866)
# ---------------------------------------------------------------------------
Expand Down