diff --git a/README.md b/README.md index 272c5f6f81..47a78dc6d3 100644 --- a/README.md +++ b/README.md @@ -524,6 +524,7 @@ These are only needed for **headless / CI extraction** (`graphify extract`). Whe | `GRAPHIFY_MAX_OUTPUT_TOKENS` | Raise output cap for dense corpora | optional — e.g. `32768` for large files | | `GRAPHIFY_API_TIMEOUT` | Per-call timeout in seconds for HTTP, claude-cli, Anthropic SDK, and Bedrock backends (default: 600) | optional — also `--api-timeout` flag | | `GRAPHIFY_MAX_RETRIES` | How many times to retry a rate-limited (429) request before giving up (default: 6; honors `Retry-After`) | optional — raise for strict per-org limits (e.g. kimi); `0` disables | +| `GRAPHIFY_MAX_RETRY_DEPTH` | How deep a truncated chunk may be bisected and re-extracted (default: 3, so up to 8x sub-calls for one chunk) | optional — lower it to cap worst-case spend; `0` disables every retry (no bisection, no hollow-response retry), so a chunk costs exactly one call | | `GRAPHIFY_FORCE` | Force graph rebuild even with fewer nodes | optional — also `--force` flag | | `GRAPHIFY_GOOGLE_WORKSPACE` | Auto-enable Google Workspace export | optional — set to `1` | | `GRAPHIFY_TRIAGE_BACKEND` | Backend for `graphify prs --triage` | optional — auto-detected from available keys | diff --git a/graphify/llm.py b/graphify/llm.py index a05c9d4632..5f40a4c1a2 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -433,6 +433,30 @@ def _resolve_max_retries(default: int = 6) -> int: return default +def _resolve_max_retry_depth(default: int = 3) -> int: + """How deep adaptive retry may bisect a truncated chunk. + + A chunk of N files can split into up to ``2**depth`` pieces, so this is the + knob that bounds worst-case cost. It used to be a Python-API kwarg only, + with no way for a `graphify extract` operator to lower it — or set it to 0 — + as a mitigation (#2880). Honour GRAPHIFY_MAX_RETRY_DEPTH. + + ``0`` means no retries of any kind: no bisection, and no same-chunk retry of + a hollow response either. It is set to cap spend, so it has to hold for + every retry path, not only the one it names — see + :func:`_extract_with_adaptive_retry`. One call per chunk, full stop. + """ + raw = os.environ.get("GRAPHIFY_MAX_RETRY_DEPTH", "").strip() + if raw: + try: + v = int(raw) + if v >= 0: + return v + except ValueError: + pass + return default + + def _thinking_disabled_via_env() -> bool: """Opt-in (GRAPHIFY_DISABLE_THINKING) to send ``{"thinking": {"type": "disabled"}}`` to reasoning-capable OpenAI-compatible models such as ``deepseek-v4-flash``. @@ -1093,9 +1117,10 @@ def _bedrock_response_text(resp: dict, default: str = "") -> str: API does not promise a text block is first: reasoning-capable models emit a ``reasoningContent`` block ahead of the answer, and ``toolUse`` or future block types can precede it too. Indexing position 0 therefore yields no text - at all for those models, which reads downstream as a hollow response, gets - reclassified as truncation, and sends the chunk into bisection that cannot - converge. Select on the block's shape instead of its position so this holds + at all for those models, which reads downstream as a hollow response and + costs the chunk a round of retries before it is failed (before #2880 it was + reclassified as truncation and bisected, which could not converge at all). + Select on the block's shape instead of its position so this holds for any model; a response whose first block is already text is unaffected. """ content = resp.get("output", {}).get("message", {}).get("content", []) @@ -1118,9 +1143,8 @@ def _response_is_hollow(raw_content: str | None, parsed: dict) -> bool: JSON prefix that fails to parse. All of these collapse to a "successful" call producing zero nodes and zero edges. Without this check the chunk is silently dropped from the corpus because no exception is raised and - `finish_reason` is `"stop"` rather than `"length"`. By flagging the - result as hollow, callers can re-route it through the same bisection - path used for context-window overflow and `finish_reason="length"`. + `finish_reason` is `"stop"` rather than `"length"`. Callers flag it with + :func:`_mark_hollow` so the adaptive-retry layer can recover it. """ if raw_content is None or not raw_content.strip(): return True @@ -1130,6 +1154,42 @@ def _response_is_hollow(raw_content: str | None, parsed: dict) -> bool: return not nodes and not edges and not hyperedges +# Backoff between same-chunk retries of a hollow response (#2880). Two entries +# ⇒ at most three calls per chunk, versus the 15 the bisection path could spend. +_HOLLOW_BACKOFF_S = (2.0, 8.0) + + +def _mark_hollow(result: dict, raw_content: str | None, backend: str | None) -> dict: + """Label a hollow response so adaptive retry retries it, without bisecting. + + Hollow and truncated are different failures with different remedies, and + labelling hollow as `finish_reason="length"` conflated them (#2880): + + - **truncated** — the model ran out of `max_completion_tokens` mid-JSON. + Bisecting is the correct recovery: smaller input ⇒ shorter output. + - **hollow** — HTTP 200 with empty/null/whitespace content, or content that + parses to zero nodes and zero edges (a rate limit, a transport hiccup, a + refusal, an agentic prose reply, a reasoning-first content block). + + Bisecting a hollow response cannot converge: both halves go to the same + misbehaving backend and come back hollow too, so one bad response cost + `2**max_retry_depth` billed calls — up to 15 per chunk at the default + depth, all of them failing. `_extract_with_adaptive_retry` retries the + *same* chunk with backoff instead. + """ + if _response_is_hollow(raw_content, result) and result.get("finish_reason") != "length": + print( + f"[graphify] {backend or 'backend'} returned a hollow response " + f"(content={'empty' if not (raw_content or '').strip() else 'no nodes/edges'}, " + f"output_tokens={result.get('output_tokens', 0)}); " + "will retry the same chunk (a hollow response is not a size problem, " + "so the chunk is not bisected).", + file=sys.stderr, + ) + result["finish_reason"] = "hollow" + return result + + def _backend_env_keys(backend: str) -> list[str]: """Return accepted API-key environment variables for a backend.""" cfg = BACKENDS[backend] @@ -1308,17 +1368,9 @@ def _call_openai_compat( # An overwhelmed local model (typically Ollama) can return HTTP 200 with # empty / null content or unparseable half-generated JSON. The call looks # successful, `finish_reason` is `"stop"`, and the chunk would be silently - # dropped from the corpus. Re-label as `"length"` so the adaptive retry - # layer bisects the chunk — same recovery as a true truncation. - if _response_is_hollow(raw_content, result) and result["finish_reason"] != "length": - print( - f"[graphify] {backend or 'backend'} returned a hollow response " - f"(content={'empty' if not (raw_content or '').strip() else 'no nodes/edges'}, " - f"output_tokens={result['output_tokens']}); " - "treating as truncation so adaptive retry can bisect the chunk.", - file=sys.stderr, - ) - result["finish_reason"] = "length" + # dropped from the corpus. Label it hollow so the adaptive retry layer + # retries the same chunk — see _mark_hollow for why not bisection. + _mark_hollow(result, raw_content, backend) output_tokens = result["output_tokens"] if output_tokens < 50 and backend == "ollama": print( @@ -1361,13 +1413,7 @@ def _call_claude(api_key: str, model: str, user_message: str, max_tokens: int = # vocabulary so the adaptive-retry layer doesn't have to know which # backend produced the result. result["finish_reason"] = "length" if resp.stop_reason == "max_tokens" else "stop" - if _response_is_hollow(raw_content, result) and result["finish_reason"] != "length": - print( - "[graphify] claude returned a hollow response; treating as " - "truncation so adaptive retry can bisect the chunk.", - file=sys.stderr, - ) - result["finish_reason"] = "length" + _mark_hollow(result, raw_content, "claude") return result @@ -1527,9 +1573,10 @@ def _call_claude_cli(user_message: str, max_tokens: int = 8192, *, deep_mode: bo # the user turn is only a raw file dump with no request, reply # conversationally ("I see the file, but there's no actual request # attached — what would you like me to do with it?"). That prose parses to - # zero nodes/edges, so _response_is_hollow flags it as truncation and the - # adaptive-retry path bisects the chunk indefinitely, never converging and - # never writing graph.json (verified against Claude Code 2.1.197). + # zero nodes/edges, so _response_is_hollow flags it and the chunk is + # retried and then failed rather than extracted (verified against Claude + # Code 2.1.197). Before #2880 it was misread as truncation and bisected + # indefinitely, never converging and never writing graph.json. # # Putting the full extraction schema plus an explicit imperative in the # user turn — and dropping --system-prompt — makes the CLI emit the JSON @@ -1575,8 +1622,8 @@ def _call_claude_cli(user_message: str, max_tokens: int = 8192, *, deep_mode: bo # Claude Code releases increasingly treat a bare file-dump prompt as an # agentic task and REPORT the extraction in prose ("Knowledge graph # extracted — 21 nodes, 20 edges…") instead of returning it; that parses to - # zero nodes, reads as truncation, and gets bisected without ever - # converging (#2076). --json-schema pins the object shape regardless of + # zero nodes and reads as hollow (#2076 — and before #2880, as truncation + # to be bisected without ever converging). --json-schema pins the shape regardless of # that framing; the user-turn prompt above stays as the fallback for older # CLIs that predate the flag. if _claude_cli_supports_json_schema(claude_cmd): @@ -1624,13 +1671,7 @@ def _call_claude_cli(user_message: str, max_tokens: int = 8192, *, deep_mode: bo result["model"] = next(iter(model_usage), "claude-code-plan") stop_reason = envelope.get("stop_reason", "") result["finish_reason"] = "length" if stop_reason == "max_tokens" else "stop" - if _response_is_hollow(raw_content, result) and result["finish_reason"] != "length": - print( - "[graphify] claude-cli returned a hollow response; treating as " - "truncation so adaptive retry can bisect the chunk.", - file=sys.stderr, - ) - result["finish_reason"] = "length" + _mark_hollow(result, raw_content, "claude-cli") return result @@ -1687,13 +1728,7 @@ def _call_azure( result["output_tokens"] = resp.usage.completion_tokens if resp.usage else 0 result["model"] = model result["finish_reason"] = resp.choices[0].finish_reason - if _response_is_hollow(raw_content, result) and result["finish_reason"] != "length": - print( - "[graphify] azure returned a hollow response; treating as " - "truncation so adaptive retry can bisect the chunk.", - file=sys.stderr, - ) - result["finish_reason"] = "length" + _mark_hollow(result, raw_content, "azure") return result @@ -1744,13 +1779,7 @@ def _call_bedrock(model: str, user_message: str, max_tokens: int = 8192, *, deep result["output_tokens"] = usage.get("outputTokens", 0) result["model"] = model result["finish_reason"] = "length" if resp.get("stopReason") == "max_tokens" else "stop" - if _response_is_hollow(text, result) and result["finish_reason"] != "length": - print( - "[graphify] bedrock returned a hollow response; treating as " - "truncation so adaptive retry can bisect the chunk.", - file=sys.stderr, - ) - result["finish_reason"] = "length" + _mark_hollow(result, text, "bedrock") return result @@ -2112,9 +2141,11 @@ def _extract_with_adaptive_retry( - hollow successful responses — the model returned HTTP 200 with empty, null, or unparseable content (typical of a local Ollama under load). - `_call_openai_compat` re-labels these as `finish_reason="length"` so they - take the same recovery path; without that the chunk would be silently - dropped from the corpus. + These do NOT bisect: a hollow response is a backend problem, not a size + problem, and both halves come back hollow from the same backend, so + bisection cannot converge and costs `2**max_depth` billed calls (#2880). + The *same* chunk is retried with backoff instead, and the chunk fails + loudly if it is still hollow. - recognized timeout exceptions — dense chunks can take long enough to hit `GRAPHIFY_API_TIMEOUT` before returning output. For `claude-cli`, @@ -2162,6 +2193,27 @@ def _split_lone_slice() -> "tuple[FileSlice, FileSlice] | None": result = extract_files_direct( chunk, backend=backend, api_key=api_key, model=model, root=root, deep_mode=deep_mode ) + # A hollow response is retried as-is, with backoff — see _mark_hollow. + # Bounded by a fixed number of attempts, so one misbehaving backend + # costs at most _HOLLOW_BACKOFF_S + 1 calls per chunk instead of the + # 2**max_depth the bisection path used to spend (#2880). + # + # max_depth=0 means "no retries", and an operator sets it to cap spend, + # so it has to hold for the hollow path too: one call per chunk, full + # stop. Bounding only the bisection depth would still let a misbehaving + # backend triple the call count of a run that asked for no retries. + for _delay in (_HOLLOW_BACKOFF_S if max_depth > 0 else ()): + if result.get("finish_reason") != "hollow": + break + print( + f"[graphify] retrying the same chunk of {len(chunk)} in {_delay:g}s " + f"after a hollow response", + file=sys.stderr, + ) + time.sleep(_delay) + result = extract_files_direct( + chunk, backend=backend, api_key=api_key, model=model, root=root, deep_mode=deep_mode + ) except Exception as exc: # noqa: BLE001 — re-raise unless it's a known context overflow or timeout is_timeout = _looks_like_timeout(exc) if not (_looks_like_context_exceeded(exc) or is_timeout): @@ -2214,6 +2266,27 @@ def _split_lone_slice() -> "tuple[FileSlice, FileSlice] | None": "_partial_files": _merged_partial_files(left, right), } + if result.get("finish_reason") == "hollow": + # Still hollow after every retry. Fail the chunk loudly rather than + # bisecting into a fan-out that cannot converge (#2880): the files are + # marked partial so the next run re-dispatches them, and they are not + # promoted to the semantic cache as authoritative. + _attempts = (len(_HOLLOW_BACKOFF_S) + 1) if max_depth > 0 else 1 + print( + f"[graphify] chunk of {len(chunk)} still hollow after " + f"{_attempts} attempt(s) — giving up on this chunk. " + f"Its files are marked for re-extraction on the next run. A hollow " + f"response usually means a rate limit, a transport hiccup, a refusal, " + f"or a model that answered in prose rather than JSON.", + file=sys.stderr, + ) + _mark_partial(result) + result["_partial_files"] = sorted( + set(_chunk_partial_files(chunk)) | set(result.get("_partial_files", []) or []) + ) + result["finish_reason"] = "stop" + return result + if result.get("finish_reason") != "length": return result @@ -2296,7 +2369,7 @@ def extract_corpus_parallel( on_chunk_done: Callable | None = None, token_budget: int | None = 60_000, max_concurrency: int = 4, - max_retry_depth: int = 3, + max_retry_depth: int | None = None, deep_mode: bool = False, cache_root: "Path | None" = None, ) -> dict: @@ -2319,10 +2392,16 @@ def extract_corpus_parallel( - When the LLM returns `finish_reason="length"` (output truncated at `max_completion_tokens`), the chunk is split in half and each half re-extracted recursively, up to `max_retry_depth` levels deep - (default 3 → max 8x expansion of one chunk). + (default 3 → max 8x expansion of one chunk). Leave it None to take + the default, overridable by GRAPHIFY_MAX_RETRY_DEPTH so an operator + can lower it without a code change (#2880). - This is signal-driven: chunks too dense to fit in one response self-heal by splitting until they do, while well-sized chunks pay - no extra cost. Set `max_retry_depth=0` to disable retries. + no extra cost. + - Hollow responses (HTTP 200, no usable content) are NOT bisected — + the same chunk is retried with backoff, then fails loudly. + - `max_retry_depth=0` disables retries of BOTH kinds: no bisection + and no same-chunk hollow retry, so a chunk costs exactly one call. `on_chunk_done(idx, total, chunk_result)` fires once per chunk as it completes (in completion order, not submission order). `idx` is the @@ -2345,6 +2424,8 @@ def extract_corpus_parallel( Accepts ``str`` paths as well as ``Path``; string entries are coerced up front so packing/slicing helpers can rely on ``Path`` semantics (#1386). """ + if max_retry_depth is None: + max_retry_depth = _resolve_max_retry_depth() files = [f if isinstance(f, (Path, FileSlice)) else Path(f) for f in files] # Split oversized splittable documents into slices that cover the whole file # before packing, so content past _FILE_CHAR_CAP is extracted instead of diff --git a/tests/test_llm_backends.py b/tests/test_llm_backends.py index c574f21375..4480eff629 100644 --- a/tests/test_llm_backends.py +++ b/tests/test_llm_backends.py @@ -545,11 +545,11 @@ def create(self, **__): monkeypatch.setitem(sys.modules, "openai", fake_module) -def test_call_openai_compat_relabels_empty_content_as_length(monkeypatch): +def test_call_openai_compat_labels_empty_content_hollow(monkeypatch): # Simulates an overwhelmed Ollama: HTTP 200, empty content, finish_reason - # "stop", zero completion tokens. Pre-fix this would silently return an - # empty fragment and the chunk would be dropped. Post-fix `finish_reason` - # is rewritten to "length" so the adaptive retry layer bisects. + # "stop", zero completion tokens. The chunk must not be dropped silently, + # but it must not be labelled "length" either: bisecting a hollow response + # cannot converge and costs 2**max_retry_depth billed calls (#2880). fake_resp = _fake_openai_response("", finish_reason="stop", completion_tokens=0) _install_fake_openai(monkeypatch, fake_resp) @@ -557,13 +557,13 @@ def test_call_openai_compat_relabels_empty_content_as_length(monkeypatch): "http://localhost:11434/v1", "ollama", "qwen2.5-coder:7b", "user msg", temperature=0, max_completion_tokens=8192, backend="ollama", ) - assert result["finish_reason"] == "length", ( - "empty content from a 'successful' call must be re-labelled so the " - "adaptive retry layer treats it as a truncation and bisects the chunk" + assert result["finish_reason"] == "hollow", ( + "empty content from a 'successful' call must be labelled hollow so the " + "adaptive retry layer retries the same chunk instead of bisecting it" ) -def test_call_openai_compat_relabels_none_content_as_length(monkeypatch): +def test_call_openai_compat_labels_none_content_hollow(monkeypatch): fake_resp = _fake_openai_response(None, finish_reason="stop") _install_fake_openai(monkeypatch, fake_resp) @@ -571,13 +571,13 @@ def test_call_openai_compat_relabels_none_content_as_length(monkeypatch): "http://localhost:11434/v1", "ollama", "qwen2.5-coder:7b", "u", temperature=0, max_completion_tokens=8192, backend="ollama", ) - assert result["finish_reason"] == "length" + assert result["finish_reason"] == "hollow" -def test_call_openai_compat_relabels_unparseable_json_as_length(monkeypatch): +def test_call_openai_compat_labels_unparseable_json_hollow(monkeypatch): # A half-generated response: `{"nodes": [{"id":` parses to {} (empty # fragment) via _parse_llm_json's JSONDecodeError fallback. That is also - # hollow and must trigger bisection. + # hollow. fake_resp = _fake_openai_response('{"nodes": [{"id":', finish_reason="stop", completion_tokens=20) _install_fake_openai(monkeypatch, fake_resp) @@ -585,6 +585,19 @@ def test_call_openai_compat_relabels_unparseable_json_as_length(monkeypatch): "http://localhost:11434/v1", "ollama", "qwen2.5-coder:7b", "u", temperature=0, max_completion_tokens=8192, backend="ollama", ) + assert result["finish_reason"] == "hollow" + + +def test_call_openai_compat_keeps_real_truncation_as_length(monkeypatch): + # A genuine truncation stays "length" — that one IS a size problem and the + # bisection path is the right recovery. + fake_resp = _fake_openai_response('{"nodes": [{"id":', finish_reason="length", completion_tokens=8192) + _install_fake_openai(monkeypatch, fake_resp) + + result = llm._call_openai_compat( + "http://localhost:11434/v1", "k", "m", + "u", temperature=0, max_completion_tokens=8192, backend="kimi", + ) assert result["finish_reason"] == "length" @@ -868,12 +881,10 @@ def test_extract_corpus_parallel_ollama_parallel_env_restores_concurrency(tmp_pa mock_pool.assert_called() -def test_adaptive_retry_bisects_on_hollow_ollama_response(tmp_path): - # End-to-end: an overwhelmed Ollama returns hollow on the full 4-file - # chunk; halves succeed. The bug being fixed is that pre-fix this - # produces zero nodes (chunk silently dropped). Post-fix the hollow - # response is relabelled `finish_reason="length"` and the existing - # bisection path recovers the full 4 nodes. +def test_adaptive_retry_bisects_on_truncated_response(tmp_path): + # End-to-end: the full 4-file chunk truncates at max_completion_tokens; + # halves fit. A truncation IS a size problem, so bisection is the right + # recovery and must keep working (contrast: hollow, below). files = [tmp_path / f"f{i}.md" for i in range(4)] for f in files: f.write_text("hello") @@ -883,8 +894,7 @@ def test_adaptive_retry_bisects_on_hollow_ollama_response(tmp_path): def fake_extract(chunk, *_, **__): calls["n"] += 1 if len(chunk) == 4: - # Hollow response: looks successful, finish_reason already - # rewritten to "length" by _call_openai_compat. + # Truncated: the model ran out of output budget mid-JSON. return { "nodes": [], "edges": [], "hyperedges": [], "input_tokens": 100, "output_tokens": 0, @@ -1320,3 +1330,148 @@ def create(self, **_): llm._call_llm("hi", backend="kimi") assert ctor_kwargs.get("timeout") == 1.0, ctor_kwargs assert ctor_kwargs.get("max_retries", 0) >= 5, ctor_kwargs + + +def test_adaptive_retry_does_not_bisect_a_hollow_response(tmp_path, monkeypatch): + """#2880: a hollow response is retried as-is, never bisected. + + Bisecting cannot converge — both halves go to the same misbehaving backend + — so it burned up to 2**max_depth billed calls per chunk. The retry must be + on the SAME chunk, and the sub-call count must stay bounded. + """ + monkeypatch.setattr(llm, "_HOLLOW_BACKOFF_S", (0.0, 0.0)) + files = [tmp_path / f"f{i}.md" for i in range(4)] + for f in files: + f.write_text("hello") + + seen: list[int] = [] + + def fake_extract(chunk, *_, **__): + seen.append(len(chunk)) + return { + "nodes": [], "edges": [], "hyperedges": [], + "input_tokens": 100, "output_tokens": 0, + "model": "m", "finish_reason": "hollow", + } + + with patch("graphify.llm.extract_files_direct", side_effect=fake_extract): + result = llm._extract_with_adaptive_retry( + files, backend="ollama", api_key="ollama", model="qwen2.5-coder:7b", + root=tmp_path, max_depth=3, + ) + + # Every call saw the whole chunk: no halving happened. + assert seen == [4, 4, 4], f"hollow must not bisect, got chunk sizes {seen}" + # ...and the files are marked for re-extraction rather than cached as done. + assert sorted(Path(p).name for p in result["_partial_files"]) == [ + "f0.md", "f1.md", "f2.md", "f3.md", + ] + assert result["finish_reason"] == "stop" + + +def test_adaptive_retry_recovers_a_transient_hollow_response(tmp_path, monkeypatch): + """A hollow response that clears on retry costs 2 calls, not 15.""" + monkeypatch.setattr(llm, "_HOLLOW_BACKOFF_S", (0.0, 0.0)) + files = [tmp_path / f"f{i}.md" for i in range(4)] + for f in files: + f.write_text("hello") + + calls = {"n": 0} + + def fake_extract(chunk, *_, **__): + calls["n"] += 1 + if calls["n"] == 1: + return { + "nodes": [], "edges": [], "hyperedges": [], + "input_tokens": 100, "output_tokens": 0, + "model": "m", "finish_reason": "hollow", + } + return _ok(nodes=[{"id": f.stem} for f in chunk]) + + with patch("graphify.llm.extract_files_direct", side_effect=fake_extract): + result = llm._extract_with_adaptive_retry( + files, backend="ollama", api_key="ollama", model="qwen2.5-coder:7b", + root=tmp_path, max_depth=3, + ) + + assert calls["n"] == 2 + assert len(result["nodes"]) == 4 + assert not result.get("_partial_files") + + +def test_max_retry_depth_zero_disables_hollow_retries_too(tmp_path, monkeypatch): + """#2880 review catch: `0` means no retries of ANY kind. + + Bounding only the bisection depth still let a misbehaving backend triple + the call count of a run whose operator had explicitly asked for no retries, + which is the opposite of what the knob is set for. + """ + monkeypatch.setattr(llm, "_HOLLOW_BACKOFF_S", (0.0, 0.0)) + files = [tmp_path / f"f{i}.md" for i in range(4)] + for f in files: + f.write_text("hello") + + calls = {"n": 0} + + def fake_extract(chunk, *_, **__): + calls["n"] += 1 + return { + "nodes": [], "edges": [], "hyperedges": [], + "input_tokens": 100, "output_tokens": 0, + "model": "m", "finish_reason": "hollow", + } + + with patch("graphify.llm.extract_files_direct", side_effect=fake_extract): + result = llm._extract_with_adaptive_retry( + files, backend="ollama", api_key="ollama", model="qwen2.5-coder:7b", + root=tmp_path, max_depth=0, + ) + + assert calls["n"] == 1, "max_depth=0 must cost exactly one call per chunk" + # The chunk still fails loudly and its files are still re-dispatched next run. + assert sorted(Path(p).name for p in result["_partial_files"]) == [ + "f0.md", "f1.md", "f2.md", "f3.md", + ] + assert result["finish_reason"] == "stop" + + +def test_hollow_retries_still_run_at_the_default_depth(tmp_path, monkeypatch): + """Guard the other side: the cap must not silently disable the retry path.""" + monkeypatch.setattr(llm, "_HOLLOW_BACKOFF_S", (0.0, 0.0)) + files = [tmp_path / f"f{i}.md" for i in range(4)] + for f in files: + f.write_text("hello") + + calls = {"n": 0} + + def fake_extract(chunk, *_, **__): + calls["n"] += 1 + return { + "nodes": [], "edges": [], "hyperedges": [], + "input_tokens": 100, "output_tokens": 0, + "model": "m", "finish_reason": "hollow", + } + + with patch("graphify.llm.extract_files_direct", side_effect=fake_extract): + llm._extract_with_adaptive_retry( + files, backend="ollama", api_key="ollama", model="qwen2.5-coder:7b", + root=tmp_path, max_depth=3, + ) + + assert calls["n"] == 3 + + +def test_max_retry_depth_reads_the_env_var(monkeypatch): + """#2880: max_retry_depth was a Python-API kwarg only, so a `graphify + extract` operator had no way to lower it as a mitigation.""" + monkeypatch.delenv("GRAPHIFY_MAX_RETRY_DEPTH", raising=False) + assert llm._resolve_max_retry_depth() == 3 + monkeypatch.setenv("GRAPHIFY_MAX_RETRY_DEPTH", "0") + assert llm._resolve_max_retry_depth() == 0 + monkeypatch.setenv("GRAPHIFY_MAX_RETRY_DEPTH", "1") + assert llm._resolve_max_retry_depth() == 1 + # garbage and negatives fall back to the default + monkeypatch.setenv("GRAPHIFY_MAX_RETRY_DEPTH", "banana") + assert llm._resolve_max_retry_depth() == 3 + monkeypatch.setenv("GRAPHIFY_MAX_RETRY_DEPTH", "-2") + assert llm._resolve_max_retry_depth() == 3