-
Notifications
You must be signed in to change notification settings - Fork 1.2k
Isolate malformed LLM batch responses #265
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -404,6 +404,27 @@ def test_run_batches_uses_message_text_for_content_blocks(self) -> None: | |
|
|
||
| assert results[0][1] == ["chunk"] | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| def test_run_batches_isolates_raw_invoke_validation_error(self) -> None: | ||
| analyzer = _RawTextAnalyzer(base_prompt="test", model=self.MODEL) | ||
|
|
||
| def _invoke(prompt: str) -> AIMessage: | ||
| if "b.py" in prompt: | ||
| LLMAnalysisResult.model_validate({"findings": 'We{"findings":[]}'}) | ||
| return AIMessage(content="ok") | ||
|
|
||
| analyzer._llm.invoke.side_effect = _invoke | ||
| batches = [ | ||
| Batch(file_path="a.py", content="code a"), | ||
| Batch(file_path="b.py", content="code b"), | ||
| Batch(file_path="c.py", content="code c"), | ||
| ] | ||
|
|
||
| results = analyzer.run_batches(batches) | ||
|
|
||
| assert {batch.file_path for batch, _ in results} == {"a.py", "c.py"} | ||
| assert [items for _, items in results] == [["ok"], ["ok"]] | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| async def test_arun_batches_uses_message_text_for_content_blocks(self) -> None: | ||
| analyzer = _RawTextAnalyzer(base_prompt="test", model=self.MODEL) | ||
|
|
@@ -416,6 +437,72 @@ async def test_arun_batches_uses_message_text_for_content_blocks(self) -> None: | |
| assert results[0][1] == ["async chunk"] | ||
|
|
||
|
|
||
| # --------------------------------------------------------------------------- | ||
| # LLMAnalyzerBase.run_batches (sync execution) | ||
| # --------------------------------------------------------------------------- | ||
|
|
||
|
|
||
| class TestRunBatches: | ||
| MODEL = "nvidia/openai/gpt-oss-120b" | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| def test_malformed_structured_batch_does_not_abort_the_others(self) -> None: | ||
|
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Good regression test for the sync path. Please add the async counterpart: an |
||
| """A malformed structured response costs only its own batch.""" | ||
|
|
||
| def _invoke(prompt: str) -> LLMAnalysisResult: | ||
| if "b.py" in prompt: | ||
| return LLMAnalysisResult.model_validate({"findings": 'We{"findings":[]}'}) | ||
| return LLMAnalysisResult( | ||
| findings=[ | ||
| LLMFinding(rule_id="T-1", message="hit", severity="LOW", start_line=1), | ||
| ] | ||
| ) | ||
|
|
||
| analyzer = LLMAnalyzerBase(base_prompt="test", model=self.MODEL) | ||
| analyzer._structured_llm.invoke.side_effect = _invoke | ||
|
|
||
| batches = [ | ||
| Batch(file_path="a.py", content="code a"), | ||
| Batch(file_path="b.py", content="code b"), | ||
| Batch(file_path="c.py", content="code c"), | ||
| ] | ||
| results = analyzer.run_batches(batches) | ||
|
|
||
| assert {batch.file_path for batch, _ in results} == {"a.py", "c.py"} | ||
| assert [items[0].rule_id for _, items in results] == ["T-1", "T-1"] | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| def test_value_error_still_propagates(self) -> None: | ||
| """ValueError signals misconfiguration, not a malformed model response.""" | ||
| analyzer = LLMAnalyzerBase(base_prompt="test", model=self.MODEL) | ||
| analyzer._structured_llm.invoke.side_effect = ValueError("no API key") | ||
|
|
||
| with pytest.raises(ValueError, match="no API key"): | ||
| analyzer.run_batches([Batch(file_path="a.py", content="code")]) | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| def test_parse_validation_error_does_not_abort_the_others(self) -> None: | ||
| analyzer = LLMAnalyzerBase(base_prompt="test", model=self.MODEL) | ||
| analyzer._structured_llm.invoke.return_value = LLMAnalysisResult(findings=[]) | ||
| original_parse = analyzer.parse_response | ||
|
|
||
| def _parse(response: object, batch: Batch) -> list[Finding]: | ||
| if batch.file_path == "b.py": | ||
| LLMAnalysisResult.model_validate({"findings": 'We{"findings":[]}'}) | ||
| return original_parse(response, batch) | ||
|
|
||
| analyzer.parse_response = _parse | ||
| batches = [ | ||
| Batch(file_path="a.py", content="code a"), | ||
| Batch(file_path="b.py", content="code b"), | ||
| Batch(file_path="c.py", content="code c"), | ||
| ] | ||
|
|
||
| results = analyzer.run_batches(batches) | ||
|
|
||
| assert {batch.file_path for batch, _ in results} == {"a.py", "c.py"} | ||
|
|
||
|
|
||
| # --------------------------------------------------------------------------- | ||
| # LLMAnalyzerBase.arun_batches (async parallel execution) | ||
| # --------------------------------------------------------------------------- | ||
|
|
@@ -622,6 +709,32 @@ async def _flaky_ainvoke(prompt: str) -> LLMAnalysisResult: | |
| results = await analyzer.arun_batches(batches) | ||
| assert {batch.file_path for batch, _ in results} == {"a.py", "c.py"} | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| async def test_malformed_structured_batch_does_not_abort_the_others(self) -> None: | ||
| """A malformed structured response is isolated even though it is a ValueError.""" | ||
|
|
||
| async def _ainvoke(prompt: str) -> LLMAnalysisResult: | ||
| if "b.py" in prompt: | ||
| return LLMAnalysisResult.model_validate({"findings": 'We{"findings":[]}'}) | ||
| return LLMAnalysisResult( | ||
| findings=[ | ||
| LLMFinding(rule_id="T-1", message="hit", severity="LOW", start_line=1), | ||
| ] | ||
| ) | ||
|
|
||
| analyzer = LLMAnalyzerBase(base_prompt="test", model=self.MODEL) | ||
| analyzer._structured_llm.ainvoke = _ainvoke | ||
|
|
||
| batches = [ | ||
| Batch(file_path="a.py", content="code a"), | ||
| Batch(file_path="b.py", content="code b"), | ||
| Batch(file_path="c.py", content="code c"), | ||
| ] | ||
| results = await analyzer.arun_batches(batches) | ||
|
|
||
| assert {batch.file_path for batch, _ in results} == {"a.py", "c.py"} | ||
| assert [items[0].rule_id for _, items in results] == ["T-1", "T-1"] | ||
|
|
||
| @patch(MOCK_PATCH_TARGET, _mock_get_chat_model) | ||
| async def test_all_batches_failed_returns_empty(self) -> None: | ||
| analyzer = LLMAnalyzerBase(base_prompt="test", model=self.MODEL) | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Inconsistency: this raw-mode branch has no
ValidationErrorcarve-out, so aValidationErrorraised here (or in a subclass's raw-mode pipeline) propagates via theValueErrorre-raise, while the structured branch above skips it. If intentional, add a comment; otherwise align the two branches.