finally getting accurate answers
Build and Push Containers / build-and-push-app (push) Successful in 1m46s
Build and Push Containers / build-and-push-db (push) Successful in 12s

This commit is contained in:
2026-09-05 10:26:39 -04:00
parent bb2803bebd
commit 766702c750
9 changed files with 1628 additions and 41 deletions
+196 -8
View File
@@ -209,17 +209,26 @@ def test_agent_tools_names_and_parameters() -> None:
# ("pass ONLY `pattern`") + the source-name-is-not-a-document
# clause (the model kept scoping grep with an ls-style source name
# — the 2026-09-03 incident loop shape, but on grep) plus the
# one-call-at-a-time discipline clause.
# one-call-at-a-time discipline clause. The 2026-09-05 incident
# (the "Qwen 3.8" sample question — the harness prior is that grep
# takes a REGEX; this grep is a fixed substring, owner-locked A5):
# the plain-substring-never-a-regex clause states the contract up
# front, so the regex-shaped first grep that does fire gets the
# teaching no-match line instead of a trusted miss.
assert grep["description"] == (
"Search the indexed documents for an exact string "
"(case-insensitive) and return up to 20 matching lines "
"as `source/path:line: text` — a locator, not a "
"context-adder: read the winner with `read`. For a "
"normal search pass ONLY `pattern` — it searches every "
"document and that is how you search the knowledge "
"base; never pass a source name as `path` (a source "
"name is not a document). Call one tool at a time — "
"wait for this result before your next call."
"context-adder: read the winner with `read`. The "
"pattern is a plain substring, NEVER a regex — if a "
"pattern with regex syntax (like '.*' or '\\.') comes "
"back with no matches, retry with the plain text you "
"expect to see. For a normal search pass ONLY `pattern` "
"— it searches every document and that is how you "
"search the knowledge base; never pass a source name "
"as `path` (a source name is not a document). Call one "
"tool at a time — wait for this result before your "
"next call."
)
grep_params = grep["parameters"]
assert grep_params["type"] == "object"
@@ -227,7 +236,8 @@ def test_agent_tools_names_and_parameters() -> None:
assert set(grep_params["properties"]) == {"pattern", "path"}
assert all(p["type"] == "string" for p in grep_params["properties"].values())
assert grep_params["properties"]["pattern"]["description"] == (
"The exact text to search for (a plain substring, not a regex)"
"The exact text to search for (a plain substring, "
"not a regex — no '.*', no '\\.', no character classes)"
)
# Phase 72 (task 02): the bare-path contract is stated up front;
# task 05 (live gate iterations 1-8): the one-known-document clause
@@ -1164,6 +1174,184 @@ def test_grep_truncates_match_lines_at_200_chars(monkeypatch: pytest.MonkeyPatch
assert holder.tool_calls == 1
# ---------- grep no-match teaching: the regex-shaped pattern
# (the 2026-09-05 "Qwen 3.8" incident — the harness prior is that
# grep takes a REGEX; this grep is a fixed substring, owner-locked
# A5, and the contract does not change) ----------
def test_plain_form_reduces_regex_to_literal_text() -> None:
"""The plain-form hint: the pattern reduced to literal text — the
incident's exact recovery (``qwen.*3\\.8`` → ``qwen3.8``) plus the
edge cases (raw ``.*`` runs dropped before unescape, so an escaped
dot survives; first alternative only; classes/quantifiers/parens/
anchors gone; whitespace preserved; pure metacharacters → ``""``).
"""
assert agent.plain_form(r"qwen.*3\.8") == "qwen3.8" # the incident
assert agent.plain_form(r"qwen 3\.8") == "qwen 3.8"
assert agent.plain_form(r"Qwen 3\.8") == "Qwen 3.8" # case kept
assert agent.plain_form(r"qwen3\.8") == "qwen3.8"
assert agent.plain_form(r"llama\.cpp") == "llama.cpp" # escaped dot kept
assert agent.plain_form(r"qwen[0-9]+") == "qwen" # class + quantifier
assert agent.plain_form("a|b") == "a" # first alternative only
assert agent.plain_form(r"\d+") == "" # no literal text — no hint
assert agent.plain_form(r".*") == "" # pure wildcard — no hint
assert agent.plain_form(r"(qwen)3\.8") == "qwen3.8" # group contents kept
assert agent.plain_form("a{2,3}b") == "ab"
assert agent.plain_form(r"^qwen$") == "qwen" # anchors dropped
assert agent.plain_form(r"a\.b") == "a.b" # escaped dot is a literal
assert agent.plain_form("plain") == "plain" # identity for plain text
def test_looks_like_regex_detection() -> None:
"""One metacharacter anywhere marks the pattern regex-shaped; a
plain substring (even with a space) does not."""
for p in (
r"qwen.*3\.8", r"qwen 3\.8", "qwen+", "a?b", "x|y", "(a)", "[a-z]", "a^b", "b$c", "a{2}"
):
assert agent.looks_like_regex(p) is True, p
for p in ("qwen 3.8", "qwen3.8", "plain substring", ""):
assert agent.looks_like_regex(p) is False, p
def test_grep_no_match_regex_pattern_gets_teaching_line(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The incident's shape: a regex-shaped pattern that (necessarily)
misses gets the TEACHING no-match line — the plain-substring
contract stated, the plain-form retry hint handed over. Still a
counted result; the context is untouched (locked A5)."""
d1 = _doc("Alpha", "a/one.md", "One", "qwen3.8-27b juggernaut\nno regex text")
monkeypatch.setattr(agent, "all_documents", lambda db: [d1])
holder = AgentHolder()
llm = ScriptedLLM(
[
ToolCallPiece(
id="call_1", name="grep", arguments={"pattern": r"qwen.*3\.8"}
)
],
[StreamPiece("content", "ans")],
)
asyncio.run(_run(llm, holder, _settings()))
assert llm.requests[1][0][3]["content"] == agent.NO_MATCHES_REGEX.format(
pattern=r"qwen.*3\.8", plain="qwen3.8"
)
assert llm.requests[1][0][3]["content"] == (
"No matches for 'qwen.*3\\.8'. grep matches a plain substring "
"(case-insensitive), not a regex — '.*', '\\.' and the like are "
"literal text here, so that pattern can never match. Retry with "
"the plain text you expect to see (e.g. 'qwen3.8')."
)
assert holder.tool_calls == 1 # a no-match with teaching is still a result
assert holder.read_docs == [] # locked A5: a grep adds no context
def test_grep_no_match_regex_scoped_gets_teaching_line(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The scoped teaching variant: the resolved identity is echoed, the
hint handed over."""
d1 = _doc("Alpha", "a/one.md", "One", "nothing regex-shaped here")
def _find(db: Any, source: str, path: str) -> Document | None:
return d1 if (source, path) == ("Alpha", "a/one.md") else None
monkeypatch.setattr(agent, "find_document", _find)
holder = AgentHolder()
llm = ScriptedLLM(
[
ToolCallPiece(
id="call_1",
name="grep",
arguments={"pattern": r"qwen 3\.8", "path": "Alpha/a/one.md"},
)
],
[StreamPiece("content", "ans")],
)
asyncio.run(_run(llm, holder, _settings()))
assert llm.requests[1][0][3]["content"] == (
"No matches for 'qwen 3\\.8' in Alpha/a/one.md. grep matches a "
"plain substring (case-insensitive), not a regex — retry with "
"the plain text you expect to see (e.g. 'qwen 3.8')."
)
assert holder.tool_calls == 1
assert holder.read_docs == []
def test_grep_no_match_plain_pattern_keeps_ordinary_line(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A no-match for a PLAIN pattern (no metacharacters — "qwen 3.8" with
the space included) keeps the ordinary line byte-identical: the
teaching never fires for a well-formed pattern (the retrieval side —
the name-hit lexical signal — is what covers that case)."""
d1 = _doc("Alpha", "a/one.md", "One", "qwen3.8-27b juggernaut")
monkeypatch.setattr(agent, "all_documents", lambda db: [d1])
holder = AgentHolder()
llm = ScriptedLLM(
[ToolCallPiece(id="call_1", name="grep", arguments={"pattern": "qwen 3.8"})],
[StreamPiece("content", "ans")],
)
asyncio.run(_run(llm, holder, _settings()))
assert llm.requests[1][0][3]["content"] == (
"No matches for 'qwen 3.8' in the knowledge base."
)
assert holder.tool_calls == 1
def test_grep_matched_regex_pattern_returns_matches_not_teaching(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A pattern with metacharacters that MATCHES literally gets the
ordinary match output — the teaching can never suppress a real hit
(the detection keys on a NO-MATCH only)."""
d1 = _doc("S", "a.md", "A", "the C++ compiler is here")
monkeypatch.setattr(agent, "all_documents", lambda db: [d1])
holder = AgentHolder()
llm = ScriptedLLM(
[ToolCallPiece(id="call_1", name="grep", arguments={"pattern": "C++"})],
[StreamPiece("content", "ans")],
)
asyncio.run(_run(llm, holder, _settings()))
assert llm.requests[1][0][3]["content"] == "S/a.md:1: the C++ compiler is here"
assert holder.tool_calls == 1
def test_grep_no_match_regex_reducing_to_empty_falls_back(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A regex-shaped pattern with no literal text left after the
reduction (``.*``) gets the ORDINARY line — no empty hint."""
d1 = _doc("Alpha", "a/one.md", "One", "any text at all")
monkeypatch.setattr(agent, "all_documents", lambda db: [d1])
holder = AgentHolder()
llm = ScriptedLLM(
[ToolCallPiece(id="call_1", name="grep", arguments={"pattern": r".*"})],
[StreamPiece("content", "ans")],
)
asyncio.run(_run(llm, holder, _settings()))
assert llm.requests[1][0][3]["content"] == (
"No matches for '.*' in the knowledge base."
)
assert holder.tool_calls == 1
def test_no_matches_regex_templates_pin() -> None:
"""The teaching templates are verbatim pins (the model-facing copy —
the mock E2E keys off the plain-substring clause)."""
assert agent.NO_MATCHES_REGEX == (
"No matches for '{pattern}'. grep matches a plain substring "
"(case-insensitive), not a regex — '.*', '\\.' and the like are "
"literal text here, so that pattern can never match. Retry with "
"the plain text you expect to see (e.g. '{plain}')."
)
assert agent.NO_MATCHES_REGEX_SCOPED == (
"No matches for '{pattern}' in {source}/{path}. grep matches a "
"plain substring (case-insensitive), not a regex — retry with "
"the plain text you expect to see (e.g. '{plain}')."
)
def test_grep_scoped_to_one_document(monkeypatch: pytest.MonkeyPatch) -> None:
"""Scoped grep: only the named document is loaded (find_document on
the first-slash split), ``all_documents`` never runs, and the match