Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ Ten v2 CLI commands: `list [--project] [--all-projects] [--recent N] [--order ti

The `SearchEngine` interface is the port; `internal/storage` is the adapter. Database opened lazily. `OpenReadOnly()` provides read-only access for external consumers.

Opt-in lexical term dropping is owned by `internal/storage/relaxation.go`; `docs/search.md#opt-in-lexical-relaxation` defines protected units, the two-unprotected-term floor, fixed scope, provenance and zero-overlap limits. Unfiltered IDF uses the same query-echo eligibility as unfiltered result pages, counted in SQL via `directBackscrollSearchEchoSQL` (keep lockstep with `isDirectBackscrollSearchEcho`; do not reintroduce a Go-side row scan). Ordinary search behavior/output must remain unchanged.
Opt-in lexical term dropping is owned by `internal/storage/relaxation.go`; `docs/search.md#opt-in-lexical-relaxation` defines protected units, the two-unprotected-term floor, fixed scope, provenance and zero-overlap limits. Unfiltered IDF uses the same query-echo eligibility as unfiltered result pages, counted in SQL via `directBackscrollSearchEchoSQL` (keep lockstep with `isDirectBackscrollSearchEcho`) — except the Codex `shell` wrapper form, whose JSON-encoded argv is unbounded for SQL GLOB: `recallFrequency` subtracts those rows with the PR #87 broad-SQL-prefilter (`text LIKE 'shell %'`) plus strict-Go-predicate split (`pendingSearchEchoShellMatches`), and `isDirectBackscrollSearchEcho` uses the same helper. Never reintroduce a general Go-side row scan or a GLOB enumeration of JSON separator byte-sequences. Ordinary search behavior/output must remain unchanged.

### Core Pipeline

Expand Down
129 changes: 129 additions & 0 deletions cmd/backscroll/echo_shell_zero_query_gap_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
package main

import (
"database/sql"
"encoding/json"
"fmt"
"path/filepath"
"strings"
"testing"

"github.com/pablontiv/backscroll/internal/startuplock"
)

// Spike/regression for the Codex `shell` wrapper form of the zero-valued
// search_echo query-time gap. The exec_command half of this gap was fixed in
// PR #86; the requeue side learned the shell form in PR #87, but
// isDirectBackscrollSearchEcho / directBackscrollSearchEchoSQL (the functions
// that exclude a row from unfiltered result pages and --relax IDF counting
// RIGHT NOW, before reparse converges it) never gained shell handling.
//
// The fixture is a real CodexReader.Parse round-trip: the rollout below is
// ingested by the actual Codex reader, its serialized shape is asserted, and
// only then is search_echo forced back to 0 to reproduce the pre-#80 state.
func TestZeroValuedCodexShellEchoExcludedBeforeReplay(t *testing.T) {
e := newQueryEchoE2E(t)
e.writeRecord("target.jsonl", "target", "target", 0, "violet handshake quartz marker", false)
for i := 0; i < 4; i++ {
e.writeRecord(fmt.Sprintf("noise-%d.jsonl", i), fmt.Sprintf("noise-%d", i), "noise", 1+i, "adaptation rollout distractor", false)
}

codexRoot := e.addReaderManifest("codex", "codex")
// Real Codex shell calls carry extra keys (workdir, timeout_ms), so the
// serialized `command=` token is not the only key=value and the strict
// predicate must locate it inside the sorted token list.
for i := 0; i < 8; i++ {
id := fmt.Sprintf("codex-shell-%d", i)
args, _ := json.Marshal(map[string]any{
"command": []string{"bash", "-lc", "backscroll search --text 'violet handshake'"},
"workdir": "/synthetic/query-echo-e2e",
"timeout_ms": 10000,
})
writeCodexRollout(t, filepath.Join(codexRoot, id+".jsonl"), id, 10+i,
map[string]any{"type": "function_call", "name": "shell", "call_id": id, "arguments": string(args)})
}
// Already-fixed exec_command control in the same forced-zero state: it
// must stay excluded by both the Go and the SQL/IDF paths.
for i := 0; i < 4; i++ {
id := fmt.Sprintf("codex-exec-%d", i)
args, _ := json.Marshal(map[string]any{"cmd": "backscroll search --text 'violet handshake'"})
writeCodexRollout(t, filepath.Join(codexRoot, id+".jsonl"), id, 20+i,
map[string]any{"type": "function_call", "name": "exec_command", "call_id": id, "arguments": string(args)})
}
e.run("status", "--json")

// Same gap with every separator JSON-escaped (control characters like
// U+0009 must escape as two-byte \t in JSON, so the serialized text has
// no argv whitespace at all and only two strings.Fields tokens). These
// rows exercised a page/IDF disagreement: recallFrequency's shell
// prefilter has no token-count floor, but the result-page predicate did.
for i := 0; i < 2; i++ {
id := fmt.Sprintf("codex-tab-%d", i)
args, _ := json.Marshal(map[string]any{
"command": []string{"sh", "-c", "backscroll\tsearch\t--text\tviolet\thandshake"},
})
writeCodexRollout(t, filepath.Join(codexRoot, id+".jsonl"), id, 30+i,
map[string]any{"type": "function_call", "name": "shell", "call_id": id, "arguments": string(args)})
}
e.run("status", "--json")

db, err := sql.Open("sqlite", e.database)
if err != nil {
t.Fatal(err)
}
var serialized, tabSerialized string
if err := db.QueryRow(`SELECT text FROM search_items WHERE source_path LIKE ? AND content_type='tool' LIMIT 1`, "%codex-shell-0.jsonl").Scan(&serialized); err != nil {
_ = db.Close()
t.Fatal(err)
}
const wantSerialized = `shell command=["bash","-lc","backscroll search --text 'violet handshake'"] timeout_ms=10000 workdir=/synthetic/query-echo-e2e`
if serialized != wantSerialized {
_ = db.Close()
t.Fatalf("Codex shell SerializeToolInput shape=%q want %q", serialized, wantSerialized)
}
if err := db.QueryRow(`SELECT text FROM search_items WHERE source_path LIKE ? AND content_type='tool' LIMIT 1`, "%codex-tab-0.jsonl").Scan(&tabSerialized); err != nil {
_ = db.Close()
t.Fatal(err)
}
const wantTabSerialized = `shell command=["sh","-c","backscroll\tsearch\t--text\tviolet\thandshake"]`
if tabSerialized != wantTabSerialized {
_ = db.Close()
t.Fatalf("Codex tab shell SerializeToolInput shape=%q want %q", tabSerialized, wantTabSerialized)
}
if got := len(strings.Fields(tabSerialized)); got != 2 {
_ = db.Close()
t.Fatalf("tab-escaped shell row must have exactly two whitespace-separated tokens to exercise the guard, got %d", got)
}
if _, err := db.Exec(`UPDATE search_items SET search_echo=0 WHERE content_type='tool' AND source_path LIKE ?`, "%codex-%.jsonl"); err != nil {
_ = db.Close()
t.Fatal(err)
}
if err := db.Close(); err != nil {
t.Fatal(err)
}

// Hold the startup lock so the following searches run as followers on the
// last committed snapshot: the query-time fallback is what must exclude the
// zeroed rows, with no owner replay to converge them first.
lease, acquired, err := startuplock.TryAcquire(e.database)
if err != nil || !acquired {
t.Fatalf("acquire parent startup lock: acquired=%v err=%v", acquired, err)
}
defer func() {
if err := lease.Release(); err != nil {
t.Errorf("release parent startup lock: %v", err)
}
}()

got := e.searchJSON("violet handshake", "", 20)
for _, row := range got {
if row.ContentType == "tool" && strings.HasPrefix(filepath.Base(row.FilePath), "codex-") {
t.Errorf("zero-valued Codex echo leaked into unfiltered recall: %v", queryEchoShape(got))
}
}

out := e.run("search", "--text", "violet handshake adaptation", "--all-projects", "--lexical-only", "--relax", "--robot", "--fields", "minimal", "--max-tokens", "200")
if !strings.Contains(out, "result_0_filepath="+filepath.Join(e.fixtures, "target.jsonl")+"\n") || !strings.Contains(out, `result_0_dropped_terms=["adaptation"]`) {
t.Errorf("zero-valued shell query echoes inverted unfiltered --relax IDF\nstdout=%s", out)
}
}
15 changes: 13 additions & 2 deletions docs/search.md
Original file line number Diff line number Diff line change
Expand Up @@ -175,6 +175,17 @@ rows use the UUID-less per-file reload path. An index that stored those calls as
`search_echo=0` before their readers marked echoes re-enters the same bounded
replay while the surviving source still has a serialized direct search call;
paired outputs are marked by identity on that reparse, not by output shape.
While such a source awaits replay, the query-time exclusion already keeps its
zero-valued call rows out of unfiltered result pages and unfiltered `--relax`
IDF counting, recognizing the same serialized forms. Result pages are filtered
in Go: the `bash` and `exec_command` forms by their three-token prefix shape,
and the Codex `shell` form by a `shell` first-token gate followed by decoding
the JSON-encoded argv (its separator byte-sequences are unbounded for
text-shape matching, and no token-count floor may precede the gate — an
all-escaped argv serializes to just two whitespace-separated tokens). IDF
counting evaluates the `bash`/`exec_command` prefixes in SQL and applies the
same `shell` decode in Go over a broad `text LIKE 'shell %'` prefilter, so
both paths exclude the same rows.
Subsequent source expiry, `rebuild`, and supported canonical recovery preserve
proven pairing evidence. The general extraction epoch is unchanged.

Expand Down Expand Up @@ -210,7 +221,7 @@ backscroll search --text '"violet handshake" quartz marker adaptation' --relax -
The deterministic sequence is:

1. Run strict AND search. Unmarked queries keep the existing sanitizer, ranking and snippets. Leading `+term` marks a term that cannot be dropped; quoted spans are protected phrase units. For queries containing these protected units, strict matching keeps every unit without dynamic stopword removal. Quotes preserve FTS phrase order; ordinary unquoted terms retain Porter stemming/prefix matching (trigram matching for tools).
2. Only if that stage has zero eligible rows, drop one **unprotected** term at a time, lowest IDF first. For a fixed corpus, this is highest document frequency first. Frequencies are counted with the actual tokenizer's MATCH expression over the applicable index(es), globally rather than within the result scope (project, path, dates, tags). Unfiltered IDF uses the same echo eligibility as unfiltered result pages: direct Backscroll retrieval-call tool rows (`search_echo=1`, a serialized `bash command=backscroll search ...` invocation, or a serialized `exec_command cmd=backscroll search ...` invocation, each matched as that three-token prefix regardless of what follows) do not inflate document frequency. Explicit `--content-type tool` keeps those rows in both the page and the IDF count. Equal frequencies drop in original query order. Zero-frequency terms have highest IDF and are not specially discarded. A term that appears only in those excluded echo rows is absent from the unfiltered corpus, so its document frequency is 0 and `--relax` drops it last — the same as any other zero-frequency extra term that can prevent recovery at the two-term floor. Each retry still requires every retained unit, bypassing dynamic stopwords so the retained core cannot silently disappear.
2. Only if that stage has zero eligible rows, drop one **unprotected** term at a time, lowest IDF first. For a fixed corpus, this is highest document frequency first. Frequencies are counted with the actual tokenizer's MATCH expression over the applicable index(es), globally rather than within the result scope (project, path, dates, tags). Unfiltered IDF uses the same echo eligibility as unfiltered result pages: direct Backscroll retrieval-call tool rows (`search_echo=1`, a serialized `bash command=backscroll search ...` or `exec_command cmd=backscroll search ...` invocation, each matched as that three-token prefix regardless of what follows, or a serialized Codex `shell` call whose JSON-encoded argv is exactly `[<shell>, "-c" | "-lc", "backscroll search ..."]`, selected by a `shell` text-prefix gate and then matched by decoding the argv, since the JSON separator byte-sequences are unbounded for SQL pattern matching) do not inflate document frequency. Explicit `--content-type tool` keeps those rows in both the page and the IDF count. Equal frequencies drop in original query order. Zero-frequency terms have highest IDF and are not specially discarded. A term that appears only in those excluded echo rows is absent from the unfiltered corpus, so its document frequency is 0 and `--relax` drops it last — the same as any other zero-frequency extra term that can prevent recovery at the two-term floor. Each retry still requires every retained unit, bypassing dynamic stopwords so the retained core cannot silently disappear.
3. Stop at the first stage with results, or before fewer than **two distinct unprotected terms** remain. Protected terms are additional to that floor. Case-insensitive repeated spellings count once, and a keep marker on any occurrence protects that unit. Queries with at most two unprotected terms perform strict search only. There is no single-term/empty fallback and no global OR.

Stemming/phrase-expansion stages are skipped: stemming is already available and protected phrases must not weaken. Scope widening is always skipped. Project (including cwd-inferred project), source, source-path, content-type, role, dates, and tags are retained at every stage. Tool searches remain on their trigram index even when relaxation is explicitly requested. Existing unfiltered echo exclusion and cross-index RRF still apply.
Expand All @@ -232,7 +243,7 @@ result_0_dropped_terms=["adaptation"]

This addresses **recoverable term overload**, not vocabulary invention. In the native regression, the target contains `violet handshake`, while unrelated records make `adaptation` a common term; strict search misses and dropping that term recovers the target. In contrast, the existing synthetic `s2_conversational_paraphrase` and `s5_progressive_refinement` targets share no surviving content terms with their original queries. The refinement `violet handshake` adds new vocabulary; this feature does not promise to recover those zero-overlap cases. An absent, high-IDF extra term can also prevent recovery at the two-term floor. No production p95 or broad semantic-recall gain is claimed from the small fixture corpus.

Regression owners: `cmd/backscroll/search_relaxation_test.go` (native input-to-output recovery and unchanged strict controls), `cmd/backscroll/search_relaxation_echo_idf_e2e_test.go` (unfiltered IDF ignores query-echo rows), `cmd/backscroll/search_relaxation_output_test.go` (budgeted provenance and early validation), `internal/storage/relaxation_test.go` (IDF order, protected core, scope and paging), and `internal/storage/relaxation_echo_idf_test.go` (echo-eligibility of unfiltered IDF, including echo-only DF=0).
Regression owners: `cmd/backscroll/search_relaxation_test.go` (native input-to-output recovery and unchanged strict controls), `cmd/backscroll/search_relaxation_echo_idf_e2e_test.go` (unfiltered IDF ignores query-echo rows), `cmd/backscroll/echo_shell_zero_query_gap_test.go` (zero-valued Codex `shell` echoes excluded from unfiltered pages and IDF before replay), `cmd/backscroll/search_relaxation_output_test.go` (budgeted provenance and early validation), `internal/storage/relaxation_test.go` (IDF order, protected core, scope and paging), and `internal/storage/relaxation_echo_idf_test.go` (echo-eligibility of unfiltered IDF, including echo-only DF=0 and the Codex `shell` argv boundary cases).

## Exit Codes

Expand Down
11 changes: 10 additions & 1 deletion internal/storage/queries.go
Original file line number Diff line number Diff line change
Expand Up @@ -1145,7 +1145,16 @@ func (d *Database) filterShellEchoZeroPaths(paths []string) ([]string, error) {
}

// pendingSearchEchoShellMatches is the Go-side check for whether a stored
// Codex shell tool row is a direct `backscroll search` call.
// Codex shell tool row is a direct `backscroll search` call. It is shared by
// the requeue path (filterShellEchoZeroPaths) and by the query-time exclusion
// paths (isDirectBackscrollSearchEcho and recallFrequency's IDF counting).
// Each call site first gates on the serialized text starting with the `shell`
// tool-name token — whitespace-delimited and case-insensitive in Go, `text
// LIKE 'shell %'` in SQL — and then applies this decode, with no token-count
// floor anywhere: an argv whose separators are all JSON control escapes
// serializes to just two whitespace-separated tokens, and a floor on only one
// path made pages and IDF disagree. Serializer-produced rows are therefore
// accepted identically by all three.
//
// SerializeToolInput emits the rollout's `arguments` as a space-joined
// `key=value` token list with keys sorted alphabetically. Real Codex shell
Expand Down
26 changes: 26 additions & 0 deletions internal/storage/relaxation.go
Original file line number Diff line number Diff line change
Expand Up @@ -225,5 +225,31 @@ func (d *Database) recallFrequency(term recallTerm, contentType string) (int, er
if err != nil {
return 0, fmt.Errorf("measure relaxation term frequency: %w", err)
}
// directBackscrollSearchEchoSQL cannot recognize the Codex shell wrapper
// form: its argv is JSON-encoded and the separator byte-sequences are
// unbounded for SQL GLOB. Subtract those rows with the same
// broad-SQL-prefilter plus strict-Go-predicate split the requeue path
// uses (see pendingSearchEchoShellMatches), so unfiltered IDF counts the
// exact row set isDirectBackscrollSearchEcho keeps.
shellRows, err := d.db.Query(
"SELECT si.text FROM ("+matched+") matched JOIN search_items si ON si.id = matched.rowid WHERE si.content_type = 'tool' AND COALESCE(si.search_echo, 0) = 0 AND si.text LIKE 'shell %'",
args...,
)
if err != nil {
return 0, fmt.Errorf("measure relaxation term frequency: %w", err)
}
defer func() { _ = shellRows.Close() }()
for shellRows.Next() {
var text string
if err := shellRows.Scan(&text); err != nil {
return 0, fmt.Errorf("measure relaxation term frequency: %w", err)
}
if pendingSearchEchoShellMatches(text) {
count--
}
}
if err := shellRows.Err(); err != nil {
return 0, fmt.Errorf("measure relaxation term frequency: %w", err)
}
return count, nil
}
Loading
Loading