-
Notifications
You must be signed in to change notification settings - Fork 18
Cache IDF corpus statistics to cut DB round-trips on repeated recall #55
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Open
lkxdsb
wants to merge
3
commits into
afx-team:main
Choose a base branch
from
lkxdsb:feat/idf-cache-stats-cache
base: main
Could not load branches
Branch not found: {{ refName }}
Loading
Could not load tags
Nothing to show
Loading
Are you sure you want to change the base?
Some commits from the old base branch may be removed from the timeline,
and old review comments may become outdated.
Open
Changes from 2 commits
Commits
File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Some comments aren't visible on the classic Files Changed page.
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,128 @@ | ||
| """Feel-the-issue demo for #54 — show the IDF cache's real DB impact. | ||
|
|
||
| Runs the SAME search workload twice against a real HebbMind + SQLiteMemoryStore: | ||
|
|
||
| A) With the cache (default). | ||
| B) With the cache neutralised (TTL=0, so every call re-fetches). | ||
|
|
||
| and prints the number of real ``corpus_size`` / ``keyword_doc_freqs`` DB calls | ||
| in each case. That delta is the work issue #54 removes — made visible, not | ||
| asserted. Intended for a human to read the output, so it narrates as it goes. | ||
| """ | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| import shutil | ||
| import sys | ||
| import tempfile | ||
| from pathlib import Path | ||
|
|
||
| sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | ||
|
|
||
| from hebb import HebbMind | ||
| from hebb.config.settings import Settings | ||
| from hebb.storage.sqlite_store import SQLiteMemoryStore | ||
|
|
||
| # A realistic recall workload: 3 back-to-back queries over one corpus, | ||
| # mirroring RecallAgent.recall's queries[:3] loop. The queries share words | ||
| # ("python", "retrieval") so the cache can reuse DF across them — the exact | ||
| # case issue #54 calls out. | ||
| QUERIES = [ | ||
| "how does python handle concurrency and the gil", | ||
| "python gil and retrieval tradeoffs", | ||
| "python async retrieval patterns", | ||
| ] | ||
|
|
||
|
|
||
| class Counter: | ||
| def __init__(self, store: SQLiteMemoryStore) -> None: | ||
| self.corpus = 0 | ||
| self.df = 0 | ||
| self._s = store | ||
| self._oc = store.corpus_size | ||
| self._od = store.keyword_doc_freqs | ||
| store.corpus_size = self._c # type: ignore[method-assign] | ||
| store.keyword_doc_freqs = self._d # type: ignore[method-assign] | ||
|
|
||
| async def _c(self, partition_ids=None) -> int: | ||
| self.corpus += 1 | ||
| return await self._oc(partition_ids) | ||
|
|
||
| async def _d(self, terms, partition_ids=None) -> dict[str, int]: | ||
| self.df += 1 | ||
| return await self._od(terms, partition_ids) | ||
|
|
||
|
|
||
| def _seed(hc: HebbMind) -> None: | ||
| hc.add("Python's GIL serializes bytecode execution in one process.", partition="mem_semantic") | ||
| hc.add("RAG retrieves documents then conditions the LLM on them.", partition="mem_semantic") | ||
| hc.add("asyncio lets Python do concurrent IO despite the GIL.", partition="mem_semantic") | ||
| hc.add("I prefer dark mode and 2-space indents.", partition="mem_preference") | ||
|
|
||
|
|
||
| def _run_workload(hc: HebbMind) -> tuple[int, int]: | ||
| store = hc._searcher.store # type: ignore[assignment] | ||
| c = Counter(store) | ||
| # Scenario 1: RecallAgent's 3-query pass (overlapping tokens). | ||
| for q in QUERIES: | ||
| hc.search(q) | ||
| # Scenario 2: the same query retried 4 times (UI retry / agent retry) — | ||
| # the other case issue #54 names. With the cache, only the first hits. | ||
| for _ in range(4): | ||
| hc.search("python gil retrieval") | ||
| return c.corpus, c.df | ||
|
|
||
|
|
||
| def _new_home() -> Path: | ||
| home = Path(tempfile.mkdtemp()) | ||
| return home | ||
|
|
||
|
|
||
| def main() -> int: | ||
| print("=" * 64) | ||
| print("Issue #54 demo: does caching cut DB round-trips on recall?") | ||
| print("=" * 64) | ||
|
|
||
| # ---- A) With cache (default, TTL=60s) ------------------------------- # | ||
| print("\n[A] WITH cache (TTL=60s, the fix):") | ||
| home_a = _new_home() | ||
| sa = Settings(home_dir=home_a, llm_model="openai/gpt-4o-mini", | ||
| embedding_provider="noop", embedding_dim=3) | ||
| hc = HebbMind(config=sa) | ||
| _seed(hc) | ||
| corpus_a, df_a = _run_workload(hc) | ||
| print(f" 3 recall queries + 4 retried same-query searches") | ||
| print(f" → {corpus_a} corpus_size SQL, {df_a} DF SQL") | ||
| hc.close() | ||
| shutil.rmtree(home_a, ignore_errors=True) | ||
|
|
||
| # ---- B) Without cache (TTL=0 → every call re-fetches) --------------- # | ||
| print("\n[B] WITHOUT cache (TTL=0 → re-fetch every time):") | ||
| home_b = _new_home() | ||
| sb = Settings(home_dir=home_b, llm_model="openai/gpt-4o-mini", | ||
| embedding_provider="noop", embedding_dim=3) | ||
| hc = HebbMind(config=sb) | ||
| _seed(hc) | ||
| hc._searcher._idf_cache_ttl = 0.0 # neutralise: all entries expire instantly | ||
| corpus_b, df_b = _run_workload(hc) | ||
| print(f" 3 recall queries + 4 retried same-query searches") | ||
| print(f" → {corpus_b} corpus_size SQL, {df_b} DF SQL") | ||
| hc.close() | ||
| shutil.rmtree(home_b, ignore_errors=True) | ||
|
|
||
| # ---- Verdict -------------------------------------------------------- # | ||
| print("\n" + "=" * 64) | ||
| print("VERDICT") | ||
| print("=" * 64) | ||
| print(f" corpus_size SQL: with cache {corpus_a} vs without {corpus_b}") | ||
| print(f" DF SQL: with cache {df_a} vs without {df_b}") | ||
| saved_c = corpus_b - corpus_a | ||
| saved_d = df_b - df_a | ||
| print(f"\n #54 saves {saved_c} corpus_size round-trip(s) and {saved_d} DF round-trip(s)") | ||
| print(" on this 3-query recall pass. That is the waste the cache removes.") | ||
| print("=" * 64) | ||
| return 0 if (saved_c >= 0 and saved_d >= 0) else 1 | ||
|
|
||
|
|
||
| if __name__ == "__main__": | ||
| sys.exit(main()) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,142 @@ | ||
| """End-to-end verification of the IDF cache via the real HebbMind facade (issue #54). | ||
|
|
||
| Unlike ``verify_idf_cache_real_store.py`` (which drives ``SQLiteMemoryStore`` + | ||
| ``MemorySearcher`` directly), this starts the actual public SDK entrypoint — | ||
| ``HebbMind`` — exactly the way README's five-minute tour does: ``hc.add(...)`` | ||
| then ``hc.search(...)``. The whole stack (storage + embedder + graph + hybrid | ||
| searcher) runs in-process; only the embedding model is swapped for the noop | ||
| provider so no model download / GPU is needed. | ||
|
|
||
| Asserts, against the genuine component chain: | ||
|
|
||
| 1. After two identical ``search()`` calls, the searcher's IDF statistic caches | ||
| are populated (corpus_size: one entry per partition scope; df: one per | ||
| (token, partition)) — i.e. the production code path actually fills the cache. | ||
| 2. The second ``search()`` reuses cached stats — verified by counting real | ||
| ``SQLiteMemoryStore`` method calls (each = a real SQL round-trip). | ||
| 3. Staleness is bounded, not silent: a memory written AFTER the cache is filled | ||
| is invisible to IDF within the TTL (expected by design), and reappears once | ||
| the TTL elapses. Proves the cache is honest about its trade-off. | ||
|
|
||
| Runnable script (scripts/ is ruff-excluded), prints PASS/FAIL, exits non-zero | ||
| on regression. | ||
| """ | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| import asyncio | ||
| import sys | ||
| import tempfile | ||
| import time | ||
| from pathlib import Path | ||
|
|
||
| sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | ||
|
|
||
| from hebb import HebbMind | ||
| from hebb.config.settings import Settings | ||
| from hebb.storage.sqlite_store import SQLiteMemoryStore | ||
|
|
||
| _QUERY = "alpha beta" | ||
|
|
||
|
|
||
| class IDFReadCounter: | ||
| """Count real calls to the live store's IDF-read methods.""" | ||
|
|
||
| def __init__(self, store: SQLiteMemoryStore) -> None: | ||
| self.corpus = 0 | ||
| self.df = 0 | ||
| self._store = store | ||
| self._orig_c = store.corpus_size | ||
| self._orig_d = store.keyword_doc_freqs | ||
| store.corpus_size = self._c # type: ignore[method-assign] | ||
| store.keyword_doc_freqs = self._d # type: ignore[method-assign] | ||
|
|
||
| async def _c(self, partition_ids=None) -> int: | ||
| self.corpus += 1 | ||
| return await self._orig_c(partition_ids) | ||
|
|
||
| async def _d(self, terms, partition_ids=None) -> dict[str, int]: | ||
| self.df += 1 | ||
| return await self._orig_d(terms, partition_ids) | ||
|
|
||
|
|
||
| def main() -> int: | ||
| failures: list[str] = [] | ||
| home = Path(tempfile.mkdtemp()) | ||
| settings = Settings( | ||
| home_dir=home, | ||
| llm_model="openai/gpt-4o-mini", | ||
| embedding_provider="noop", | ||
| embedding_dim=3, | ||
| ) | ||
|
|
||
| with HebbMind(config=settings) as hc: | ||
| hc.add("alpha beta rule the early corpus", partition="mem_semantic") | ||
| hc.add("beta dominates the common terms here", partition="mem_semantic") | ||
|
|
||
| searcher = hc._searcher # the live MemorySearcher the facade owns | ||
| store = searcher.store # type: ignore[assignment] | ||
| counter = IDFReadCounter(store) | ||
|
|
||
| # ---- Check 1: two identical searches, second reuses the cache ------- # | ||
| hc.search(_QUERY) | ||
| first_c, first_d = counter.corpus, counter.df | ||
| hc.search(_QUERY) | ||
| second_c, second_d = counter.corpus, counter.df | ||
| d_c, d_d = second_c - first_c, second_d - first_d | ||
|
|
||
| cache_populated = len(searcher._corpus_size_cache) >= 1 and len(searcher._df_cache) >= 1 | ||
| second_reused = d_c == 0 and d_d == 0 | ||
| print( | ||
| "Check 1 — real HebbMind.search() twice:\n" | ||
| f" caches populated? corpus_size={len(searcher._corpus_size_cache)}, df={len(searcher._df_cache)}\n" | ||
| f" 1st → {first_c} corpus / {first_d} df calls; 2nd → +{d_c} / +{d_d} (expect 0/0)" | ||
| ) | ||
| if not cache_populated: | ||
| failures.append("Check 1a: production path did not populate the IDF cache.") | ||
| if not second_reused: | ||
| failures.append(f"Check 1b: 2nd search added {d_c}/{d_d} IDF calls (expected 0/0).") | ||
|
|
||
| # ---- Check 2: staleness within TTL, then self-heals after TTL ------- # | ||
| # Write a 3rd memory. Within the TTL the cached corpus_size is stale | ||
| # (still 2), so the IDF weighter keeps using the old N — by design. | ||
| hc.add("gamma is a rare word", partition="mem_semantic") | ||
| # Read the ACTUAL partition key the production path cached under (the | ||
| # facade's search scope may be None or a partition set — don't assume). | ||
| pk = next(iter(searcher._corpus_size_cache.keys())) | ||
| cached_n = searcher._idf_corpus_size_get(time.monotonic(), pk) | ||
| stale_within_ttl = cached_n == 2 # real is now 3 | ||
| print( | ||
| f"\nCheck 2 — staleness bounded by TTL:\n" | ||
| f" after 3rd write, cached corpus_size = {cached_n} (real 3) → stale within TTL? {stale_within_ttl}" | ||
| ) | ||
| if not stale_within_ttl: | ||
| failures.append("Check 2a: cache did not show expected bounded staleness.") | ||
|
|
||
| # Backdate the cache entry so the TTL elapses, then search → must | ||
| # refetch the NEW corpus size (3). | ||
| entry = searcher._corpus_size_cache.get(pk) | ||
| if entry is not None: | ||
| searcher._corpus_size_cache[pk] = (entry[0], time.monotonic() - 1) | ||
| # Also expire the df entries so they refetch against the new corpus. | ||
| for k in list(searcher._df_cache.keys()): | ||
| v = searcher._df_cache[k] | ||
| searcher._df_cache[k] = (v[0], time.monotonic() - 1) | ||
| counter.corpus = 0 | ||
| hc.search("alpha gamma") # gamma is the newly-added token | ||
| refreshed = searcher._idf_corpus_size_get(time.monotonic(), pk) | ||
| refetched = refreshed == 3 | ||
| print(f" after forced TTL expiry + search, refetched corpus_size = {refreshed} (expect 3) → self-healed? {refetched}") | ||
| if not refetched: | ||
| failures.append("Check 2b: cache did not refetch the new corpus_size after TTL expiry.") | ||
|
|
||
| print() | ||
| if failures: | ||
| print("RESULT: FAIL — " + " | ".join(failures)) | ||
| return 1 | ||
| print("RESULT: PASS — real HebbMind facade fills/reuses/heals the IDF cache end-to-end.") | ||
| return 0 | ||
|
|
||
|
|
||
| if __name__ == "__main__": | ||
| sys.exit(main()) | ||
Oops, something went wrong.
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
Uh oh!
There was an error while loading. Please reload this page.