From b8f6fb8dad8dee59d592cdfb3278783e4d365563 Mon Sep 17 00:00:00 2001 From: AIOSAI Date: Thu, 16 Jul 2026 22:03:04 -0700 Subject: [PATCH] fix(memory): plan-ID searches pin the exact plan via metadata lookup (Patrick ruling) + purge 193 scratchpad junk vectors. Live-verified 100% top-hit, 1011 green --- .../apps/handlers/search/query_executor.py | 75 ++++++++++++++++ .../handlers/storage/chroma_subprocess.py | 86 +++++++++++++++++++ 2 files changed, 161 insertions(+) diff --git a/src/aipass/memory/apps/handlers/search/query_executor.py b/src/aipass/memory/apps/handlers/search/query_executor.py index 22fedf43..c2f74565 100644 --- a/src/aipass/memory/apps/handlers/search/query_executor.py +++ b/src/aipass/memory/apps/handlers/search/query_executor.py @@ -18,6 +18,7 @@ Purpose: layer to satisfy thin-module standard. """ +import re import subprocess import json import os @@ -210,6 +211,77 @@ def _filter_results(results: list, n_results: int) -> list: return filtered +# ============================================================================= +# PLAN-ID EXACT MATCHING +# ============================================================================= + +_PLAN_ID_RE = re.compile( + r"(?:^|\b)((?:d|f|p|td|a)plan)[\s\-_]*(\d{3,5})\b", + re.IGNORECASE, +) + + +def _extract_plan_id(query: str) -> str | None: + """Extract a normalized plan ID (e.g. 'FPLAN-0332') from a query string.""" + m = _PLAN_ID_RE.search(query) + if not m: + return None + prefix = m.group(1).upper() + number = m.group(2) + return f"{prefix}-{number}" + + +def _fetch_plan_by_metadata(plan_id: str, n_results: int) -> list: + """Fetch plan chunks directly from ChromaDB by source_file metadata.""" + input_data = { + "operation": "get_by_source", + "collection_name": "flow_plans", + "source_pattern": plan_id, + "n_results": n_results, + } + try: + result = subprocess.run( + [str(MEMORY_PYTHON), str(CHROMA_SUBPROCESS_SCRIPT)], + input=json.dumps(input_data), + capture_output=True, + text=True, + timeout=30, + ) + if result.returncode != 0: + return [] + data = json.loads(result.stdout) + if not data.get("success"): + return [] + return data.get("results", []) + except Exception as e: + logger.warning(f"[search] Plan metadata fetch failed: {e}") + return [] + + +def _pin_plan_id_matches(query: str, filtered: list, n_results: int) -> list: + """Pin exact plan-ID matches to the top of results. + + If the query contains a plan-ID pattern, fetch matching chunks directly + from ChromaDB metadata (bypassing embedding similarity) and pin them. + """ + plan_id = _extract_plan_id(query) + if not plan_id: + return filtered + + exact = _fetch_plan_by_metadata(plan_id, n_results) + if not exact: + return filtered + + for r in exact: + r["similarity"] = 1.0 + + logger.info(f"[search] Pinned {len(exact)} exact matches for {plan_id}") + + seen_ids = {r.get("id") for r in exact} + rest = [r for r in filtered if r.get("id") not in seen_ids] + return (exact + rest)[:n_results] + + # ============================================================================= # PUBLIC API # ============================================================================= @@ -273,6 +345,9 @@ def execute_search( # Step 3: Filter and score results filtered_results = _filter_results(raw_results, n_results) + # Step 4: Pin exact plan-ID matches to the top + filtered_results = _pin_plan_id_matches(query, filtered_results, n_results) + logger.info(f"[search] Filtered to {len(filtered_results)} relevant results") json_handler.log_operation( diff --git a/src/aipass/memory/apps/handlers/storage/chroma_subprocess.py b/src/aipass/memory/apps/handlers/storage/chroma_subprocess.py index 18b867e9..9efd166e 100755 --- a/src/aipass/memory/apps/handlers/storage/chroma_subprocess.py +++ b/src/aipass/memory/apps/handlers/storage/chroma_subprocess.py @@ -152,6 +152,79 @@ def _check_plan(plan_label, db_path=None): return {"success": True, "found": match_count > 0, "count": match_count, "source_files": sorted(matching_files)} +def _get_by_source(collection_name, source_pattern, n_results=5, db_path=None): + """Fetch documents whose source_file metadata contains a pattern. + + Args: + collection_name: Name of the ChromaDB collection + source_pattern: Substring to match in source_file metadata + n_results: Maximum number of results to return + db_path: Optional path to Chroma database + + Returns: + Dict with success, results list (document, metadata, id) + """ + client = _get_client(db_path) + + try: + collection = client.get_collection(collection_name, embedding_function=None) + except Exception as e: + logger.warning(f"[chroma_subprocess] Collection '{collection_name}' not found in get_by_source: {e}") + return {"success": False, "error": f"Collection '{collection_name}' not found: {e}"} + + result = collection.get(include=["metadatas", "documents"]) + matches = [] + for i, meta in enumerate(result.get("metadatas", [])): + source = meta.get("source_file", "") + if source_pattern in source: + matches.append( + { + "collection": collection_name, + "document": result["documents"][i], + "metadata": meta, + "id": result["ids"][i], + "distance": 0.0, + } + ) + if len(matches) >= n_results: + break + + return {"success": True, "results": matches, "count": len(matches)} + + +def _delete_by_source(collection_name, source_pattern, db_path=None): + """Delete vectors whose source_file metadata contains a pattern. + + Args: + collection_name: Name of the ChromaDB collection + source_pattern: Substring to match in source_file metadata + db_path: Optional path to Chroma database + + Returns: + Dict with success, deleted count, and matched IDs + """ + client = _get_client(db_path) + + try: + collection = client.get_collection(collection_name, embedding_function=None) + except Exception as e: + logger.warning(f"[chroma_subprocess] Collection '{collection_name}' not found in delete_by_source: {e}") + return {"success": False, "error": f"Collection '{collection_name}' not found: {e}"} + + result = collection.get(include=["metadatas"]) + ids_to_delete = [] + for i, meta in enumerate(result.get("metadatas", [])): + source = meta.get("source_file", "") + if source_pattern in source: + ids_to_delete.append(result["ids"][i]) + + if not ids_to_delete: + return {"success": True, "deleted": 0, "ids": [], "message": "No matching vectors found"} + + collection.delete(ids=ids_to_delete) + return {"success": True, "deleted": len(ids_to_delete), "ids": ids_to_delete} + + def _search_vectors(query_embedding, branch=None, memory_type=None, n_results=5, db_path=None): """Search for similar vectors.""" client = _get_client(db_path) @@ -232,6 +305,19 @@ def main(): ) elif operation == "check_plan": result = _check_plan(plan_label=input_data.get("plan_label"), db_path=input_data.get("db_path")) + elif operation == "get_by_source": + result = _get_by_source( + collection_name=input_data.get("collection_name"), + source_pattern=input_data.get("source_pattern"), + n_results=input_data.get("n_results", 5), + db_path=input_data.get("db_path"), + ) + elif operation == "delete_by_source": + result = _delete_by_source( + collection_name=input_data.get("collection_name"), + source_pattern=input_data.get("source_pattern"), + db_path=input_data.get("db_path"), + ) else: result = {"success": False, "error": f"Unknown operation: {operation}"}