fix(memory): plan-ID searches pin the exact plan via metadata lookup (Patrick ruling) + purge 193 scratchpad junk vectors. Live-verified 100% top-hit, 1011 green
This commit is contained in:
@@ -18,6 +18,7 @@ Purpose:
|
||||
layer to satisfy thin-module standard.
|
||||
"""
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
import json
|
||||
import os
|
||||
@@ -210,6 +211,77 @@ def _filter_results(results: list, n_results: int) -> list:
|
||||
return filtered
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# PLAN-ID EXACT MATCHING
|
||||
# =============================================================================
|
||||
|
||||
_PLAN_ID_RE = re.compile(
|
||||
r"(?:^|\b)((?:d|f|p|td|a)plan)[\s\-_]*(\d{3,5})\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _extract_plan_id(query: str) -> str | None:
|
||||
"""Extract a normalized plan ID (e.g. 'FPLAN-0332') from a query string."""
|
||||
m = _PLAN_ID_RE.search(query)
|
||||
if not m:
|
||||
return None
|
||||
prefix = m.group(1).upper()
|
||||
number = m.group(2)
|
||||
return f"{prefix}-{number}"
|
||||
|
||||
|
||||
def _fetch_plan_by_metadata(plan_id: str, n_results: int) -> list:
|
||||
"""Fetch plan chunks directly from ChromaDB by source_file metadata."""
|
||||
input_data = {
|
||||
"operation": "get_by_source",
|
||||
"collection_name": "flow_plans",
|
||||
"source_pattern": plan_id,
|
||||
"n_results": n_results,
|
||||
}
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[str(MEMORY_PYTHON), str(CHROMA_SUBPROCESS_SCRIPT)],
|
||||
input=json.dumps(input_data),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return []
|
||||
data = json.loads(result.stdout)
|
||||
if not data.get("success"):
|
||||
return []
|
||||
return data.get("results", [])
|
||||
except Exception as e:
|
||||
logger.warning(f"[search] Plan metadata fetch failed: {e}")
|
||||
return []
|
||||
|
||||
|
||||
def _pin_plan_id_matches(query: str, filtered: list, n_results: int) -> list:
|
||||
"""Pin exact plan-ID matches to the top of results.
|
||||
|
||||
If the query contains a plan-ID pattern, fetch matching chunks directly
|
||||
from ChromaDB metadata (bypassing embedding similarity) and pin them.
|
||||
"""
|
||||
plan_id = _extract_plan_id(query)
|
||||
if not plan_id:
|
||||
return filtered
|
||||
|
||||
exact = _fetch_plan_by_metadata(plan_id, n_results)
|
||||
if not exact:
|
||||
return filtered
|
||||
|
||||
for r in exact:
|
||||
r["similarity"] = 1.0
|
||||
|
||||
logger.info(f"[search] Pinned {len(exact)} exact matches for {plan_id}")
|
||||
|
||||
seen_ids = {r.get("id") for r in exact}
|
||||
rest = [r for r in filtered if r.get("id") not in seen_ids]
|
||||
return (exact + rest)[:n_results]
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# PUBLIC API
|
||||
# =============================================================================
|
||||
@@ -273,6 +345,9 @@ def execute_search(
|
||||
# Step 3: Filter and score results
|
||||
filtered_results = _filter_results(raw_results, n_results)
|
||||
|
||||
# Step 4: Pin exact plan-ID matches to the top
|
||||
filtered_results = _pin_plan_id_matches(query, filtered_results, n_results)
|
||||
|
||||
logger.info(f"[search] Filtered to {len(filtered_results)} relevant results")
|
||||
|
||||
json_handler.log_operation(
|
||||
|
||||
@@ -152,6 +152,79 @@ def _check_plan(plan_label, db_path=None):
|
||||
return {"success": True, "found": match_count > 0, "count": match_count, "source_files": sorted(matching_files)}
|
||||
|
||||
|
||||
def _get_by_source(collection_name, source_pattern, n_results=5, db_path=None):
|
||||
"""Fetch documents whose source_file metadata contains a pattern.
|
||||
|
||||
Args:
|
||||
collection_name: Name of the ChromaDB collection
|
||||
source_pattern: Substring to match in source_file metadata
|
||||
n_results: Maximum number of results to return
|
||||
db_path: Optional path to Chroma database
|
||||
|
||||
Returns:
|
||||
Dict with success, results list (document, metadata, id)
|
||||
"""
|
||||
client = _get_client(db_path)
|
||||
|
||||
try:
|
||||
collection = client.get_collection(collection_name, embedding_function=None)
|
||||
except Exception as e:
|
||||
logger.warning(f"[chroma_subprocess] Collection '{collection_name}' not found in get_by_source: {e}")
|
||||
return {"success": False, "error": f"Collection '{collection_name}' not found: {e}"}
|
||||
|
||||
result = collection.get(include=["metadatas", "documents"])
|
||||
matches = []
|
||||
for i, meta in enumerate(result.get("metadatas", [])):
|
||||
source = meta.get("source_file", "")
|
||||
if source_pattern in source:
|
||||
matches.append(
|
||||
{
|
||||
"collection": collection_name,
|
||||
"document": result["documents"][i],
|
||||
"metadata": meta,
|
||||
"id": result["ids"][i],
|
||||
"distance": 0.0,
|
||||
}
|
||||
)
|
||||
if len(matches) >= n_results:
|
||||
break
|
||||
|
||||
return {"success": True, "results": matches, "count": len(matches)}
|
||||
|
||||
|
||||
def _delete_by_source(collection_name, source_pattern, db_path=None):
|
||||
"""Delete vectors whose source_file metadata contains a pattern.
|
||||
|
||||
Args:
|
||||
collection_name: Name of the ChromaDB collection
|
||||
source_pattern: Substring to match in source_file metadata
|
||||
db_path: Optional path to Chroma database
|
||||
|
||||
Returns:
|
||||
Dict with success, deleted count, and matched IDs
|
||||
"""
|
||||
client = _get_client(db_path)
|
||||
|
||||
try:
|
||||
collection = client.get_collection(collection_name, embedding_function=None)
|
||||
except Exception as e:
|
||||
logger.warning(f"[chroma_subprocess] Collection '{collection_name}' not found in delete_by_source: {e}")
|
||||
return {"success": False, "error": f"Collection '{collection_name}' not found: {e}"}
|
||||
|
||||
result = collection.get(include=["metadatas"])
|
||||
ids_to_delete = []
|
||||
for i, meta in enumerate(result.get("metadatas", [])):
|
||||
source = meta.get("source_file", "")
|
||||
if source_pattern in source:
|
||||
ids_to_delete.append(result["ids"][i])
|
||||
|
||||
if not ids_to_delete:
|
||||
return {"success": True, "deleted": 0, "ids": [], "message": "No matching vectors found"}
|
||||
|
||||
collection.delete(ids=ids_to_delete)
|
||||
return {"success": True, "deleted": len(ids_to_delete), "ids": ids_to_delete}
|
||||
|
||||
|
||||
def _search_vectors(query_embedding, branch=None, memory_type=None, n_results=5, db_path=None):
|
||||
"""Search for similar vectors."""
|
||||
client = _get_client(db_path)
|
||||
@@ -232,6 +305,19 @@ def main():
|
||||
)
|
||||
elif operation == "check_plan":
|
||||
result = _check_plan(plan_label=input_data.get("plan_label"), db_path=input_data.get("db_path"))
|
||||
elif operation == "get_by_source":
|
||||
result = _get_by_source(
|
||||
collection_name=input_data.get("collection_name"),
|
||||
source_pattern=input_data.get("source_pattern"),
|
||||
n_results=input_data.get("n_results", 5),
|
||||
db_path=input_data.get("db_path"),
|
||||
)
|
||||
elif operation == "delete_by_source":
|
||||
result = _delete_by_source(
|
||||
collection_name=input_data.get("collection_name"),
|
||||
source_pattern=input_data.get("source_pattern"),
|
||||
db_path=input_data.get("db_path"),
|
||||
)
|
||||
else:
|
||||
result = {"success": False, "error": f"Unknown operation: {operation}"}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user