feat: port Memory Bank module to AIPass repo (FPLAN-0003)
Port internal Memory Bank system into src/aipass/memory/ with adapted imports and paths. All 17 Python files compile clean, 13 existing tests still pass. Ported components: - Entry point: apps/memory.py (drone @memory command routing) - Modules: rollover.py (orchestration), search.py (query routing) - Handlers: detector, extractor, line_counter, json_handler, indexer, embedder, chroma, vector_search, normalize, manager, dashboard_push, central_writer, memory_watcher, chroma_subprocess Adaptations from internal system: - Removed all /home/aipass/ hardcoded paths - Removed sys.path manipulation hacks - Replaced prax logger with stdlib logging - Replaced cli console/header with Rich (already a dep) - Changed BRANCH_REGISTRY → AIPASS_REGISTRY references - Made chromadb/sentence-transformers optional (try/except) - Removed private branch registry concept - Subprocess Python uses sys.executable with AIPASS_MEMORY_PYTHON override Still needs: drone registration, spawn template updates, container testing, integration tests Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
08f84233ae
commit
da4fa82771
Executable
+645
@@ -0,0 +1,645 @@
|
||||
|
||||
# ===================AIPASS====================
|
||||
# META DATA HEADER
|
||||
# Name: rollover.py - Rollover Orchestration Module
|
||||
# Date: 2025-11-16
|
||||
# Version: 0.2.0
|
||||
# Category: memory/modules
|
||||
#
|
||||
# CHANGELOG (Max 5 entries):
|
||||
# - v0.2.0 (2026-03-06): Adapted for AIPass public repo - removed internal deps
|
||||
# - v0.1.0 (2025-11-16): Initial version - orchestrate rollover workflow
|
||||
#
|
||||
# CODE STANDARDS:
|
||||
# - Thin orchestration: Delegate all logic to handlers
|
||||
# - No business logic: Only coordinate workflow
|
||||
# - handle_command() pattern
|
||||
# =============================================
|
||||
|
||||
"""
|
||||
Rollover Orchestration Module
|
||||
|
||||
Coordinates the memory rollover workflow by calling handlers in sequence:
|
||||
1. Detect rollover triggers (monitor/detector)
|
||||
2. Extract oldest memories (rollover/extractor)
|
||||
3. Generate embeddings (vector/embedder)
|
||||
4. Store in Chroma (storage/chroma)
|
||||
|
||||
Purpose:
|
||||
Thin orchestration layer - no business logic implementation.
|
||||
All domain logic lives in handlers.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import logging
|
||||
import subprocess
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import List, Dict
|
||||
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
from rich import box
|
||||
|
||||
# =============================================================================
|
||||
# INFRASTRUCTURE SETUP
|
||||
# =============================================================================
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
console = Console()
|
||||
|
||||
# Handler imports (relative within the memory package)
|
||||
from ..handlers.monitor import detector
|
||||
from ..handlers.rollover import extractor
|
||||
from ..handlers.vector import embedder
|
||||
from ..handlers.tracking import line_counter
|
||||
|
||||
# ChromaDB storage via subprocess
|
||||
# Resolve paths relative to the handlers directory
|
||||
_HANDLERS_DIR = Path(__file__).resolve().parent.parent / "handlers"
|
||||
CHROMA_SUBPROCESS_SCRIPT = _HANDLERS_DIR / "storage" / "chroma_subprocess.py"
|
||||
|
||||
# Use system python by default; can be overridden via environment variable
|
||||
import os
|
||||
MEMORY_PYTHON = os.environ.get("AIPASS_MEMORY_PYTHON", sys.executable)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# REPO ROOT DISCOVERY
|
||||
# =============================================================================
|
||||
|
||||
def _find_repo_root() -> Path:
|
||||
"""Walk up from this file to find the repo root (contains AIPASS_REGISTRY.json)."""
|
||||
current = Path(__file__).resolve().parent
|
||||
for parent in [current] + list(current.parents):
|
||||
if (parent / "AIPASS_REGISTRY.json").exists():
|
||||
return parent
|
||||
return Path.cwd()
|
||||
|
||||
|
||||
_REPO_ROOT = _find_repo_root()
|
||||
|
||||
|
||||
# No other module imports (modules don't import modules)
|
||||
|
||||
|
||||
def _store_vectors_subprocess(branch: str, memory_type: str, embeddings: list,
|
||||
documents: list, metadatas: list, db_path: str | Path | None = None) -> dict:
|
||||
"""
|
||||
Store vectors via subprocess.
|
||||
|
||||
This ensures ChromaDB compatibility regardless of calling Python version.
|
||||
"""
|
||||
# Convert numpy arrays to lists for JSON serialization
|
||||
embeddings_serializable = [
|
||||
emb.tolist() if hasattr(emb, 'tolist') else emb
|
||||
for emb in embeddings
|
||||
]
|
||||
|
||||
input_data = {
|
||||
'operation': 'store_vectors',
|
||||
'branch': branch,
|
||||
'memory_type': memory_type,
|
||||
'embeddings': embeddings_serializable,
|
||||
'documents': documents,
|
||||
'metadatas': metadatas,
|
||||
'db_path': str(db_path) if db_path else None
|
||||
}
|
||||
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[str(MEMORY_PYTHON), str(CHROMA_SUBPROCESS_SCRIPT)],
|
||||
input=json.dumps(input_data),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
return {'success': False, 'error': result.stderr or 'Subprocess failed'}
|
||||
|
||||
return json.loads(result.stdout)
|
||||
except subprocess.TimeoutExpired:
|
||||
return {'success': False, 'error': 'Storage operation timed out'}
|
||||
except json.JSONDecodeError as e:
|
||||
return {'success': False, 'error': f'Invalid JSON response: {e}'}
|
||||
except Exception as e:
|
||||
return {'success': False, 'error': str(e)}
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# COMMAND HANDLERS
|
||||
# =============================================================================
|
||||
|
||||
def handle_command(command: str, args: List[str]) -> bool: # noqa: ARG001
|
||||
"""
|
||||
Handle rollover commands
|
||||
|
||||
Commands supported:
|
||||
- rollover: Execute rollover for triggered branches
|
||||
- status: Show rollover statistics
|
||||
- check: Check which branches need rollover
|
||||
- sync-lines: Update line count metadata for all branches
|
||||
|
||||
Args:
|
||||
command: Command name
|
||||
args: Additional arguments
|
||||
|
||||
Returns:
|
||||
True if command handled, False otherwise
|
||||
"""
|
||||
if command in ('--help', '-h', 'help'):
|
||||
print_help()
|
||||
return True
|
||||
|
||||
if command == 'rollover':
|
||||
execute_rollover()
|
||||
return True
|
||||
|
||||
elif command == 'status':
|
||||
show_status()
|
||||
return True
|
||||
|
||||
elif command == 'check':
|
||||
check_triggers()
|
||||
return True
|
||||
|
||||
elif command == 'sync-lines':
|
||||
sync_line_counts()
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def print_help() -> None:
|
||||
"""Display rollover module help"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Rollover Module - Memory Rollover Orchestration[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
console.print("[bold]USAGE:[/bold]")
|
||||
console.print(" python3 -m aipass.memory.apps.modules.rollover <command>")
|
||||
console.print()
|
||||
console.print("[bold]COMMANDS:[/bold]")
|
||||
console.print(" [cyan]rollover[/cyan] Execute rollover for files over 600 lines")
|
||||
console.print(" [cyan]status[/cyan] Show rollover statistics for all branches")
|
||||
console.print(" [cyan]check[/cyan] Check which files need rollover (dry run)")
|
||||
console.print(" [cyan]sync-lines[/cyan] Update line count metadata for all branches")
|
||||
console.print(" [cyan]help[/cyan] Show this help message")
|
||||
console.print()
|
||||
console.print("[bold]WORKFLOW:[/bold]")
|
||||
console.print(" 1. Detect files over 600 lines")
|
||||
console.print(" 2. Extract oldest entries (target ~500 lines)")
|
||||
console.print(" 3. Generate embeddings via sentence-transformers")
|
||||
console.print(" 4. Store vectors in local + global ChromaDB")
|
||||
console.print()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# ROLLOVER ORCHESTRATION
|
||||
# =============================================================================
|
||||
|
||||
def execute_rollover() -> bool:
|
||||
"""
|
||||
Execute rollover workflow for all triggered branches
|
||||
|
||||
Workflow:
|
||||
1. Check all branches for triggers
|
||||
2. For each trigger:
|
||||
- Extract oldest 100 entries
|
||||
- Generate embeddings
|
||||
- Store in Chroma
|
||||
3. Report results
|
||||
"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Memory - Rollover Execution[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
|
||||
# Step 1: Detect triggers
|
||||
console.print("[cyan]Checking for rollover triggers...[/cyan]")
|
||||
triggers_result = detector.check_all_branches()
|
||||
|
||||
if not triggers_result['success']:
|
||||
logger.error(f"[rollover] Failed to check branches: {triggers_result.get('error', 'Unknown error')}")
|
||||
console.print("[red]x[/red] Failed to check for rollover triggers")
|
||||
return False
|
||||
|
||||
triggers = triggers_result.get('triggers', [])
|
||||
if not triggers:
|
||||
console.print("[green]>[/green] No files need rollover")
|
||||
logger.info("[rollover] No rollover triggers detected")
|
||||
return True
|
||||
|
||||
console.print(f"[green]>[/green] Found {len(triggers)} files ready for rollover")
|
||||
logger.info(f"[rollover] Found {len(triggers)} files ready for rollover")
|
||||
console.print()
|
||||
|
||||
# Process each trigger
|
||||
success_count = 0
|
||||
failed = []
|
||||
|
||||
for trigger in triggers:
|
||||
console.print(f"[yellow]Processing:[/yellow] {trigger}")
|
||||
|
||||
# Step 1: CREATE BACKUP (safety net)
|
||||
backup_result = extractor.create_rollover_backup(trigger.file_path)
|
||||
|
||||
if not backup_result['success']:
|
||||
error_msg = backup_result.get('error', 'Backup failed')
|
||||
logger.error(f"[rollover] Backup failed for {trigger}: {error_msg}")
|
||||
failed.append((trigger, "backup", error_msg))
|
||||
continue # Don't proceed without backup
|
||||
|
||||
logger.info(f"[rollover] {backup_result.get('message')}")
|
||||
|
||||
# Step 2: Extract memories (auto-calculates percentage)
|
||||
extract_result = extractor.extract_with_metadata(trigger.file_path)
|
||||
|
||||
if not extract_result['success']:
|
||||
error_msg = extract_result.get('error', 'Unknown error')
|
||||
logger.error(f"[rollover] Extraction failed for {trigger}: {error_msg}")
|
||||
|
||||
# RESTORE from backup
|
||||
restore_result = extractor.restore_from_backup(trigger.file_path)
|
||||
if restore_result['success']:
|
||||
logger.info("[rollover] Restored from backup after extraction failure")
|
||||
|
||||
failed.append((trigger, "extraction", error_msg))
|
||||
continue
|
||||
|
||||
memories = extract_result.get('entries', [])
|
||||
branch = extract_result.get('branch', '')
|
||||
memory_type = extract_result.get('type', 'unknown')
|
||||
old_lines = extract_result.get('old_lines', 0)
|
||||
new_lines = extract_result.get('new_lines', 0)
|
||||
|
||||
if not branch:
|
||||
logger.error(f"[rollover] No branch found in extraction result for {trigger}")
|
||||
failed.append((trigger, "extraction", "No branch in result"))
|
||||
continue
|
||||
|
||||
logger.info(f"[rollover] Extracted {len(memories)} items from {trigger} ({old_lines} -> {new_lines} lines)")
|
||||
|
||||
# Convert memory items to text for vectorization
|
||||
texts = _extract_text_from_memories(memories)
|
||||
|
||||
# Step 3: Generate embeddings
|
||||
embed_result = embedder.encode_batch(texts)
|
||||
|
||||
if not embed_result['success']:
|
||||
error_msg = embed_result.get('error', 'Unknown error')
|
||||
logger.error(f"[rollover] Embedding failed for {trigger}: {error_msg}")
|
||||
|
||||
# RESTORE from backup
|
||||
restore_result = extractor.restore_from_backup(trigger.file_path)
|
||||
if restore_result['success']:
|
||||
logger.info("[rollover] Restored from backup after embedding failure")
|
||||
|
||||
failed.append((trigger, "embedding", error_msg))
|
||||
continue
|
||||
|
||||
embeddings = embed_result.get('embeddings', [])
|
||||
if not embeddings:
|
||||
logger.error(f"[rollover] No embeddings generated for {trigger}")
|
||||
failed.append((trigger, "embedding", "No embeddings in result"))
|
||||
continue
|
||||
|
||||
logger.info(f"[rollover] Generated {len(embeddings)} embeddings for {trigger}")
|
||||
|
||||
# Step 4: Prepare metadata for vectorization
|
||||
metadatas = []
|
||||
for memory in memories:
|
||||
metadata = memory.get('_metadata', {})
|
||||
metadata['timestamp'] = memory.get('timestamp', '')
|
||||
metadatas.append(metadata)
|
||||
|
||||
# Step 5: Store in LOCAL branch Chroma (via subprocess)
|
||||
# Type assertions for Pylance (validated above with early returns)
|
||||
branch_str: str = branch
|
||||
memory_type_str: str = memory_type
|
||||
embeddings_list: list = embeddings
|
||||
|
||||
local_chroma_path = _get_branch_local_chroma_path(branch_str)
|
||||
local_store_result = None
|
||||
|
||||
if local_chroma_path:
|
||||
local_store_result = _store_vectors_subprocess(
|
||||
branch=branch_str,
|
||||
memory_type=memory_type_str,
|
||||
embeddings=embeddings_list,
|
||||
documents=texts,
|
||||
metadatas=metadatas,
|
||||
db_path=str(local_chroma_path)
|
||||
)
|
||||
|
||||
if not local_store_result['success']:
|
||||
logger.warning(f"[rollover] Local storage failed for {branch}: {local_store_result.get('error')}")
|
||||
# Continue anyway - global storage is primary
|
||||
else:
|
||||
logger.info(f"[rollover] Stored {len(embeddings)} vectors in local Chroma for {branch}")
|
||||
|
||||
# Step 6: Store in GLOBAL Memory Chroma (via subprocess)
|
||||
global_store_result = _store_vectors_subprocess(
|
||||
branch=branch_str,
|
||||
memory_type=memory_type_str,
|
||||
embeddings=embeddings_list,
|
||||
documents=texts,
|
||||
metadatas=metadatas
|
||||
# db_path=None means global
|
||||
)
|
||||
|
||||
if not global_store_result['success']:
|
||||
error_msg = global_store_result.get('error', 'Unknown error')
|
||||
logger.error(f"[rollover] Global storage failed for {trigger}: {error_msg}")
|
||||
|
||||
# RESTORE from backup (CRITICAL - file was modified but storage failed)
|
||||
restore_result = extractor.restore_from_backup(trigger.file_path)
|
||||
if restore_result['success']:
|
||||
logger.info("[rollover] Restored from backup after storage failure")
|
||||
else:
|
||||
logger.error(f"[rollover] CRITICAL: Failed to restore from backup: {restore_result.get('error')}")
|
||||
|
||||
failed.append((trigger, "global_storage", error_msg))
|
||||
continue
|
||||
|
||||
logger.info(f"[rollover] Stored {len(embeddings)} vectors in global Chroma for {branch}")
|
||||
|
||||
# Step 7: Update line count metadata
|
||||
update_result = line_counter.update_line_count(trigger.file_path)
|
||||
if update_result['success']:
|
||||
logger.info(f"[rollover] Updated line count metadata for {trigger.file_path.name}")
|
||||
else:
|
||||
logger.warning(f"[rollover] Failed to update line count for {trigger.file_path.name}: {update_result.get('error')}")
|
||||
|
||||
# Success!
|
||||
success_count += 1
|
||||
global_collection = global_store_result.get('collection')
|
||||
global_total = global_store_result.get('total_vectors')
|
||||
|
||||
# Report both local and global storage
|
||||
local_status = "> local" if local_store_result and local_store_result['success'] else "x local"
|
||||
console.print(
|
||||
f" [green]>[/green] Rolled over {len(memories)} items -> {global_collection} "
|
||||
f"({old_lines} -> {new_lines} lines, global: {global_total} vectors, {local_status})"
|
||||
)
|
||||
logger.info(f"[rollover] Successfully rolled over {trigger}: {len(memories)} items, {old_lines} -> {new_lines} lines")
|
||||
|
||||
# Report results
|
||||
console.print()
|
||||
if success_count > 0:
|
||||
console.print(f"[green]>[/green] Rollover complete: {success_count}/{len(triggers)} successful")
|
||||
logger.info(f"[rollover] Rollover complete: {success_count}/{len(triggers)} successful")
|
||||
|
||||
if failed:
|
||||
console.print()
|
||||
console.print("[red]Failed operations:[/red]")
|
||||
for trigger, stage, err in failed:
|
||||
console.print(f" [red]x[/red] {trigger} - {stage}: {err}")
|
||||
logger.error(f"[rollover] {len(failed)} operations failed")
|
||||
|
||||
return success_count > 0
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# PATH HELPERS
|
||||
# =============================================================================
|
||||
|
||||
def _get_branch_local_chroma_path(branch_name: str) -> Path | None:
|
||||
"""
|
||||
Get local .chroma path for branch
|
||||
|
||||
Args:
|
||||
branch_name: Branch name (e.g., "SEED", "AIPASS")
|
||||
|
||||
Returns:
|
||||
Path to branch's local .chroma directory, or None if branch not found
|
||||
"""
|
||||
# Read registry to get branch path
|
||||
if not branch_name:
|
||||
return None
|
||||
|
||||
registry = detector._read_registry()
|
||||
|
||||
for branch in registry:
|
||||
if branch.get('name', '').upper() == branch_name.upper():
|
||||
branch_path = Path(branch.get('path', ''))
|
||||
if branch_path.exists():
|
||||
chroma_path = branch_path / '.chroma'
|
||||
# Auto-create .chroma directory if missing
|
||||
if not chroma_path.exists():
|
||||
chroma_path.mkdir(parents=True, exist_ok=True)
|
||||
logger.info(f"[rollover] Created local .chroma directory for {branch_name}")
|
||||
return chroma_path
|
||||
|
||||
logger.warning(f"[rollover] Branch {branch_name} not found in registry")
|
||||
return None
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# TEXT EXTRACTION HELPERS
|
||||
# =============================================================================
|
||||
|
||||
def _extract_text_from_memories(memories: List[Dict]) -> List[str]:
|
||||
"""
|
||||
Extract text content from memory items for vectorization
|
||||
|
||||
Memory items have different structures:
|
||||
- sessions: 'activities' array (join into text)
|
||||
- observations: might have 'content' or 'text' field
|
||||
- generic: convert to JSON string
|
||||
|
||||
Args:
|
||||
memories: List of memory items
|
||||
|
||||
Returns:
|
||||
List of text strings for embedding
|
||||
"""
|
||||
texts = []
|
||||
|
||||
for memory in memories:
|
||||
# Try common text fields
|
||||
if 'activities' in memory and isinstance(memory['activities'], list):
|
||||
# Sessions type - join activities
|
||||
text = '\n'.join(str(a) for a in memory['activities'])
|
||||
elif 'content' in memory:
|
||||
text = str(memory['content'])
|
||||
elif 'text' in memory:
|
||||
text = str(memory['text'])
|
||||
elif 'message' in memory:
|
||||
text = str(memory['message'])
|
||||
else:
|
||||
# Fallback - convert to string representation
|
||||
text = str(memory)
|
||||
|
||||
texts.append(text)
|
||||
|
||||
return texts
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# LINE COUNT SYNC
|
||||
# =============================================================================
|
||||
|
||||
def sync_line_counts() -> None:
|
||||
"""
|
||||
Update line count metadata for all branch memory files.
|
||||
|
||||
Reads actual line counts and updates document_metadata.status.current_lines
|
||||
for all *.local.json and *.observations.json files in AIPASS_REGISTRY.
|
||||
"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Memory - Sync Line Counts[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
|
||||
console.print("[cyan]Updating line counts for all memory files...[/cyan]")
|
||||
console.print()
|
||||
|
||||
result = line_counter.update_all_memory_files()
|
||||
|
||||
if result['success']:
|
||||
console.print(f"[green]>[/green] Updated {result['updated']} files")
|
||||
if result['failed'] > 0:
|
||||
console.print(f"[yellow]![/yellow] {result['failed']} files failed:")
|
||||
for branch, mem_type, error in result.get('failures', []):
|
||||
console.print(f" [red]x[/red] {branch}.{mem_type}: {error}")
|
||||
logger.info(f"[rollover] Synced line counts: {result['updated']} updated, {result['failed']} failed")
|
||||
else:
|
||||
console.print("[red]x[/red] Failed to sync line counts")
|
||||
logger.error("[rollover] Failed to sync line counts")
|
||||
|
||||
console.print()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# STATUS & CHECKING
|
||||
# =============================================================================
|
||||
|
||||
def get_rollover_stats() -> dict:
|
||||
"""Return rollover stats from detector (module-layer passthrough)."""
|
||||
return detector.get_rollover_stats()
|
||||
|
||||
|
||||
def show_status() -> None:
|
||||
"""
|
||||
Show rollover statistics for all branches
|
||||
|
||||
Displays:
|
||||
- Files checked
|
||||
- Files ready for rollover
|
||||
- Per-branch status (current/max lines)
|
||||
"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Memory - Rollover Status[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
|
||||
# Get stats from detector
|
||||
stats_result = detector.get_rollover_stats()
|
||||
|
||||
if not stats_result['success']:
|
||||
console.print(f"[red]x[/red] Failed to get status: {stats_result.get('error', 'Unknown error')}")
|
||||
logger.error(f"[rollover] Failed to get status: {stats_result.get('error')}")
|
||||
return
|
||||
|
||||
stats = stats_result
|
||||
|
||||
# Summary
|
||||
console.print(f"[cyan]Branches:[/cyan] {stats['total_branches']}")
|
||||
console.print(f"[cyan]Files checked:[/cyan] {stats['files_checked']}")
|
||||
console.print(f"[cyan]Ready for rollover:[/cyan] {stats['files_ready']}")
|
||||
console.print()
|
||||
|
||||
# Per-branch details
|
||||
if stats['branches']:
|
||||
console.print("[yellow]Branch Details:[/yellow]")
|
||||
console.print()
|
||||
|
||||
for branch_name, branch_stats in stats['branches'].items():
|
||||
console.print(f" [bold]{branch_name}[/bold]")
|
||||
|
||||
for memory_type, file_stats in branch_stats.items():
|
||||
current = file_stats['current']
|
||||
max_lines = file_stats['max']
|
||||
ready = file_stats['ready']
|
||||
remaining = file_stats['remaining']
|
||||
|
||||
status_marker = "[red]![/red]" if ready else "[green]OK[/green]"
|
||||
status_text = "READY" if ready else f"{remaining} remaining"
|
||||
|
||||
console.print(
|
||||
f" {status_marker} {memory_type}: {current}/{max_lines} lines ({status_text})"
|
||||
)
|
||||
|
||||
console.print()
|
||||
|
||||
|
||||
def check_triggers() -> None:
|
||||
"""
|
||||
Check which branches need rollover (without executing)
|
||||
|
||||
Displays list of files that hit rollover threshold
|
||||
"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Memory - Rollover Check[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
|
||||
triggers_result = detector.check_all_branches()
|
||||
|
||||
if not triggers_result['success']:
|
||||
console.print(f"[red]x[/red] Failed to check triggers: {triggers_result.get('error', 'Unknown error')}")
|
||||
logger.error(f"[rollover] Failed to check triggers: {triggers_result.get('error')}")
|
||||
return
|
||||
|
||||
triggers = triggers_result.get('triggers', [])
|
||||
|
||||
if not triggers:
|
||||
console.print("[green]>[/green] No files need rollover")
|
||||
return
|
||||
|
||||
console.print(f"[yellow]Found {len(triggers)} files ready for rollover:[/yellow]")
|
||||
console.print()
|
||||
|
||||
for trigger in triggers:
|
||||
console.print(f" * {trigger}")
|
||||
|
||||
console.print()
|
||||
console.print("[dim]Run 'drone @memory rollover' to process these files[/dim]")
|
||||
console.print()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# STANDALONE EXECUTION
|
||||
# =============================================================================
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
|
||||
# Handle --help before argparse (module standard)
|
||||
if len(sys.argv) < 2 or sys.argv[1] in ('--help', '-h', 'help'):
|
||||
handle_command('help', [])
|
||||
sys.exit(0)
|
||||
|
||||
# Execute command via handle_command
|
||||
command = sys.argv[1]
|
||||
if not handle_command(command, sys.argv[2:]):
|
||||
console.print(f"[red]Unknown command:[/red] {command}")
|
||||
console.print("Run with [cyan]help[/cyan] for available commands")
|
||||
sys.exit(1)
|
||||
Executable
+383
@@ -0,0 +1,383 @@
|
||||
|
||||
# ===================AIPASS====================
|
||||
# META DATA HEADER
|
||||
# Name: search.py - Search Orchestration Module
|
||||
# Date: 2025-11-27
|
||||
# Version: 0.2.0
|
||||
# Category: memory/modules
|
||||
#
|
||||
# CHANGELOG (Max 5 entries):
|
||||
# - v0.2.0 (2026-03-06): Adapted for AIPass public repo - removed internal deps
|
||||
# - v0.1.0 (2025-11-27): Initial version - orchestrate semantic search
|
||||
#
|
||||
# CODE STANDARDS:
|
||||
# - Thin orchestration: Delegate all logic to handlers
|
||||
# - No business logic: Only coordinate workflow
|
||||
# - handle_command() pattern
|
||||
# =============================================
|
||||
|
||||
"""
|
||||
Search Orchestration Module
|
||||
|
||||
Coordinates semantic search workflow by calling handlers in sequence:
|
||||
1. Encode query text to embedding (vector/embedder)
|
||||
2. Search Chroma collections (storage/chroma via subprocess)
|
||||
3. Format and display results (Rich panels)
|
||||
|
||||
Purpose:
|
||||
Thin orchestration layer - no business logic implementation.
|
||||
All domain logic lives in handlers.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import logging
|
||||
import subprocess
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import List
|
||||
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
from rich import box
|
||||
|
||||
# =============================================================================
|
||||
# INFRASTRUCTURE SETUP
|
||||
# =============================================================================
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
console = Console()
|
||||
|
||||
# Handler imports (relative within the memory package)
|
||||
from ..handlers.vector import embedder
|
||||
|
||||
# ChromaDB search via subprocess
|
||||
_HANDLERS_DIR = Path(__file__).resolve().parent.parent / "handlers"
|
||||
CHROMA_SUBPROCESS_SCRIPT = _HANDLERS_DIR / "storage" / "chroma_subprocess.py"
|
||||
|
||||
# Use system python by default; can be overridden via environment variable
|
||||
MEMORY_PYTHON = os.environ.get("AIPASS_MEMORY_PYTHON", sys.executable)
|
||||
|
||||
|
||||
def _search_vectors_subprocess(
|
||||
query_embedding: list,
|
||||
branch: str | None = None,
|
||||
memory_type: str | None = None,
|
||||
n_results: int = 5,
|
||||
db_path: str | Path | None = None
|
||||
) -> dict:
|
||||
"""
|
||||
Search vectors via subprocess.
|
||||
|
||||
This ensures ChromaDB compatibility regardless of calling Python version.
|
||||
"""
|
||||
input_data = {
|
||||
'operation': 'search_vectors',
|
||||
'query_embedding': query_embedding,
|
||||
'branch': branch,
|
||||
'memory_type': memory_type,
|
||||
'n_results': n_results,
|
||||
'db_path': str(db_path) if db_path else None
|
||||
}
|
||||
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[str(MEMORY_PYTHON), str(CHROMA_SUBPROCESS_SCRIPT)],
|
||||
input=json.dumps(input_data),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
return {'success': False, 'error': result.stderr or 'Subprocess failed'}
|
||||
|
||||
return json.loads(result.stdout)
|
||||
except subprocess.TimeoutExpired:
|
||||
return {'success': False, 'error': 'Search operation timed out'}
|
||||
except json.JSONDecodeError as e:
|
||||
return {'success': False, 'error': f'Invalid JSON response: {e}'}
|
||||
except Exception as e:
|
||||
return {'success': False, 'error': str(e)}
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# COMMAND HANDLERS
|
||||
# =============================================================================
|
||||
|
||||
def handle_command(command: str, args: List[str]) -> bool:
|
||||
"""
|
||||
Handle search commands
|
||||
|
||||
Commands supported:
|
||||
- search <query>: Execute semantic search across all branches
|
||||
- help: Show search help
|
||||
|
||||
Args:
|
||||
command: Command name
|
||||
args: Additional arguments (query text, options)
|
||||
|
||||
Returns:
|
||||
True if command handled, False otherwise
|
||||
"""
|
||||
if command in ('--help', '-h', 'help'):
|
||||
print_help()
|
||||
return True
|
||||
|
||||
if command == 'search':
|
||||
if not args:
|
||||
console.print("[red]Error:[/red] Search query required")
|
||||
console.print("Usage: search <query> [--branch BRANCH] [--type TYPE] [--n N]")
|
||||
return True
|
||||
|
||||
# Parse arguments
|
||||
query_parts = []
|
||||
branch = None
|
||||
memory_type = None
|
||||
n_results = 5
|
||||
|
||||
i = 0
|
||||
while i < len(args):
|
||||
if args[i] == '--branch' and i + 1 < len(args):
|
||||
branch = args[i + 1]
|
||||
i += 2
|
||||
elif args[i] == '--type' and i + 1 < len(args):
|
||||
memory_type = args[i + 1]
|
||||
i += 2
|
||||
elif args[i] == '--n' and i + 1 < len(args):
|
||||
try:
|
||||
n_results = int(args[i + 1])
|
||||
except ValueError:
|
||||
console.print(f"[red]Error:[/red] Invalid number: {args[i + 1]}")
|
||||
return True
|
||||
i += 2
|
||||
else:
|
||||
query_parts.append(args[i])
|
||||
i += 1
|
||||
|
||||
query = ' '.join(query_parts)
|
||||
if not query:
|
||||
console.print("[red]Error:[/red] Search query required")
|
||||
return True
|
||||
|
||||
execute_search(query, branch=branch, memory_type=memory_type, n_results=n_results)
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def print_help() -> None:
|
||||
"""Display search module help"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Search Module - Semantic Memory Search[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
console.print("[bold]USAGE:[/bold]")
|
||||
console.print(" python3 -m aipass.memory.apps.modules.search search <query> [options]")
|
||||
console.print()
|
||||
console.print("[bold]COMMANDS:[/bold]")
|
||||
console.print(" [cyan]search <query>[/cyan] Search across all memory collections")
|
||||
console.print(" [cyan]help[/cyan] Show this help message")
|
||||
console.print()
|
||||
console.print("[bold]OPTIONS:[/bold]")
|
||||
console.print(" [cyan]--branch BRANCH[/cyan] Filter by branch (e.g., SEED, CLI)")
|
||||
console.print(" [cyan]--type TYPE[/cyan] Filter by memory type (observations, local)")
|
||||
console.print(" [cyan]--n N[/cyan] Number of results (default: 5)")
|
||||
console.print()
|
||||
console.print("[bold]EXAMPLES:[/bold]")
|
||||
console.print(" # Search all branches")
|
||||
console.print(" [dim]drone @memory search \"error handling patterns\"[/dim]")
|
||||
console.print()
|
||||
console.print(" # Search specific branch")
|
||||
console.print(" [dim]drone @memory search \"registry bugs\" --branch SEED[/dim]")
|
||||
console.print()
|
||||
console.print(" # Search specific memory type")
|
||||
console.print(" [dim]drone @memory search \"collaboration\" --type observations --n 10[/dim]")
|
||||
console.print()
|
||||
console.print("[bold]HOW IT WORKS:[/bold]")
|
||||
console.print(" 1. Convert query to 384-dim embedding (all-MiniLM-L6-v2)")
|
||||
console.print(" 2. Search ChromaDB collections for similar vectors")
|
||||
console.print(" 3. Display top N most relevant memories")
|
||||
console.print()
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# SEARCH ORCHESTRATION
|
||||
# =============================================================================
|
||||
|
||||
def execute_search(query: str, branch: str | None = None, memory_type: str | None = None, n_results: int = 5) -> bool:
|
||||
"""
|
||||
Execute semantic search and display results
|
||||
|
||||
Workflow:
|
||||
1. Encode query to embedding vector
|
||||
2. Search ChromaDB via subprocess
|
||||
3. Format and display results with Rich
|
||||
|
||||
Args:
|
||||
query: Search query text
|
||||
branch: Optional branch filter
|
||||
memory_type: Optional memory type filter
|
||||
n_results: Number of results to return
|
||||
|
||||
Returns:
|
||||
True if search successful, False otherwise
|
||||
"""
|
||||
console.print()
|
||||
console.print(Panel.fit(
|
||||
"[bold cyan]Memory - Semantic Search[/bold cyan]",
|
||||
border_style="cyan",
|
||||
box=box.ROUNDED
|
||||
))
|
||||
console.print()
|
||||
|
||||
# Step 1: Encode query
|
||||
console.print(f"[cyan]Query:[/cyan] {query}")
|
||||
if branch:
|
||||
console.print(f"[cyan]Branch:[/cyan] {branch}")
|
||||
if memory_type:
|
||||
console.print(f"[cyan]Type:[/cyan] {memory_type}")
|
||||
console.print()
|
||||
|
||||
console.print("[dim]Encoding query...[/dim]")
|
||||
embed_result = embedder.encode_batch([query])
|
||||
|
||||
if not embed_result['success']:
|
||||
error_msg = embed_result.get('error', 'Unknown error')
|
||||
logger.error(f"[search] Failed to encode query: {error_msg}")
|
||||
console.print(f"[red]x[/red] Failed to encode query: {error_msg}")
|
||||
return False
|
||||
|
||||
embeddings = embed_result.get('embeddings', [])
|
||||
if not embeddings:
|
||||
console.print("[red]x[/red] No embedding generated")
|
||||
return False
|
||||
|
||||
query_embedding = embeddings[0]
|
||||
# Convert numpy array to list for JSON serialization
|
||||
if hasattr(query_embedding, 'tolist'):
|
||||
query_embedding = query_embedding.tolist()
|
||||
|
||||
logger.info(f"[search] Encoded query to {len(query_embedding)}-dim vector")
|
||||
|
||||
# Step 2: Search via subprocess
|
||||
console.print("[dim]Searching collections...[/dim]")
|
||||
search_result = _search_vectors_subprocess(
|
||||
query_embedding=query_embedding,
|
||||
branch=branch,
|
||||
memory_type=memory_type,
|
||||
n_results=n_results
|
||||
)
|
||||
|
||||
if not search_result['success']:
|
||||
error_msg = search_result.get('error', 'Unknown error')
|
||||
logger.error(f"[search] Search failed: {error_msg}")
|
||||
console.print(f"[red]x[/red] Search failed: {error_msg}")
|
||||
return False
|
||||
|
||||
results = search_result.get('results', [])
|
||||
collections_searched = search_result.get('collections_searched', 0)
|
||||
total_results = search_result.get('total_results', 0)
|
||||
|
||||
logger.info(f"[search] Found {total_results} results across {collections_searched} collections")
|
||||
|
||||
# Step 3: Display results
|
||||
console.print(f"[green]>[/green] Found {total_results} results in {collections_searched} collections")
|
||||
console.print()
|
||||
|
||||
if not results:
|
||||
console.print("[yellow]No matching memories found[/yellow]")
|
||||
console.print()
|
||||
console.print("[dim]Try:[/dim]")
|
||||
console.print(" * Different search terms")
|
||||
console.print(" * Broader query without filters")
|
||||
console.print(" * Check if memories have been rolled over (drone @memory status)")
|
||||
return True
|
||||
|
||||
# Minimum similarity threshold - filter out irrelevant results
|
||||
MIN_SIMILARITY_THRESHOLD = 0.40 # 40% minimum relevance
|
||||
|
||||
# Filter and process results
|
||||
filtered_results = []
|
||||
for result in results[:n_results]:
|
||||
document = result.get('document', '')
|
||||
distance = result.get('distance', 0)
|
||||
|
||||
# Calculate similarity (ChromaDB L2 distance: 0=identical, ~2=very different)
|
||||
similarity = max(0, 1 - (distance / 2))
|
||||
|
||||
# Skip empty documents and low-relevance results
|
||||
if not document or not document.strip():
|
||||
continue
|
||||
if similarity < MIN_SIMILARITY_THRESHOLD:
|
||||
continue
|
||||
|
||||
result['similarity'] = similarity
|
||||
filtered_results.append(result)
|
||||
|
||||
if not filtered_results:
|
||||
console.print("[yellow]No relevant memories found[/yellow]")
|
||||
console.print()
|
||||
console.print("[dim]The search found some results but none were relevant enough (>40% similarity).[/dim]")
|
||||
console.print("[dim]Try more specific search terms related to your AIPass work.[/dim]")
|
||||
return True
|
||||
|
||||
for i, result in enumerate(filtered_results, 1):
|
||||
collection = result.get('collection', 'unknown')
|
||||
document = result.get('document', '')
|
||||
metadata = result.get('metadata', {})
|
||||
similarity = result.get('similarity', 0)
|
||||
|
||||
# Parse collection name
|
||||
parts = collection.split('_')
|
||||
branch_name = parts[0].upper() if parts else 'UNKNOWN'
|
||||
mem_type = parts[1] if len(parts) > 1 else 'unknown'
|
||||
|
||||
# Build metadata display
|
||||
meta_lines = []
|
||||
if 'timestamp' in metadata:
|
||||
meta_lines.append(f"[dim]Time:[/dim] {metadata['timestamp']}")
|
||||
if 'source' in metadata:
|
||||
meta_lines.append(f"[dim]Source:[/dim] {metadata['source']}")
|
||||
|
||||
meta_text = " | ".join(meta_lines) if meta_lines else ""
|
||||
|
||||
# Create panel for each result
|
||||
panel_title = f"Result {i} - {branch_name} ({mem_type}) - Similarity: {similarity:.2%}"
|
||||
|
||||
panel_content = document
|
||||
if meta_text:
|
||||
panel_content += f"\n\n{meta_text}"
|
||||
|
||||
console.print(Panel(
|
||||
panel_content,
|
||||
title=panel_title,
|
||||
title_align="left",
|
||||
border_style="cyan" if similarity > 0.7 else "blue" if similarity > 0.5 else "dim"
|
||||
))
|
||||
|
||||
console.print()
|
||||
logger.info(f"[search] Displayed {len(filtered_results)} results")
|
||||
|
||||
return True
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# STANDALONE EXECUTION
|
||||
# =============================================================================
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Handle --help before argparse (module standard)
|
||||
if len(sys.argv) < 2 or sys.argv[1] in ('--help', '-h', 'help'):
|
||||
handle_command('help', [])
|
||||
sys.exit(0)
|
||||
|
||||
# Execute command via handle_command
|
||||
command = sys.argv[1]
|
||||
if not handle_command(command, sys.argv[2:]):
|
||||
console.print(f"[red]Unknown command:[/red] {command}")
|
||||
console.print("Run with [cyan]help[/cyan] for available commands")
|
||||
sys.exit(1)
|
||||
Reference in New Issue
Block a user