diff --git a/REQUIREMENTS.md b/REQUIREMENTS.md index 017c2ed..7bf9219 100644 --- a/REQUIREMENTS.md +++ b/REQUIREMENTS.md @@ -2,7 +2,7 @@ > **Note:** This document is automatically generated and verified against the live test suite by `scripts/generate_requirements.py` and `tests/backend/test_requirements_sync.py`. -**Test Verification Baseline:** **1003 Automated Tests** (658 Pytest Backend + 295 Vitest Frontend + 50 Playwright E2E). +**Test Verification Baseline:** **1032 Automated Tests** (679 Pytest Backend + 303 Vitest Frontend + 50 Playwright E2E). --- @@ -460,7 +460,7 @@ classDiagram - `test_split_by_length` - `test_chunk_markdown` -#### `tests/backend/test_chunker_languages.py` (16 tests) +#### `tests/backend/test_chunker_languages.py` (18 tests) - `test_language_detection` - `test_get_tree_sitter_parser_caching_and_fallbacks` - `test_extract_symbols_unsupported_language` @@ -477,6 +477,8 @@ classDiagram - `test_get_file_outline_helper` - `test_markdown_chunking_with_subchunks` - `test_markdown_chunking_with_nested_headings_and_empty` +- `test_call_extraction_constructors_and_generics` +- `test_toplevel_call_source_symbol_preservation` #### `tests/backend/test_db_and_tools.py` (17 tests) - `test_db_path_and_init` @@ -626,13 +628,16 @@ classDiagram - `test_sync_single_git_repo_vector_upsert_failure` - `test_sync_local_paths_vector_upsert_failure` -#### `tests/backend/test_mcp_v2.py` (6 tests) +#### `tests/backend/test_mcp_v2.py` (9 tests) - `test_fastmcp_tools_registered` - `test_fastmcp_resources_and_prompts` - `test_fastmcp_tool_execution` - `test_fastmcp_resource_read` - `test_fastmcp_prompt_get` - `test_fastmcp_streamable_http_transport` +- `test_search_code_with_dense_weight_and_score_breakdown` +- `test_search_code_empty_and_missing_ast_boundaries` +- `test_search_docs_with_dense_weight_and_score_breakdown` #### `tests/backend/test_multi_git_providers.py` (9 tests) - `test_detect_git_provider` @@ -657,7 +662,7 @@ classDiagram - `test_api_get_omni_search_symbols_and_files` - `test_navigator_tree_has_no_empty_folder_root` - _Verify get_navigator_tree sanitizes URI schemes and never produces empty name root folders._ -#### `tests/backend/test_navigator_service.py` (11 tests) +#### `tests/backend/test_navigator_service.py` (12 tests) - `test_db` - `test_navigator_tree_construction` - `test_navigator_tree_all_repos` @@ -668,18 +673,26 @@ classDiagram - `test_symbol_impact_retrieval` - `test_symbol_impact_not_found` - `test_no_outgoing_calls_in_callers` +- `test_class_symbol_impact_aggregation` - `test_real_codebase_symbol_extraction_and_navigation` -#### `tests/backend/test_schemas.py` (2 tests) +#### `tests/backend/test_schemas.py` (4 tests) - `test_code_symbol_creation` - `test_search_request_defaults` +- `test_search_request_dense_weight_boundaries` +- `test_search_request_search_mode_boundaries` -#### `tests/backend/test_search.py` (4 tests) +#### `tests/backend/test_search.py` (9 tests) - `test_execute_hybrid_search_empty_query` - `test_execute_hybrid_search_delegation` +- `test_execute_hybrid_search_ast_enrichment` +- `test_execute_hybrid_search_ast_enrichment_interval_fallback` +- `test_execute_hybrid_search_ast_enrichment_resilience` +- `test_execute_hybrid_search_ast_enrichment_skipped_for_docs` - `test_execute_hybrid_search_exception` - `test_execute_hybrid_search_end_to_end_real` - _Validates REAL hybrid retrieval without mocking get_vector_store or execute_hybrid_search. Asserts both Markdown and PDF docs are matched under doc_type='doc'._ +- `test_api_test_search_endpoint` #### `tests/backend/test_tools.py` (3 tests) - `test_dynamic_catalog_description` @@ -793,7 +806,7 @@ Asserts both Markdown and PDF docs are matched under doc_type='doc'._ - `test_test_connection_active_embedded` - _Verify test_connection succeeds on active embedded Qdrant store without file lock conflict._ - `test_switch_same_embedded_directory` - _Verify switch_vector_store succeeds when switching collection on the same embedded Qdrant directory._ -#### `tests/backend/test_vector_store_qdrant.py` (32 tests) +#### `tests/backend/test_vector_store_qdrant.py` (40 tests) - `TestQdrantVectorStoreInit::test_init_in_memory_or_embedded` - `TestQdrantVectorStoreInit::test_init_remote_success` - `TestQdrantVectorStoreInit::test_init_remote_fallback_to_embedded_on_connection_error` @@ -806,6 +819,10 @@ Asserts both Markdown and PDF docs are matched under doc_type='doc'._ - `TestQdrantVectorStoreOperations::test_search_dense_and_hybrid_rrf` - `TestQdrantVectorStoreOperations::test_search_weighted_score_fusion_range_and_boost` - `TestQdrantVectorStoreOperations::test_search_weighted_score_fusion_alpha_weighting` +- `TestQdrantVectorStoreOperations::test_search_explicit_dense_weight_and_score_decomposition` +- `TestQdrantVectorStoreOperations::test_search_modes_case_insensitivity_and_whitespace` +- `TestQdrantVectorStoreOperations::test_search_modes_semantic_and_lexical` +- `TestQdrantVectorStoreOperations::test_search_mode_lexical_empty_sparse_no_dense_fallback` - `TestQdrantVectorStoreOperations::test_search_dense_fallback_without_sparse` - `TestQdrantVectorStoreOperations::test_delete_by_path` - `TestQdrantVectorStoreOperations::test_delete_by_repo` @@ -822,6 +839,10 @@ Asserts both Markdown and PDF docs are matched under doc_type='doc'._ - `test_search_dense_and_hybrid_rrf` - `test_search_weighted_score_fusion_range_and_boost` - `test_search_weighted_score_fusion_alpha_weighting` +- `test_search_explicit_dense_weight_and_score_decomposition` +- `test_search_modes_case_insensitivity_and_whitespace` +- `test_search_modes_semantic_and_lexical` +- `test_search_mode_lexical_empty_sparse_no_dense_fallback` - `test_search_dense_fallback_without_sparse` - `test_delete_by_path` - `test_delete_by_repo` @@ -1203,12 +1224,13 @@ and leaves the prior indexed state intact without data loss._ - renders vector database health badge in header when vector_db_status is present - renders ChromaDB provider and unhealthy status badge in header -#### `CodeNavigator.test.tsx` (5 tests) +#### `CodeNavigator.test.tsx` (6 tests) - renders toolbar, hero layout, and fetches initial tree data - handles density mode switching and persists to localStorage - loads file outline on file selection and symbol impact on symbol selection - supports caller click-through navigation jumping to caller file and symbol - handles repo switcher change and re-fetches tree +- handles external navigation and permits subsequent repo changes without loop #### `DiagnosticsViewer.test.tsx` (10 tests) - renders log records, badges, and controls @@ -1307,7 +1329,7 @@ and leaves the prior indexed state intact without data loss._ - calls onSelectCallee when a clickable callee is clicked for cross-file navigation - renders loading state when loading is true -#### `NavigatorOmniSearch.test.tsx` (8 tests) +#### `NavigatorOmniSearch.test.tsx` (10 tests) - renders omni-search input with placeholder - fetches matches when user types and displays floating overlay - navigates with keyboard and selects on Enter @@ -1316,6 +1338,8 @@ and leaves the prior indexed state intact without data loss._ - handles fetch error gracefully without crashing - closes dropdown when clicking outside the container - supports ArrowUp navigation within bounds +- renders container prefix and highlights query match in symbol and path +- focuses search input when pressing Ctrl+K #### `NavigatorOutline.test.tsx` (9 tests) - renders empty placeholder when outline is null or empty @@ -1405,13 +1429,18 @@ and leaves the prior indexed state intact without data loss._ - cancels sync by calling /admin/api/repos/{id}/cancel-sync - closes EventSource on unmount -#### `SearchInspector.test.tsx` (6 tests) +#### `SearchInspector.test.tsx` (11 tests) - renders initial prompt and inputs -- performs search and renders matching hit cards +- switches search mode and updates UI controls +- adjusts hybrid split slider and presets +- performs search and renders matching hit cards with score breakdowns and signature +- invokes onOpenInNavigator when Open in Navigator button is clicked - displays empty results message when no hits found - handles search API failure with error display - performs doc search with repo filter and renders documentation hits -- renders search query form and hit card headers with responsive classes +- copies code snippet when Copy button is clicked +- renders hits with missing AST metadata and null scores without error +- handles network error during search gracefully #### `Settings.test.tsx` (29 tests) - renders vector database panel, auto-sync panel, multi-provider token boxes, rate limits, and host vault list diff --git a/app/api/routers/repositories.py b/app/api/routers/repositories.py index cc32a6d..27770f6 100644 --- a/app/api/routers/repositories.py +++ b/app/api/routers/repositories.py @@ -293,19 +293,35 @@ async def api_test_search(payload: SearchRequest): if not query: return JSONResponse(status_code=400, content={"error": "Query required"}) + search_mode = payload.search_mode or "hybrid" hits = search_service.execute_hybrid_search( query_text=query, doc_type=payload.type, repo=payload.repo, - limit=payload.limit or 6 + language=payload.language, + category=payload.category, + tag=payload.tag, + limit=payload.limit or 6, + dense_weight=payload.dense_weight, + search_mode=search_mode ) results = [] for h in hits: results.append({ "score": round(getattr(h, "score", 0.0), 4), + "dense_score": getattr(h, "dense_score", None), + "sparse_score": getattr(h, "sparse_score", None), + "dense_rank": getattr(h, "dense_rank", None), + "sparse_rank": getattr(h, "sparse_rank", None), "payload": getattr(h, "payload", {}) }) - return {"query": query, "type": payload.type, "results": results} + return { + "query": query, + "type": payload.type, + "search_mode": search_mode, + "dense_weight": payload.dense_weight, + "results": results + } except Exception as e: logger.error(f"Error testing search: {e}") return JSONResponse(status_code=500, content={"error": "Failed to execute search test."}) diff --git a/app/mcp/handlers/search_handlers.py b/app/mcp/handlers/search_handlers.py index fbc2cd9..0081cbc 100644 --- a/app/mcp/handlers/search_handlers.py +++ b/app/mcp/handlers/search_handlers.py @@ -23,9 +23,11 @@ async def handle_search_code( query: Annotated[str, Field(description="Natural language question or code concept (e.g. 'JWT token authentication handler').")], repo: Annotated[Optional[str], Field(description="Optional repository name/alias to filter by.")] = None, language: Annotated[Optional[str], Field(description="Optional language filter (e.g. 'python', 'typescript', 'go').")] = None, - limit: Annotated[int, Field(description="Max number of code blocks to return (default 5).")] = 5 + limit: Annotated[int, Field(description="Max number of code blocks to return (default 5).")] = 5, + dense_weight: Annotated[Optional[float], Field(description="Weight between 0.0 (pure lexical BM25) and 1.0 (pure semantic vector). Default is 0.5 balanced.")] = None, + mode: Annotated[Optional[str], Field(description="Search mode: 'hybrid' (default), 'semantic', or 'lexical'.")] = "hybrid" ) -> str: - """Hybrid semantic and BM25 search over code functions, classes, and logic snippets with line numbers and GitHub links.""" + """Hybrid semantic and BM25 search over code functions, classes, and logic snippets with line numbers, AST signatures, and GitHub links.""" query = query.strip() if query else "" if not query: return "Error: search query cannot be empty." @@ -33,20 +35,44 @@ async def handle_search_code( try: from app.services.auth import enforce_tool_permission, Role enforce_tool_permission(Role.VIEWER) - hits = _get_tools_attr("execute_hybrid_search", execute_hybrid_search)(query_text=query, doc_type="code", repo=repo, language=language, limit=limit) + hits = _get_tools_attr("execute_hybrid_search", execute_hybrid_search)( + query_text=query, + doc_type="code", + repo=repo, + language=language, + limit=limit, + dense_weight=dense_weight, + search_mode=mode or "hybrid" + ) if not hits: return f"No matching code snippets found for query: '{query}'." formatted = [] for hit in hits: p = hit.payload - header = f"### [{p.get('repo')}] {p.get('rel_path')} (Lines {p.get('start_line')}-{p.get('end_line')})" - if p.get("symbol"): - header += f" - Symbol: `{p.get('symbol')}`" + header = f"### [{p.get('repo')}] {p.get('rel_path')} (Lines {p.get('start_line')}-{p.get('end_line')})\n" + + sym = p.get("full_symbol") or p.get("symbol") + kind = p.get("kind") + if sym: + kind_str = f" (`{kind}`)" if kind else "" + header += f"- **Symbol**: `{sym}`{kind_str}\n" + + sig = p.get("signature") + if sig: + header += f"- **Signature**: `{sig}`\n" + link_url = p.get("permalink_url") or p.get("github_url") if link_url: - header += f"\nSource Link: {link_url}" - header += f"\nRelevance Score: {hit.score:.4f} ({hit.score * 100:.1f}%)\n" + header += f"- **Source Link**: {link_url}\n" + + score_val = float(hit.score) if isinstance(getattr(hit, "score", None), (int, float)) else 0.0 + score_str = f"{score_val:.4f} ({score_val * 100:.1f}%)" + d_val = getattr(hit, "dense_score", None) + s_val = getattr(hit, "sparse_score", None) + if isinstance(d_val, (int, float)) and isinstance(s_val, (int, float)): + score_str += f" [Semantic: {float(d_val) * 100:.1f}% | Lexical: {float(s_val) * 100:.1f}%]" + header += f"- Relevance Score: {score_str}\n\n" lang = p.get("language", "") block = f"{header}```{lang}\n{p.get('content')}\n```" @@ -63,7 +89,9 @@ async def handle_search_docs( repo: Annotated[Optional[str], Field(description="Optional repository/vault filter.")] = None, category: Annotated[Optional[str], Field(description="Optional category filter.")] = None, tag: Annotated[Optional[str], Field(description="Optional tag filter.")] = None, - limit: Annotated[int, Field(description="Max documents to return (default 5).")] = 5 + limit: Annotated[int, Field(description="Max documents to return (default 5).")] = 5, + dense_weight: Annotated[Optional[float], Field(description="Weight between 0.0 (pure lexical BM25) and 1.0 (pure semantic vector). Default is 0.5 balanced.")] = None, + mode: Annotated[Optional[str], Field(description="Search mode: 'hybrid' (default), 'semantic', or 'lexical'.")] = "hybrid" ) -> str: """Hybrid search across system documentation, markdown notes, architectural decisions, and runbooks.""" query = query.strip() if query else "" @@ -73,7 +101,16 @@ async def handle_search_docs( try: from app.services.auth import enforce_tool_permission, Role enforce_tool_permission(Role.VIEWER) - hits = _get_tools_attr("execute_hybrid_search", execute_hybrid_search)(query_text=query, doc_type="doc", repo=repo, category=category, tag=tag, limit=limit) + hits = _get_tools_attr("execute_hybrid_search", execute_hybrid_search)( + query_text=query, + doc_type="doc", + repo=repo, + category=category, + tag=tag, + limit=limit, + dense_weight=dense_weight, + search_mode=mode or "hybrid" + ) if not hits: return f"No matching documentation found for query: '{query}'." @@ -84,17 +121,24 @@ async def handle_search_docs( header = f"### [{p.get('repo')}] {p.get('rel_path')}" if p.get("heading") and p.get("heading") != "Root": header += f" -> {p.get('heading')}" + header += "\n" if tags_str: - header += f"\nTags: {tags_str}" + header += f"- **Tags**: {tags_str}\n" link_url = p.get("permalink_url") or p.get("github_url") if link_url: - header += f"\nSource Link: {link_url}" - header += f"\nRelevance Score: {hit.score:.4f} ({hit.score * 100:.1f}%)\n" + header += f"- **Source Link**: {link_url}\n" + + score_val = float(hit.score) if isinstance(getattr(hit, "score", None), (int, float)) else 0.0 + score_str = f"{score_val:.4f} ({score_val * 100:.1f}%)" + d_val = getattr(hit, "dense_score", None) + s_val = getattr(hit, "sparse_score", None) + if isinstance(d_val, (int, float)) and isinstance(s_val, (int, float)): + score_str += f" [Semantic: {float(d_val) * 100:.1f}% | Lexical: {float(s_val) * 100:.1f}%]" + header += f"- Relevance Score: {score_str}\n\n" block = f"{header}---\n{p.get('content')}" formatted.append(block) - return "\n\n========================\n\n".join(formatted) except Exception as e: logger.error(f"search_docs failed: {e}") diff --git a/app/models/schemas.py b/app/models/schemas.py index c1fb443..7bb95b7 100644 --- a/app/models/schemas.py +++ b/app/models/schemas.py @@ -1,5 +1,5 @@ from pydantic import BaseModel, Field -from typing import List, Optional, Dict, Any, Tuple +from typing import List, Optional, Dict, Any, Tuple, Literal # Database & Sync Models class RepoConfig(BaseModel): @@ -117,6 +117,8 @@ class SearchRequest(BaseModel): tag: Optional[str] = None limit: int = 5 exact: bool = True + dense_weight: Optional[float] = Field(default=None, ge=0.0, le=1.0) + search_mode: Optional[Literal["hybrid", "semantic", "lexical"]] = "hybrid" class SyncRequest(BaseModel): repo: Optional[str] = None diff --git a/app/services/chunking/symbol_extractor.py b/app/services/chunking/symbol_extractor.py index f1de461..aa7e19e 100644 --- a/app/services/chunking/symbol_extractor.py +++ b/app/services/chunking/symbol_extractor.py @@ -154,12 +154,12 @@ def traverse(node, parent_symbol: Optional[str] = None): elif node.type in CALL_NODE_TYPES: target = extract_target_from_call_node(node, source_bytes) - if target and target not in ("self", "this", "super"): + if target and target not in ("self", "this", "super", "new", "var"): # Clean method prefix if full_symbol has parent active_src = parent_symbol if parent_symbol else file_symbol - # if current active symbol is a method like Foo.bar, extract just bar or Foo.bar - if active_src and "." in active_src: - active_src_name = active_src.split(".")[-1] + # If current active symbol is a method like Foo.bar, extract just bar + if parent_symbol and "." in parent_symbol: + active_src_name = parent_symbol.split(".")[-1] else: active_src_name = active_src relationships.append({ diff --git a/app/services/chunking/text_chunker.py b/app/services/chunking/text_chunker.py index 66cc5d1..ee0ad76 100644 --- a/app/services/chunking/text_chunker.py +++ b/app/services/chunking/text_chunker.py @@ -216,9 +216,18 @@ def extract_node_name(node, source_bytes: bytes) -> Optional[str]: def extract_target_from_call_node(node, source_bytes: bytes) -> Optional[str]: """Extract function, method, constructor, or macro name from a call node.""" - fn_node = node.child_by_field_name("function") or node.child_by_field_name("method") or node.child_by_field_name("expression") + fn_node = ( + node.child_by_field_name("function") + or node.child_by_field_name("method") + or node.child_by_field_name("expression") + or node.child_by_field_name("type") + or node.child_by_field_name("constructor") + ) if not fn_node and len(node.children) > 0: - fn_node = node.children[0] + for child in node.children: + if child.type not in ("new", "(", ")", ";", "{", "}", "[", "]"): + fn_node = child + break if fn_node: call_str = source_bytes[fn_node.start_byte:fn_node.end_byte].decode("utf-8", errors="ignore").strip() @@ -228,22 +237,17 @@ def extract_target_from_call_node(node, source_bytes: bytes) -> Optional[str]: # Clean method call like self.foo() or obj.bar() or math.sqrt() -> get target symbol name if "(" in call_str: call_str = call_str.split("(")[0].strip() - if "." in call_str: - parts = [p for p in call_str.split(".") if p] - if parts: - return parts[-1] - if "::" in call_str: - parts = [p for p in call_str.split("::") if p] - if parts: - return parts[-1] - if "->" in call_str: - parts = [p for p in call_str.split("->") if p] - if parts: - return parts[-1] - if "\\" in call_str: - parts = [p for p in call_str.split("\\") if p] - if parts: - return parts[-1] + # Strip generics e.g. Foo or Bar -> Foo or Bar + call_str = re.sub(r'<.*?>', '', call_str).strip() + call_str = call_str.strip('; >') + for sep in ('.', '::', '->', '\\'): + if sep in call_str: + parts = [p for p in call_str.split(sep) if p] + if parts: + call_str = parts[-1] + call_str = re.sub(r'<.*?>', '', call_str).strip('; >()') + if not call_str or call_str in ("new", "var", "self", "this", "super", "return", "throw", "yield", "void"): + return None return call_str return None diff --git a/app/services/indexing/git_syncer.py b/app/services/indexing/git_syncer.py index a899de2..f83875c 100644 --- a/app/services/indexing/git_syncer.py +++ b/app/services/indexing/git_syncer.py @@ -345,6 +345,8 @@ def sync_single_git_repo(repo_id: int): for s in batch_symbols: if "inserted_id" in s: sym_map[(s["repo"], s["filepath"], s["name"])] = s["inserted_id"] + if s.get("full_symbol"): + sym_map[(s["repo"], s["filepath"], s["full_symbol"])] = s["inserted_id"] rel_tuples = [] for r in batch_relationships: diff --git a/app/services/indexing/local_syncer.py b/app/services/indexing/local_syncer.py index 40a0dbe..32ae5e8 100644 --- a/app/services/indexing/local_syncer.py +++ b/app/services/indexing/local_syncer.py @@ -181,6 +181,8 @@ def sync_local_paths(): for s in all_symbols: if "inserted_id" in s: sym_map[(s["repo"], s["filepath"], s["name"])] = s["inserted_id"] + if s.get("full_symbol"): + sym_map[(s["repo"], s["filepath"], s["full_symbol"])] = s["inserted_id"] rel_tuples = [] for r in all_relationships: diff --git a/app/services/navigator.py b/app/services/navigator.py index b1266d7..9972c00 100644 --- a/app/services/navigator.py +++ b/app/services/navigator.py @@ -218,10 +218,39 @@ def get_symbol_impact(repo: str, symbol_id: int) -> Optional[Dict[str, Any]]: repo_filter_clause = "" if target_repo == "__all__" else " AND r.repo = ?" repo_params = [] if target_repo == "__all__" else [target_repo] + clean_sym_fp = _clean_path(sym["filepath"]) + + # Check for child member symbols in the same file (e.g. methods of a class or members of a struct/interface) + member_rows = conn.execute( + """ + SELECT id, name, full_symbol + FROM ast_symbols + WHERE repo = ? + AND (filepath = ? OR filepath = ? OR filepath LIKE ?) + AND start_line > ? AND end_line <= ? AND id != ? + """, + (sym["repo"], sym["filepath"], f"/{clean_sym_fp}", f"%/{clean_sym_fp}", sym["start_line"], sym["end_line"], sym["id"]) + ).fetchall() + + target_names = {sym["name"]} + if sym["full_symbol"]: + target_names.add(sym["full_symbol"]) + for m in member_rows: + target_names.add(m["name"]) + if m["full_symbol"]: + target_names.add(m["full_symbol"]) + target_names_list = list(target_names) + + exclude_source_names = list(target_names) + exclude_source_ids = [sym["id"]] + [m["id"] for m in member_rows] + # 1. Fetch incoming callers: - # Matches relationships where this symbol is called/used. - # Never includes outgoing calls made by this symbol. - # Resolves source_symbol_id from ast_symbols if missing, and groups multiple calls from same caller. + # Matches relationships where this symbol (or any of its member methods) is called/used. + # Excludes self-calls originating from within this symbol or its member methods. + callers_placeholders = ",".join(["?"] * len(target_names_list)) + ex_name_placeholders = ",".join(["?"] * len(exclude_source_names)) + ex_id_placeholders = ",".join(["?"] * len(exclude_source_ids)) + callers_query = f""" SELECT MIN(r.id) as id, @@ -240,20 +269,25 @@ def get_symbol_impact(repo: str, symbol_id: int) -> Optional[Dict[str, Any]]: FROM ast_symbols ) src_sym ON ( r.source_symbol_id = src_sym.id - OR (r.source_symbol = src_sym.name AND (r.source_filepath = src_sym.filepath OR r.source_filepath LIKE '%/' || src_sym.filepath) AND r.repo = src_sym.repo) + OR ((r.source_symbol = src_sym.name OR r.source_symbol = src_sym.full_symbol) AND (r.source_filepath = src_sym.filepath OR r.source_filepath LIKE '%/' || src_sym.filepath) AND r.repo = src_sym.repo) ) AND src_sym.rn = 1 - WHERE (r.target_symbol = ? OR (r.target_symbol = ? AND ? != '')){repo_filter_clause} + WHERE r.target_symbol IN ({callers_placeholders}) + AND (r.source_symbol_id IS NULL OR r.source_symbol_id NOT IN ({ex_id_placeholders})) + AND r.source_symbol NOT IN ({ex_name_placeholders}) + AND r.relationship_type != 'IMPORTS'{repo_filter_clause} GROUP BY r.source_filepath, r.source_symbol, r.relationship_type ORDER BY r.source_filepath, MIN(r.line_number) ASC """ - full_sym = sym["full_symbol"] or "" - caller_params = [sym["name"], full_sym, full_sym] + repo_params + caller_params = target_names_list + exclude_source_ids + exclude_source_names + repo_params callers = conn.execute(callers_query, caller_params).fetchall() # 2. Fetch outgoing dependencies (callees): - # Matches calls originating from this symbol. - # Resolves target_filepath and target_symbol_id from ast_symbols so links work across usages! - # Groups repeated calls to the same target and sorts resolved codebase targets to the top. + # Matches calls originating from this symbol or any of its member methods. + callee_source_ids = [sym["id"]] + [m["id"] for m in member_rows] + callee_source_names = list(target_names) + src_id_placeholders = ",".join(["?"] * len(callee_source_ids)) + src_name_placeholders = ",".join(["?"] * len(callee_source_names)) + callees_query = f""" SELECT MIN(r.id) as id, @@ -274,17 +308,19 @@ def get_symbol_impact(repo: str, symbol_id: int) -> Optional[Dict[str, Any]]: AND (r.repo = tgt_sym.repo OR ? = '__all__') AND tgt_sym.rn = 1 ) - WHERE (r.source_symbol_id = ? OR (r.source_symbol = ? AND (r.source_filepath = ? OR r.source_filepath LIKE ?))) + WHERE (r.source_symbol_id IN ({src_id_placeholders}) + OR (r.source_symbol IN ({src_name_placeholders}) AND (r.source_filepath = ? OR r.source_filepath = ? OR r.source_filepath LIKE ?))) AND r.relationship_type != 'IMPORTS'{repo_filter_clause} GROUP BY r.target_symbol, r.relationship_type ORDER BY CASE WHEN tgt_sym.filepath IS NOT NULL THEN 0 ELSE 1 END ASC, MIN(r.line_number) ASC """ - callee_params = [target_repo, sym["id"], sym["name"], sym["filepath"], f"%/{_clean_path(sym['filepath'])}"] + repo_params + callee_params = [target_repo] + callee_source_ids + callee_source_names + [sym["filepath"], f"/{clean_sym_fp}", f"%/{clean_sym_fp}"] + repo_params callees = conn.execute(callees_query, callee_params).fetchall() # 3. Fetch imports: + # Imports in codebases are module/file-level. Include both symbol-specific imports (if any) and containing file-level imports. imports_query = f""" SELECT MIN(r.id) as id, @@ -293,12 +329,18 @@ def get_symbol_impact(repo: str, symbol_id: int) -> Optional[Dict[str, Any]]: COUNT(*) as import_count, GROUP_CONCAT(DISTINCT r.line_number) as all_lines FROM ast_relationships r - WHERE (r.source_symbol_id = ? OR (r.source_symbol = ? AND (r.source_filepath = ? OR r.source_filepath LIKE ?))) + WHERE (r.source_symbol_id = ? + OR (r.source_symbol = ? AND (r.source_filepath = ? OR r.source_filepath = ? OR r.source_filepath LIKE ?)) + OR (r.source_filepath = ? OR r.source_filepath = ? OR r.source_filepath LIKE ?)) AND r.relationship_type = 'IMPORTS'{repo_filter_clause} GROUP BY r.target_symbol ORDER BY MIN(r.line_number) ASC """ - import_params = [sym["id"], sym["name"], sym["filepath"], f"%/{_clean_path(sym['filepath'])}"] + repo_params + import_params = [ + sym["id"], + sym["name"], sym["filepath"], f"/{clean_sym_fp}", f"%/{clean_sym_fp}", + sym["filepath"], f"/{clean_sym_fp}", f"%/{clean_sym_fp}" + ] + repo_params imports = conn.execute(imports_query, import_params).fetchall() # 4. Fetch API route mapping: @@ -344,23 +386,28 @@ def get_omni_search(repo: str, query: str, limit: int = 25) -> Dict[str, Any]: sym_sql = f""" SELECT id, repo, filepath, name, full_symbol, kind, start_line, end_line, signature FROM ast_symbols - WHERE (name LIKE ? OR full_symbol LIKE ?){repo_clause} + WHERE (name LIKE ? OR full_symbol LIKE ? OR signature LIKE ?){repo_clause} LIMIT ? """ - sym_params = [like_q, like_q] + repo_params + [limit] + sym_params = [like_q, like_q, like_q] + repo_params + [limit] for row in conn.execute(sym_sql, sym_params).fetchall(): sym_name = row["name"] or "" + full_sym = row["full_symbol"] or sym_name sym_lower = sym_name.lower() + full_lower = full_sym.lower() - if sym_lower == lower_q: + if sym_lower == lower_q or full_lower == lower_q: score = 0.99 label = "99% AST exact match" - elif sym_lower.startswith(lower_q): + elif sym_lower.startswith(lower_q) or full_lower.startswith(lower_q): score = 0.94 label = "94% AST prefix match" - else: + elif lower_q in sym_lower or lower_q in full_lower: score = 0.88 label = "88% AST symbol match" + else: + score = 0.82 + label = "82% Signature match" preview = row["signature"] or f"{row['kind']} {sym_name}" matches.append({ @@ -368,6 +415,7 @@ def get_omni_search(repo: str, query: str, limit: int = 25) -> Dict[str, Any]: "type": "symbol", "symbol_id": row["id"], "name": sym_name, + "full_symbol": full_sym, "kind": row["kind"], "filepath": _clean_path(row["filepath"]), "repo": row["repo"], diff --git a/app/services/search.py b/app/services/search.py index c3cd234..4e08fad 100644 --- a/app/services/search.py +++ b/app/services/search.py @@ -238,6 +238,73 @@ def render_tree_node(node, prefix="", is_last=True): return "\n".join(lines) +def _enrich_code_ast_metadata(results: List[VectorSearchResult]) -> None: + """Enriches code hits with AST symbol details (signature, clean kind, ast_symbol_id).""" + if not results: + return + + code_results = [r for r in results if r.payload.get("doc_type") == "code" or r.payload.get("symbol")] + if not code_results: + return + + try: + with get_db_connection() as conn: + for r in code_results: + p = r.payload + repo = p.get("repo") + rel_path = p.get("rel_path") or p.get("path") + sym_name = p.get("symbol") + start_l = p.get("start_line") + end_l = p.get("end_line") + + clean_fp = rel_path.lstrip("/").replace("\\", "/") if rel_path else "" + + row = None + # Priority 1: match by exact symbol and repo and filepath/lines + if sym_name and repo: + member_name = sym_name.split(".")[-1] + row = conn.execute( + """ + SELECT id, name, full_symbol, kind, signature, start_line, end_line + FROM ast_symbols + WHERE repo = ? + AND (name = ? OR full_symbol = ? OR name = ?) + AND (filepath = ? OR filepath = ? OR filepath LIKE ?) + ORDER BY (CASE WHEN full_symbol = ? THEN 1 WHEN name = ? THEN 2 ELSE 3 END), + abs(start_line - ?) ASC + LIMIT 1 + """, + (repo, sym_name, sym_name, member_name, clean_fp, f"/{clean_fp}", f"%/{clean_fp}", sym_name, member_name, start_l or 0) + ).fetchone() + + # Priority 2: match by file and overlapping line span if not found + if not row and repo and clean_fp and start_l is not None and end_l is not None: + row = conn.execute( + """ + SELECT id, name, full_symbol, kind, signature, start_line, end_line + FROM ast_symbols + WHERE repo = ? + AND (filepath = ? OR filepath = ? OR filepath LIKE ?) + AND start_line <= ? AND end_line >= ? + ORDER BY (end_line - start_line) ASC + LIMIT 1 + """, + (repo, clean_fp, f"/{clean_fp}", f"%/{clean_fp}", end_l, start_l) + ).fetchone() + + if row: + if row["signature"] and not p.get("signature"): + p["signature"] = row["signature"] + if row["full_symbol"] and not p.get("full_symbol"): + p["full_symbol"] = row["full_symbol"] + if row["kind"] and (not p.get("kind") or p.get("kind") == "module"): + p["kind"] = row["kind"] + if "ast_symbol_id" not in p and row["id"]: + p["ast_symbol_id"] = row["id"] + except Exception as e: + logger.debug(f"AST enrichment skipped due to error: {e}") + + def execute_hybrid_search( query_text: str, doc_type: Optional[str] = None, @@ -245,23 +312,29 @@ def execute_hybrid_search( language: Optional[str] = None, category: Optional[str] = None, tag: Optional[str] = None, - limit: int = 5 + limit: int = 5, + dense_weight: Optional[float] = None, + search_mode: str = "hybrid" ) -> List[VectorSearchResult]: - """Executes vector search via the configured active VectorStore backend.""" + """Executes vector search via the configured active VectorStore backend and enriches code hits with AST metadata.""" if not query_text or not query_text.strip(): return [] try: store = get_vector_store() - return store.search( + results = store.search( query_text=query_text.strip(), doc_type=doc_type, repo=repo, language=language, category=category, tag=tag, - limit=limit + limit=limit, + dense_weight=dense_weight, + search_mode=search_mode ) + _enrich_code_ast_metadata(results) + return results except Exception as e: logger.error(f"Error executing vector search: {e}") return [] diff --git a/app/services/vector_store/base.py b/app/services/vector_store/base.py index c106bfd..2df3ad5 100644 --- a/app/services/vector_store/base.py +++ b/app/services/vector_store/base.py @@ -70,6 +70,10 @@ class VectorSearchResult(BaseModel): """Represents a ranked search result item returned from a vector store.""" id: str score: float = 0.0 + dense_score: Optional[float] = None + sparse_score: Optional[float] = None + dense_rank: Optional[int] = None + sparse_rank: Optional[int] = None payload: Dict[str, Any] = Field(default_factory=dict) def to_dict(self) -> Dict[str, Any]: @@ -77,6 +81,10 @@ def to_dict(self) -> Dict[str, Any]: return { "id": self.id, "score": self.score, + "dense_score": self.dense_score, + "sparse_score": self.sparse_score, + "dense_rank": self.dense_rank, + "sparse_rank": self.sparse_rank, "payload": self.payload } @@ -117,7 +125,9 @@ def search( language: Optional[str] = None, category: Optional[str] = None, tag: Optional[str] = None, - limit: int = 5 + limit: int = 5, + dense_weight: Optional[float] = None, + search_mode: str = "hybrid" ) -> List[VectorSearchResult]: """Performs vector search returning ranked results.""" pass diff --git a/app/services/vector_store/chroma_store.py b/app/services/vector_store/chroma_store.py index e4f7335..33d4929 100644 --- a/app/services/vector_store/chroma_store.py +++ b/app/services/vector_store/chroma_store.py @@ -252,7 +252,9 @@ def search( language: Optional[str] = None, category: Optional[str] = None, tag: Optional[str] = None, - limit: int = 5 + limit: int = 5, + dense_weight: Optional[float] = None, + search_mode: str = "hybrid" ) -> List[VectorSearchResult]: """Performs dense vector search returning ranked results.""" if not query_text or not query_text.strip(): @@ -312,7 +314,11 @@ def search( results.append( VectorSearchResult( id=str(doc_id), - score=score, + score=round(score, 4), + dense_score=round(score, 4), + sparse_score=0.0, + dense_rank=i + 1, + sparse_rank=None, payload=payload ) ) diff --git a/app/services/vector_store/pgvector_store.py b/app/services/vector_store/pgvector_store.py index ef30996..1bc7e67 100644 --- a/app/services/vector_store/pgvector_store.py +++ b/app/services/vector_store/pgvector_store.py @@ -272,6 +272,8 @@ def search( category: Optional[str] = None, tag: Optional[str] = None, limit: int = 5, + dense_weight: Optional[float] = None, + search_mode: str = "hybrid", ) -> List[VectorSearchResult]: """ Performs cosine distance vector search returning ranked results. @@ -322,7 +324,7 @@ def search( rows = conn.execute(search_sql, params).mappings().fetchall() results: List[VectorSearchResult] = [] - for row in rows: + for rank_idx, row in enumerate(rows, start=1): payload = row["payload"] if isinstance(payload, str): try: @@ -333,10 +335,15 @@ def search( payload = {} score = float(row["score"]) if row["score"] is not None else 0.0 + clamped_score = max(0.0, min(1.0, score)) results.append( VectorSearchResult( id=str(row["id"]), - score=score, + score=round(clamped_score, 4), + dense_score=round(clamped_score, 4), + sparse_score=0.0, + dense_rank=rank_idx, + sparse_rank=None, payload=payload, ) ) diff --git a/app/services/vector_store/qdrant_store.py b/app/services/vector_store/qdrant_store.py index 9f0e73e..0307625 100644 --- a/app/services/vector_store/qdrant_store.py +++ b/app/services/vector_store/qdrant_store.py @@ -291,9 +291,11 @@ def search( language: Optional[str] = None, category: Optional[str] = None, tag: Optional[str] = None, - limit: int = 5 + limit: int = 5, + dense_weight: Optional[float] = None, + search_mode: str = "hybrid" ) -> List[VectorSearchResult]: - """Performs vector search returning ranked results using Dense + Sparse normalized weighted fusion.""" + """Performs vector search returning ranked results using Dense, Sparse, or configurable Weighted Fusion.""" if not query_text or not query_text.strip(): return [] @@ -302,9 +304,6 @@ def search( logger.warning(f"Collection '{self.collection_name}' does not exist in Qdrant.") return [] - dense_vec = get_dense_embedding(query_text.strip()) - sparse_vec = get_sparse_embedding(query_text.strip()) - must_conditions = [] if doc_type: if doc_type == "doc": @@ -322,73 +321,44 @@ def search( query_filter = qmodels.Filter(must=must_conditions) if must_conditions else None - try: - alpha = float(os.getenv("HYBRID_DENSE_WEIGHT", "0.7")) - except (ValueError, TypeError): - alpha = 0.7 - alpha = max(0.0, min(1.0, alpha)) - - candidate_limit = max(limit * 5, 50) - - if sparse_vec is not None and len(sparse_vec.indices) > 0: - batch_response = self.client.query_batch_points( - collection_name=self.collection_name, - requests=[ - qmodels.QueryRequest( - query=dense_vec, - using="dense", - limit=candidate_limit, - filter=query_filter, - with_payload=True, - ), - qmodels.QueryRequest( - query=sparse_vec, - using="sparse", - limit=candidate_limit, - filter=query_filter, - with_payload=True, - ), - ], - ) - - dense_pts = {str(p.id): p for p in batch_response[0].points} - sparse_pts = {str(p.id): p for p in batch_response[1].points} + mode = (search_mode or "hybrid").lower().strip() + if mode not in ("hybrid", "semantic", "lexical"): + mode = "hybrid" - sparse_scores = [p.score for p in sparse_pts.values() if p.score is not None] - max_sparse = max(sparse_scores, default=1.0) - if max_sparse <= 0.0: - max_sparse = 1.0 - - all_uids = set(dense_pts.keys()).union(sparse_pts.keys()) - scored_results: List[VectorSearchResult] = [] - - for uid in all_uids: - d_score = max(0.0, min(1.0, float(dense_pts[uid].score))) if uid in dense_pts and dense_pts[uid].score is not None else 0.0 - s_score = (float(sparse_pts[uid].score) / max_sparse) if uid in sparse_pts and sparse_pts[uid].score is not None and max_sparse > 0 else 0.0 - s_score = max(0.0, min(1.0, s_score)) + if mode == "semantic": + alpha = 1.0 + elif mode == "lexical": + alpha = 0.0 + elif dense_weight is not None: + try: + alpha = float(dense_weight) + except (ValueError, TypeError): + alpha = 0.5 + alpha = max(0.0, min(1.0, alpha)) + else: + try: + env_alpha = os.getenv("HYBRID_DENSE_WEIGHT") + alpha = float(env_alpha) if env_alpha is not None else 0.5 + except (ValueError, TypeError): + alpha = 0.5 + alpha = max(0.0, min(1.0, alpha)) - if uid in dense_pts and uid in sparse_pts: - final_score = alpha * d_score + (1.0 - alpha) * s_score - elif uid in dense_pts: - final_score = alpha * d_score - else: - final_score = (1.0 - alpha) * s_score + # Compute embeddings lazily based on mode to avoid unnecessary inference + dense_vec = get_dense_embedding(query_text.strip()) if mode in ("hybrid", "semantic") else None + sparse_vec = get_sparse_embedding(query_text.strip()) if mode in ("hybrid", "lexical") else None - final_score = max(0.0, min(1.0, final_score)) - pt = dense_pts.get(uid) or sparse_pts.get(uid) - payload = (pt.payload or {}) if pt else {} + candidate_limit = max(limit * 5, 50) - scored_results.append( - VectorSearchResult( - id=uid, - score=round(final_score, 4), - payload=payload, - ) - ) + # If pure lexical search was explicitly requested but query produced no sparse indices, return empty + if mode == "lexical": + if sparse_vec is None or len(sparse_vec.indices) == 0: + logger.warning(f"Lexical search requested for '{query_text}', but no sparse BM25 tokens were generated.") + return [] - scored_results.sort(key=lambda r: r.score, reverse=True) - return scored_results[:limit] - else: + # Pure semantic search or hybrid fallback if no sparse vector is available + if mode == "semantic" or (mode == "hybrid" and (sparse_vec is None or len(sparse_vec.indices) == 0)): + if dense_vec is None: + dense_vec = get_dense_embedding(query_text.strip()) response = self.client.query_points( collection_name=self.collection_name, query=dense_vec, @@ -398,17 +368,120 @@ def search( with_payload=True, ) results: List[VectorSearchResult] = [] - for pt in response.points: + for rank_idx, pt in enumerate(response.points, start=1): raw_score = float(pt.score) if pt.score is not None else 0.0 clamped_score = max(0.0, min(1.0, raw_score)) results.append( VectorSearchResult( id=str(pt.id), score=round(clamped_score, 4), + dense_score=round(clamped_score, 4), + sparse_score=0.0, + dense_rank=rank_idx, + sparse_rank=None, payload=pt.payload or {}, ) ) return results + + # Pure lexical search + if mode == "lexical": + response = self.client.query_points( + collection_name=self.collection_name, + query=sparse_vec, + using="sparse", + query_filter=query_filter, + limit=limit, + with_payload=True, + ) + sparse_scores = [p.score for p in response.points if p.score is not None] + max_sparse = max(sparse_scores, default=1.0) + if max_sparse <= 0.0: + max_sparse = 1.0 + + results: List[VectorSearchResult] = [] + for rank_idx, pt in enumerate(response.points, start=1): + raw_s = float(pt.score) if pt.score is not None else 0.0 + norm_s = max(0.0, min(1.0, raw_s / max_sparse)) + results.append( + VectorSearchResult( + id=str(pt.id), + score=round(norm_s, 4), + dense_score=0.0, + sparse_score=round(norm_s, 4), + dense_rank=None, + sparse_rank=rank_idx, + payload=pt.payload or {}, + ) + ) + return results + + # Hybrid mode: query both dense and sparse representations + batch_response = self.client.query_batch_points( + collection_name=self.collection_name, + requests=[ + qmodels.QueryRequest( + query=dense_vec, + using="dense", + limit=candidate_limit, + filter=query_filter, + with_payload=True, + ), + qmodels.QueryRequest( + query=sparse_vec, + using="sparse", + limit=candidate_limit, + filter=query_filter, + with_payload=True, + ), + ], + ) + + dense_pts = {str(p.id): p for p in batch_response[0].points} + sparse_pts = {str(p.id): p for p in batch_response[1].points} + + # Map ranks + dense_ranks = {str(p.id): idx for idx, p in enumerate(batch_response[0].points, start=1)} + sparse_ranks = {str(p.id): idx for idx, p in enumerate(batch_response[1].points, start=1)} + + sparse_scores = [p.score for p in sparse_pts.values() if p.score is not None] + max_sparse = max(sparse_scores, default=1.0) + if max_sparse <= 0.0: + max_sparse = 1.0 + + all_uids = set(dense_pts.keys()).union(sparse_pts.keys()) + scored_results: List[VectorSearchResult] = [] + + for uid in all_uids: + d_score = max(0.0, min(1.0, float(dense_pts[uid].score))) if uid in dense_pts and dense_pts[uid].score is not None else 0.0 + s_score = (float(sparse_pts[uid].score) / max_sparse) if uid in sparse_pts and sparse_pts[uid].score is not None and max_sparse > 0 else 0.0 + s_score = max(0.0, min(1.0, s_score)) + + if uid in dense_pts and uid in sparse_pts: + final_score = alpha * d_score + (1.0 - alpha) * s_score + elif uid in dense_pts: + final_score = alpha * d_score + else: + final_score = (1.0 - alpha) * s_score + + final_score = max(0.0, min(1.0, final_score)) + pt = dense_pts.get(uid) or sparse_pts.get(uid) + payload = (pt.payload or {}) if pt else {} + + scored_results.append( + VectorSearchResult( + id=uid, + score=round(final_score, 4), + dense_score=round(d_score, 4), + sparse_score=round(s_score, 4), + dense_rank=dense_ranks.get(uid), + sparse_rank=sparse_ranks.get(uid), + payload=payload, + ) + ) + + scored_results.sort(key=lambda r: r.score, reverse=True) + return scored_results[:limit] except Exception as e: logger.error(f"Error searching Qdrant collection '{self.collection_name}': {e}") return [] diff --git a/docs/superpowers/plans/2026-10-02-search-inspector-hybrid-scoring.md b/docs/superpowers/plans/2026-10-02-search-inspector-hybrid-scoring.md new file mode 100644 index 0000000..cb7991b --- /dev/null +++ b/docs/superpowers/plans/2026-10-02-search-inspector-hybrid-scoring.md @@ -0,0 +1,125 @@ +# Hybrid Search Inspector & Dynamic Scoring Calibration Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Implement dynamic hybrid search scoring calibration, score breakdown transparency (dense vs sparse), AST context enrichment (signatures, kind, symbol), and a redesigned Web Search Inspector with interactive split sliders, repo dropdowns, and 1-click navigation into Code Navigator. + +**Architecture:** Extend `VectorSearchResult` and `VectorStore.search` to track `dense_score` and `sparse_score` and accept `dense_weight` / `search_mode`. In `search_service`, join code hits with SQLite `ast_symbols` to inject function signatures. Expose these controls via MCP tools (`search_code`, `search_docs`), `/admin/api/search/test`, and the React frontend (`SearchInspector.tsx`). + +**Tech Stack:** Python 3.11, FastAPI, Qdrant Client, SQLite, React 19, TypeScript, Vitest, Vite. + +## Global Constraints +- **CRITICAL**: Never push directly to `main`. Work on branch `feat/search-inspector-hybrid-scoring`. +- **CRITICAL**: No version bump or GitHub release. Keep package versions unchanged. +- Preserve full code chunk bodies in search returns for AI agents. +- All tests must pass before opening the PR. + +--- + +### Task 1: Vector Store & Backend Scoring Calibration + +**Files:** +- Modify: `app/services/vector_store/base.py` +- Modify: `app/services/vector_store/qdrant_store.py` +- Modify: `app/services/vector_store/chroma_store.py` +- Modify: `app/services/vector_store/pgvector_store.py` +- Modify: `app/services/search.py` +- Test: `tests/test_vector_store.py` + +**Interfaces:** +- `VectorSearchResult` gains fields: `dense_score: Optional[float] = None`, `sparse_score: Optional[float] = None`. +- `VectorStore.search(query_text, doc_type=None, repo=None, language=None, category=None, tag=None, limit=5, dense_weight=None, search_mode="hybrid") -> List[VectorSearchResult]`. +- `execute_hybrid_search(..., dense_weight=None, search_mode="hybrid") -> List[VectorSearchResult]`. + +- [ ] **Step 1: Write unit tests for dynamic weights and score decomposition** + - Add tests in `tests/test_vector_store.py` verifying `dense_weight=0.0` (pure lexical), `dense_weight=1.0` (pure semantic), `dense_weight=0.5`, and that `dense_score` and `sparse_score` are recorded. +- [ ] **Step 2: Update `VectorSearchResult` in `app/services/vector_store/base.py`** + - Add `dense_score` and `sparse_score` attributes. +- [ ] **Step 3: Update `qdrant_store.py` search logic** + - Handle `dense_weight` parameter with default 0.5. + - Handle `search_mode` (`"hybrid"`, `"semantic"`, `"lexical"`). + - Record normalized `dense_score` and `sparse_score` on each `VectorSearchResult`. +- [ ] **Step 4: Update Chroma and PgVector store backends for signature compatibility** +- [ ] **Step 5: Enrich code search results with AST signatures in `app/services/search.py`** + - When returning code hits, query SQLite `ast_symbols` for extracted `signature`, `full_symbol`, and `ast_symbol_id`. +- [ ] **Step 6: Run tests and verify** + - Run pytest on `tests/test_vector_store.py` and `tests/test_search_service.py`. +- [ ] **Step 7: Commit Task 1 changes** + +--- + +### Task 2: API & MCP Tools Enhancement + +**Files:** +- Modify: `app/models/schemas.py` +- Modify: `app/api/routers/repositories.py` +- Modify: `app/mcp/handlers/search_handlers.py` +- Modify: `app/mcp/tools.py` +- Test: `tests/test_mcp_tools.py` + +**Interfaces:** +- `SearchRequest` schema accepts `dense_weight: Optional[float]` and `search_mode: Optional[str] = "hybrid"`. +- `handle_search_code` accepts `dense_weight` and `mode`, returning markdown with AST signature, kind, score breakdown, and source link. +- `handle_search_docs` accepts `dense_weight` and `mode`, returning markdown with score breakdown and source link. + +- [ ] **Step 1: Update `SearchRequest` schema in `app/models/schemas.py`** +- [ ] **Step 2: Update `/admin/api/search/test` in `app/api/routers/repositories.py`** + - Pass `dense_weight` and `search_mode` to `search_service`. + - Include `dense_score`, `sparse_score`, `search_mode`, and `dense_weight` in API response. +- [ ] **Step 3: Update MCP tool handlers in `app/mcp/handlers/search_handlers.py` & `app/mcp/tools.py`** + - Add `dense_weight` and `mode` parameters. + - Enrich formatted markdown output with AST signature, kind, and score breakdown. +- [ ] **Step 4: Run MCP and router unit tests** + - Run `pytest tests/test_mcp_tools.py` and API tests. +- [ ] **Step 5: Commit Task 2 changes** + +--- + +### Task 3: Redesign Search & Inspector Web UI + +**Files:** +- Modify: `frontend/src/SearchInspector.tsx` +- Modify: `frontend/src/types.ts` +- Modify: `frontend/src/App.tsx` +- Modify: `frontend/src/styles/components.css` +- Test: `frontend/src/tests/SearchInspector.test.tsx` + +**Interfaces:** +- Mode buttons: `Hybrid Fusion`, `Semantic (Dense)`, `Lexical (BM25)`. +- Split slider: range input $[0.0, 1.0]$ with quick preset buttons (`Balanced 50/50`, `Semantic 70/30`, `Keyword 30/70`, `Pure Semantic 100/0`, `Pure Keyword 0/100`). +- Repo dropdown loaded dynamically from `/admin/api/repos` + limit filter (`5`, `10`, `25`). +- Score chips: Hybrid %, Dense %, BM25 %. +- "Open in Navigator" button invoking navigation callback to open file at symbol lines. +- Code blocks with line numbering and copy action. + +- [ ] **Step 1: Update TypeScript types in `frontend/src/types.ts`** + - Add `dense_score`, `sparse_score`, `signature`, `ast_symbol_id` to search hit types. +- [ ] **Step 2: Update `frontend/src/SearchInspector.tsx`** + - Implement mode toggle (`hybrid` | `semantic` | `lexical`). + - Implement dynamic slider with presets. + - Fetch repo list from `/admin/api/repos` for dropdown. + - Add "Open in Navigator" bridging callback. + - Add line numbers and copy code functionality. +- [ ] **Step 3: Wire up navigation bridge in `frontend/src/App.tsx`** + - Pass callback or navigate state from `SearchInspector` to `CodeNavigator`. +- [ ] **Step 4: Add styles in `frontend/src/styles/components.css`** +- [ ] **Step 5: Update Vitest tests in `frontend/src/tests/SearchInspector.test.tsx`** + - Run `npm --prefix frontend test` and ensure all suites pass. +- [ ] **Step 6: Commit Task 3 changes** + +--- + +### Task 4: End-to-End Build, Local Verification & PR Creation + +**Files:** +- Build frontend: `npm --prefix frontend run build` +- Deploy to local container and verify +- Git push and GitHub PR creation + +- [ ] **Step 1: Build frontend bundle and copy to local container** +- [ ] **Step 2: Verify live local container on port 8021 and dev preview on 5173** + - Test `/admin/api/search/test` with various `dense_weight` and `search_mode` values. + - Verify UI renders scores, split slider, repo dropdown, and "Open in Navigator". +- [ ] **Step 3: Push branch `feat/search-inspector-hybrid-scoring` to GitHub** +- [ ] **Step 4: Open Pull Request against `main`** +- [ ] **Step 5: Monitor and confirm CI checks pass** diff --git a/docs/superpowers/specs/2026-10-02-search-inspector-hybrid-scoring-design.md b/docs/superpowers/specs/2026-10-02-search-inspector-hybrid-scoring-design.md new file mode 100644 index 0000000..38ffb75 --- /dev/null +++ b/docs/superpowers/specs/2026-10-02-search-inspector-hybrid-scoring-design.md @@ -0,0 +1,166 @@ +# Design Spec: Hybrid Search Inspector & Dynamic Scoring Calibration + +**Date:** 2026-10-02 +**Branch:** `feat/search-inspector-hybrid-scoring` +**Status:** In Review + +--- + +## 1. Problem Statement & Motivation +ContextCortex combines dense vector embeddings (BGE-Small cosine similarity) and sparse lexical embeddings (FastEmbed / BM25 SPLADE) to perform hybrid search across indexed repositories and documentation. + +Currently, hybrid scoring suffers from several limitations: +1. **Lexical / BM25 Overpowering / Suppression**: In `qdrant_store.py`, fusion is hardcoded as `0.7 * dense + 0.3 * sparse`. An exact keyword match (BM25 score 1.0) is capped at `0.30`, so mediocre semantic matches (0.72 * 0.7 = 0.504) easily overpower exact identifier hits. Furthermore, users cannot customize the dense/sparse weight split. +2. **Opaque Scores**: Both the Web Search Inspector and MCP search tools return a single aggregated score. Users and AI agents cannot see whether a match came from semantic concept matching or lexical keyword hits. +3. **Missing AST Signatures in Search Returns**: While search returns the symbol name and line numbers, it lacks extracted function signatures (parameters, return types) and container hierarchy, forcing AI agents to parse raw code bodies to understand the API surface. +4. **Isolated Search Inspector**: In the admin UI (`SearchInspector.tsx`), results are rendered in raw `
` tags without line numbers, repo selection is an unassisted text box rather than a populated dropdown, and there is no direct action to inspect a matched code snippet inside the Code Navigator.
+
+---
+
+## 2. Goals & Non-Goals
+
+### Goals
+- **Configurable Hybrid Split**: Support a dynamic `dense_weight` / $\alpha \in [0.0, 1.0]$ in backend search queries, MCP tools, and the Web UI, with an interactive slider and quick presets (`Balanced 50/50`, `Semantic 70/30`, `Keyword 30/70`, `Pure Semantic 100/0`, `Pure Keyword 0/100`).
+- **Score Transparency & Diagnostics**: Expose individual component scores (`dense_score` and `sparse_score`) alongside the fused `score` across `VectorSearchResult`, `/admin/api/search/test`, and MCP tool outputs.
+- **Search Mode Selection**: Allow querying in `hybrid` (fused), `semantic` (dense only), or `lexical` (BM25 only) mode.
+- **AST Context Enrichment**: Automatically enrich code search returns with extracted signatures, container paths, clean AST kind badges, and line spans from `ast_symbols`.
+- **Search-to-Navigator Bridging**: Provide a 1-click "Open in Navigator" action on every code search hit in the UI, seamlessly switching to the Code Navigator tab with the file and symbol highlighted.
+- **Search Inspector UI Polish**: Replace plain text repo filter with a dropdown populated from `/admin/api/repos`, add a limit selector (`5`, `10`, `25`), and format code snippets with line numbers and a copy button.
+
+### Non-Goals
+- No changes to embedding models or vector index schemas (BGE-Small and FastEmbed stay as configured).
+- No version bump or GitHub release until explicitly requested.
+- No changes to `main` branch directly.
+
+---
+
+## 3. Architecture & Detailed Design
+
+### 3.1 Vector Store & Scoring Model
+In `app/services/vector_store/base.py`:
+- Update `VectorSearchResult`:
+  ```python
+  class VectorSearchResult(BaseModel):
+      id: str
+      score: float = 0.0
+      dense_score: Optional[float] = None
+      sparse_score: Optional[float] = None
+      dense_rank: Optional[int] = None
+      sparse_rank: Optional[int] = None
+      payload: Dict[str, Any] = Field(default_factory=dict)
+  ```
+
+In `app/services/vector_store/qdrant_store.py`:
+- `search()` accepts `dense_weight: Optional[float] = None` and `search_mode: str = "hybrid"`.
+- Determine effective weight:
+  - If `search_mode == "semantic"`: $\alpha = 1.0$ (query dense only).
+  - If `search_mode == "lexical"`: $\alpha = 0.0$ (query sparse only).
+  - If `dense_weight` is provided: $\alpha = \max(0.0, \min(1.0, \text{dense\_weight}))$.
+  - Otherwise fallback to `HYBRID_DENSE_WEIGHT` environment variable or default `0.5` (balanced).
+- When computing fusion:
+  - Store normalized `d_score` and normalized `s_score` on each `VectorSearchResult`.
+  - Calculate `final_score = alpha * d_score + (1.0 - alpha) * s_score`.
+
+### 3.2 Search Service & AST Context Enrichment
+In `app/services/search.py`:
+- `execute_hybrid_search(query_text, doc_type, repo, language, category, tag, limit, dense_weight=None, search_mode="hybrid")`.
+- When `doc_type == "code"`:
+  - For each result, query SQLite `ast_symbols` for matching `repo`, `filepath` (normalized), and overlapping `start_line` / `end_line`.
+  - Enrich payload with `signature`, `full_symbol`, and `ast_symbol_id` if available.
+
+### 3.3 HTTP API (`/admin/api/search/test`)
+In `app/api/routers/repositories.py`:
+- Update `SearchRequest` schema to accept:
+  - `dense_weight: Optional[float] = Field(default=None, ge=0.0, le=1.0)`
+  - `search_mode: Optional[str] = Field(default="hybrid")`
+- Response payload:
+  ```json
+  {
+    "query": "...",
+    "type": "code",
+    "search_mode": "hybrid",
+    "dense_weight": 0.5,
+    "results": [
+      {
+        "score": 0.7842,
+        "dense_score": 0.8210,
+        "sparse_score": 0.6950,
+        "payload": {
+          "repo": "mcp-router-code",
+          "rel_path": "Components/Providers/ProvidersController.cs",
+          "symbol": "ProvidersController.GetAuthProviders",
+          "signature": "[HttpGet(\"auth\")] public async Task GetAuthProviders()",
+          "kind": "method_declaration",
+          "start_line": 195,
+          "end_line": 212,
+          "ast_symbol_id": 24010,
+          "content": "..."
+        }
+      }
+    ]
+  }
+  ```
+
+### 3.4 MCP Tools (`search_code`, `search_docs`)
+In `app/mcp/handlers/search_handlers.py`:
+- Update `handle_search_code`:
+  - New parameters:
+    - `dense_weight: Annotated[Optional[float], Field(description="Weight between 0.0 (pure lexical BM25) and 1.0 (pure semantic vector). Default is 0.5 balanced.")] = None`
+    - `mode: Annotated[Optional[str], Field(description="Search mode: 'hybrid' (default), 'semantic', or 'lexical'.")] = "hybrid"`
+  - Rich markdown header:
+    ````markdown
+    ### [mcp-router-code] Components/Providers/ProvidersController.cs (Lines 195-212)
+    - **Symbol**: `ProvidersController.GetAuthProviders` (`method`)
+    - **Signature**: `[HttpGet("auth")] public async Task GetAuthProviders()`
+    - **Score**: 78.4% (Semantic: 82.1% | Lexical: 69.5%)
+    - **Source Link**: https://github.com/...#L195-L212
+
+    ```c_sharp
+    [HttpGet("auth")]
+    public async Task GetAuthProviders()
+    ...
+    ```
+    ````
+- Update `handle_search_docs` similarly with `dense_weight` and `mode`.
+
+### 3.5 Web UI (`SearchInspector.tsx`)
+1. **Search Mode Segmented Control**:
+   - `[ Hybrid Fusion ]` | `[ Semantic (Dense) ]` | `[ Lexical (BM25) ]`
+2. **Interactive Hybrid Split Slider (in Hybrid mode)**:
+   - Slider ranging from 0.0 to 1.0 with dynamic label: `Semantic X% / Lexical Y%`.
+   - Preset chips: `Balanced (50/50)`, `Semantic Bias (70/30)`, `Lexical Bias (30/70)`.
+3. **Repo Dropdown & Limit Filter**:
+   - Fetches repos from `/admin/api/repos` to provide a clean ``: `5`, `10`, `25`.
+4. **Enhanced Search Hit Cards**:
+   - Header with: Repo badge, `rel_path`, AST kind badge, Symbol badge, line span, and git host link.
+   - Signature box displaying the extracted syntax signature if present.
+   - Score chips:
+     `Hybrid: 78.4%`
+     `Dense: 82.1%`
+     `BM25: 69.5%`
+   - "Open in Navigator" action button that invokes a callback or updates URL/context to switch tabs to Navigator, selecting the repo, file, and targeting the line span.
+   - Code block with line number column and "Copy Code" button.
+
+---
+
+## 4. Testing & Verification Plan
+
+### Backend Tests
+- `tests/test_vector_store.py`:
+  - Test `search()` with `search_mode="hybrid"`, `"semantic"`, and `"lexical"`.
+  - Test custom `dense_weight=0.0` (pure BM25), `dense_weight=1.0` (pure dense), and `dense_weight=0.5`.
+  - Verify `dense_score` and `sparse_score` are accurately recorded on `VectorSearchResult`.
+- `tests/test_search_service.py` & `tests/test_mcp_tools.py`:
+  - Verify AST signature enrichment joins correctly.
+  - Verify `search_code` and `search_docs` MCP outputs contain the new markdown format and score breakdown.
+- API route test: `POST /admin/api/search/test` verifies handling of `dense_weight` and `search_mode`.
+
+### Frontend Tests
+- `frontend/src/tests/SearchInspector.test.tsx`:
+  - Verify rendering of search mode buttons and hybrid split slider.
+  - Verify preset buttons update the slider value.
+  - Verify search submission passes `dense_weight` and `search_mode`.
+  - Verify score breakdown badges and AST signature rendering.
+  - Verify "Open in Navigator" triggers tab transition.
+- Full Vitest suite passes without regression.
diff --git a/frontend/dist/index.html b/frontend/dist/index.html
index 53204b2..98ac8a0 100644
--- a/frontend/dist/index.html
+++ b/frontend/dist/index.html
@@ -17,7 +17,7 @@
         } catch (e) {}
       })();
     
-    
+    
     
     
     
@@ -33,7 +33,7 @@
     
     
     
-    
+    
   
   
     
diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 9d5b99e..c55f9ec 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -14,6 +14,13 @@ function App() { const [activeTab, setActiveTab] = useState('overview'); const [isMobileNavOpen, setIsMobileNavOpen] = useState(false); const [stats, setStats] = useState(null); + const [navigatorTarget, setNavigatorTarget] = useState<{ + repo: string; + path: string; + symbolId?: number; + startLine?: number; + endLine?: number; + } | null>(null); const loadStats = async () => { try { @@ -127,11 +134,27 @@ function App() {
{activeTab === 'overview' && } - {(activeTab === 'navigator' || activeTab === 'topology') && } + {(activeTab === 'navigator' || activeTab === 'topology') && ( + setNavigatorTarget(null)} + /> + )} {activeTab === 'git-repos' && } {activeTab === 'files-storage' && } {activeTab === 'ingestion-catalog' && } - {activeTab === 'search-inspector' && } + {activeTab === 'search-inspector' && ( + { + setNavigatorTarget({ repo, path, symbolId, startLine, endLine }); + setActiveTab('navigator'); + }} + /> + )} {activeTab === 'settings' && } {activeTab === 'diagnostics' && }
diff --git a/frontend/src/CodeNavigator.tsx b/frontend/src/CodeNavigator.tsx index c3dc183..185bf86 100644 --- a/frontend/src/CodeNavigator.tsx +++ b/frontend/src/CodeNavigator.tsx @@ -1,4 +1,4 @@ -import React, { useState, useEffect, useCallback } from 'react'; +import React, { useState, useEffect, useCallback, useRef } from 'react'; import type { DensityMode, NavigatorTreeNode, @@ -22,6 +22,9 @@ export interface CodeNavigatorProps { initialRepo?: string; initialPath?: string; initialSymbolId?: number; + initialStartLine?: number; + initialEndLine?: number; + onNavigationConsumed?: () => void; } interface HistoryItem { @@ -36,6 +39,9 @@ export const CodeNavigator: React.FC = ({ initialRepo = '__all__', initialPath, initialSymbolId, + initialStartLine, + initialEndLine, + onNavigationConsumed, }) => { // Persistence for density mode const [density, setDensity] = useState(() => { @@ -238,6 +244,40 @@ export const CodeNavigator: React.FC = ({ [fetchImpact] ); + // Track last consumed external navigation to prevent navigation trap + const lastNavRef = useRef(null); + + // Handle external navigation (e.g. from SearchInspector) + useEffect(() => { + if (!initialPath && !initialSymbolId && (!initialRepo || initialRepo === '__all__')) { + return; + } + const navKey = `${initialRepo || ''}:${initialPath || ''}:${initialSymbolId || ''}:${initialStartLine || ''}:${initialEndLine || ''}`; + if (lastNavRef.current === navKey) { + return; + } + lastNavRef.current = navKey; + + const targetRepo = initialRepo && initialRepo !== '__all__' ? initialRepo : selectedRepo; + if (initialRepo && initialRepo !== '__all__' && initialRepo !== selectedRepo) { + setSelectedRepo(initialRepo); + } + if (initialPath) { + setSelectedPath(initialPath); + fetchFileContent(targetRepo, initialPath); + fetchOutline(targetRepo, initialPath); + if (initialStartLine !== undefined) setTargetStartLine(initialStartLine); + if (initialEndLine !== undefined) setTargetEndLine(initialEndLine); + setActiveInspectorTab('reader'); + } + if (initialSymbolId) { + setSelectedSymbolId(initialSymbolId); + fetchImpact(targetRepo, initialSymbolId); + setActiveInspectorTab('intelligence'); + } + onNavigationConsumed?.(); + }, [initialRepo, initialPath, initialSymbolId, initialStartLine, initialEndLine, fetchFileContent, fetchOutline, fetchImpact, onNavigationConsumed]); + // Push item into navigation history const pushHistory = useCallback( (item: HistoryItem) => { diff --git a/frontend/src/SearchInspector.tsx b/frontend/src/SearchInspector.tsx index 43a374d..010a9d4 100644 --- a/frontend/src/SearchInspector.tsx +++ b/frontend/src/SearchInspector.tsx @@ -1,33 +1,84 @@ -import { useState } from 'react'; +import { useState, useEffect } from 'react'; import type { FormEvent } from 'react'; import type { SearchHit } from './types'; import { useToast } from './ToastContext'; +import { getKindBadgeClass } from './components/navigator/NavigatorOutline'; -export default function SearchInspector() { +export interface SearchInspectorProps { + onOpenInNavigator?: (repo: string, path: string, symbolId?: number, startLine?: number, endLine?: number) => void; +} + +export function cleanKind(kind?: string): string { + if (!kind) return ''; + const k = kind.toLowerCase(); + if (k.includes('class')) return 'class'; + if (k.includes('method')) return 'method'; + if (k.includes('func')) return 'func'; + if (k.includes('interface')) return 'interface'; + if (k.includes('struct')) return 'struct'; + if (k.includes('enum')) return 'enum'; + if (k.includes('module')) return 'module'; + return kind.replace(/_declaration|_item/g, ''); +} + +export default function SearchInspector({ onOpenInNavigator }: SearchInspectorProps = {}) { const toast = useToast(); const [query, setQuery] = useState(''); const [type, setType] = useState('code'); const [repo, setRepo] = useState(''); - + const [limit, setLimit] = useState(5); + const [searchMode, setSearchMode] = useState<'hybrid' | 'semantic' | 'lexical'>('hybrid'); + const [denseWeight, setDenseWeight] = useState(0.5); + + const [availableRepos, setAvailableRepos] = useState([]); const [isSearching, setIsSearching] = useState(false); const [results, setResults] = useState(null); const [error, setError] = useState(null); + const [copiedIdx, setCopiedIdx] = useState(null); + + useEffect(() => { + const fetchRepos = async () => { + try { + const res = await fetch('/admin/api/repos'); + if (res.ok) { + const data = await res.json(); + if (Array.isArray(data)) { + setAvailableRepos(data.map((r: any) => typeof r === 'string' ? r : r.name).filter(Boolean)); + } + } + } catch (e) { + console.error('Failed to load repositories list:', e); + } + }; + fetchRepos(); + }, []); + + const runSearchTest = async (e?: FormEvent) => { + if (e) e.preventDefault(); + if (!query.trim()) return; - const runSearchTest = async (e: FormEvent) => { - e.preventDefault(); setIsSearching(true); setError(null); setResults(null); try { + const payload: Record = { + query: query.trim(), + type, + repo: repo.trim() || null, + limit, + search_mode: searchMode, + dense_weight: searchMode === 'hybrid' ? denseWeight : (searchMode === 'semantic' ? 1.0 : 0.0) + }; + const res = await fetch('/admin/api/search/test', { method: 'POST', headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ query: query.trim(), type, repo: repo.trim() || null }) + body: JSON.stringify(payload) }); const data = await res.json(); if (!res.ok) throw new Error(data.error || 'Search failed'); - + setResults(data.results || []); } catch (err: any) { setError(err.message); @@ -37,28 +88,157 @@ export default function SearchInspector() { } }; + const handleCopyCode = async (idx: number, content: string) => { + try { + if (navigator?.clipboard?.writeText) { + await navigator.clipboard.writeText(content); + setCopiedIdx(idx); + setTimeout(() => setCopiedIdx(null), 2000); + } + } catch { + toast.error('Failed to copy to clipboard'); + } + }; + return (

Live Hybrid Search Inspector

-

Test hybrid search results across code and documentation directly from the browser.

+

+ Test hybrid, semantic, and lexical search results with configurable scoring calibration and AST context. +

+ + {/* Search Mode Segmented Control */} +
+
+ + + +
+
+ + {/* Configurable Hybrid Split Slider (only shown in hybrid mode) */} + {searchMode === 'hybrid' && ( +
+
+ Hybrid Scoring Split: + + Semantic: {Math.round(denseWeight * 100)}% / Lexical: {Math.round((1 - denseWeight) * 100)}% + +
+ setDenseWeight(parseFloat(e.target.value))} + className="hybrid-range-input" + aria-label="Hybrid dense weight slider" + /> +
+ + + + + +
+
+ )} -
+
- setQuery(e.target.value)} /> + setQuery(e.target.value)} + />
- setType(e.target.value)}> + +
- - setRepo(e.target.value)} /> + + setRepo(e.target.value)} + /> + + + {availableRepos.map(r => ( + + ))} + +
+
+ +
+ )} + +
+ + {/* AST Signature snippet if present */} + {p.signature && ( +
+ Signature: + {p.signature} +
+ )} +
{p.content}
); }) )} - diff --git a/frontend/src/components/navigator/NavigatorCodeViewer.tsx b/frontend/src/components/navigator/NavigatorCodeViewer.tsx index 9fd6ba4..d55f924 100644 --- a/frontend/src/components/navigator/NavigatorCodeViewer.tsx +++ b/frontend/src/components/navigator/NavigatorCodeViewer.tsx @@ -413,7 +413,7 @@ export const NavigatorCodeViewer: React.FC = ({
diff --git a/frontend/src/components/navigator/NavigatorOmniSearch.tsx b/frontend/src/components/navigator/NavigatorOmniSearch.tsx index 009b5ea..117e04d 100644 --- a/frontend/src/components/navigator/NavigatorOmniSearch.tsx +++ b/frontend/src/components/navigator/NavigatorOmniSearch.tsx @@ -24,6 +24,59 @@ export function getMatchBadgeClass(type: OmniSearchMatchKind): string { } } +export function HighlightMatch({ text, query }: { text: string; query: string }) { + if (!query || !query.trim() || !text) { + return <>{text}; + } + const trimmed = query.trim(); + const escaped = trimmed.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); + const regex = new RegExp(`(${escaped})`, 'gi'); + const parts = text.split(regex); + if (parts.length <= 1) { + return <>{text}; + } + return ( + <> + {text} + + + ); +} + +export function formatKind(kind?: string): string { + if (!kind) return ''; + const k = kind.toLowerCase(); + if (k.includes('method')) return 'method'; + if (k.includes('func')) return 'func'; + if (k.includes('class')) return 'class'; + if (k.includes('interface')) return 'interface'; + if (k.includes('struct')) return 'struct'; + if (k.includes('type')) return 'type'; + if (k.includes('enum')) return 'enum'; + if (k.includes('property') || k.includes('field')) return 'prop'; + if (k === 'route') return 'route'; + if (k === 'file') return 'file'; + if (k === 'code') return 'code'; + return k.replace(/_declaration|_definition|_specifier/g, ''); +} + +export function formatDisplayPath(path: string): string { + if (path.includes('://')) { + return path.split('://')[1]; + } + return path; +} + export const NavigatorOmniSearch: React.FC = ({ repo, onSelectResult, @@ -39,6 +92,19 @@ export const NavigatorOmniSearch: React.FC = ({ const inputRef = useRef(null); const abortControllerRef = useRef(null); + // Global Ctrl+K / Cmd+K listener + useEffect(() => { + const handleGlobalKey = (e: KeyboardEvent) => { + if ((e.ctrlKey || e.metaKey) && e.key.toLowerCase() === 'k') { + e.preventDefault(); + inputRef.current?.focus(); + inputRef.current?.select(); + } + }; + window.addEventListener('keydown', handleGlobalKey); + return () => window.removeEventListener('keydown', handleGlobalKey); + }, []); + // Debounced search useEffect(() => { const trimmed = query.trim(); @@ -172,6 +238,12 @@ export const NavigatorOmniSearch: React.FC = ({ aria-expanded={isOpen} /> + {!query && ( + + )} + {loading && ( = ({ )}
- {/* Floating Absolute Overlay */} + {/* Floating Centered Overlay */} {isOpen && (
= ({ const isSelected = index === activeIndex; const badgeClass = getMatchBadgeClass(item.type); + // Determine container prefix if full_symbol has parent + let containerPrefix = ''; + let symbolName = item.name; + if (item.full_symbol && item.full_symbol.includes('.') && item.type === 'symbol') { + const lastDot = item.full_symbol.lastIndexOf('.'); + containerPrefix = item.full_symbol.substring(0, lastDot + 1); + symbolName = item.full_symbol.substring(lastDot + 1); + } + return (
= ({ onMouseEnter={() => setActiveIndex(index)} role="option" aria-selected={isSelected} + title={item.full_symbol || item.name} >
{item.type.toUpperCase()} - {item.name} + {containerPrefix && ( + + + + )} + + + {item.kind && item.kind !== item.type && ( - ({item.kind}) + + {formatKind(item.kind)} + )}
- {item.score_label} + + {item.score_label} +
- - {item.filepath} - {item.start_line > 0 && `:${item.start_line}`} + + + {item.start_line > 0 && ( + :{item.start_line} + )} {item.preview && item.preview !== item.name && ( - {item.preview} + <> + + + + + )}
diff --git a/frontend/src/components/navigator/types.ts b/frontend/src/components/navigator/types.ts index 02c1e9a..387acb2 100644 --- a/frontend/src/components/navigator/types.ts +++ b/frontend/src/components/navigator/types.ts @@ -126,6 +126,7 @@ export interface OmniSearchResultItem { id: string; type: OmniSearchMatchKind; name: string; + full_symbol?: string; kind?: string; filepath: string; repo?: string; diff --git a/frontend/src/styles/components.css b/frontend/src/styles/components.css index e735f5b..591e8dc 100644 --- a/frontend/src/styles/components.css +++ b/frontend/src/styles/components.css @@ -391,6 +391,133 @@ input:focus, select:focus, textarea:focus { white-space: pre-wrap; word-break: break-all; color: var(--text); + margin-top: 8px; +} + +/* Search Inspector Controls */ +.search-mode-segmented { + display: inline-flex; + background: rgba(255, 255, 255, 0.05); + border: 1px solid var(--border-card); + border-radius: 8px; + padding: 3px; + gap: 4px; + margin-bottom: 14px; +} + +.search-mode-btn { + background: transparent; + border: none; + color: var(--text-muted); + font-size: 0.85rem; + padding: 6px 14px; + border-radius: 6px; + cursor: pointer; + display: flex; + align-items: center; + gap: 6px; + transition: all 0.15s ease-in-out; +} + +.search-mode-btn:hover { + color: var(--text); + background: rgba(255, 255, 255, 0.06); +} + +.search-mode-btn.active { + background: var(--primary); + color: #ffffff; + font-weight: 600; + box-shadow: 0 2px 6px rgba(0, 0, 0, 0.2); +} + +.hybrid-slider-container { + background: rgba(255, 255, 255, 0.02); + border: 1px solid var(--border-card); + border-radius: 8px; + padding: 12px 16px; + margin-bottom: 16px; +} + +.hybrid-slider-header { + display: flex; + justify-content: space-between; + align-items: center; + font-size: 0.85rem; + margin-bottom: 8px; +} + +.hybrid-split-value { + font-family: var(--font-family-mono); + color: var(--accent); + font-weight: 600; +} + +.hybrid-range-input { + width: 100%; + accent-color: var(--primary); + cursor: pointer; +} + +.hybrid-presets { + display: flex; + flex-wrap: wrap; + gap: 6px; + margin-top: 10px; +} + +.btn-preset { + background: rgba(255, 255, 255, 0.04); + border: 1px solid var(--border-card); + color: var(--text-muted); + border-radius: 4px; + padding: 3px 8px; + font-size: 0.75rem; + cursor: pointer; + transition: all 0.15s; +} + +.btn-preset:hover { + color: var(--text); + border-color: var(--primary); + background: rgba(var(--primary-rgb), 0.1); +} + +.btn-preset.active { + background: rgba(var(--primary-rgb), 0.25); + border-color: var(--primary); + color: var(--primary); + font-weight: 600; +} + +.search-hit-signature { + background: rgba(255, 255, 255, 0.03); + border-left: 3px solid var(--primary); + padding: 6px 10px; + border-radius: 0 4px 4px 0; + font-family: var(--font-family-mono); + font-size: 0.82rem; + color: var(--accent); + margin-bottom: 6px; + overflow-x: auto; +} + +.search-hit-actions { + display: flex; + align-items: center; + gap: 8px; +} + +.badge-info { + background: rgba(56, 189, 248, 0.15); + color: #38bdf8; + border: 1px solid rgba(56, 189, 248, 0.3); +} + +.badge-warning { + background: rgba(251, 191, 36, 0.15); + color: #fbbf24; + border: 1px solid rgba(251, 191, 36, 0.3); } /* Footer */ diff --git a/frontend/src/styles/navigator.css b/frontend/src/styles/navigator.css index 2c0ec8d..333a368 100644 --- a/frontend/src/styles/navigator.css +++ b/frontend/src/styles/navigator.css @@ -393,7 +393,7 @@ .nav-toolbar-center { flex: 1; - max-width: 400px; + max-width: 580px; min-width: 200px; } @@ -531,30 +531,55 @@ to { transform: rotate(360deg); } } -/* Floating Dropdown - Absolute overlay with high z-index */ +.nav-omni-kbd { + font-size: 0.65rem; + font-family: var(--font-mono, monospace); + background: rgba(255, 255, 255, 0.08); + color: var(--text-muted, #94a3b8); + padding: 2px 6px; + border-radius: 4px; + border: 1px solid rgba(255, 255, 255, 0.12); + letter-spacing: 0.05em; + flex-shrink: 0; + margin-left: 6px; + user-select: none; +} + +/* Floating Dropdown - Spacious Centered Command Palette Overlay */ .nav-omni-dropdown { position: absolute; - top: calc(100% + 4px); - left: 0; - right: 0; - z-index: 50; + top: calc(100% + 6px); + left: 50%; + transform: translateX(-50%); + width: max(100%, 640px); + max-width: min(92vw, 760px); + z-index: 100; background: var(--bg-card, #0d2c2f); border: 1px solid var(--border-card, #15474d); border-radius: 8px; - box-shadow: 0 15px 30px rgba(0, 0, 0, 0.6), 0 5px 15px rgba(0, 0, 0, 0.4); - max-height: 420px; + box-shadow: 0 20px 45px rgba(0, 0, 0, 0.75), 0 8px 20px rgba(0, 0, 0, 0.5); + max-height: 480px; display: flex; flex-direction: column; overflow: hidden; - backdrop-filter: blur(8px); + backdrop-filter: blur(12px); +} + +@media (max-width: 768px) { + .nav-omni-dropdown { + left: 0; + transform: none; + width: 100%; + max-width: 100%; + } } .nav-omni-dropdown-header { display: flex; align-items: center; justify-content: space-between; - padding: 8px 12px; - background: rgba(0, 0, 0, 0.3); + padding: 8px 14px; + background: rgba(0, 0, 0, 0.35); border-bottom: 1px solid var(--border-card, #15474d); font-size: 0.72rem; font-weight: 600; @@ -570,54 +595,56 @@ } .nav-omni-empty { - padding: 16px; + padding: 20px 16px; text-align: center; - font-size: 0.8rem; + font-size: 0.82rem; color: var(--text-muted, #94a3b8); } .nav-omni-list { overflow-y: auto; - padding: 4px; + padding: 6px; } .nav-omni-item { display: flex; flex-direction: column; - padding: 8px 10px; + padding: 9px 12px; border-radius: 6px; cursor: pointer; - transition: background-color 0.12s ease; - margin-bottom: 2px; + transition: all 0.15s ease; + margin-bottom: 3px; border: 1px solid transparent; + gap: 5px; } .nav-omni-item:hover, .nav-omni-item.active { - background: rgba(8, 145, 178, 0.15); - border-color: rgba(8, 145, 178, 0.35); + background: rgba(14, 165, 233, 0.12); + border-color: rgba(56, 189, 248, 0.35); } .nav-omni-item-top { display: flex; align-items: center; justify-content: space-between; - margin-bottom: 4px; - gap: 8px; + gap: 12px; } .nav-omni-item-title-group { display: flex; align-items: center; - gap: 6px; + gap: 8px; + flex: 1; + min-width: 0; overflow: hidden; } .nav-omni-badge { font-size: 0.65rem; font-weight: 700; - padding: 1px 5px; - border-radius: 3px; + padding: 2px 6px; + border-radius: 4px; text-transform: uppercase; letter-spacing: 0.04em; font-family: var(--font-mono, monospace); @@ -659,53 +686,95 @@ color: #cbd5e1; } -.nav-omni-item-name { +.nav-omni-item-container { font-size: 0.82rem; + color: #94a3b8; + font-family: var(--font-mono, monospace); + flex-shrink: 0; + white-space: nowrap; +} + +.nav-omni-item-name { + font-size: 0.85rem; font-weight: 600; - color: var(--text, #f8fafc); + color: #f8fafc; font-family: var(--font-mono, monospace); white-space: nowrap; overflow: hidden; text-overflow: ellipsis; + min-width: 0; } .nav-omni-item-kind { - font-size: 0.72rem; - color: var(--text-muted, #94a3b8); + font-size: 0.65rem; + font-weight: 500; + color: #94a3b8; + background: rgba(255, 255, 255, 0.06); + border: 1px solid rgba(255, 255, 255, 0.09); + padding: 1px 5px; + border-radius: 4px; + text-transform: lowercase; + flex-shrink: 0; } .nav-omni-score-pill { - font-size: 0.68rem; + font-size: 0.65rem; font-weight: 600; - padding: 1px 7px; - border-radius: 10px; - background: rgba(16, 185, 129, 0.15); + padding: 2px 7px; + border-radius: 6px; + background: rgba(16, 185, 129, 0.12); color: #34d399; - border: 1px solid rgba(52, 211, 153, 0.3); + border: 1px solid rgba(52, 211, 153, 0.25); white-space: nowrap; flex-shrink: 0; } +.nav-omni-match { + background: rgba(56, 189, 248, 0.25); + color: #38bdf8; + font-weight: 700; + border-radius: 2px; + padding: 0 1px; +} + .nav-omni-item-bottom { display: flex; align-items: center; gap: 8px; - font-size: 0.72rem; - color: var(--text-muted, #94a3b8); + font-size: 0.73rem; + color: #94a3b8; font-family: var(--font-mono, monospace); overflow: hidden; + line-height: 1.3; } .nav-omni-item-path { - color: var(--primary, #38bdf8); + color: #38bdf8; + flex-shrink: 0; + max-width: 48%; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.nav-omni-line-badge { + color: #64748b; + font-weight: 600; +} + +.nav-omni-sep { + color: #475569; + font-size: 0.65rem; flex-shrink: 0; } .nav-omni-item-preview { - color: var(--text-muted, #94a3b8); + color: #cbd5e1; white-space: nowrap; overflow: hidden; text-overflow: ellipsis; + flex: 1; + min-width: 0; } /* ========================================================================== @@ -840,6 +909,7 @@ padding: 0 4px; max-width: 100%; box-sizing: border-box; + border-left: 3px solid transparent; } .nav-code-line-row:hover { @@ -870,18 +940,28 @@ min-width: 0; } -/* Target line highlight */ +/* Enclosing symbol scope range highlight */ .nav-code-line-target { - background: rgba(14, 165, 233, 0.18) !important; - border-left: 3px solid #38bdf8 !important; + background: rgba(14, 165, 233, 0.04) !important; + border-left: 3px solid rgba(56, 189, 248, 0.22) !important; } .nav-code-line-target .nav-code-line-number { + color: #64748b; +} + +/* Primary declaration / target line highlight */ +.nav-code-line-primary { + background: rgba(14, 165, 233, 0.16) !important; + border-left: 3px solid #38bdf8 !important; +} + +.nav-code-line-primary .nav-code-line-number { color: #38bdf8 !important; font-weight: 700; } -.nav-code-line-target .nav-code-line-content { +.nav-code-line-primary .nav-code-line-content { font-weight: 500; } diff --git a/frontend/src/tests/CodeNavigator.test.tsx b/frontend/src/tests/CodeNavigator.test.tsx index 494fa62..827b590 100644 --- a/frontend/src/tests/CodeNavigator.test.tsx +++ b/frontend/src/tests/CodeNavigator.test.tsx @@ -311,4 +311,33 @@ describe('CodeNavigator Container', () => { ); }); }); + + it('handles external navigation and permits subsequent repo changes without loop', async () => { + const onConsumed = vi.fn(); + render( + + ); + + await waitFor(() => { + expect(onConsumed).toHaveBeenCalled(); + expect(screen.getByText('main.py')).toBeInTheDocument(); + }); + + // Now user switches repo to repo-web + const repoSelect = screen.getByRole('combobox', { name: /repository/i }); + fireEvent.change(repoSelect, { target: { value: 'repo-web' } }); + + await waitFor(() => { + expect(repoSelect).toHaveValue('repo-web'); + expect((globalThis as any).fetch).toHaveBeenCalledWith( + expect.stringContaining('/admin/api/navigator/tree?repo=repo-web') + ); + }); + }); }); diff --git a/frontend/src/tests/NavigatorOmniSearch.test.tsx b/frontend/src/tests/NavigatorOmniSearch.test.tsx index 992bebe..36c3251 100644 --- a/frontend/src/tests/NavigatorOmniSearch.test.tsx +++ b/frontend/src/tests/NavigatorOmniSearch.test.tsx @@ -221,4 +221,62 @@ describe('NavigatorOmniSearch Component', () => { expect(onSelect).toHaveBeenCalledWith(mockMatches[0]); }); + + it('renders container prefix and highlights query match in symbol and path', async () => { + const symbolWithContainer: OmniSearchResultItem = { + id: 'sym_nested', + type: 'symbol', + symbol_id: 15, + name: 'GetAllProviders', + full_symbol: 'ProvidersController.GetAllProviders', + kind: 'method_declaration', + filepath: 'mcp-router-code://Components/Providers/ProvidersController.cs', + repo: 'test-repo', + start_line: 25, + end_line: 60, + score: 0.95, + score_label: '95% Prefix match', + preview: 'public async Task GetAllProviders()', + }; + + vi.spyOn(globalThis, 'fetch').mockResolvedValue({ + ok: true, + json: async () => ({ + query: 'Providers', + repo: 'test-repo', + total_matches: 1, + matches: [symbolWithContainer], + }), + } as Response); + + render(); + + const input = screen.getByPlaceholderText(/search files, symbols, routes, or code text/i); + fireEvent.change(input, { target: { value: 'Providers' } }); + + await screen.findByText('GetAllProviders'); + + // Container prefix should be displayed + expect(screen.getByText('ProvidersController.')).toBeInTheDocument(); + + // Kind should be cleaned from method_declaration to method + expect(screen.getByText('method')).toBeInTheDocument(); + + // Path should be cleaned of protocol scheme + expect(screen.getByText('Components/Providers/ProvidersController.cs')).toBeInTheDocument(); + + // Query match marks should exist + const marks = document.querySelectorAll('mark.nav-omni-match'); + expect(marks.length).toBeGreaterThan(0); + }); + + it('focuses search input when pressing Ctrl+K', () => { + render(); + const input = screen.getByPlaceholderText(/search files, symbols, routes, or code text/i); + + expect(document.activeElement).not.toBe(input); + fireEvent.keyDown(window, { key: 'k', ctrlKey: true }); + expect(document.activeElement).toBe(input); + }); }); + diff --git a/frontend/src/tests/SearchInspector.test.tsx b/frontend/src/tests/SearchInspector.test.tsx index 6c0d98c..2cbf260 100644 --- a/frontend/src/tests/SearchInspector.test.tsx +++ b/frontend/src/tests/SearchInspector.test.tsx @@ -7,10 +7,17 @@ import type { SearchHit } from '../types'; const mockHits: SearchHit[] = [ { score: 0.825, + dense_score: 0.880, + sparse_score: 0.720, + dense_rank: 1, + sparse_rank: 2, payload: { repo: 'knowledge-rag-mcp', rel_path: 'app/services/indexer.py', symbol: 'IndexerService.sync', + signature: 'async def sync(self) -> bool:', + kind: 'method_declaration', + ast_symbol_id: 101, start_line: 45, end_line: 80, github_url: 'https://github.com/example/knowledge-rag-mcp/blob/main/app/services/indexer.py#L45-L80', @@ -22,6 +29,12 @@ const mockHits: SearchHit[] = [ describe('SearchInspector Component', () => { beforeEach(() => { vi.clearAllMocks(); + (globalThis as any).fetch = vi.fn().mockImplementation((url: string) => { + if (typeof url === 'string' && url.includes('/admin/api/repos')) { + return Promise.resolve({ ok: true, json: async () => [{ name: 'repo-1' }, { name: 'repo-2' }] }); + } + return Promise.resolve({ ok: true, json: async () => ({ results: [] }) }); + }); }); it('renders initial prompt and inputs', () => { @@ -33,9 +46,56 @@ describe('SearchInspector Component', () => { expect(screen.getByText('Live Hybrid Search Inspector')).toBeInTheDocument(); expect(screen.getByText('Enter a query above to test hybrid retrieval.')).toBeInTheDocument(); + expect(screen.getByTestId('mode-hybrid-btn')).toBeInTheDocument(); + expect(screen.getByTestId('mode-semantic-btn')).toBeInTheDocument(); + expect(screen.getByTestId('mode-lexical-btn')).toBeInTheDocument(); + expect(screen.getByLabelText('Hybrid dense weight slider')).toBeInTheDocument(); }); - it('performs search and renders matching hit cards', async () => { + it('switches search mode and updates UI controls', () => { + render( + + + + ); + + // Click Semantic + fireEvent.click(screen.getByTestId('mode-semantic-btn')); + expect(screen.queryByLabelText('Hybrid dense weight slider')).not.toBeInTheDocument(); + + // Click Lexical + fireEvent.click(screen.getByTestId('mode-lexical-btn')); + expect(screen.queryByLabelText('Hybrid dense weight slider')).not.toBeInTheDocument(); + + // Click Hybrid again + fireEvent.click(screen.getByTestId('mode-hybrid-btn')); + expect(screen.getByLabelText('Hybrid dense weight slider')).toBeInTheDocument(); + }); + + it('adjusts hybrid split slider and presets', () => { + render( + + + + ); + + const slider = screen.getByLabelText('Hybrid dense weight slider') as HTMLInputElement; + expect(slider.value).toBe('0.5'); + + // Click Semantic Bias preset + fireEvent.click(screen.getByRole('button', { name: /Semantic Bias/i })); + expect(slider.value).toBe('0.7'); + + // Click Keyword Bias preset + fireEvent.click(screen.getByRole('button', { name: /Keyword Bias/i })); + expect(slider.value).toBe('0.3'); + + // Click Balanced preset + fireEvent.click(screen.getByRole('button', { name: /Balanced/i })); + expect(slider.value).toBe('0.5'); + }); + + it('performs search and renders matching hit cards with score breakdowns and signature', async () => { (globalThis as any).fetch = vi.fn().mockResolvedValue({ ok: true, json: async () => ({ results: mockHits }) @@ -58,18 +118,57 @@ describe('SearchInspector Component', () => { '/admin/api/search/test', expect.objectContaining({ method: 'POST', - body: JSON.stringify({ query: 'IndexerService', type: 'code', repo: null }) + body: JSON.stringify({ + query: 'IndexerService', + type: 'code', + repo: null, + limit: 5, + search_mode: 'hybrid', + dense_weight: 0.5 + }) }) ); expect(screen.getByText('knowledge-rag-mcp')).toBeInTheDocument(); expect(screen.getByText('app/services/indexer.py')).toBeInTheDocument(); expect(screen.getByText('IndexerService.sync')).toBeInTheDocument(); expect(screen.getByText('Score: 82.5% (0.8250)')).toBeInTheDocument(); - expect(screen.getByText(/async def sync/)).toBeInTheDocument(); + expect(screen.getByText('Semantic: 88.0%')).toBeInTheDocument(); + expect(screen.getByText('Lexical: 72.0%')).toBeInTheDocument(); + expect(screen.getByText(/async def sync\(self\) -> bool:/)).toBeInTheDocument(); expect(screen.getByText('View on GitHub')).toBeInTheDocument(); }); }); + it('invokes onOpenInNavigator when Open in Navigator button is clicked', async () => { + const handleOpenInNavigator = vi.fn(); + (globalThis as any).fetch = vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ results: mockHits }) + }); + + render( + + + + ); + + fireEvent.change(screen.getByPlaceholderText(/e.g. JWT token/i), { target: { value: 'IndexerService' } }); + fireEvent.click(screen.getByRole('button', { name: /Search/i })); + + await waitFor(() => { + expect(screen.getByRole('button', { name: /Open in Navigator/i })).toBeInTheDocument(); + }); + + fireEvent.click(screen.getByRole('button', { name: /Open in Navigator/i })); + expect(handleOpenInNavigator).toHaveBeenCalledWith( + 'knowledge-rag-mcp', + 'app/services/indexer.py', + 101, + 45, + 80 + ); + }); + it('displays empty results message when no hits found', async () => { (globalThis as any).fetch = vi.fn().mockResolvedValue({ ok: true, @@ -140,7 +239,7 @@ describe('SearchInspector Component', () => { ); fireEvent.change(screen.getByPlaceholderText(/e.g. JWT token/i), { target: { value: 'system design' } }); - fireEvent.change(screen.getByRole('combobox'), { target: { value: 'doc' } }); + fireEvent.change(screen.getByRole('combobox', { name: /Target Type/i }), { target: { value: 'doc' } }); fireEvent.change(screen.getByPlaceholderText(/All Repos/i), { target: { value: 'docs-vault' } }); fireEvent.click(screen.getByRole('button', { name: /Search/i })); @@ -150,7 +249,14 @@ describe('SearchInspector Component', () => { '/admin/api/search/test', expect.objectContaining({ method: 'POST', - body: JSON.stringify({ query: 'system design', type: 'doc', repo: 'docs-vault' }) + body: JSON.stringify({ + query: 'system design', + type: 'doc', + repo: 'docs-vault', + limit: 5, + search_mode: 'hybrid', + dense_weight: 0.5 + }) }) ); expect(screen.getByText('docs-vault')).toBeInTheDocument(); @@ -159,32 +265,97 @@ describe('SearchInspector Component', () => { }); }); - it('renders search query form and hit card headers with responsive classes', async () => { + it('copies code snippet when Copy button is clicked', async () => { + const writeTextMock = vi.fn().mockResolvedValue(undefined); + Object.assign(navigator, { + clipboard: { + writeText: writeTextMock + } + }); + (globalThis as any).fetch = vi.fn().mockResolvedValue({ ok: true, json: async () => ({ results: mockHits }) }); - const { container } = render( + render( ); - const formRow = container.querySelector('.form-row'); - expect(formRow).toBeInTheDocument(); + fireEvent.change(screen.getByPlaceholderText(/e.g. JWT token/i), { target: { value: 'IndexerService' } }); + fireEvent.click(screen.getByRole('button', { name: /Search/i })); + + await waitFor(() => { + expect(screen.getByRole('button', { name: /Copy/i })).toBeInTheDocument(); + }); - const queryInput = screen.getByPlaceholderText(/e.g. JWT token/i); - fireEvent.change(queryInput, { target: { value: 'IndexerService' } }); + fireEvent.click(screen.getByRole('button', { name: /Copy/i })); + expect(writeTextMock).toHaveBeenCalledWith('async def sync(self):\n pass'); + await waitFor(() => { + expect(screen.getByText('Copied')).toBeInTheDocument(); + }); + }); + + it('renders hits with missing AST metadata and null scores without error', async () => { + const rawHits = [ + { + id: 'raw-1', + score: 0.65, + dense_score: null, + sparse_score: null, + payload: { + repo: 'scripts-repo', + rel_path: 'deploy.sh', + content: '#!/bin/bash\necho deploy' + } + } + ]; + + (globalThis as any).fetch = vi.fn().mockImplementation((url: string) => { + if (typeof url === 'string' && url.includes('/admin/api/repos')) { + return Promise.resolve({ ok: true, json: async () => [{ name: 'scripts-repo' }] }); + } + return Promise.resolve({ ok: true, json: async () => ({ results: rawHits }) }); + }); + + render( + + + + ); + + fireEvent.change(screen.getByPlaceholderText(/e.g. JWT token/i), { target: { value: 'deploy' } }); fireEvent.click(screen.getByRole('button', { name: /Search/i })); await waitFor(() => { - const hitCard = container.querySelector('.search-hit-card'); - expect(hitCard).toBeInTheDocument(); - expect(container.querySelector('.search-hit-header')).toBeInTheDocument(); - expect(container.querySelector('.search-hit-code')).toBeInTheDocument(); + expect(screen.getAllByText('scripts-repo').length).toBeGreaterThanOrEqual(1); + expect(screen.getByText('deploy.sh')).toBeInTheDocument(); + expect(screen.getByText(/65\.0%/)).toBeInTheDocument(); + expect(screen.queryByText(/Signature:/i)).not.toBeInTheDocument(); }); }); -}); + it('handles network error during search gracefully', async () => { + (globalThis as any).fetch = vi.fn().mockImplementation((url: string) => { + if (typeof url === 'string' && url.includes('/admin/api/repos')) { + return Promise.resolve({ ok: true, json: async () => [] }); + } + return Promise.reject(new Error('Network offline')); + }); + + render( + + + + ); + + fireEvent.change(screen.getByPlaceholderText(/e.g. JWT token/i), { target: { value: 'test' } }); + fireEvent.click(screen.getByRole('button', { name: /Search/i })); + await waitFor(() => { + expect(screen.getAllByText(/Network offline/i).length).toBeGreaterThanOrEqual(1); + }); + }); +}); diff --git a/frontend/src/types.ts b/frontend/src/types.ts index ab36863..0f3f7be 100644 --- a/frontend/src/types.ts +++ b/frontend/src/types.ts @@ -135,13 +135,25 @@ export interface BrowseData { export interface SearchHit { score: number; + dense_score?: number | null; + sparse_score?: number | null; + dense_rank?: number | null; + sparse_rank?: number | null; payload: { repo: string; rel_path: string; symbol?: string; + full_symbol?: string; + signature?: string; + kind?: string; + ast_symbol_id?: number; + language?: string; start_line: number; end_line: number; github_url?: string; + permalink_url?: string; + heading?: string; + tags?: string[]; content: string; }; } diff --git a/tests/backend/test_chunker_languages.py b/tests/backend/test_chunker_languages.py index b4526f2..cb0279a 100644 --- a/tests/backend/test_chunker_languages.py +++ b/tests/backend/test_chunker_languages.py @@ -258,3 +258,38 @@ def test_markdown_chunking_with_nested_headings_and_empty(): assert "Section 1" in headings assert "Subsection 1.1" in headings assert "Section 2" in headings + + +def test_call_extraction_constructors_and_generics(): + code = """using System; + +namespace App +{ + public class Startup + { + public void Configure() + { + var s = new MyService(); + app.UseMiddleware(); + } + } +} +""" + res = extract_symbols_and_chunks(code, "Startup.cs", repo="test") + targets = [r.target_symbol for r in res.relationships if r.relationship_type == "CALLS"] + assert "MyService" in targets + assert "UseMiddleware" in targets + assert "new" not in targets + assert "var" not in targets + + +def test_toplevel_call_source_symbol_preservation(): + code = """import math +print(math.sqrt(16)) +run_task() +""" + res = extract_symbols_and_chunks(code, "main.py", repo="test") + sources = [r.source_symbol for r in res.relationships if r.relationship_type == "CALLS"] + assert all(src == "main.py" for src in sources) + assert "py" not in sources + diff --git a/tests/backend/test_mcp_v2.py b/tests/backend/test_mcp_v2.py index 615c274..6cc5005 100644 --- a/tests/backend/test_mcp_v2.py +++ b/tests/backend/test_mcp_v2.py @@ -103,3 +103,112 @@ async def test_fastmcp_streamable_http_transport(): ) assert resp.status_code == 200 assert "ContextCortex" in resp.text + + +@pytest.mark.asyncio +async def test_search_code_with_dense_weight_and_score_breakdown(temp_mcp_db): + mock_hit = MagicMock() + mock_hit.score = 0.82 + mock_hit.dense_score = 0.88 + mock_hit.sparse_score = 0.74 + mock_hit.payload = { + "repo": "demo-repo", + "rel_path": "services/auth.py", + "start_line": 20, + "end_line": 35, + "symbol": "AuthService.login", + "signature": "def login(self, username: str, password: str) -> Token:", + "kind": "method_declaration", + "github_url": "https://github.com/demo/auth.py#L20-L35", + "language": "python", + "content": "def login(...): return token" + } + + with patch("app.mcp.tools.execute_hybrid_search", return_value=[mock_hit]) as mock_exec: + res, structured = await mcp_server.call_tool( + "search_code", + {"query": "user login", "dense_weight": 0.6, "mode": "hybrid"} + ) + assert len(res) == 1 + text = res[0].text + assert "services/auth.py" in text + assert "AuthService.login" in text + assert "def login(self, username: str, password: str)" in text + assert "Semantic: 88.0%" in text + assert "Lexical: 74.0%" in text + mock_exec.assert_called_once_with( + query_text="user login", + doc_type="code", + repo=None, + language=None, + limit=5, + dense_weight=0.6, + search_mode="hybrid" + ) + +@pytest.mark.asyncio +async def test_search_code_empty_and_missing_ast_boundaries(temp_mcp_db): + # 1. Empty results returns polite notice + with patch("app.mcp.tools.execute_hybrid_search", return_value=[]): + res, _ = await mcp_server.call_tool("search_code", {"query": "nonexistent_func"}) + assert "No matching code snippets found for query: 'nonexistent_func'" in res[0].text + + # 2. Hit without AST signature/symbol renders cleanly without crashing + mock_hit_no_ast = MagicMock() + mock_hit_no_ast.score = 0.70 + mock_hit_no_ast.dense_score = 0.70 + mock_hit_no_ast.sparse_score = 0.0 + mock_hit_no_ast.payload = { + "repo": "raw-repo", + "rel_path": "scripts/build.sh", + "start_line": 1, + "end_line": 10, + "symbol": None, + "signature": None, + "content": "#!/usr/bin/env bash\necho 'building'" + } + with patch("app.mcp.tools.execute_hybrid_search", return_value=[mock_hit_no_ast]): + res, _ = await mcp_server.call_tool("search_code", {"query": "build.sh"}) + assert len(res) == 1 + text = res[0].text + assert "scripts/build.sh" in text + assert "echo 'building'" in text + assert "Signature" not in text + +@pytest.mark.asyncio +async def test_search_docs_with_dense_weight_and_score_breakdown(temp_mcp_db): + mock_doc = MagicMock() + mock_doc.score = 0.82 + mock_doc.dense_score = 0.80 + mock_doc.sparse_score = 0.85 + mock_doc.payload = { + "repo": "docs-repo", + "rel_path": "docs/architecture.md", + "title": "Architecture Guide", + "heading": "Microservices", + "start_line": 12, + "end_line": 30, + "content": "ContextCortex microservices routing and gateways" + } + + with patch("app.mcp.tools.execute_hybrid_search", return_value=[mock_doc]) as mock_exec: + res, _ = await mcp_server.call_tool( + "search_docs", + {"query": "microservices", "dense_weight": 0.4, "mode": "lexical"} + ) + assert len(res) == 1 + text = res[0].text + assert "docs/architecture.md" in text + assert "Microservices" in text + assert "Relevance Score: 0.8200 (82.0%) [Semantic: 80.0% | Lexical: 85.0%]" in text + mock_exec.assert_called_once_with( + query_text="microservices", + doc_type="doc", + repo=None, + category=None, + tag=None, + limit=5, + dense_weight=0.4, + search_mode="lexical" + ) + diff --git a/tests/backend/test_navigator_service.py b/tests/backend/test_navigator_service.py index 6c54f3f..8ec5942 100644 --- a/tests/backend/test_navigator_service.py +++ b/tests/backend/test_navigator_service.py @@ -288,6 +288,53 @@ def test_no_outgoing_calls_in_callers(test_db): assert len(res["imports"]) == 0 +def test_class_symbol_impact_aggregation(test_db): + # Seed a class with methods, callers, callees, and file-level imports + conn = sqlite3.connect(test_db) + conn.executescript(""" + INSERT INTO indexed_files (filepath, repo, doc_type, language) + VALUES ('app/services/user_service.py', 'test-repo', 'code', 'python'); + + INSERT INTO ast_symbols (id, repo, filepath, name, full_symbol, kind, start_line, end_line, signature, language) + VALUES (10, 'test-repo', 'app/services/user_service.py', 'UserService', 'UserService', 'class_definition', 1, 50, 'class UserService:', 'python'), + (11, 'test-repo', 'app/services/user_service.py', 'get_user', 'UserService.get_user', 'function_definition', 5, 20, 'def get_user(self, id):', 'python'), + (12, 'test-repo', 'app/services/user_service.py', 'save_user', 'UserService.save_user', 'function_definition', 22, 45, 'def save_user(self, user):', 'python'); + + -- File-level import + INSERT INTO ast_relationships (id, repo, source_symbol_id, source_filepath, source_symbol, target_symbol, relationship_type, line_number) + VALUES (20, 'test-repo', NULL, 'app/services/user_service.py', 'user_service.py', 'typing', 'IMPORTS', 1); + + -- External caller calling UserService.get_user + INSERT INTO ast_relationships (id, repo, source_symbol_id, source_filepath, source_symbol, target_symbol, relationship_type, line_number) + VALUES (21, 'test-repo', NULL, 'app/api/endpoints.py', 'handle_request', 'get_user', 'CALLS', 10); + + -- Method save_user calling external function + INSERT INTO ast_relationships (id, repo, source_symbol_id, source_filepath, source_symbol, target_symbol, relationship_type, line_number) + VALUES (22, 'test-repo', 12, 'app/services/user_service.py', 'save_user', 'db_commit', 'CALLS', 30); + """) + conn.commit() + conn.close() + + # Inspect the class UserService (id=10) + res = get_symbol_impact("test-repo", 10) + assert res is not None + assert res["symbol"]["name"] == "UserService" + + # Aggregated callers should find handle_request calling get_user + assert len(res["callers"]) == 1 + assert res["callers"][0]["source_symbol"] == "handle_request" + assert res["callers"][0]["target_symbol"] == "get_user" + + # Aggregated callees should find db_commit called from save_user + assert len(res["callees"]) == 1 + assert res["callees"][0]["target_symbol"] == "db_commit" + + # File-level imports should be available for the class + assert len(res["imports"]) == 1 + assert res["imports"][0]["target_symbol"] == "typing" + + + def test_real_codebase_symbol_extraction_and_navigation(tmp_path): from app.services.chunking import extract_symbols_and_chunks diff --git a/tests/backend/test_schemas.py b/tests/backend/test_schemas.py index 2696017..36d34ac 100644 --- a/tests/backend/test_schemas.py +++ b/tests/backend/test_schemas.py @@ -19,3 +19,41 @@ def test_search_request_defaults(): assert req.type == "code" assert req.limit == 5 assert req.exact is True + assert req.dense_weight is None + assert req.search_mode == "hybrid" + +def test_search_request_dense_weight_boundaries(): + from pydantic import ValidationError + + # Valid boundary values: 0.0, 0.5, 1.0 + r0 = SearchRequest(query="test", dense_weight=0.0) + assert r0.dense_weight == 0.0 + r_mid = SearchRequest(query="test", dense_weight=0.5) + assert r_mid.dense_weight == 0.5 + r1 = SearchRequest(query="test", dense_weight=1.0) + assert r1.dense_weight == 1.0 + + # Negative out-of-bounds (< 0.0) + with pytest.raises(ValidationError): + SearchRequest(query="test", dense_weight=-0.001) + + # Upper out-of-bounds (> 1.0) + with pytest.raises(ValidationError): + SearchRequest(query="test", dense_weight=1.001) + + # Non-numeric + with pytest.raises(ValidationError): + SearchRequest(query="test", dense_weight="not-a-number") + +def test_search_request_search_mode_boundaries(): + from pydantic import ValidationError + + # Valid modes + for mode in ["hybrid", "semantic", "lexical"]: + r = SearchRequest(query="test", search_mode=mode) + assert r.search_mode == mode + + # Invalid modes must be rejected by Literal schema validation + for invalid in ["unknown", "HYBRID", "dense", "sparse", ""]: + with pytest.raises(ValidationError): + SearchRequest(query="test", search_mode=invalid) diff --git a/tests/backend/test_search.py b/tests/backend/test_search.py index d64a053..3ae0a68 100644 --- a/tests/backend/test_search.py +++ b/tests/backend/test_search.py @@ -25,7 +25,9 @@ def test_execute_hybrid_search_delegation(): language="python", category="core", tag="auth", - limit=5 + limit=5, + dense_weight=0.7, + search_mode="hybrid" ) assert len(results) == 1 @@ -39,9 +41,147 @@ def test_execute_hybrid_search_delegation(): language="python", category="core", tag="auth", - limit=5 + limit=5, + dense_weight=0.7, + search_mode="hybrid" ) +def test_execute_hybrid_search_ast_enrichment(): + mock_hit = VectorSearchResult( + id="code-hit-1", + score=0.88, + payload={ + "repo": "test-repo", + "doc_type": "code", + "rel_path": "src/auth.py", + "symbol": "AuthService.validate_token", + "start_line": 10, + "end_line": 25, + "content": "def validate_token(self, token): pass" + } + ) + with patch("app.services.search.get_vector_store") as mock_get_store, \ + patch("app.services.search.get_db_connection") as mock_db: + mock_store = MagicMock() + mock_store.search.return_value = [mock_hit] + mock_get_store.return_value = mock_store + + mock_conn = MagicMock() + mock_row = { + "id": 1234, + "name": "validate_token", + "full_symbol": "AuthService.validate_token", + "kind": "method_declaration", + "signature": "def validate_token(self, token: str) -> bool:", + "start_line": 10, + "end_line": 25, + } + mock_conn.execute.return_value.fetchone.return_value = mock_row + mock_db.return_value.__enter__.return_value = mock_conn + + results = execute_hybrid_search( + query_text="validate token", + doc_type="code", + repo="test-repo" + ) + + assert len(results) == 1 + p = results[0].payload + assert p["signature"] == "def validate_token(self, token: str) -> bool:" + assert p["kind"] == "method_declaration" + assert p["ast_symbol_id"] == 1234 + +def test_execute_hybrid_search_ast_enrichment_interval_fallback(): + # Priority 2: symbol is None or doesn't match by name, but line interval overlaps + mock_hit = VectorSearchResult( + id="code-hit-2", + score=0.85, + payload={ + "repo": "test-repo", + "doc_type": "code", + "rel_path": "src/utils.py", + "symbol": None, + "start_line": 50, + "end_line": 65, + "content": "def helper(): pass" + } + ) + with patch("app.services.search.get_vector_store") as mock_get_store, \ + patch("app.services.search.get_db_connection") as mock_db: + mock_store = MagicMock() + mock_store.search.return_value = [mock_hit] + mock_get_store.return_value = mock_store + + mock_conn = MagicMock() + # First query (exact symbol) returns None + # Second query (line interval) returns matching symbol + mock_row = { + "id": 5678, + "name": "helper", + "full_symbol": "Utils.helper", + "kind": "function_declaration", + "signature": "def helper(val: int) -> int:", + "start_line": 48, + "end_line": 68, + } + mock_conn.execute.return_value.fetchone.return_value = mock_row + mock_db.return_value.__enter__.return_value = mock_conn + + results = execute_hybrid_search("helper", doc_type="code") + assert len(results) == 1 + p = results[0].payload + assert p["signature"] == "def helper(val: int) -> int:" + assert p["kind"] == "function_declaration" + assert p["full_symbol"] == "Utils.helper" + assert p["ast_symbol_id"] == 5678 + +def test_execute_hybrid_search_ast_enrichment_resilience(): + # When SQLite has no matching symbol or raises an error, search must NOT crash + mock_hit = VectorSearchResult( + id="code-hit-3", + score=0.75, + payload={ + "repo": "test-repo", + "doc_type": "code", + "rel_path": "src/unknown.py", + "symbol": "UnknownClass", + "content": "class UnknownClass: pass" + } + ) + with patch("app.services.search.get_vector_store") as mock_get_store, \ + patch("app.services.search.get_db_connection") as mock_db: + mock_store = MagicMock() + mock_store.search.return_value = [mock_hit] + mock_get_store.return_value = mock_store + + # Simulate SQLite operational failure + mock_conn = MagicMock() + mock_conn.execute.side_effect = Exception("database disk image is malformed") + mock_db.return_value.__enter__.return_value = mock_conn + + results = execute_hybrid_search("unknown", doc_type="code") + assert len(results) == 1 + # Payload remains preserved without failure + assert results[0].payload["symbol"] == "UnknownClass" + assert "signature" not in results[0].payload + +def test_execute_hybrid_search_ast_enrichment_skipped_for_docs(): + # AST enrichment must be completely bypassed for doc_type="doc" + mock_hit = VectorSearchResult( + id="doc-hit-1", + score=0.89, + payload={"repo": "docs", "doc_type": "doc", "rel_path": "README.md"} + ) + with patch("app.services.search.get_vector_store") as mock_get_store, \ + patch("app.services.search.get_db_connection") as mock_db: + mock_store = MagicMock() + mock_store.search.return_value = [mock_hit] + mock_get_store.return_value = mock_store + + results = execute_hybrid_search("readme", doc_type="doc") + assert len(results) == 1 + mock_db.assert_not_called() + def test_execute_hybrid_search_exception(): with patch("app.services.search.get_vector_store") as mock_get_store: mock_store = MagicMock() @@ -126,3 +266,62 @@ def test_execute_hybrid_search_end_to_end_real(tmp_path, monkeypatch): assert len(pdf_results) >= 1 assert any(h.payload.get("rel_path") == "docs/manual.pdf" for h in pdf_results) + +def test_api_test_search_endpoint(): + from fastapi import FastAPI + from fastapi.testclient import TestClient + from app.api.routers.repositories import router as repo_router + + test_app = FastAPI() + test_app.include_router(repo_router) + client = TestClient(test_app) + + mock_hit = VectorSearchResult( + id="sym-1", + score=0.85, + dense_score=0.90, + sparse_score=0.75, + dense_rank=1, + sparse_rank=2, + payload={"repo": "test-repo", "rel_path": "index.ts", "symbol": "runApp"} + ) + + with patch("app.services.search.execute_hybrid_search", return_value=[mock_hit]) as mock_exec: + resp = client.post( + "/admin/api/search/test", + json={ + "query": "run app", + "type": "code", + "repo": "test-repo", + "dense_weight": 0.65, + "search_mode": "hybrid", + "limit": 10 + } + ) + assert resp.status_code == 200 + data = resp.json() + assert data["query"] == "run app" + assert data["dense_weight"] == 0.65 + assert data["search_mode"] == "hybrid" + assert len(data["results"]) == 1 + res0 = data["results"][0] + assert res0["score"] == 0.85 + assert res0["dense_score"] == 0.90 + assert res0["sparse_score"] == 0.75 + assert res0["dense_rank"] == 1 + assert res0["sparse_rank"] == 2 + assert res0["payload"]["symbol"] == "runApp" + + mock_exec.assert_called_once_with( + query_text="run app", + doc_type="code", + repo="test-repo", + language=None, + category=None, + tag=None, + limit=10, + dense_weight=0.65, + search_mode="hybrid" + ) + + diff --git a/tests/backend/test_vector_store_qdrant.py b/tests/backend/test_vector_store_qdrant.py index 3270eec..c073bf9 100644 --- a/tests/backend/test_vector_store_qdrant.py +++ b/tests/backend/test_vector_store_qdrant.py @@ -260,6 +260,128 @@ def test_search_weighted_score_fusion_alpha_weighting(self, memory_store, monkey score_low = results_low_dense[0].score assert 0.0 <= score_low <= 1.0 + def test_search_explicit_dense_weight_and_score_decomposition(self, memory_store): + doc = VectorDocument( + id=str(uuid.uuid4()), + text="Distributed cache invalidation protocols and Redis clusters.", + repo="cache-system", + path="/docs/cache.md", + ) + memory_store.upsert_documents([doc]) + + # Explicit dense weight 0.8 + res = memory_store.search("Redis cache clusters", limit=5, dense_weight=0.8) + assert len(res) == 1 + hit = res[0] + assert hit.dense_score is not None + assert hit.sparse_score is not None + assert hit.score > 0 + assert hit.dense_rank == 1 + assert hit.sparse_rank == 1 + + # Explicit dense weight 0.0 (pure lexical) + res_lexical = memory_store.search("Redis cache clusters", limit=5, dense_weight=0.0) + assert len(res_lexical) == 1 + assert res_lexical[0].score == res_lexical[0].sparse_score + + # Explicit dense weight 1.0 (pure semantic) + res_semantic = memory_store.search("Redis cache clusters", limit=5, dense_weight=1.0) + assert len(res_semantic) == 1 + assert res_semantic[0].score == res_semantic[0].dense_score + + # Out-of-bounds negative dense weight (< 0.0) is clamped to 0.0 + res_neg = memory_store.search("Redis cache clusters", limit=5, dense_weight=-0.5) + assert len(res_neg) == 1 + assert res_neg[0].score == res_neg[0].sparse_score + + # Out-of-bounds excessive dense weight (> 1.0) is clamped to 1.0 + res_pos = memory_store.search("Redis cache clusters", limit=5, dense_weight=2.0) + assert len(res_pos) == 1 + assert res_pos[0].score == res_pos[0].dense_score + + # Non-numeric string falls back to default 0.5 without exception + res_str = memory_store.search("Redis cache clusters", limit=5, dense_weight="invalid") + assert len(res_str) == 1 + assert res_str[0].score > 0 + + def test_search_modes_case_insensitivity_and_whitespace(self, memory_store): + doc = VectorDocument( + id=str(uuid.uuid4()), + text="Kafka event streaming and partition consumer groups.", + repo="streaming", + path="/docs/kafka.md", + ) + memory_store.upsert_documents([doc]) + + # Uppercase and leading/trailing whitespace + res_sem = memory_store.search("Kafka event", limit=5, search_mode=" SEMANTIC ") + assert len(res_sem) == 1 + assert res_sem[0].dense_score > 0 + assert res_sem[0].sparse_score == 0.0 + + res_lex = memory_store.search("Kafka consumer", limit=5, search_mode="LEXICAL") + assert len(res_lex) == 1 + assert res_lex[0].sparse_score > 0 + assert res_lex[0].dense_score == 0.0 + + # Unrecognized search mode falls back to hybrid + res_unrec = memory_store.search("Kafka streaming", limit=5, search_mode="non_existent_mode") + assert len(res_unrec) == 1 + assert res_unrec[0].score > 0 + + def test_search_modes_semantic_and_lexical(self, memory_store, monkeypatch): + doc = VectorDocument( + id=str(uuid.uuid4()), + text="Kubernetes ingress controller configuration and TLS termination.", + repo="k8s-infra", + path="/docs/ingress.md", + ) + memory_store.upsert_documents([doc]) + + # Semantic mode + res_sem = memory_store.search("TLS ingress", limit=5, search_mode="semantic") + assert len(res_sem) == 1 + assert res_sem[0].dense_score > 0 + assert res_sem[0].sparse_score == 0.0 + + # Lexical mode + res_lex = memory_store.search("ingress controller", limit=5, search_mode="lexical") + assert len(res_lex) == 1 + assert res_lex[0].sparse_score > 0 + assert res_lex[0].dense_score == 0.0 + + # Ensure lazy embedding inference: semantic mode doesn't compute sparse embedding + from app.services.vector_store import qdrant_store + sparse_called = [] + monkeypatch.setattr(qdrant_store, "get_sparse_embedding", lambda q: sparse_called.append(q)) + memory_store.search("TLS ingress", limit=5, search_mode="semantic") + assert len(sparse_called) == 0 + + # Ensure lexical mode doesn't compute dense embedding + dense_called = [] + monkeypatch.setattr(qdrant_store, "get_dense_embedding", lambda q: dense_called.append(q)) + memory_store.search("ingress controller", limit=5, search_mode="lexical") + assert len(dense_called) == 0 + + def test_search_mode_lexical_empty_sparse_no_dense_fallback(self, memory_store, monkeypatch): + doc = VectorDocument( + id=str(uuid.uuid4()), + text="General documentation without specific match.", + repo="core", + path="/docs/general.md", + ) + memory_store.upsert_documents([doc]) + + from app.services.vector_store import qdrant_store + class EmptySparse: + indices = [] + values = [] + + monkeypatch.setattr(qdrant_store, "get_sparse_embedding", lambda _: EmptySparse()) + # Pure lexical search must return empty list and NOT fall back to dense semantic results + res = memory_store.search("general", search_mode="lexical") + assert res == [] + def test_search_dense_fallback_without_sparse(self, memory_store): doc = VectorDocument( id=str(uuid.uuid4()),