diff --git a/FQDN_NORMALIZATION_FIX.py b/FQDN_NORMALIZATION_FIX.py index 196d6e2e..7c5fc3f4 100644 --- a/FQDN_NORMALIZATION_FIX.py +++ b/FQDN_NORMALIZATION_FIX.py @@ -1,6 +1,11 @@ """ FQDN Normalization Fix - Enhanced Component ID Resolution +NOTE: This file is a standalone reference/patch proposal for +codewiki/src/be/cluster_modules.py. It is kept at the repository root +temporarily for review purposes; its logic should be integrated into +codewiki/src/be/cluster_modules.py (or this file removed) once merged. + This file contains the proposed fix for cluster_modules.py to handle: 1. LLM-added "deps." prefixes 2. Fuzzy substring matching for nested paths @@ -133,6 +138,7 @@ def normalize_component_ids_enhanced( if '.' in comp_id: # Try matching last 2-4 segments segments = comp_id.split('.') + suffix_matches = [] for n in range(2, min(5, len(segments) + 1)): suffix = '.'.join(segments[-n:]) suffix_matches = [ @@ -314,3 +320,4 @@ def build_short_id_to_fqdn_map_enhanced(components: Dict) -> Dict[str, str]: logger.warning(f" ⚠️ Failed to normalize {total_failed} component IDs") logger.info("") """ + diff --git a/codewiki/src/be/dependency_analyzer/analysis/call_graph_analyzer.py b/codewiki/src/be/dependency_analyzer/analysis/call_graph_analyzer.py index 7175cd9b..db524edf 100644 --- a/codewiki/src/be/dependency_analyzer/analysis/call_graph_analyzer.py +++ b/codewiki/src/be/dependency_analyzer/analysis/call_graph_analyzer.py @@ -231,13 +231,16 @@ def _analyze_c_file(self, file_path: str, content: str, repo_dir: str): """ from codewiki.src.be.dependency_analyzer.analyzers.c import analyze_c_file - functions, relationships = analyze_c_file(file_path, content, repo_path=repo_dir) + try: + functions, relationships = analyze_c_file(file_path, content, repo_path=repo_dir) - for func in functions: - func_id = func.id if func.id else f"{file_path}:{func.name}" - self.functions[func_id] = func + for func in functions: + func_id = func.id if func.id else f"{file_path}:{func.name}" + self.functions[func_id] = func - self.call_relationships.extend(relationships) + self.call_relationships.extend(relationships) + except Exception as e: + logger.error(f"Failed to analyze C file {file_path}: {e}", exc_info=True) def _analyze_cpp_file(self, file_path: str, content: str, repo_dir: str): """ @@ -249,15 +252,18 @@ def _analyze_cpp_file(self, file_path: str, content: str, repo_dir: str): """ from codewiki.src.be.dependency_analyzer.analyzers.cpp import analyze_cpp_file - functions, relationships = analyze_cpp_file( - file_path, content, repo_path=repo_dir - ) + try: + functions, relationships = analyze_cpp_file( + file_path, content, repo_path=repo_dir + ) - for func in functions: - func_id = func.id if func.id else f"{file_path}:{func.name}" - self.functions[func_id] = func + for func in functions: + func_id = func.id if func.id else f"{file_path}:{func.name}" + self.functions[func_id] = func - self.call_relationships.extend(relationships) + self.call_relationships.extend(relationships) + except Exception as e: + logger.error(f"Failed to analyze C++ file {file_path}: {e}", exc_info=True) def _analyze_java_file(self, file_path: str, content: str, repo_dir: str): """ diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py index 27a69bf2..058aba98 100644 --- a/codewiki/src/be/llm_services.py +++ b/codewiki/src/be/llm_services.py @@ -136,7 +136,12 @@ def create_main_model(config: Config) -> OpenAIModel: provider=OpenAIProvider( base_url=base_url, api_key=api_key, - # default_headers removed - use http_client if needed + # NOTE: pydantic-ai's OpenAIProvider takes only base_url, api_key, + # openai_client and http_client - there is no default_headers + # parameter (verified against pydantic-ai 2.40.0), so passing one + # raises TypeError. To send anthropic-version here, build an + # AsyncOpenAI client with default_headers and pass it as + # openai_client=. ), settings=OpenAIModelSettings(**settings_dict) ) @@ -186,7 +191,12 @@ def create_fallback_model(config: Config) -> OpenAIModel: provider=OpenAIProvider( base_url=base_url, api_key=api_key, - # default_headers removed - use http_client if needed + # NOTE: pydantic-ai's OpenAIProvider takes only base_url, api_key, + # openai_client and http_client - there is no default_headers + # parameter (verified against pydantic-ai 2.40.0), so passing one + # raises TypeError. To send anthropic-version here, build an + # AsyncOpenAI client with default_headers and pass it as + # openai_client=. ), settings=OpenAIModelSettings(**settings_dict) ) @@ -250,7 +260,12 @@ def create_cluster_model(config: Config) -> OpenAIModel: provider=OpenAIProvider( base_url=base_url, api_key=api_key, - # default_headers removed - use http_client if needed + # NOTE: pydantic-ai's OpenAIProvider takes only base_url, api_key, + # openai_client and http_client - there is no default_headers + # parameter (verified against pydantic-ai 2.40.0), so passing one + # raises TypeError. To send anthropic-version here, build an + # AsyncOpenAI client with default_headers and pass it as + # openai_client=. ), settings=OpenAIModelSettings(**settings_dict) ) @@ -336,7 +351,7 @@ def create_openai_client(config: Config, model: str = None) -> OpenAI: return OpenAI( base_url=base_url, api_key=api_key, - # default_headers removed - use http_client if needed + default_headers=default_headers if default_headers else None, ) @@ -457,4 +472,4 @@ def call_llm( raise RuntimeError( f"Unexpected error calling {model_stage_name} model '{model}': " f"{type(e).__name__}: {str(e)}" - ) from e \ No newline at end of file + ) from e diff --git a/codewiki/src/fe/config.py b/codewiki/src/fe/config.py index c77d3d4a..cb041f29 100644 --- a/codewiki/src/fe/config.py +++ b/codewiki/src/fe/config.py @@ -48,4 +48,4 @@ def ensure_directories(cls): @classmethod def get_absolute_path(cls, path: str) -> str: """Get absolute path for a given relative path.""" - return os.path.abspath(path) \ No newline at end of file + return os.path.abspath(path) diff --git a/codewiki/src/fe/visualise_docs.py b/codewiki/src/fe/visualise_docs.py index c2bd9e97..fa6485ce 100644 --- a/codewiki/src/fe/visualise_docs.py +++ b/codewiki/src/fe/visualise_docs.py @@ -157,7 +157,7 @@ async def serve_doc(filename: str): try: file_path = file_path.resolve() docs_folder_resolved = Path(DOCS_FOLDER).resolve() - if not str(file_path).startswith(str(docs_folder_resolved)): + if not file_path.is_relative_to(docs_folder_resolved): raise HTTPException(status_code=403, detail="Access denied") except Exception: raise HTTPException(status_code=403, detail="Invalid file path") diff --git a/test_clustering_validation.py b/test_clustering_validation.py index ede57eae..1ce8372f 100644 --- a/test_clustering_validation.py +++ b/test_clustering_validation.py @@ -15,6 +15,46 @@ logging.basicConfig(level=logging.INFO, format='%(levelname)s: %(message)s') logger = logging.getLogger(__name__) +class TestResults: + """Accumulates test results and prints a summary.""" + + def __init__(self): + self.passed = 0 + self.failed = 0 + self.failures = [] + + def add_test(self, name: str, passed: bool, details: str = ""): + if passed: + self.passed += 1 + logger.info(f"✅ TEST PASSED: {name}") + else: + self.failed += 1 + self.failures.append((name, details)) + logger.error(f"❌ TEST FAILED: {name} {details}") + + def print_summary(self): + total = self.passed + self.failed + print("\n" + "="*70) + print("TEST SUMMARY") + print("="*70) + print(f"Total tests: {total}") + print(f"✅ Passed: {self.passed}") + print(f"❌ Failed: {self.failed}") + if total: + print(f"Success rate: {self.passed/total*100:.1f}%") + + if self.failed == 0: + print("\n🎉 ALL TESTS PASSED! Validation logic is working correctly.") + else: + print(f"\n⚠️ {self.failed} test(s) failed. Please review the validation logic.") + for name, details in self.failures: + print(f" - {name}: {details}") + + @property + def success(self): + return self.failed == 0 + + def simulate_validation(response_content: str, max_id: int): """ Simulates the validation logic from cluster_modules.py (lines 338-369) @@ -140,8 +180,7 @@ def run_tests(): print("CODEWIKI CLUSTERING VALIDATION TEST SUITE") print("="*70) - passed = 0 - failed = 0 + results = TestResults() for i, test_case in enumerate(test_cases, 1): print(f"\n{'='*70}") @@ -153,28 +192,15 @@ def run_tests(): test_case['max_id'] ) - if success == test_case['should_pass']: - logger.info(f"✅ TEST PASSED: Got expected result (success={success})") - passed += 1 - else: - logger.error(f"❌ TEST FAILED: Expected {test_case['should_pass']}, got {success}") - failed += 1 - - # Summary - print("\n" + "="*70) - print("TEST SUMMARY") - print("="*70) - print(f"Total tests: {len(test_cases)}") - print(f"✅ Passed: {passed}") - print(f"❌ Failed: {failed}") - print(f"Success rate: {passed/len(test_cases)*100:.1f}%") + results.add_test( + test_case['name'], + success == test_case['should_pass'], + f"(expected {test_case['should_pass']}, got {success})" + ) - if failed == 0: - print("\n🎉 ALL TESTS PASSED! Validation logic is working correctly.") - else: - print(f"\n⚠️ {failed} test(s) failed. Please review the validation logic.") + results.print_summary() - return failed == 0 + return results.success if __name__ == "__main__": success = run_tests()