From 110785fe3100012f5f4c08c01f5e62a8ad7df110 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:27 +0000 Subject: [PATCH 1/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/src/be/flamingo_guidelines.py | 68 ++++++++++++++------------ 1 file changed, 37 insertions(+), 31 deletions(-) diff --git a/codewiki/src/be/flamingo_guidelines.py b/codewiki/src/be/flamingo_guidelines.py index 97761d68..1afff843 100644 --- a/codewiki/src/be/flamingo_guidelines.py +++ b/codewiki/src/be/flamingo_guidelines.py @@ -23,37 +23,56 @@ VALIDATION_RULES_ENV_VAR = "VALIDATION_RULES_PATH" -def load_flamingo_guidelines() -> str: +def _load_markdown_file_from_env(env_var: str, missing_log_suffix: str, content_description: str) -> str: """ - Load Flamingo markdown guidelines from file path specified in env var. + Load markdown content from a file path specified in an environment variable. - Environment Variable: - FLAMINGO_MARKDOWN_GUIDELINES_PATH: Path to the guidelines markdown file - (downloaded during GitHub Actions workflow) + Args: + env_var: Name of the environment variable holding the file path. + missing_log_suffix: Text appended to the log message when the env var is not set. + content_description: Human-readable description of the content, used in log messages. Returns: - Guidelines content string, or empty string if not available. + File content string, or empty string if not available. """ - guidelines_path = os.environ.get(GUIDELINES_ENV_VAR) + file_path = os.environ.get(env_var) - if not guidelines_path: - logger.info(f"[CodeWiki] {GUIDELINES_ENV_VAR} not set - continuing without Flamingo guidelines") + if not file_path: + logger.info(f"[CodeWiki] {env_var} not set - continuing without {missing_log_suffix}") return "" try: - path = Path(guidelines_path) + path = Path(file_path) if not path.exists(): - logger.warning(f"[CodeWiki] Guidelines file not found: {guidelines_path}") + logger.warning(f"[CodeWiki] {content_description} file not found: {file_path}") return "" content = path.read_text(encoding='utf-8') - logger.info(f"[CodeWiki] Loaded Flamingo markdown guidelines ({len(content)} chars)") + logger.info(f"[CodeWiki] Loaded {content_description} ({len(content)} chars)") return content except Exception as e: - logger.warning(f"[CodeWiki] Failed to load guidelines: {e}") + logger.warning(f"[CodeWiki] Failed to load {content_description}: {e}") return "" +def load_flamingo_guidelines() -> str: + """ + Load Flamingo markdown guidelines from file path specified in env var. + + Environment Variable: + FLAMINGO_MARKDOWN_GUIDELINES_PATH: Path to the guidelines markdown file + (downloaded during GitHub Actions workflow) + + Returns: + Guidelines content string, or empty string if not available. + """ + return _load_markdown_file_from_env( + GUIDELINES_ENV_VAR, + "Flamingo guidelines", + "Flamingo markdown guidelines", + ) + + # Load guidelines at module import time FLAMINGO_MARKDOWN_GUIDELINES = load_flamingo_guidelines() @@ -291,24 +310,11 @@ def load_validation_rules() -> str: Returns: Validation rules content string, or empty string if not available. """ - rules_path = os.environ.get(VALIDATION_RULES_ENV_VAR) - - if not rules_path: - logger.info(f"[CodeWiki] {VALIDATION_RULES_ENV_VAR} not set - continuing without validation rules injection") - return "" - - try: - path = Path(rules_path) - if not path.exists(): - logger.warning(f"[CodeWiki] Validation rules file not found: {rules_path}") - return "" - - content = path.read_text(encoding='utf-8') - logger.info(f"[CodeWiki] Loaded markdown validation rules ({len(content)} chars)") - return content - except Exception as e: - logger.warning(f"[CodeWiki] Failed to load validation rules: {e}") - return "" + return _load_markdown_file_from_env( + VALIDATION_RULES_ENV_VAR, + "validation rules injection", + "markdown validation rules", + ) # Load validation rules at module import time From b9481953853335bdd7c56059f8136bc737f29c35 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:28 +0000 Subject: [PATCH 2/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/src/be/llm_services.py | 97 +++++++++++---------------------- 1 file changed, 31 insertions(+), 66 deletions(-) diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py index 058aba98..1ed05c6a 100644 --- a/codewiki/src/be/llm_services.py +++ b/codewiki/src/be/llm_services.py @@ -85,20 +85,22 @@ def get_model_max_token_field(stage: str = 'generation') -> str: return os.environ.get(env_var, 'max_tokens') -def create_main_model(config: Config) -> OpenAIModel: +def _create_provider_model(config: Config, prefix: str) -> OpenAIModel: """ - Create the main LLM model from configuration. + Create an LLM model for the given per-provider config prefix (e.g. 'main' or 'fallback'). NOTE: Pydantic AI currently hardcodes the 'max_tokens' parameter name in OpenAIModelSettings. For reasoning models (o3, o3-mini) that require 'max_completion_tokens', the direct API call in call_llm() uses the correct parameter name. If Pydantic AI models fail with "Unrecognized request argument supplied: max_tokens", the system will fall back to the direct API call. """ + model_name = getattr(config, f'{prefix}_model') + # Use per-provider max_tokens - max_tokens = getattr(config, 'main_max_tokens', None) or get_max_output_tokens() + max_tokens = getattr(config, f'{prefix}_max_tokens', None) or get_max_output_tokens() # Check if model supports custom temperature (use per-provider field) - temperature = getattr(config, 'main_temperature', 0.0) - temperature_supported = getattr(config, 'main_temperature_supported', True) + temperature = getattr(config, f'{prefix}_temperature', 0.0) + temperature_supported = getattr(config, f'{prefix}_temperature_supported', True) # Build settings dict - only include temperature if model supports it settings_dict = {'max_tokens': max_tokens} @@ -106,33 +108,33 @@ def create_main_model(config: Config) -> OpenAIModel: settings_dict['temperature'] = temperature # Build provider with per-provider base_url and optional api_version header - base_url = getattr(config, 'main_base_url', None) + base_url = getattr(config, f'{prefix}_base_url', None) if not base_url: raise ValueError( - "main_base_url is required in configuration for main/generation model.\n" - f"Model: {config.main_model}\n" - "Please set via CLI: --main-base-url \n" - "Or in config file: main_base_url = ''" + f"{prefix}_base_url is required in configuration for {prefix} model.\n" + f"Model: {model_name}\n" + f"Please set via CLI: --{prefix}-base-url \n" + f"Or in config file: {prefix}_base_url = ''" ) # Prepare default headers for API version (Anthropic models) default_headers = {} - api_version = getattr(config, 'main_api_version', None) + api_version = getattr(config, f'{prefix}_api_version', None) if api_version: default_headers['anthropic-version'] = api_version # Get per-provider API key - api_key = getattr(config, 'main_api_key', None) + api_key = getattr(config, f'{prefix}_api_key', None) if not api_key: raise ValueError( - "main_api_key is required in configuration for main/generation model.\n" - f"Model: {config.main_model}\n" - "Please set via CLI: --main-api-key \n" + f"{prefix}_api_key is required in configuration for {prefix} model.\n" + f"Model: {model_name}\n" + f"Please set via CLI: --{prefix}-api-key \n" "Different AI providers require different API keys." ) return OpenAIModel( - model_name=config.main_model, + model_name=model_name, provider=OpenAIProvider( base_url=base_url, api_key=api_key, @@ -147,59 +149,21 @@ def create_main_model(config: Config) -> OpenAIModel: ) -def create_fallback_model(config: Config) -> OpenAIModel: - """Create the fallback LLM model from configuration.""" - # Use per-provider max_tokens - max_tokens = getattr(config, 'fallback_max_tokens', None) or get_max_output_tokens() - # Check if model supports custom temperature (use per-provider field) - temperature = getattr(config, 'fallback_temperature', 0.0) - temperature_supported = getattr(config, 'fallback_temperature_supported', True) - - # Build settings dict - only include temperature if model supports it - settings_dict = {'max_tokens': max_tokens} - if temperature_supported: - settings_dict['temperature'] = temperature - - # Build provider with per-provider base_url and optional api_version header - base_url = getattr(config, 'fallback_base_url', None) - if not base_url: - raise ValueError( - "fallback_base_url is required in configuration for fallback model.\n" - f"Model: {config.fallback_model}\n" - "Please set via CLI: --fallback-base-url \n" - "Or in config file: fallback_base_url = ''" - ) +def create_main_model(config: Config) -> OpenAIModel: + """ + Create the main LLM model from configuration. - # Prepare default headers for API version (Anthropic models) - default_headers = {} - api_version = getattr(config, 'fallback_api_version', None) - if api_version: - default_headers['anthropic-version'] = api_version + NOTE: Pydantic AI currently hardcodes the 'max_tokens' parameter name in OpenAIModelSettings. + For reasoning models (o3, o3-mini) that require 'max_completion_tokens', the direct API call + in call_llm() uses the correct parameter name. If Pydantic AI models fail with "Unrecognized + request argument supplied: max_tokens", the system will fall back to the direct API call. + """ + return _create_provider_model(config, 'main') - # Get per-provider API key - api_key = getattr(config, 'fallback_api_key', None) - if not api_key: - raise ValueError( - "fallback_api_key is required in configuration for fallback model.\n" - f"Model: {config.fallback_model}\n" - "Please set via CLI: --fallback-api-key \n" - "Different AI providers require different API keys." - ) - return OpenAIModel( - model_name=config.fallback_model, - provider=OpenAIProvider( - base_url=base_url, - api_key=api_key, - # NOTE: pydantic-ai's OpenAIProvider takes only base_url, api_key, - # openai_client and http_client - there is no default_headers - # parameter (verified against pydantic-ai 2.40.0), so passing one - # raises TypeError. To send anthropic-version here, build an - # AsyncOpenAI client with default_headers and pass it as - # openai_client=. - ), - settings=OpenAIModelSettings(**settings_dict) - ) +def create_fallback_model(config: Config) -> OpenAIModel: + """Create the fallback LLM model from configuration.""" + return _create_provider_model(config, 'fallback') @@ -473,3 +437,4 @@ def call_llm( f"Unexpected error calling {model_stage_name} model '{model}': " f"{type(e).__name__}: {str(e)}" ) from e + From a745b86827d563845978aee0af3f9f246a953a09 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:29 +0000 Subject: [PATCH 3/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/src/fe/models.py | 23 ++++++++++++++--------- 1 file changed, 14 insertions(+), 9 deletions(-) diff --git a/codewiki/src/fe/models.py b/codewiki/src/fe/models.py index 253d7369..dba8f91f 100644 --- a/codewiki/src/fe/models.py +++ b/codewiki/src/fe/models.py @@ -5,7 +5,7 @@ from datetime import datetime from typing import Optional -from dataclasses import dataclass +from dataclasses import dataclass, asdict from pydantic import BaseModel, HttpUrl @@ -14,11 +14,12 @@ class RepositorySubmission(BaseModel): repo_url: HttpUrl -class JobStatusResponse(BaseModel): - """Pydantic model for job status API response.""" +@dataclass +class JobStatus: + """Tracks the status of a documentation generation job.""" job_id: str repo_url: str - status: str + status: str # 'queued', 'processing', 'completed', 'failed' created_at: datetime started_at: Optional[datetime] = None completed_at: Optional[datetime] = None @@ -29,12 +30,11 @@ class JobStatusResponse(BaseModel): commit_id: Optional[str] = None -@dataclass -class JobStatus: - """Tracks the status of a documentation generation job.""" +class JobStatusResponse(BaseModel): + """Pydantic model for job status API response.""" job_id: str repo_url: str - status: str # 'queued', 'processing', 'completed', 'failed' + status: str created_at: datetime started_at: Optional[datetime] = None completed_at: Optional[datetime] = None @@ -44,6 +44,11 @@ class JobStatus: main_model: Optional[str] = None commit_id: Optional[str] = None + @classmethod + def from_job_status(cls, job_status: JobStatus) -> "JobStatusResponse": + """Build a JobStatusResponse from a JobStatus instance.""" + return cls(**asdict(job_status)) + @dataclass class CacheEntry: @@ -52,4 +57,4 @@ class CacheEntry: repo_url_hash: str docs_path: str created_at: datetime - last_accessed: datetime \ No newline at end of file + last_accessed: datetime From ee304b3ec123d479e29e8cc3601ddf6b53602fd4 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:30 +0000 Subject: [PATCH 4/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/cli/commands/generate.py | 8 +------- 1 file changed, 1 insertion(+), 7 deletions(-) diff --git a/codewiki/cli/commands/generate.py b/codewiki/cli/commands/generate.py index ba469ae7..76c467df 100644 --- a/codewiki/cli/commands/generate.py +++ b/codewiki/cli/commands/generate.py @@ -11,6 +11,7 @@ import time from codewiki.cli.config_manager import ConfigManager +from codewiki.cli.commands.config import parse_patterns from codewiki.cli.utils.errors import ( ConfigurationError, RepositoryError, @@ -32,13 +33,6 @@ from codewiki.cli.models.config import AgentInstructions -def parse_patterns(patterns_str: str) -> List[str]: - """Parse comma-separated patterns into a list.""" - if not patterns_str: - return [] - return [p.strip() for p in patterns_str.split(',') if p.strip()] - - @click.command(name="generate") @click.option( "--output", From 36a3098542446e09ba814374dac2b074fc7762b2 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:31 +0000 Subject: [PATCH 5/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/src/be/dependency_analyzer/analyzers/cpp.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/codewiki/src/be/dependency_analyzer/analyzers/cpp.py b/codewiki/src/be/dependency_analyzer/analyzers/cpp.py index 20cec996..8540ac0b 100644 --- a/codewiki/src/be/dependency_analyzer/analyzers/cpp.py +++ b/codewiki/src/be/dependency_analyzer/analyzers/cpp.py @@ -373,6 +373,14 @@ def _class_has_method(self, class_node, method_name): return False def analyze_cpp_file(file_path: str, content: str, repo_path: str = None) -> Tuple[List[Node], List[CallRelationship]]: + """Analyze a C++ source file and extract nodes and call relationships. + + Note: structurally similar to analyze_c_file (c.py) and other language + analyzers in this package, since each wraps a language-specific + TreeSitter*Analyzer with the same construct-and-collect pattern. Kept + separate because the underlying analyzer classes are not identical + (different tree-sitter grammars and node handling). + """ analyzer = TreeSitterCppAnalyzer(file_path, content, repo_path) return analyzer.nodes, analyzer.call_relationships From 301eb34741299b72b51421bfbbf249e06276a8dd Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:32 +0000 Subject: [PATCH 6/7] fix(DUP-001): 10 review findings across 7 files --- .../dependency_analyzer/analyzers/javascript.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/codewiki/src/be/dependency_analyzer/analyzers/javascript.py b/codewiki/src/be/dependency_analyzer/analyzers/javascript.py index 3210d869..052a03a5 100644 --- a/codewiki/src/be/dependency_analyzer/analyzers/javascript.py +++ b/codewiki/src/be/dependency_analyzer/analyzers/javascript.py @@ -698,7 +698,16 @@ def _extract_assignment_name(self, node) -> Optional[str]: def analyze_javascript_file_treesitter( file_path: str, content: str, repo_path: str = None ) -> Tuple[List[Node], List[CallRelationship]]: - """Analyze a JavaScript file using tree-sitter.""" + """Analyze a JavaScript file using tree-sitter. + + Thin wrapper around `TreeSitterJSAnalyzer`, structurally identical to + `analyze_typescript_file_treesitter` in the sibling `typescript` module + (both build an analyzer, run `.analyze()`, and return its nodes and + relationships). Kept file-local because the analyzer classes themselves + (`TreeSitterJSAnalyzer` vs. the TypeScript analyzer) differ and are not + currently unified behind a shared interface; extracting a shared wrapper + would require introducing that shared interface first. + """ try: logger.debug(f"Tree-sitter JS analysis for {file_path}") analyzer = TreeSitterJSAnalyzer(file_path, content, repo_path) @@ -710,7 +719,3 @@ def analyze_javascript_file_treesitter( except Exception as e: logger.error(f"Error in tree-sitter JS analysis for {file_path}: {e}", exc_info=True) return [], [] - - - - From 978e6cfb6488fa7bd3d34dbea11d62294dc84301 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 05:05:33 +0000 Subject: [PATCH 7/7] fix(DUP-001): 10 review findings across 7 files --- codewiki/src/be/dependency_analyzer/analyzers/typescript.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/codewiki/src/be/dependency_analyzer/analyzers/typescript.py b/codewiki/src/be/dependency_analyzer/analyzers/typescript.py index 3da9d2df..fdde8268 100644 --- a/codewiki/src/be/dependency_analyzer/analyzers/typescript.py +++ b/codewiki/src/be/dependency_analyzer/analyzers/typescript.py @@ -22,6 +22,7 @@ import tree_sitter_typescript from codewiki.src.be.dependency_analyzer.models.core import Node, CallRelationship +from codewiki.src.be.dependency_analyzer.analyzers.javascript import analyze_javascript_file_treesitter logger = logging.getLogger(__name__) @@ -992,3 +993,4 @@ def analyze_typescript_file_treesitter( except Exception as e: logger.error(f"Error in tree-sitter TS analysis for {file_path}: {e}", exc_info=True) return [], [] +