Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -27,8 +27,19 @@ OLLAMA_PORT=11434
OLLAMA_BASE_URL=http://localhost:11434
# Modelos recomendados no PRD (Qwen 2.5 / Llama 3.1 para LLM, bge-m3 para embeddings)
OLLAMA_LLM_MODEL=qwen2.5:1.5b
OLLAMA_NUM_PREDICT=1024
OLLAMA_NUM_CTX=8192
OLLAMA_FILE_NUM_PREDICT=512
OLLAMA_SYNTHESIS_NUM_PREDICT=3072
OLLAMA_EMBEDDING_MODEL=bge-m3
OLLAMA_KEEP_ALIVE=24h
# URL do serviço RepoAnalyzer quando o backend roda em Docker Compose.
REPO_ANALYZER_URL=http://ai-service:8000
# Limite de arquivos priorizados nos perfis do RepoAnalyzer (Complete usa MAX_FILES).
ANALYZER_QUICK_FILES=8
ANALYZER_BALANCED_FILES=80
# Limite rígido de arquivos elegíveis inventariados pelo RepoAnalyzer.
MAX_FILES=1000

# --- Documentos enviados (S1-19/S1-22) ---
# Limite de tamanho por arquivo, aplicado no backend e informado ao cliente
Expand Down
3 changes: 2 additions & 1 deletion ai-service/Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,10 @@ FROM python:3.11-slim

WORKDIR /app

# Install system dependencies
# Git is required by the repository analyzer to clone public GitHub projects.
RUN apt-get update && apt-get install -y --no-install-recommends \
curl \
git \
&& rm -rf /var/lib/apt/lists/*

COPY requirements.txt .
Expand Down
6 changes: 6 additions & 0 deletions ai-service/analyzer/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,9 +15,15 @@ class AnalyzerSettings:
max_file_size_kb: int = int(os.getenv("MAX_FILE_SIZE_KB", "500"))
max_chunk_chars: int = int(os.getenv("MAX_CHUNK_CHARS", "10000"))
max_files: int = int(os.getenv("MAX_FILES", "1000"))
quick_profile_files: int = int(os.getenv("ANALYZER_QUICK_FILES", "8"))
balanced_profile_files: int = int(os.getenv("ANALYZER_BALANCED_FILES", "80"))
workspace_dir: Path = Path(os.getenv("WORKSPACE_DIR", "workspace_analyzer"))
request_timeout_seconds: int = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "300"))
ollama_keep_alive: str = os.getenv("OLLAMA_KEEP_ALIVE", "5m")
ollama_num_predict: int = max(128, int(os.getenv("OLLAMA_NUM_PREDICT", "1024")))
ollama_num_ctx: int = max(2048, int(os.getenv("OLLAMA_NUM_CTX", "8192")))
ollama_file_num_predict: int = max(128, int(os.getenv("OLLAMA_FILE_NUM_PREDICT", "512")))
ollama_synthesis_num_predict: int = max(128, int(os.getenv("OLLAMA_SYNTHESIS_NUM_PREDICT", "3072")))
ollama_think: bool = False
ollama_max_retries: int = int(os.getenv("OLLAMA_MAX_RETRIES", "2"))
ollama_retry_backoff_seconds: float = float(os.getenv("OLLAMA_RETRY_BACKOFF_SECONDS", "3"))
Expand Down
11 changes: 11 additions & 0 deletions ai-service/analyzer/models.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
from dataclasses import dataclass, field
import threading
from typing import Any


Expand All @@ -20,6 +21,7 @@ class RunState:
run_id: str
url: str
status: str = "created"
profile: str = "quick"
stage: str = ""
message: str = ""
error: str = ""
Expand All @@ -32,7 +34,16 @@ class RunState:
files_total: int = 0
files_processed: int = 0
files_ignored: int = 0
files_candidates: int = 0
files_skipped_by_scope: int = 0
language_counts: dict[str, int] = field(default_factory=dict)
current_files: list[str] = field(default_factory=list)
selected_paths: list[str] = field(default_factory=list)
completed_summaries: dict[str, str] = field(default_factory=dict)
partial_chunk_summaries: dict[str, list[str]] = field(default_factory=dict)
project_summary: str = ""
pause_event: Any = field(default_factory=threading.Event, repr=False)
cancel_event: Any = field(default_factory=threading.Event, repr=False)
worker_thread: Any = field(default=None, repr=False)

stats: dict[str, Any] = field(default_factory=dict)
67 changes: 58 additions & 9 deletions ai-service/analyzer/ollama_client.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
import re
import time
import json
from collections.abc import Callable

import httpx

Expand Down Expand Up @@ -30,6 +32,8 @@ def __init__(
think: bool = False,
max_retries: int = 2,
retry_backoff_seconds: float = 3.0,
num_predict: int = 1024,
num_ctx: int = 8192,
):
self.base_url = base_url.rstrip("/")
self.model = model
Expand All @@ -38,40 +42,85 @@ def __init__(
self.think = False
self.max_retries = max(0, max_retries)
self.retry_backoff_seconds = retry_backoff_seconds
self.num_predict = max(128, num_predict)
self.num_ctx = max(2048, num_ctx)
self._client = httpx.Client(timeout=timeout)

def close(self) -> None:
self._client.close()

def chat(self, system: str, user: str, temperature: float = 0.1) -> str:
def chat(
self,
system: str,
user: str,
temperature: float = 0.1,
should_cancel: Callable[[], None] | None = None,
num_predict: int | None = None,
accept_truncated: bool = False,
) -> str:
if not system.strip().startswith(("/nothink", "/no_think")):
system = f"/nothink\n{system}"

output_limit = max(128, num_predict or self.num_predict)
payload = {
"model": self.model,
"messages": [
{"role": "system", "content": system},
{"role": "user", "content": user},
],
"stream": False,
"stream": True,
"keep_alive": self.keep_alive,
"think": False,
"options": {
"temperature": temperature
"temperature": temperature,
"num_predict": output_limit,
"num_ctx": self.num_ctx,
}
}

last_error: Exception | None = None
for attempt in range(self.max_retries + 1):
output_limit_retries = 0
http_retries = 0
while True:
try:
response = self._client.post(f"{self.base_url}/api/chat", json=payload)
response.raise_for_status()
data = response.json()
content_parts: list[str] = []
done_reason = ""
with self._client.stream("POST", f"{self.base_url}/api/chat", json=payload) as response:
response.raise_for_status()
for line in response.iter_lines():
if should_cancel:
should_cancel()
if not line:
continue
data = json.loads(line)
content = data.get("message", {}).get("content", "")
if content:
content_parts.append(content)
if data.get("done"):
done_reason = data.get("done_reason", "")
break
if done_reason == "length":
if accept_truncated:
data = {"message": {"content": "".join(content_parts) +
"\n\n[Análise resumida limitada pelo teto de tokens; trate este trecho como parcial.]"}}
break
if output_limit_retries < 4 and output_limit < 8192:
output_limit = min(8192, output_limit * 2)
output_limit_retries += 1
http_retries = 0
payload["options"]["num_predict"] = output_limit
continue
raise OllamaError(
f"A resposta do modelo atingiu o limite de {output_limit} tokens. "
"Aumente OLLAMA_FILE_NUM_PREDICT ou OLLAMA_SYNTHESIS_NUM_PREDICT."
)
data = {"message": {"content": "".join(content_parts)}}
break
except httpx.HTTPError as exc:
last_error = exc
if attempt < self.max_retries:
time.sleep(self.retry_backoff_seconds * (attempt + 1))
if http_retries < self.max_retries:
http_retries += 1
time.sleep(self.retry_backoff_seconds * http_retries)
continue
raise OllamaError(
f"Não foi possível acessar o Ollama em {self.base_url} "
Expand Down
Loading
Loading