chore: nettoyage .gitignore - retrait venv, __pycache__, .env, data, logs, modèles du tracking

This commit is contained in:
Lowei 2026-06-16 20:40:37 +02:00
parent 7333a22bcd
commit 604a5affe0
9231 changed files with 1244 additions and 1601201 deletions

View file

@ -22,7 +22,6 @@ class LLMManager:
self.model_name = model_name
self.base_url = base_url or os.environ.get("OLLAMA_BASE_URL", "http://localhost:11434")
self.temperature = temperature
self._lock = asyncio.Lock()
def build_payload(
self,
@ -51,60 +50,59 @@ class LLMManager:
async def call_llama(self, payload: Dict[str, Any]) -> str:
"""Call Ollama Generate API and stream response."""
async with self._lock:
try:
# Automatic payload conversion for legacy compatibility (e.g. from moderation.py)
ollama_payload = {
"model": payload.get("model", self.model_name),
"prompt": payload["prompt"],
"system": payload.get("system", payload.get("system_prompt", "Tu es une IA serviable.")),
"stream": True, # Streaming pour voir la réponse en direct
"think": False, # Désactive le reasoning GPT-OSS
"format": "",
"options": {
"temperature": payload.get("temperature", self.temperature),
"num_ctx": payload.get("n_ctx", 16384),
"num_predict": payload.get("max_tokens") or payload.get("num_predict", 8192),
"repeat_penalty": payload.get("repeat_penalty") or payload.get("repeat_last_n", 1.2),
"stop": payload.get("stop", [])
}
try:
# Automatic payload conversion for legacy compatibility (e.g. from moderation.py)
ollama_payload = {
"model": payload.get("model", self.model_name),
"prompt": payload["prompt"],
"system": payload.get("system", payload.get("system_prompt", "Tu es une IA serviable.")),
"stream": True, # Streaming pour voir la réponse en direct
"think": False, # Désactive le reasoning GPT-OSS
"format": "",
"options": {
"temperature": payload.get("temperature", self.temperature),
"num_ctx": payload.get("n_ctx", 16384),
"num_predict": payload.get("max_tokens") or payload.get("num_predict", 8192),
"repeat_penalty": payload.get("repeat_penalty") or payload.get("repeat_last_n", 1.2),
"stop": payload.get("stop", [])
}
}
accumulated_reply = ""
api_url = f"{self.base_url}/api/generate"
accumulated_reply = ""
api_url = f"{self.base_url}/api/generate"
# Build headers with optional API key
headers = {}
api_key = os.environ.get("OLLAMA_API_KEY", "")
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
async with aiohttp.ClientSession() as session:
async with session.post(api_url, json=ollama_payload, headers=headers) as response:
if response.status != 200:
err_text = await response.text()
logger.error(f"Ollama API Error ({response.status}): {err_text}")
return ""
# Streaming en direct dans le terminal
async for line in response.content:
if line:
try:
chunk = json.loads(line)
resp = chunk.get("response", "")
if resp:
accumulated_reply += resp
print(resp, end="", flush=True)
if chunk.get("done"):
break
except json.JSONDecodeError:
continue
print()
return accumulated_reply.strip()
# Build headers with optional API key
headers = {}
api_key = os.environ.get("OLLAMA_API_KEY", "")
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
async with aiohttp.ClientSession() as session:
async with session.post(api_url, json=ollama_payload, headers=headers) as response:
if response.status != 200:
err_text = await response.text()
logger.error(f"Ollama API Error ({response.status}): {err_text}")
return ""
# Streaming en direct dans le terminal
async for line in response.content:
if line:
try:
chunk = json.loads(line)
resp = chunk.get("response", "")
if resp:
accumulated_reply += resp
print(resp, end="", flush=True)
if chunk.get("done"):
break
except json.JSONDecodeError:
continue
print()
return accumulated_reply.strip()
except Exception as e:
logger.error(f"Error calling Ollama: {e}")
return ""
except Exception as e:
logger.error(f"Error calling Ollama: {e}")
return ""
async def close_session(self):
pass