chore: nettoyage .gitignore - retrait venv, __pycache__, .env, data, logs, modèles du tracking
This commit is contained in:
parent
7333a22bcd
commit
604a5affe0
9231 changed files with 1244 additions and 1601201 deletions
102
core/llm.py
102
core/llm.py
|
|
@ -22,7 +22,6 @@ class LLMManager:
|
|||
self.model_name = model_name
|
||||
self.base_url = base_url or os.environ.get("OLLAMA_BASE_URL", "http://localhost:11434")
|
||||
self.temperature = temperature
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
def build_payload(
|
||||
self,
|
||||
|
|
@ -51,60 +50,59 @@ class LLMManager:
|
|||
|
||||
async def call_llama(self, payload: Dict[str, Any]) -> str:
|
||||
"""Call Ollama Generate API and stream response."""
|
||||
async with self._lock:
|
||||
try:
|
||||
# Automatic payload conversion for legacy compatibility (e.g. from moderation.py)
|
||||
ollama_payload = {
|
||||
"model": payload.get("model", self.model_name),
|
||||
"prompt": payload["prompt"],
|
||||
"system": payload.get("system", payload.get("system_prompt", "Tu es une IA serviable.")),
|
||||
"stream": True, # Streaming pour voir la réponse en direct
|
||||
"think": False, # Désactive le reasoning GPT-OSS
|
||||
"format": "",
|
||||
"options": {
|
||||
"temperature": payload.get("temperature", self.temperature),
|
||||
"num_ctx": payload.get("n_ctx", 16384),
|
||||
"num_predict": payload.get("max_tokens") or payload.get("num_predict", 8192),
|
||||
"repeat_penalty": payload.get("repeat_penalty") or payload.get("repeat_last_n", 1.2),
|
||||
"stop": payload.get("stop", [])
|
||||
}
|
||||
try:
|
||||
# Automatic payload conversion for legacy compatibility (e.g. from moderation.py)
|
||||
ollama_payload = {
|
||||
"model": payload.get("model", self.model_name),
|
||||
"prompt": payload["prompt"],
|
||||
"system": payload.get("system", payload.get("system_prompt", "Tu es une IA serviable.")),
|
||||
"stream": True, # Streaming pour voir la réponse en direct
|
||||
"think": False, # Désactive le reasoning GPT-OSS
|
||||
"format": "",
|
||||
"options": {
|
||||
"temperature": payload.get("temperature", self.temperature),
|
||||
"num_ctx": payload.get("n_ctx", 16384),
|
||||
"num_predict": payload.get("max_tokens") or payload.get("num_predict", 8192),
|
||||
"repeat_penalty": payload.get("repeat_penalty") or payload.get("repeat_last_n", 1.2),
|
||||
"stop": payload.get("stop", [])
|
||||
}
|
||||
}
|
||||
|
||||
accumulated_reply = ""
|
||||
api_url = f"{self.base_url}/api/generate"
|
||||
accumulated_reply = ""
|
||||
api_url = f"{self.base_url}/api/generate"
|
||||
|
||||
# Build headers with optional API key
|
||||
headers = {}
|
||||
api_key = os.environ.get("OLLAMA_API_KEY", "")
|
||||
if api_key:
|
||||
headers["Authorization"] = f"Bearer {api_key}"
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.post(api_url, json=ollama_payload, headers=headers) as response:
|
||||
if response.status != 200:
|
||||
err_text = await response.text()
|
||||
logger.error(f"Ollama API Error ({response.status}): {err_text}")
|
||||
return ""
|
||||
|
||||
# Streaming en direct dans le terminal
|
||||
async for line in response.content:
|
||||
if line:
|
||||
try:
|
||||
chunk = json.loads(line)
|
||||
resp = chunk.get("response", "")
|
||||
if resp:
|
||||
accumulated_reply += resp
|
||||
print(resp, end="", flush=True)
|
||||
if chunk.get("done"):
|
||||
break
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
print()
|
||||
return accumulated_reply.strip()
|
||||
|
||||
# Build headers with optional API key
|
||||
headers = {}
|
||||
api_key = os.environ.get("OLLAMA_API_KEY", "")
|
||||
if api_key:
|
||||
headers["Authorization"] = f"Bearer {api_key}"
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.post(api_url, json=ollama_payload, headers=headers) as response:
|
||||
if response.status != 200:
|
||||
err_text = await response.text()
|
||||
logger.error(f"Ollama API Error ({response.status}): {err_text}")
|
||||
return ""
|
||||
|
||||
# Streaming en direct dans le terminal
|
||||
async for line in response.content:
|
||||
if line:
|
||||
try:
|
||||
chunk = json.loads(line)
|
||||
resp = chunk.get("response", "")
|
||||
if resp:
|
||||
accumulated_reply += resp
|
||||
print(resp, end="", flush=True)
|
||||
if chunk.get("done"):
|
||||
break
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
print()
|
||||
return accumulated_reply.strip()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error calling Ollama: {e}")
|
||||
return ""
|
||||
except Exception as e:
|
||||
logger.error(f"Error calling Ollama: {e}")
|
||||
return ""
|
||||
|
||||
async def close_session(self):
|
||||
pass
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue