feat: support Gemma 4 chat and agent trials

This commit is contained in:
Hamza Ayed
2026-09-29 23:19:25 +03:00
parent 18d17eba50
commit 874c7152d1
3 changed files with 28 additions and 4 deletions
+12 -2
View File
@@ -47,6 +47,10 @@ class ChatRequest(BaseModel):
class AgentRequest(BaseModel):
task: str = Field(description="مهمة قصيرة للوكيل المحلي")
model: str | None = Field(
default=None,
description="اسم نموذج Ollama؛ اتركه فارغًا لاستخدام النموذج الافتراضي",
)
class StoredMessage(BaseModel):
@@ -106,7 +110,7 @@ def get_local_user() -> dict[str, str]:
def chat_payload(request: ChatRequest, *, stream: bool) -> dict[str, Any]:
model = request.model or os.getenv("LOCAL_MODEL", "qwen2.5:1.5b-instruct-q4_K_M")
return {
payload: dict[str, Any] = {
"model": model,
"messages": (
[message.model_dump() for message in request.messages]
@@ -130,6 +134,10 @@ def chat_payload(request: ChatRequest, *, stream: bool) -> dict[str, Any]:
),
"stream": stream,
}
# Disable Gemma 4 thinking so Ollama places the answer in `content` for clients.
if model.lower().startswith("gemma4"):
payload["reasoning_effort"] = "none"
return payload
@app.post("/v1/chat/completions")
@@ -283,7 +291,7 @@ async def run_agent(request: AgentRequest) -> dict[str, Any]:
pass
base_url = os.getenv("LOCAL_LLM_BASE_URL", "http://127.0.0.1:11434/v1").rstrip("/")
model = os.getenv("LOCAL_MODEL", "qwen2.5:1.5b-instruct-q4_K_M")
model = request.model or os.getenv("LOCAL_MODEL", "qwen2.5:1.5b-instruct-q4_K_M")
payload = {
"model": model,
"messages": [
@@ -292,6 +300,8 @@ async def run_agent(request: AgentRequest) -> dict[str, Any]:
],
"stream": False,
}
if model.lower().startswith("gemma4"):
payload["reasoning_effort"] = "none"
completion = await get_completion(payload, base_url)
return {
"task": request.task,