-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathmain.py
More file actions
141 lines (106 loc) · 3.94 KB
/
Copy pathmain.py
File metadata and controls
141 lines (106 loc) · 3.94 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
import time
import logging
import os
os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE"
import xgboost
# Ensure torch and transformers (and Triton) are loaded BEFORE TensorFlow/Keras
# to prevent native library symbol conflicts on WSL/CUDA.
# Skip entirely on macOS — importing torch loads native OpenMP libs that
# cannot be unloaded and cause a segfault when XGBoost loads its own copy.
import platform
if platform.system() != "Darwin":
try:
import torch
if torch.cuda.is_available():
try:
import triton # type: ignore
except ImportError:
pass
import transformers
except ImportError:
pass
from contextlib import asynccontextmanager
from fastapi import FastAPI
# Add agent-orchestration to sys.path so that conversation.*, orchestrator.*,
# agents.*, schemas.*, and location.* are importable as root-level packages.
import sys
from pathlib import Path
_AGENT_DIR = Path(__file__).resolve().parents[1] / "agent-orchestration"
if str(_AGENT_DIR) not in sys.path:
sys.path.insert(0, str(_AGENT_DIR))
from backend.api.routes.orca import router as orca_router
from backend.api.routes.alerts import router as alerts_router
from backend.core.config import MODEL_BACKEND
from backend.services.orca_service import OrcaService, OrcaSessionStore
logger = logging.getLogger(__name__)
_startup_time: float = 0.0
_total_requests: int = 0
@asynccontextmanager
async def lifespan(app: FastAPI):
global _startup_time
_startup_time = time.time()
logger.info("STARTUP: entering lifespan")
logger.info("STARTUP: loading Qwen model (auto-detecting backend)...")
from conversation.model import get_conversation_model
model = get_conversation_model()
backend_name = getattr(model, "backend", "cuda")
logger.info(f"STARTUP: Qwen loaded (backend={backend_name})")
logger.info("STARTUP: Warming up Qwen model caches...")
try:
from conversation.prompts import EXTRACTION_SYSTEM_PROMPT, RESPONSE_SYSTEM_PROMPT_TEMPLATE
model.extract(EXTRACTION_SYSTEM_PROMPT, "warmup query test")
model.generate_text(RESPONSE_SYSTEM_PROMPT_TEMPLATE.format(language_desc="English."), "warmup")
logger.info("STARTUP: Qwen warmup complete")
except Exception as e:
logger.error(f"STARTUP: Failed to warm up Qwen model: {e}")
logger.info("STARTUP: creating engine")
from backend.dependencies.engine import create_orca_engine
engine = create_orca_engine()
logger.info("STARTUP: engine created")
sessions = OrcaSessionStore()
app.state.orca_model = model
app.state.orca_engine = engine
app.state.orca_service = OrcaService(
model,
engine,
sessions,
)
logger.info("STARTUP: service ready, yielding to FastAPI")
yield
logger.info("SHUTDOWN: cleaning up")
if MODEL_BACKEND == "cuda":
try:
import torch
if torch.cuda.is_available():
torch.cuda.empty_cache()
except Exception:
pass
app = FastAPI(
title="ORCA API",
description="Marine Ecosystem Reasoning with Collaborative Agents",
version="0.1.0",
lifespan=lifespan,
)
from fastapi.middleware.cors import CORSMiddleware
from backend.core.config import settings
if settings.ORCA_COMMAND_CENTER_ENABLED:
app.add_middleware(
CORSMiddleware,
allow_origins=[settings.ORCA_COMMAND_CENTER_ORIGIN],
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
)
@app.get("/api/v1/health")
def health_check():
import backend.api.dependencies.rate_limit as rl
uptime = round(time.time() - _startup_time, 1) if _startup_time else 0.0
return {
"status": "ok",
"service": "ORCA API",
"model_backend": MODEL_BACKEND,
"uptime_seconds": uptime,
"active_concurrent_requests": rl._active_requests,
}
app.include_router(orca_router)
app.include_router(alerts_router)