server
server
¶
OpenAI-compatible GLiNER2 server for notebook and development use.
Classes:
| Name | Description |
|---|---|
DetectParams |
Inference settings used to group compatible concurrent requests. |
DetectJob |
One queued detection request. |
BatchDetector |
Coalesce requests and serialize model access through one worker. |
Functions:
| Name | Description |
|---|---|
lifespan |
Load the model and own the batching worker for the server lifespan. |
list_models |
Return readiness and immutable checkpoint identity. |
chat_completions |
Run GLiNER2 through the detector chat-completion contract. |
main |
Run the lightweight notebook/development server. |
DetectParams(labels, threshold, chunk_length, overlap, flat_ner, inference_batch_size)
dataclass
¶
Inference settings used to group compatible concurrent requests.
DetectJob(text, params, future)
dataclass
¶
One queued detection request.
BatchDetector(*, max_requests=32, wait_seconds=0.01)
¶
Coalesce requests and serialize model access through one worker.
Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
def __init__(self, *, max_requests: int = 32, wait_seconds: float = 0.01) -> None:
self._max_requests = max_requests
self._wait_seconds = wait_seconds
self._queue: asyncio.Queue[DetectJob | None] = asyncio.Queue()
self._executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="gliner2-infer")
self._worker_task: asyncio.Task[None] | None = None
lifespan(_)
async
¶
Load the model and own the batching worker for the server lifespan.
Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@asynccontextmanager
async def lifespan(_: FastAPI) -> AsyncIterator[None]:
"""Load the model and own the batching worker for the server lifespan."""
global model, selected_device
selected_device = resolve_device(os.getenv(DEVICE_ENV, "auto"))
logger.info("Loading %s at revision %s on %s", MODEL_ID, MODEL_REVISION, selected_device)
model = await asyncio.to_thread(load_model, selected_device)
detector.start()
logger.info("Local GLiNER2 is ready")
try:
yield
finally:
await detector.stop()
list_models(request)
¶
Return readiness and immutable checkpoint identity.
Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@app.get("/v1/models")
def list_models(request: Request) -> dict[str, Any]:
"""Return readiness and immutable checkpoint identity."""
_authorize(request)
return {
"object": "list",
"data": [
{
"id": MODEL_ID,
"object": "model",
"revision": MODEL_REVISION,
"device": selected_device,
}
],
}
chat_completions(request)
async
¶
Run GLiNER2 through the detector chat-completion contract.
Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@app.post("/v1/chat/completions")
async def chat_completions(request: Request) -> dict[str, Any]:
"""Run GLiNER2 through the detector chat-completion contract."""
_authorize(request)
if model is None:
raise HTTPException(status_code=503, detail="GLiNER2 is not loaded")
body = await request.json()
text = _extract_text(body.get("messages", []))
labels = body.get("labels") or []
params = DetectParams(
labels=tuple(str(label) for label in labels),
threshold=float(body.get("threshold", 0.3)),
chunk_length=int(body.get("chunk_length", DEFAULT_CHUNK_LENGTH)),
overlap=int(body.get("overlap", DEFAULT_OVERLAP)),
flat_ner=bool(body.get("flat_ner", DEFAULT_FLAT_NER)),
inference_batch_size=int(body.get("batch_size", DEFAULT_INFERENCE_BATCH_SIZE)),
)
try:
entities = await detector.detect(text, params)
except ValueError as exc:
raise HTTPException(status_code=422, detail=str(exc)) from exc
content = json.dumps({"entities": entities})
return {
"id": f"chatcmpl-{uuid.uuid4().hex[:12]}",
"object": "chat.completion",
"created": int(time.time()),
"model": body.get("model", MODEL_ID),
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": content},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
}
main()
¶
Run the lightweight notebook/development server.
Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
def main() -> None:
"""Run the lightweight notebook/development server."""
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
parser = argparse.ArgumentParser(description="Notebook/development GLiNER2 server for Anonymizer.")
parser.add_argument("--host", default=DEFAULT_HOST)
parser.add_argument("--port", type=int, default=DEFAULT_PORT)
parser.add_argument("--fd", type=int, default=None, help=argparse.SUPPRESS)
args = parser.parse_args()
if args.fd is None:
uvicorn.run(app, host=args.host, port=args.port)
return
with socket.socket(fileno=args.fd) as inherited_listener:
config = uvicorn.Config(app)
uvicorn.Server(config).run(sockets=[inherited_listener])