Skip to content

server

server

OpenAI-compatible GLiNER2 server for notebook and development use.

Classes:

Name Description
DetectParams

Inference settings used to group compatible concurrent requests.

DetectJob

One queued detection request.

BatchDetector

Coalesce requests and serialize model access through one worker.

Functions:

Name Description
lifespan

Load the model and own the batching worker for the server lifespan.

list_models

Return readiness and immutable checkpoint identity.

chat_completions

Run GLiNER2 through the detector chat-completion contract.

main

Run the lightweight notebook/development server.

DetectParams(labels, threshold, chunk_length, overlap, flat_ner, inference_batch_size) dataclass

Inference settings used to group compatible concurrent requests.

DetectJob(text, params, future) dataclass

One queued detection request.

BatchDetector(*, max_requests=32, wait_seconds=0.01)

Coalesce requests and serialize model access through one worker.

Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
def __init__(self, *, max_requests: int = 32, wait_seconds: float = 0.01) -> None:
    self._max_requests = max_requests
    self._wait_seconds = wait_seconds
    self._queue: asyncio.Queue[DetectJob | None] = asyncio.Queue()
    self._executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="gliner2-infer")
    self._worker_task: asyncio.Task[None] | None = None

lifespan(_) async

Load the model and own the batching worker for the server lifespan.

Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@asynccontextmanager
async def lifespan(_: FastAPI) -> AsyncIterator[None]:
    """Load the model and own the batching worker for the server lifespan."""
    global model, selected_device
    selected_device = resolve_device(os.getenv(DEVICE_ENV, "auto"))
    logger.info("Loading %s at revision %s on %s", MODEL_ID, MODEL_REVISION, selected_device)
    model = await asyncio.to_thread(load_model, selected_device)
    detector.start()
    logger.info("Local GLiNER2 is ready")
    try:
        yield
    finally:
        await detector.stop()

list_models(request)

Return readiness and immutable checkpoint identity.

Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@app.get("/v1/models")
def list_models(request: Request) -> dict[str, Any]:
    """Return readiness and immutable checkpoint identity."""
    _authorize(request)
    return {
        "object": "list",
        "data": [
            {
                "id": MODEL_ID,
                "object": "model",
                "revision": MODEL_REVISION,
                "device": selected_device,
            }
        ],
    }

chat_completions(request) async

Run GLiNER2 through the detector chat-completion contract.

Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
@app.post("/v1/chat/completions")
async def chat_completions(request: Request) -> dict[str, Any]:
    """Run GLiNER2 through the detector chat-completion contract."""
    _authorize(request)
    if model is None:
        raise HTTPException(status_code=503, detail="GLiNER2 is not loaded")
    body = await request.json()
    text = _extract_text(body.get("messages", []))
    labels = body.get("labels") or []
    params = DetectParams(
        labels=tuple(str(label) for label in labels),
        threshold=float(body.get("threshold", 0.3)),
        chunk_length=int(body.get("chunk_length", DEFAULT_CHUNK_LENGTH)),
        overlap=int(body.get("overlap", DEFAULT_OVERLAP)),
        flat_ner=bool(body.get("flat_ner", DEFAULT_FLAT_NER)),
        inference_batch_size=int(body.get("batch_size", DEFAULT_INFERENCE_BATCH_SIZE)),
    )
    try:
        entities = await detector.detect(text, params)
    except ValueError as exc:
        raise HTTPException(status_code=422, detail=str(exc)) from exc
    content = json.dumps({"entities": entities})
    return {
        "id": f"chatcmpl-{uuid.uuid4().hex[:12]}",
        "object": "chat.completion",
        "created": int(time.time()),
        "model": body.get("model", MODEL_ID),
        "choices": [
            {
                "index": 0,
                "message": {"role": "assistant", "content": content},
                "finish_reason": "stop",
            }
        ],
        "usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
    }

main()

Run the lightweight notebook/development server.

Source code in src/anonymizer/notebooks/local_inference/gliner2/server.py
def main() -> None:
    """Run the lightweight notebook/development server."""
    logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
    parser = argparse.ArgumentParser(description="Notebook/development GLiNER2 server for Anonymizer.")
    parser.add_argument("--host", default=DEFAULT_HOST)
    parser.add_argument("--port", type=int, default=DEFAULT_PORT)
    parser.add_argument("--fd", type=int, default=None, help=argparse.SUPPRESS)
    args = parser.parse_args()
    if args.fd is None:
        uvicorn.run(app, host=args.host, port=args.port)
        return
    with socket.socket(fileno=args.fd) as inherited_listener:
        config = uvicorn.Config(app)
        uvicorn.Server(config).run(sockets=[inherited_listener])