From eace553b81ab801f73632edf6c5e62dac044c6b5 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 06:46:34 +0530 Subject: [PATCH 01/16] feat: add TypeSafe API compatibility --- .github/workflows/publish.yml | 8 +- README.md | 43 +- app/main.py | 220 +++--- openapi.json | 1245 ++++++--------------------------- 4 files changed, 331 insertions(+), 1185 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index e4fa568..b9bcba3 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -32,7 +32,7 @@ jobs: - name: Validate OpenAPI schema run: | - uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/predict' in s['paths'] and '/predict/bulk' in s['paths']; print('openapi ok:', list(s['paths']))" + uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' not in s['paths']; print('openapi ok:', list(s['paths']))" # Pull requests: build the image once and check it boots. test: @@ -54,7 +54,7 @@ jobs: - name: Container smoke test run: | - docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/predict' in s['paths']; print('container ok')" + docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths']; print('container ok')" # Nightly / manual / release: real model download + inference. smoke: @@ -82,9 +82,9 @@ jobs: sleep 2 done curl -sf http://localhost:8000/healthz - curl -sf -X POST http://localhost:8000/predict \ + curl -sf -X POST http://localhost:8000/v1/systemone \ -H 'X-API-Key: test' -H 'Content-Type: application/json' \ - -d '{"state":"I was billed twice. Please refund.","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}' + -d '{"state":"I was billed twice. Please refund.","model":"laya-english","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}' docker logs laya-api docker rm -f laya-api diff --git a/README.md b/README.md index c33e592..dbb0ece 100644 --- a/README.md +++ b/README.md @@ -22,12 +22,6 @@ Built and published for `linux/amd64` and `linux/arm64`. HTTP Basic, compared in constant time. - 📑 **Interactive OpenAPI Docs**: Swagger UI (`/docs`), ReDoc (`/redoc`) and the raw schema at `/openapi.json`. -- 🧰 **Built-in Presets**: ready-made question sets (`triage`, `email`, `guard`, - `moderation`, `router`). -- 📦 **Bulk Inference**: `/predict/bulk` over many states, with per-state - questions/model and isolated errors. -- 🔎 **Detection & Email Helpers**: `/detect` (script/language) and - `/email/state` (clean + structure an email). - 🧩 **Flexible State**: string, JSON object, or conversation turns; criteria values may be any JSON (dicts/lists/numbers are rendered as compact JSON). - 🛡️ **Non-Root**: runs as unprivileged `appuser` (uid `10001`). @@ -70,12 +64,7 @@ curl http://localhost:8000/healthz | Method | Path | Auth | Description | | --- | --- | --- | --- | | `GET` | `/healthz` | no | Liveness + resident models | -| `GET` | `/models` | yes* | Available and loaded checkpoints | -| `GET` | `/presets` | yes* | Built-in question sets | -| `POST` | `/detect` | yes* | Script/language detection (what routing uses) | -| `POST` | `/email/state` | yes* | Clean + structure an email as a state | -| `POST` | `/predict` | yes* | Typed questions over one state | -| `POST` | `/predict/bulk` | yes* | Same questions over many states | +| `POST` | `/v1/systemone` | yes* | TypeSafe-compatible typed prediction | \* Enforced only when `API_KEYS` and/or `BASIC_AUTH` is set. Any of these works: @@ -87,11 +76,12 @@ curl http://localhost:8000/healthz --- -## 🧠 Predicting +## 🧠 TypeSafe Prediction ```bash -curl -X POST localhost:8000/predict -H 'X-API-Key: key1' -H 'Content-Type: application/json' -d '{ +curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Content-Type: application/json' -d '{ "state": "I was billed twice. Please refund the duplicate today.", + "model": "laya-english", "questions": { "department": {"type": "choice", "instructions": "Which team?", "criteria": {"billing": "refunds", "technical": "bugs", "sales": "purchases"}}, "urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["not urgent", "soon", "critical"]}, @@ -102,14 +92,13 @@ curl -X POST localhost:8000/predict -H 'X-API-Key: key1' -H 'Content-Type: appli ```json { - "model": "laya-rl-agent", + "model": "laya-english", "answers": { "department": {"type": "choice", "choice": "billing", "probabilities": {"billing": 0.96, "technical": 0.02, "sales": 0.02}, "confidence": 0.82}, "urgency": {"type": "score", "score": 1.36, "legend": {"0": "not urgent", "1": "soon", "2": "critical"}, "confidence": 0.09}, - "refund": {"type": "noul", "noul": 0.82, "confidence": 0.82} + "refund": {"type": "noul", "noul": 0.82} }, - "usage": {"input_tokens": 132, "output_tokens": 0}, - "routing": {"model": "english", "reason": "English Latin text"} + "usage": {"input_tokens": 132, "output_tokens": 0} } ``` @@ -117,16 +106,17 @@ curl -X POST localhost:8000/predict -H 'X-API-Key: key1' -H 'Content-Type: appli values may be strings or any JSON value (dicts/lists/numbers are rendered as compact JSON), and `noul` accepts optional `{"true": ..., "false": ...}` text. -**Routing** — omit `model` to auto-select by language (see `/detect`), or pin -`"model": "english" | "multilingual" | "typed-decisions"`. The response includes -`routing` with the chosen checkpoint and reason. +### TypeSafe API Compatibility -**Presets** — skip `questions` and pass `"preset": "triage"` (one of `triage`, -`email`, `guard`, `moderation`, `router`); list them at `GET /presets`. +`POST /v1/systemone` accepts the TypeSafe API request shape and supports these +Laya model names: -**Bulk** — use `states` with shared `questions`/`preset`/`model`, or `items` to -override questions and model per state. Returns `{"count": N, "results": [...]}` -with per-state errors isolated as `{"ok": false, "error": "..."}`. +```text +laya-english | laya-multilingual | laya-typed-decisions +``` + +It uses `Authorization: Bearer ` and returns the documented TypeSafe +`model`, `answers`, and `usage` fields. --- @@ -136,7 +126,6 @@ with per-state errors isolated as `{"ok": false, "error": "..."}`. | --- | --- | --- | | `API_KEYS` | *(empty)* | Comma-separated keys; empty disables API-key auth. | | `BASIC_AUTH` | *(empty)* | Comma-separated `user:password` pairs. | -| `MAX_BULK_ITEMS` | `256` | Max states per `/predict/bulk`. | | `PORT` | `8000` | HTTP port (host and container). | | `MODELS` | `english` | Checkpoints to preload: `english`, `multilingual`, `typed-decisions`. | | `MODEL_ID` | `convaiinnovations/laya` | Optional repo override (mirror/local path). | diff --git a/app/main.py b/app/main.py index e37647e..99e8d45 100644 --- a/app/main.py +++ b/app/main.py @@ -37,6 +37,11 @@ AVAILABLE_MODELS = ("english", "multilingual", "typed-decisions") ModelName = Literal["english", "multilingual", "typed-decisions"] +TypeSafeModelName = Literal[ + "laya-english", + "laya-multilingual", + "laya-typed-decisions", +] _state: dict[str, Any] = {} _lock = threading.Lock() @@ -329,6 +334,76 @@ class PredictResponse(BaseModel): ) +TypeSafeInstruction = str | dict[str, Any] | list[Any] + + +class TypeSafeChoiceQuestion(BaseModel): + type: Literal["choice"] + instructions: TypeSafeInstruction + criteria: dict[str, str | dict[str, Any] | list[Any] | None] + + +class TypeSafeScoreQuestion(BaseModel): + type: Literal["score"] + instructions: TypeSafeInstruction + criteria: list[str | dict[str, Any] | list[Any]] + + +class TypeSafeNoulQuestion(BaseModel): + type: Literal["noul"] + instructions: TypeSafeInstruction + criteria: dict[str, str | dict[str, Any] | list[Any] | None] | None = None + + +TypeSafeQuestion = Annotated[ + TypeSafeChoiceQuestion | TypeSafeScoreQuestion | TypeSafeNoulQuestion, + Field(discriminator="type"), +] + + +class TypeSafeRequest(BaseModel): + state: State + model: TypeSafeModelName + questions: dict[str, TypeSafeQuestion] + + +class TypeSafeNoulAnswer(BaseModel): + type: Literal["noul"] + noul: float + + +class TypeSafeChoiceAnswer(BaseModel): + type: Literal["choice"] + choice: str + probabilities: dict[str, float] + confidence: float + + +class TypeSafeScoreAnswer(BaseModel): + type: Literal["score"] + score: float + legend: dict[str, str] + probabilities: dict[str, float] + confidence: float + + +TypeSafeAnswer = Annotated[ + TypeSafeNoulAnswer | TypeSafeChoiceAnswer | TypeSafeScoreAnswer, + Field(discriminator="type"), +] + + +class TypeSafeUsage(BaseModel): + input_tokens: int + output_tokens: int + + +class TypeSafeResponse(BaseModel): + model: str + answers: dict[str, TypeSafeAnswer] + usage: TypeSafeUsage + + class BulkItemResult(BaseModel): ok: bool result: PredictResponse | None = None @@ -428,6 +503,13 @@ def _resolve( return _dump(questions or {}) +_TYPESAFE_MODELS: dict[TypeSafeModelName, ModelName] = { + "laya-english": "english", + "laya-multilingual": "multilingual", + "laya-typed-decisions": "typed-decisions", +} + + _AUTH_RESPONSES: dict[int | str, dict[str, Any]] = { 401: {"description": "Missing or invalid credentials"}, 503: {"description": "Model not loaded yet"}, @@ -435,7 +517,12 @@ def _resolve( # ------------------------------------------------------------------------- endpoints -@app.get("/healthz", tags=["meta"], response_model=HealthResponse, summary="Health check") +@app.get( + "/healthz", + include_in_schema=False, + response_model=HealthResponse, + summary="Health check", +) def healthz() -> HealthResponse: router = _state.get("router") return HealthResponse( @@ -451,130 +538,29 @@ def healthz() -> HealthResponse: ) -@app.get( - "/models", - tags=["meta"], - response_model=ModelsResponse, - summary="List checkpoints", - dependencies=[Depends(require_auth)], - responses=_AUTH_RESPONSES, -) -def list_models() -> ModelsResponse: - """Available checkpoints and which are currently resident.""" - router = _router() - return ModelsResponse(available=list(AVAILABLE_MODELS), loaded=list(router.loaded)) - - -@app.get( - "/presets", - tags=["meta"], - summary="List built-in question presets", - dependencies=[Depends(require_auth)], - responses=_AUTH_RESPONSES, -) -def list_presets() -> dict[str, dict[str, Any]]: - """Return the ready-to-use question sets (triage/email/guard/moderation/router).""" - return _presets() - - -@app.post( - "/detect", - tags=["meta"], - summary="Detect script and language", - dependencies=[Depends(require_auth)], - responses=_AUTH_RESPONSES, -) -def detect(request: DetectRequest) -> dict[str, Any]: - """Report script, best-effort language and `is_english` for a state (what routing uses).""" - from laya import detect_language - - return detect_language(request.state) - - @app.post( - "/email/state", - tags=["meta"], - summary="Build a clean email state", - dependencies=[Depends(require_auth)], - responses=_AUTH_RESPONSES, -) -def email_state_endpoint(request: EmailStateRequest) -> dict[str, Any]: - """Clean an email body and structure it as a state for `/predict`.""" - from laya import email_state - - return email_state(request.subject, request.body, sender=request.sender, clean=request.clean) - - -@app.post( - "/predict", + "/v1/systemone", tags=["predict"], - response_model=PredictResponse, - response_model_exclude_none=True, - summary="Predict one state", + response_model=TypeSafeResponse, + summary="TypeSafe-compatible prediction", dependencies=[Depends(require_auth)], responses=_AUTH_RESPONSES, ) -def predict(request: PredictRequest) -> PredictResponse: - """Run typed questions over a single state and return calibrated answers.""" +def typesafe_predict(request: TypeSafeRequest) -> TypeSafeResponse: + """Evaluate a TypeSafe-shaped request using a Laya checkpoint.""" router = _router() - questions = _resolve(request.questions, request.preset) with _lock: - result = router.predict(request.state, questions, model=request.model) - return PredictResponse.model_validate(result) - - -@app.post( - "/predict/bulk", - tags=["predict"], - response_model=BulkPredictResponse, - response_model_exclude_none=True, - summary="Predict many states", - dependencies=[Depends(require_auth)], - responses=_AUTH_RESPONSES, -) -def predict_bulk(request: BulkPredictRequest) -> BulkPredictResponse: - """Run typed questions over many states. - - Accepts either `states` with shared `questions`/`preset`, or `items` where each - state can override its questions and model. Laya's public API predicts one state - at a time (its internals can batch, but the public `Agent.predict` cannot), so - this loops. Per-state errors are isolated and returned inline. - """ - if request.items: - jobs = [ - ( - item.state, - _resolve(item.questions or request.questions, request.preset), - item.model or request.model, - ) - for item in request.items - ] - else: - questions = _resolve(request.questions, request.preset) - jobs = [(state, questions, request.model) for state in request.states or []] - - if len(jobs) > MAX_BULK_ITEMS: - raise HTTPException( - status_code=422, - detail=f"Too many states: {len(jobs)} > MAX_BULK_ITEMS={MAX_BULK_ITEMS}", + result = router.predict( + request.state, + _dump(request.questions), + model=_TYPESAFE_MODELS[request.model], ) - - router = _router() - results: list[BulkItemResult] = [] - with _lock: - for state, questions, model in jobs: - try: - results.append( - BulkItemResult( - ok=True, - result=PredictResponse.model_validate( - router.predict(state, questions, model=model) - ), - ) - ) - except Exception as exc: # noqa: BLE001 - report per-item failure - results.append(BulkItemResult(ok=False, error=str(exc))) - return BulkPredictResponse(count=len(results), results=results) + validated = PredictResponse.model_validate(result) + return TypeSafeResponse( + model=request.model, + answers=validated.answers, + usage=validated.usage, + ) # ------------------------------------------------------------------------ openapi diff --git a/openapi.json b/openapi.json index bbdf8dc..8e3200e 100644 --- a/openapi.json +++ b/openapi.json @@ -6,265 +6,19 @@ "version": "0.5.0" }, "paths": { - "/healthz": { - "get": { - "tags": [ - "meta" - ], - "summary": "Health check", - "operationId": "healthz_healthz_get", - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HealthResponse" - } - } - } - } - } - } - }, - "/models": { - "get": { - "tags": [ - "meta" - ], - "summary": "List checkpoints", - "description": "Available checkpoints and which are currently resident.", - "operationId": "list_models_models_get", - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/ModelsResponse" - } - } - } - }, - "401": { - "description": "Missing or invalid credentials" - }, - "503": { - "description": "Model not loaded yet" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - }, - "/presets": { - "get": { - "tags": [ - "meta" - ], - "summary": "List built-in question presets", - "description": "Return the ready-to-use question sets (triage/email/guard/moderation/router).", - "operationId": "list_presets_presets_get", - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "additionalProperties": { - "additionalProperties": true, - "type": "object" - }, - "type": "object", - "title": "Response List Presets Presets Get" - } - } - } - }, - "401": { - "description": "Missing or invalid credentials" - }, - "503": { - "description": "Model not loaded yet" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - }, - "/detect": { - "post": { - "tags": [ - "meta" - ], - "summary": "Detect script and language", - "description": "Report script, best-effort language and `is_english` for a state (what routing uses).", - "operationId": "detect_detect_post", - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/DetectRequest" - } - } - }, - "required": true - }, - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "additionalProperties": true, - "type": "object", - "title": "Response Detect Detect Post" - } - } - } - }, - "401": { - "description": "Missing or invalid credentials" - }, - "503": { - "description": "Model not loaded yet" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - }, - "/email/state": { - "post": { - "tags": [ - "meta" - ], - "summary": "Build a clean email state", - "description": "Clean an email body and structure it as a state for `/predict`.", - "operationId": "email_state_endpoint_email_state_post", - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/EmailStateRequest" - } - } - }, - "required": true - }, - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "additionalProperties": true, - "type": "object", - "title": "Response Email State Endpoint Email State Post" - } - } - } - }, - "401": { - "description": "Missing or invalid credentials" - }, - "503": { - "description": "Model not loaded yet" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - }, - "/predict": { + "/v1/systemone": { "post": { "tags": [ "predict" ], - "summary": "Predict one state", - "description": "Run typed questions over a single state and return calibrated answers.", - "operationId": "predict_predict_post", + "summary": "TypeSafe-compatible prediction", + "description": "Evaluate a TypeSafe-shaped request using a Laya checkpoint.", + "operationId": "typesafe_predict_v1_systemone_post", "requestBody": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/PredictRequest" + "$ref": "#/components/schemas/TypeSafeRequest" } } }, @@ -276,7 +30,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/PredictResponse" + "$ref": "#/components/schemas/TypeSafeResponse" } } } @@ -289,541 +43,31 @@ }, "422": { "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - }, - "/predict/bulk": { - "post": { - "tags": [ - "predict" - ], - "summary": "Predict many states", - "description": "Run typed questions over many states.\n\nAccepts either `states` with shared `questions`/`preset`, or `items` where each\nstate can override its questions and model. Laya's public API predicts one state\nat a time (its internals can batch, but the public `Agent.predict` cannot), so\nthis loops. Per-state errors are isolated and returned inline.", - "operationId": "predict_bulk_predict_bulk_post", - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/BulkPredictRequest" - } - } - }, - "required": true - }, - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/BulkPredictResponse" - } - } - } - }, - "401": { - "description": "Missing or invalid credentials" - }, - "503": { - "description": "Model not loaded yet" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - }, - "security": [ - { - "ApiKeyAuth": [] - }, - { - "BearerAuth": [] - }, - { - "BasicAuth": [] - } - ] - } - } - }, - "components": { - "schemas": { - "ActionResult": { - "properties": { - "act_probability": { - "type": "number", - "title": "Act Probability", - "description": "Probability the model acted at all." - } - }, - "type": "object", - "required": [ - "act_probability" - ], - "title": "ActionResult" - }, - "BulkItem": { - "properties": { - "state": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ], - "title": "State" - }, - "questions": { - "anyOf": [ - { - "additionalProperties": { - "oneOf": [ - { - "$ref": "#/components/schemas/ChoiceQuestion" - }, - { - "$ref": "#/components/schemas/ScoreQuestion" - }, - { - "$ref": "#/components/schemas/NoulQuestion" - } - ], - "discriminator": { - "propertyName": "type", - "mapping": { - "choice": "#/components/schemas/ChoiceQuestion", - "noul": "#/components/schemas/NoulQuestion", - "score": "#/components/schemas/ScoreQuestion" - } - } - }, - "type": "object" - }, - { - "type": "null" - } - ], - "title": "Questions", - "description": "Overrides the request-level questions for this item." - }, - "model": { - "anyOf": [ - { - "type": "string", - "enum": [ - "english", - "multilingual", - "typed-decisions" - ] - }, - { - "type": "null" - } - ], - "title": "Model", - "description": "Overrides the request-level model for this item." - } - }, - "type": "object", - "required": [ - "state" - ], - "title": "BulkItem", - "description": "One state with optional per-item questions and model override." - }, - "BulkItemResult": { - "properties": { - "ok": { - "type": "boolean", - "title": "Ok" - }, - "result": { - "anyOf": [ - { - "$ref": "#/components/schemas/PredictResponse" - }, - { - "type": "null" - } - ] - }, - "error": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Error" - } - }, - "type": "object", - "required": [ - "ok" - ], - "title": "BulkItemResult" - }, - "BulkPredictRequest": { - "properties": { - "states": { - "anyOf": [ - { - "items": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ] - }, - "type": "array", - "minItems": 1 - }, - { - "type": "null" - } - ], - "title": "States", - "description": "States to classify with the shared `questions`/`preset`." - }, - "questions": { - "anyOf": [ - { - "additionalProperties": { - "oneOf": [ - { - "$ref": "#/components/schemas/ChoiceQuestion" - }, - { - "$ref": "#/components/schemas/ScoreQuestion" - }, - { - "$ref": "#/components/schemas/NoulQuestion" - } - ], - "discriminator": { - "propertyName": "type", - "mapping": { - "choice": "#/components/schemas/ChoiceQuestion", - "noul": "#/components/schemas/NoulQuestion", - "score": "#/components/schemas/ScoreQuestion" - } - } - }, - "type": "object" - }, - { - "type": "null" - } - ], - "title": "Questions", - "description": "Questions applied to each state." - }, - "preset": { - "anyOf": [ - { - "type": "string", - "enum": [ - "triage", - "email", - "guard", - "moderation", - "router" - ] - }, - { - "type": "null" - } - ], - "title": "Preset", - "description": "Preset questions for each state." - }, - "model": { - "anyOf": [ - { - "type": "string", - "enum": [ - "english", - "multilingual", - "typed-decisions" - ] - }, - { - "type": "null" - } - ], - "title": "Model", - "description": "Pin a checkpoint for all states; omit to auto-route." - }, - "items": { - "anyOf": [ - { - "items": { - "$ref": "#/components/schemas/BulkItem" - }, - "type": "array", - "minItems": 1 - }, - { - "type": "null" - } - ], - "title": "Items", - "description": "Per-state items, each optionally overriding questions and model." - } - }, - "type": "object", - "title": "BulkPredictRequest", - "examples": [ - { - "questions": { - "department": { - "criteria": { - "billing": "invoices, payments, refunds", - "sales": "new purchases", - "technical": "bugs and outages" - }, - "instructions": "Which team should handle this request?", - "type": "choice" - }, - "refund": { - "instructions": "Does the customer ask for money back?", - "type": "noul" - }, - "urgency": { - "criteria": [ - "not urgent", - "soon", - "critical" - ], - "instructions": "How urgent is this request?", - "type": "score" - } - }, - "states": [ - "I was billed twice. Please refund the duplicate today.", - "The app crashes on launch." - ] - }, - { - "items": [ - { - "state": "I was billed twice." - }, - { - "model": "english", - "state": "The app crashes." - } - ], - "preset": "triage" - } - ] - }, - "BulkPredictResponse": { - "properties": { - "count": { - "type": "integer", - "title": "Count" - }, - "results": { - "items": { - "$ref": "#/components/schemas/BulkItemResult" - }, - "type": "array", - "title": "Results" - } - }, - "type": "object", - "required": [ - "count", - "results" - ], - "title": "BulkPredictResponse" - }, - "ChoiceAnswer": { - "properties": { - "type": { - "type": "string", - "const": "choice", - "title": "Type" - }, - "choice": { - "type": "string", - "title": "Choice" - }, - "probabilities": { - "additionalProperties": { - "type": "number" - }, - "type": "object", - "title": "Probabilities" - }, - "confidence": { - "type": "number", - "title": "Confidence" - }, - "action": { - "$ref": "#/components/schemas/ActionResult" - } - }, - "type": "object", - "required": [ - "type", - "choice", - "probabilities", - "confidence", - "action" - ], - "title": "ChoiceAnswer" - }, - "ChoiceQuestion": { - "properties": { - "type": { - "type": "string", - "const": "choice", - "title": "Type" - }, - "instructions": { - "type": "string", - "title": "Instructions", - "description": "What to decide, phrased as a question." - }, - "criteria": { - "anyOf": [ - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - }, - { - "type": "null" - } - ], - "title": "Criteria", - "description": "Option label -> description, or a plain list of labels. Descriptions may be strings or any JSON value (rendered as compact JSON).", - "examples": [ - { - "billing": "invoices, payments, refunds", - "technical": [ - "bugs", - "outages" - ] - } - ] - } - }, - "type": "object", - "required": [ - "type", - "instructions" - ], - "title": "ChoiceQuestion", - "description": "Pick one labelled option, e.g. a routing department." - }, - "DetectRequest": { - "properties": { - "state": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ], - "title": "State" - } - }, - "type": "object", - "required": [ - "state" - ], - "title": "DetectRequest" - }, - "EmailStateRequest": { - "properties": { - "subject": { - "type": "string", - "title": "Subject", - "default": "" - }, - "body": { - "type": "string", - "title": "Body" - }, - "sender": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } - ], - "title": "Sender" - }, - "clean": { - "type": "boolean", - "title": "Clean", - "description": "Strip quoted history, signatures, disclaimers.", - "default": true + } } }, - "type": "object", - "required": [ - "body" - ], - "title": "EmailStateRequest" - }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + } + }, + "components": { + "schemas": { "HTTPValidationError": { "properties": { "detail": { @@ -837,84 +81,93 @@ "type": "object", "title": "HTTPValidationError" }, - "HealthResponse": { + "TypeSafeChoiceAnswer": { "properties": { - "status": { + "type": { "type": "string", - "title": "Status" - }, - "model_loaded": { - "type": "boolean", - "title": "Model Loaded" + "const": "choice", + "title": "Type" }, - "device": { + "choice": { "type": "string", - "title": "Device" - }, - "auth_enabled": { - "type": "boolean", - "title": "Auth Enabled" - }, - "auth_methods": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Auth Methods" + "title": "Choice" }, - "models": { - "items": { - "type": "string" + "probabilities": { + "additionalProperties": { + "type": "number" }, - "type": "array", - "title": "Models", - "description": "Checkpoints currently resident." + "type": "object", + "title": "Probabilities" }, - "available_models": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Available Models" + "confidence": { + "type": "number", + "title": "Confidence" } }, "type": "object", "required": [ - "status", - "model_loaded", - "device", - "auth_enabled", - "auth_methods", - "models", - "available_models" + "type", + "choice", + "probabilities", + "confidence" ], - "title": "HealthResponse" + "title": "TypeSafeChoiceAnswer" }, - "ModelsResponse": { + "TypeSafeChoiceQuestion": { "properties": { - "available": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Available" + "type": { + "type": "string", + "const": "choice", + "title": "Type" }, - "loaded": { - "items": { - "type": "string" + "instructions": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], + "title": "Instructions" + }, + "criteria": { + "additionalProperties": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + }, + { + "type": "null" + } + ] }, - "type": "array", - "title": "Loaded" + "type": "object", + "title": "Criteria" } }, "type": "object", "required": [ - "available", - "loaded" + "type", + "instructions", + "criteria" ], - "title": "ModelsResponse" + "title": "TypeSafeChoiceQuestion" }, - "NoulAnswer": { + "TypeSafeNoulAnswer": { "properties": { "type": { "type": "string", @@ -924,25 +177,16 @@ "noul": { "type": "number", "title": "Noul" - }, - "confidence": { - "type": "number", - "title": "Confidence" - }, - "action": { - "$ref": "#/components/schemas/ActionResult" } }, "type": "object", "required": [ "type", - "noul", - "confidence", - "action" + "noul" ], - "title": "NoulAnswer" + "title": "TypeSafeNoulAnswer" }, - "NoulQuestion": { + "TypeSafeNoulQuestion": { "properties": { "type": { "type": "string", @@ -950,40 +194,6 @@ "title": "Type" }, "instructions": { - "type": "string", - "title": "Instructions" - }, - "criteria": { - "anyOf": [ - { - "additionalProperties": true, - "type": "object" - }, - { - "type": "null" - } - ], - "title": "Criteria", - "description": "Optional `false`/`true` descriptions (any JSON value).", - "examples": [ - { - "false": "a legitimate email", - "true": "phishing or fraud" - } - ] - } - }, - "type": "object", - "required": [ - "type", - "instructions" - ], - "title": "NoulQuestion", - "description": "Yes/no decision without a learned neutral class (n-o-u-l)." - }, - "PredictRequest": { - "properties": { - "state": { "anyOf": [ { "type": "string" @@ -997,32 +207,28 @@ "type": "array" } ], - "title": "State", - "description": "Text, JSON object, or conversation turns to analyse." + "title": "Instructions" }, - "questions": { + "criteria": { "anyOf": [ { "additionalProperties": { - "oneOf": [ + "anyOf": [ { - "$ref": "#/components/schemas/ChoiceQuestion" + "type": "string" }, { - "$ref": "#/components/schemas/ScoreQuestion" + "additionalProperties": true, + "type": "object" }, { - "$ref": "#/components/schemas/NoulQuestion" - } - ], - "discriminator": { - "propertyName": "type", - "mapping": { - "choice": "#/components/schemas/ChoiceQuestion", - "noul": "#/components/schemas/NoulQuestion", - "score": "#/components/schemas/ScoreQuestion" + "items": {}, + "type": "array" + }, + { + "type": "null" } - } + ] }, "type": "object" }, @@ -1030,86 +236,78 @@ "type": "null" } ], - "title": "Questions", - "description": "Map of question id -> typed question definition." - }, - "preset": { + "title": "Criteria" + } + }, + "type": "object", + "required": [ + "type", + "instructions" + ], + "title": "TypeSafeNoulQuestion" + }, + "TypeSafeRequest": { + "properties": { + "state": { "anyOf": [ { - "type": "string", - "enum": [ - "triage", - "email", - "guard", - "moderation", - "router" - ] + "type": "string" }, { - "type": "null" + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" } ], - "title": "Preset", - "description": "Built-in question set to use instead of `questions`." + "title": "State" }, "model": { - "anyOf": [ - { - "type": "string", - "enum": [ - "english", - "multilingual", - "typed-decisions" - ] - }, - { - "type": "null" - } + "type": "string", + "enum": [ + "laya-english", + "laya-multilingual", + "laya-typed-decisions" ], - "title": "Model", - "description": "Pin a checkpoint; omit to auto-route by language." + "title": "Model" + }, + "questions": { + "additionalProperties": { + "oneOf": [ + { + "$ref": "#/components/schemas/TypeSafeChoiceQuestion" + }, + { + "$ref": "#/components/schemas/TypeSafeScoreQuestion" + }, + { + "$ref": "#/components/schemas/TypeSafeNoulQuestion" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "choice": "#/components/schemas/TypeSafeChoiceQuestion", + "noul": "#/components/schemas/TypeSafeNoulQuestion", + "score": "#/components/schemas/TypeSafeScoreQuestion" + } + } + }, + "type": "object", + "title": "Questions" } }, "type": "object", "required": [ - "state" + "state", + "model", + "questions" ], - "title": "PredictRequest", - "examples": [ - { - "questions": { - "department": { - "criteria": { - "billing": "invoices, payments, refunds", - "sales": "new purchases", - "technical": "bugs and outages" - }, - "instructions": "Which team should handle this request?", - "type": "choice" - }, - "refund": { - "instructions": "Does the customer ask for money back?", - "type": "noul" - }, - "urgency": { - "criteria": [ - "not urgent", - "soon", - "critical" - ], - "instructions": "How urgent is this request?", - "type": "score" - } - }, - "state": "I was billed twice. Please refund the duplicate today." - }, - { - "preset": "triage", - "state": "Hi, we were billed twice for March." - } - ] + "title": "TypeSafeRequest" }, - "PredictResponse": { + "TypeSafeResponse": { "properties": { "model": { "type": "string", @@ -1119,21 +317,21 @@ "additionalProperties": { "oneOf": [ { - "$ref": "#/components/schemas/ChoiceAnswer" + "$ref": "#/components/schemas/TypeSafeNoulAnswer" }, { - "$ref": "#/components/schemas/ScoreAnswer" + "$ref": "#/components/schemas/TypeSafeChoiceAnswer" }, { - "$ref": "#/components/schemas/NoulAnswer" + "$ref": "#/components/schemas/TypeSafeScoreAnswer" } ], "discriminator": { "propertyName": "type", "mapping": { - "choice": "#/components/schemas/ChoiceAnswer", - "noul": "#/components/schemas/NoulAnswer", - "score": "#/components/schemas/ScoreAnswer" + "choice": "#/components/schemas/TypeSafeChoiceAnswer", + "noul": "#/components/schemas/TypeSafeNoulAnswer", + "score": "#/components/schemas/TypeSafeScoreAnswer" } } }, @@ -1141,55 +339,18 @@ "title": "Answers" }, "usage": { - "$ref": "#/components/schemas/Usage" - }, - "routing": { - "anyOf": [ - { - "$ref": "#/components/schemas/Routing" - }, - { - "type": "null" - } - ], - "description": "How the checkpoint was chosen (auto-route or pinned)." + "$ref": "#/components/schemas/TypeSafeUsage" } }, - "additionalProperties": true, "type": "object", "required": [ "model", "answers", "usage" ], - "title": "PredictResponse" - }, - "Routing": { - "properties": { - "model": { - "type": "string", - "title": "Model" - }, - "reason": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Reason" - } - }, - "additionalProperties": true, - "type": "object", - "required": [ - "model" - ], - "title": "Routing" + "title": "TypeSafeResponse" }, - "ScoreAnswer": { + "TypeSafeScoreAnswer": { "properties": { "type": { "type": "string", @@ -1198,8 +359,7 @@ }, "score": { "type": "number", - "title": "Score", - "description": "Expected value on the 0-based criteria scale." + "title": "Score" }, "legend": { "additionalProperties": { @@ -1218,9 +378,6 @@ "confidence": { "type": "number", "title": "Confidence" - }, - "action": { - "$ref": "#/components/schemas/ActionResult" } }, "type": "object", @@ -1229,12 +386,11 @@ "score", "legend", "probabilities", - "confidence", - "action" + "confidence" ], - "title": "ScoreAnswer" + "title": "TypeSafeScoreAnswer" }, - "ScoreQuestion": { + "TypeSafeScoreQuestion": { "properties": { "type": { "type": "string", @@ -1242,23 +398,39 @@ "title": "Type" }, "instructions": { - "type": "string", + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], "title": "Instructions" }, "criteria": { - "items": {}, - "type": "array", - "title": "Criteria", - "description": "Ordered levels, lowest first. Each may be any JSON value.", - "examples": [ - [ - "low", + "items": { + "anyOf": [ { - "level": "high", - "sla_minutes": 60 + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" } ] - ] + }, + "type": "array", + "title": "Criteria" } }, "type": "object", @@ -1267,10 +439,9 @@ "instructions", "criteria" ], - "title": "ScoreQuestion", - "description": "Rate on an ordered scale, e.g. urgency." + "title": "TypeSafeScoreQuestion" }, - "Usage": { + "TypeSafeUsage": { "properties": { "input_tokens": { "type": "integer", @@ -1286,7 +457,7 @@ "input_tokens", "output_tokens" ], - "title": "Usage" + "title": "TypeSafeUsage" }, "ValidationError": { "properties": { From c96e30c4fadc2a4b9cb4ca707d5dd6f8402a6050 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 10:41:52 +0530 Subject: [PATCH 02/16] build: publish model-specific images --- Dockerfile | 12 +++++++----- Makefile | 6 +++++- README.md | 16 ++++++++++++++++ 3 files changed, 28 insertions(+), 6 deletions(-) diff --git a/Dockerfile b/Dockerfile index 5fdcf4c..319f928 100644 --- a/Dockerfile +++ b/Dockerfile @@ -23,20 +23,22 @@ RUN --mount=type=cache,target=/root/.cache/uv \ COPY --chown=appuser:appuser app ./app +# Prepare the cache before downloading so the model layer is not duplicated by +# a later recursive chown. +RUN mkdir -p /data/hf && chown -R appuser:appuser /app /data/hf + +USER appuser + # Optional: bake the checkpoint(s) into the image for instant/offline startup. # Build with --build-arg PRELOAD_MODEL=1 (adds ~1 GB, needs network at build). ARG PRELOAD_MODEL=0 ARG MODELS=english +ENV MODELS=${MODELS} RUN if [ "$PRELOAD_MODEL" = "1" ]; then \ MODELS="$MODELS" uv run --no-dev python -c \ "import os; from laya import Router; Router().preload([m.strip() for m in os.environ['MODELS'].split(',') if m.strip()])"; \ fi -# Own the app and the HF cache mount so a fresh named volume inherits appuser. -RUN mkdir -p /data/hf && chown -R appuser:appuser /app /data/hf - -USER appuser - EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=5 \ diff --git a/Makefile b/Makefile index a854939..494d385 100644 --- a/Makefile +++ b/Makefile @@ -1,6 +1,6 @@ PORT ?= 8000 -.PHONY: help run format check fix openapi up down build logs +.PHONY: help run format check fix openapi up down build build-model logs help: @echo "make run - run the API locally (uvicorn --reload on $(PORT))" @@ -11,6 +11,7 @@ help: @echo "make up - docker compose up --build (detached)" @echo "make down - docker compose down" @echo "make build - docker build the image" + @echo "make build-model MODEL=english - build an image with one model baked in" @echo "make logs - follow container logs" run: @@ -37,5 +38,8 @@ down: build: docker build -t laya-api:latest . +build-model: + docker build --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$(MODEL) -t laya-server:$(MODEL) . + logs: docker compose logs -f diff --git a/README.md b/README.md index dbb0ece..802d426 100644 --- a/README.md +++ b/README.md @@ -44,6 +44,22 @@ docker run -d -p 8000:8000 -e API_KEYS=key1 -v hf-cache:/data/hf ghcr.io/chneau/ First start downloads the checkpoint (~1 GB) into the `hf-cache` volume. Bake it into the image for instant/offline startup with `PRELOAD_MODEL=1`. +To publish one architecture-specific image per model to Docker Hub: + +```sh +for model in english multilingual typed-decisions; do + for arch in amd64 arm64; do + docker buildx build --platform linux/$arch --push \ + --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$model \ + -t auenkr/laya-server:laya-$model-$arch . + done +done +``` + +This publishes six tags: `laya-english-amd64`, `laya-english-arm64`, +`laya-multilingual-amd64`, `laya-multilingual-arm64`, +`laya-typed-decisions-amd64`, and `laya-typed-decisions-arm64`. + ### Check Health ```bash From a3a60f054cc7ad07d22d8adffa6e9fb1a324226c Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 13:52:47 +0530 Subject: [PATCH 03/16] deploy: add single-model Kubernetes deployment --- deploy/laya-deployment.yaml | 53 +++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) create mode 100644 deploy/laya-deployment.yaml diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml new file mode 100644 index 0000000..03af56b --- /dev/null +++ b/deploy/laya-deployment.yaml @@ -0,0 +1,53 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: laya + namespace: laya-classifer + labels: + app.kubernetes.io/name: laya +spec: + replicas: 1 + strategy: + type: Recreate + selector: + matchLabels: + app.kubernetes.io/name: laya + template: + metadata: + labels: + app.kubernetes.io/name: laya + spec: + containers: + - name: laya + image: auenkr/laya-server:laya-english-amd64 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8000 + protocol: TCP + env: + - name: PORT + value: "8000" + readinessProbe: + httpGet: + path: /healthz + port: http + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 12 + livenessProbe: + httpGet: + path: /healthz + port: http + initialDelaySeconds: 120 + periodSeconds: 30 + timeoutSeconds: 5 + failureThreshold: 5 + resources: + requests: + cpu: "1" + memory: 2Gi + limits: + cpu: "2" + memory: 4Gi From dd3338a3ba4bc771892bc776ef91c91b57fb1af2 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 14:21:49 +0530 Subject: [PATCH 04/16] build: make baked model images offline --- Dockerfile | 5 +++++ README.md | 3 ++- deploy/laya-deployment.yaml | 6 +++--- 3 files changed, 10 insertions(+), 4 deletions(-) diff --git a/Dockerfile b/Dockerfile index 319f928..72a8c22 100644 --- a/Dockerfile +++ b/Dockerfile @@ -39,6 +39,11 @@ RUN if [ "$PRELOAD_MODEL" = "1" ]; then \ "import os; from laya import Router; Router().preload([m.strip() for m in os.environ['MODELS'].split(',') if m.strip()])"; \ fi +# The published images must run from their baked cache without contacting the Hub. +ENV HF_HUB_OFFLINE=1 \ + TRANSFORMERS_OFFLINE=1 \ + HF_DATASETS_OFFLINE=1 + EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=5 \ diff --git a/README.md b/README.md index 802d426..4700f64 100644 --- a/README.md +++ b/README.md @@ -42,7 +42,8 @@ docker run -d -p 8000:8000 -e API_KEYS=key1 -v hf-cache:/data/hf ghcr.io/chneau/ ``` First start downloads the checkpoint (~1 GB) into the `hf-cache` volume. Bake it -into the image for instant/offline startup with `PRELOAD_MODEL=1`. +into the image for instant/offline startup with `PRELOAD_MODEL=1`; published +model images include the complete cache and run with Hugging Face offline mode. To publish one architecture-specific image per model to Docker Hub: diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml index 03af56b..5b4caab 100644 --- a/deploy/laya-deployment.yaml +++ b/deploy/laya-deployment.yaml @@ -20,7 +20,7 @@ spec: containers: - name: laya image: auenkr/laya-server:laya-english-amd64 - imagePullPolicy: IfNotPresent + imagePullPolicy: Always ports: - name: http containerPort: 8000 @@ -46,8 +46,8 @@ spec: failureThreshold: 5 resources: requests: - cpu: "1" - memory: 2Gi + cpu: "250m" + memory: 1Gi limits: cpu: "2" memory: 4Gi From 6dc693cfc1127210c02770de7f0c2ca9c3b7c57b Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 14:32:20 +0530 Subject: [PATCH 05/16] deploy: pin offline English image digest --- deploy/laya-deployment.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml index 5b4caab..7658860 100644 --- a/deploy/laya-deployment.yaml +++ b/deploy/laya-deployment.yaml @@ -19,7 +19,7 @@ spec: spec: containers: - name: laya - image: auenkr/laya-server:laya-english-amd64 + image: auenkr/laya-server@sha256:1ead3c2d7e96ecb4c5c734e70f53584d752bff12811ba9b115cff180a49f5f57 imagePullPolicy: Always ports: - name: http From 107c76a954d2d2b40b5d09f9e44874fd12e3fc64 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 16:16:55 +0530 Subject: [PATCH 06/16] deploy: use published English image tag --- deploy/laya-deployment.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml index 7658860..5b4caab 100644 --- a/deploy/laya-deployment.yaml +++ b/deploy/laya-deployment.yaml @@ -19,7 +19,7 @@ spec: spec: containers: - name: laya - image: auenkr/laya-server@sha256:1ead3c2d7e96ecb4c5c734e70f53584d752bff12811ba9b115cff180a49f5f57 + image: auenkr/laya-server:laya-english-amd64 imagePullPolicy: Always ports: - name: http From 8bc139fd75c588fd4cab8f047fac2a9eda6f7630 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 17:09:36 +0530 Subject: [PATCH 07/16] fix: serialize TypeSafe prediction responses --- app/main.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/app/main.py b/app/main.py index 99e8d45..5e79b39 100644 --- a/app/main.py +++ b/app/main.py @@ -558,8 +558,8 @@ def typesafe_predict(request: TypeSafeRequest) -> TypeSafeResponse: validated = PredictResponse.model_validate(result) return TypeSafeResponse( model=request.model, - answers=validated.answers, - usage=validated.usage, + answers={key: answer.model_dump() for key, answer in validated.answers.items()}, + usage=validated.usage.model_dump(), ) From d423f1e4ff1cd64a2ee9866e88a3e3fdebbd2ea8 Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 17:34:01 +0530 Subject: [PATCH 08/16] deploy: switch to multilingual model --- deploy/laya-deployment.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml index 5b4caab..82e1852 100644 --- a/deploy/laya-deployment.yaml +++ b/deploy/laya-deployment.yaml @@ -19,7 +19,7 @@ spec: spec: containers: - name: laya - image: auenkr/laya-server:laya-english-amd64 + image: auenkr/laya-server:laya-multilingual-amd64 imagePullPolicy: Always ports: - name: http From 55b939d2f556af9c22f199491d72444c5cffcb9b Mon Sep 17 00:00:00 2001 From: Golden Kumar Date: Thu, 24 Sep 2026 17:52:46 +0530 Subject: [PATCH 09/16] deploy: pin corrected multilingual image --- deploy/laya-deployment.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml index 82e1852..b7a3b85 100644 --- a/deploy/laya-deployment.yaml +++ b/deploy/laya-deployment.yaml @@ -19,7 +19,7 @@ spec: spec: containers: - name: laya - image: auenkr/laya-server:laya-multilingual-amd64 + image: auenkr/laya-server@sha256:25b91e992824408009e9eb357f7dc879fb5ba95cab757c14c2e0209e19426396 imagePullPolicy: Always ports: - name: http From f6be4ad3f27dbdb09ed9f9ad23a91886ec1a080f Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:09:02 +0100 Subject: [PATCH 10/16] refactor: support both native endpoints and TypeSafe /v1/systemone, remove deploy manifest --- .github/workflows/publish.yml | 4 +- Dockerfile | 5 - Makefile | 2 +- README.md | 94 ++- app/main.py | 144 +++- deploy/laya-deployment.yaml | 53 -- openapi.json | 1355 +++++++++++++++++++++++++++++++-- 7 files changed, 1512 insertions(+), 145 deletions(-) delete mode 100644 deploy/laya-deployment.yaml diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index b9bcba3..1d637d8 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -32,7 +32,7 @@ jobs: - name: Validate OpenAPI schema run: | - uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' not in s['paths']; print('openapi ok:', list(s['paths']))" + uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' in s['paths']; print('openapi ok:', list(s['paths']))" # Pull requests: build the image once and check it boots. test: @@ -54,7 +54,7 @@ jobs: - name: Container smoke test run: | - docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths']; print('container ok')" + docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' in s['paths']; print('container ok')" # Nightly / manual / release: real model download + inference. smoke: diff --git a/Dockerfile b/Dockerfile index 72a8c22..319f928 100644 --- a/Dockerfile +++ b/Dockerfile @@ -39,11 +39,6 @@ RUN if [ "$PRELOAD_MODEL" = "1" ]; then \ "import os; from laya import Router; Router().preload([m.strip() for m in os.environ['MODELS'].split(',') if m.strip()])"; \ fi -# The published images must run from their baked cache without contacting the Hub. -ENV HF_HUB_OFFLINE=1 \ - TRANSFORMERS_OFFLINE=1 \ - HF_DATASETS_OFFLINE=1 - EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=5 \ diff --git a/Makefile b/Makefile index 494d385..9bb9d23 100644 --- a/Makefile +++ b/Makefile @@ -39,7 +39,7 @@ build: docker build -t laya-api:latest . build-model: - docker build --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$(MODEL) -t laya-server:$(MODEL) . + docker build --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$(MODEL) -t laya-api:$(MODEL) . logs: docker compose logs -f diff --git a/README.md b/README.md index 4700f64..610cfc4 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ Dockerized [Laya](https://huggingface.co/convaiinnovations/laya) prediction service: loads one or more checkpoints behind a router, then serves typed decisions (choice / score / noul) over HTTP, auto-routed by language or pinned -with `model`. +with `model`. Also supports TypeSafe API compatibility via `/v1/systemone`. Built and published for `linux/amd64` and `linux/arm64`. @@ -18,10 +18,18 @@ Built and published for `linux/amd64` and `linux/arm64`. `typed-decisions`, auto-selected by language or pinned per request. - 🎯 **Typed Decisions**: `choice` / `score` / `noul` questions with calibrated probabilities, confidence and action probability. +- 🤝 **TypeSafe API Compatibility**: drop-in `/v1/systemone` endpoint compatible + with TypeSafe request/response schemas. - 🔐 **Timing-Safe Auth**: API keys (`X-API-Key` / `Authorization: Bearer`) and HTTP Basic, compared in constant time. - 📑 **Interactive OpenAPI Docs**: Swagger UI (`/docs`), ReDoc (`/redoc`) and the raw schema at `/openapi.json`. +- 🧰 **Built-in Presets**: ready-made question sets (`triage`, `email`, `guard`, + `moderation`, `router`). +- 📦 **Bulk Inference**: `/predict/bulk` over many states, with per-state + questions/model and isolated errors. +- 🔎 **Detection & Email Helpers**: `/detect` (script/language) and + `/email/state` (clean + structure an email). - 🧩 **Flexible State**: string, JSON object, or conversation turns; criteria values may be any JSON (dicts/lists/numbers are rendered as compact JSON). - 🛡️ **Non-Root**: runs as unprivileged `appuser` (uid `10001`). @@ -42,25 +50,12 @@ docker run -d -p 8000:8000 -e API_KEYS=key1 -v hf-cache:/data/hf ghcr.io/chneau/ ``` First start downloads the checkpoint (~1 GB) into the `hf-cache` volume. Bake it -into the image for instant/offline startup with `PRELOAD_MODEL=1`; published -model images include the complete cache and run with Hugging Face offline mode. +into the image for instant/offline startup with `PRELOAD_MODEL=1`: -To publish one architecture-specific image per model to Docker Hub: - -```sh -for model in english multilingual typed-decisions; do - for arch in amd64 arm64; do - docker buildx build --platform linux/$arch --push \ - --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$model \ - -t auenkr/laya-server:laya-$model-$arch . - done -done +```bash +make build-model MODEL=english ``` -This publishes six tags: `laya-english-amd64`, `laya-english-arm64`, -`laya-multilingual-amd64`, `laya-multilingual-arm64`, -`laya-typed-decisions-amd64`, and `laya-typed-decisions-arm64`. - ### Check Health ```bash @@ -81,6 +76,12 @@ curl http://localhost:8000/healthz | Method | Path | Auth | Description | | --- | --- | --- | --- | | `GET` | `/healthz` | no | Liveness + resident models | +| `GET` | `/models` | yes* | Available and loaded checkpoints | +| `GET` | `/presets` | yes* | Built-in question sets | +| `POST` | `/detect` | yes* | Script/language detection (what routing uses) | +| `POST` | `/email/state` | yes* | Clean + structure an email as a state | +| `POST` | `/predict` | yes* | Typed questions over one state | +| `POST` | `/predict/bulk` | yes* | Same questions over many states | | `POST` | `/v1/systemone` | yes* | TypeSafe-compatible typed prediction | \* Enforced only when `API_KEYS` and/or `BASIC_AUTH` is set. Any of these works: @@ -93,12 +94,13 @@ curl http://localhost:8000/healthz --- -## 🧠 TypeSafe Prediction +## 🧠 Predicting + +### Standard Prediction (`POST /predict`) ```bash -curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Content-Type: application/json' -d '{ +curl -X POST localhost:8000/predict -H 'X-API-Key: key1' -H 'Content-Type: application/json' -d '{ "state": "I was billed twice. Please refund the duplicate today.", - "model": "laya-english", "questions": { "department": {"type": "choice", "instructions": "Which team?", "criteria": {"billing": "refunds", "technical": "bugs", "sales": "purchases"}}, "urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["not urgent", "soon", "critical"]}, @@ -109,13 +111,14 @@ curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Con ```json { - "model": "laya-english", + "model": "laya-rl-agent", "answers": { "department": {"type": "choice", "choice": "billing", "probabilities": {"billing": 0.96, "technical": 0.02, "sales": 0.02}, "confidence": 0.82}, "urgency": {"type": "score", "score": 1.36, "legend": {"0": "not urgent", "1": "soon", "2": "critical"}, "confidence": 0.09}, - "refund": {"type": "noul", "noul": 0.82} + "refund": {"type": "noul", "noul": 0.82, "confidence": 0.82} }, - "usage": {"input_tokens": 132, "output_tokens": 0} + "usage": {"input_tokens": 132, "output_tokens": 0}, + "routing": {"model": "english", "reason": "English Latin text"} } ``` @@ -123,17 +126,47 @@ curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Con values may be strings or any JSON value (dicts/lists/numbers are rendered as compact JSON), and `noul` accepts optional `{"true": ..., "false": ...}` text. -### TypeSafe API Compatibility +**Routing** — omit `model` to auto-select by language (see `/detect`), or pin +`"model": "english" | "multilingual" | "typed-decisions"`. The response includes +`routing` with the chosen checkpoint and reason. + +**Presets** — skip `questions` and pass `"preset": "triage"` (one of `triage`, +`email`, `guard`, `moderation`, `router`); list them at `GET /presets`. + +**Bulk** — use `states` with shared `questions`/`preset`/`model`, or `items` to +override questions and model per state. Returns `{"count": N, "results": [...]}` +with per-state errors isolated as `{"ok": false, "error": "..."}`. + +--- -`POST /v1/systemone` accepts the TypeSafe API request shape and supports these -Laya model names: +### TypeSafe Compatible Prediction (`POST /v1/systemone`) + +```bash +curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Content-Type: application/json' -d '{ + "state": "I was billed twice. Please refund the duplicate today.", + "model": "laya-english", + "questions": { + "department": {"type": "choice", "instructions": "Which team?", "criteria": {"billing": "refunds", "technical": "bugs", "sales": "purchases"}}, + "urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["not urgent", "soon", "critical"]}, + "refund": {"type": "noul", "instructions": "Does the customer ask for money back?"} + } +}' +``` -```text -laya-english | laya-multilingual | laya-typed-decisions +```json +{ + "model": "laya-english", + "answers": { + "department": {"type": "choice", "choice": "billing", "probabilities": {"billing": 0.96, "technical": 0.02, "sales": 0.02}, "confidence": 0.82}, + "urgency": {"type": "score", "score": 1.36, "legend": {"0": "not urgent", "1": "soon", "2": "critical"}, "confidence": 0.09}, + "refund": {"type": "noul", "noul": 0.82} + }, + "usage": {"input_tokens": 132, "output_tokens": 0} +} ``` -It uses `Authorization: Bearer ` and returns the documented TypeSafe -`model`, `answers`, and `usage` fields. +Supported model names for TypeSafe requests: +`laya-english`, `laya-multilingual`, `laya-typed-decisions`. --- @@ -143,6 +176,7 @@ It uses `Authorization: Bearer ` and returns the documented TypeSafe | --- | --- | --- | | `API_KEYS` | *(empty)* | Comma-separated keys; empty disables API-key auth. | | `BASIC_AUTH` | *(empty)* | Comma-separated `user:password` pairs. | +| `MAX_BULK_ITEMS` | `256` | Max states per `/predict/bulk`. | | `PORT` | `8000` | HTTP port (host and container). | | `MODELS` | `english` | Checkpoints to preload: `english`, `multilingual`, `typed-decisions`. | | `MODEL_ID` | `convaiinnovations/laya` | Optional repo override (mirror/local path). | diff --git a/app/main.py b/app/main.py index 5e79b39..f5a9023 100644 --- a/app/main.py +++ b/app/main.py @@ -490,9 +490,12 @@ def _router() -> Any: return router -def _dump(questions: dict[str, Question]) -> dict[str, dict[str, Any]]: +def _dump(questions: dict[str, Any]) -> dict[str, dict[str, Any]]: """Laya's runtime expects plain dicts, not Pydantic models.""" - return {qid: q.model_dump(exclude_none=True) for qid, q in questions.items()} + return { + qid: q.model_dump(exclude_none=True) if hasattr(q, "model_dump") else q + for qid, q in questions.items() + } def _resolve( @@ -517,12 +520,7 @@ def _resolve( # ------------------------------------------------------------------------- endpoints -@app.get( - "/healthz", - include_in_schema=False, - response_model=HealthResponse, - summary="Health check", -) +@app.get("/healthz", tags=["meta"], response_model=HealthResponse, summary="Health check") def healthz() -> HealthResponse: router = _state.get("router") return HealthResponse( @@ -538,9 +536,135 @@ def healthz() -> HealthResponse: ) +@app.get( + "/models", + tags=["meta"], + response_model=ModelsResponse, + summary="List checkpoints", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def list_models() -> ModelsResponse: + """Available checkpoints and which are currently resident.""" + router = _router() + return ModelsResponse(available=list(AVAILABLE_MODELS), loaded=list(router.loaded)) + + +@app.get( + "/presets", + tags=["meta"], + summary="List built-in question presets", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def list_presets() -> dict[str, dict[str, Any]]: + """Return the ready-to-use question sets (triage/email/guard/moderation/router).""" + return _presets() + + @app.post( - "/v1/systemone", + "/detect", + tags=["meta"], + summary="Detect script and language", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def detect(request: DetectRequest) -> dict[str, Any]: + """Report script, best-effort language and `is_english` for a state (what routing uses).""" + from laya import detect_language + + return detect_language(request.state) + + +@app.post( + "/email/state", + tags=["meta"], + summary="Build a clean email state", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def email_state_endpoint(request: EmailStateRequest) -> dict[str, Any]: + """Clean an email body and structure it as a state for `/predict`.""" + from laya import email_state + + return email_state(request.subject, request.body, sender=request.sender, clean=request.clean) + + +@app.post( + "/predict", + tags=["predict"], + response_model=PredictResponse, + response_model_exclude_none=True, + summary="Predict one state", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def predict(request: PredictRequest) -> PredictResponse: + """Run typed questions over a single state and return calibrated answers.""" + router = _router() + questions = _resolve(request.questions, request.preset) + with _lock: + result = router.predict(request.state, questions, model=request.model) + return PredictResponse.model_validate(result) + + +@app.post( + "/predict/bulk", tags=["predict"], + response_model=BulkPredictResponse, + response_model_exclude_none=True, + summary="Predict many states", + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) +def predict_bulk(request: BulkPredictRequest) -> BulkPredictResponse: + """Run typed questions over many states. + + Accepts either `states` with shared `questions`/`preset`, or `items` where each + state can override its questions and model. Laya's public API predicts one state + at a time (its internals can batch, but the public `Agent.predict` cannot), so + this loops. Per-state errors are isolated and returned inline. + """ + if request.items: + jobs = [ + ( + item.state, + _resolve(item.questions or request.questions, request.preset), + item.model or request.model, + ) + for item in request.items + ] + else: + questions = _resolve(request.questions, request.preset) + jobs = [(state, questions, request.model) for state in request.states or []] + + if len(jobs) > MAX_BULK_ITEMS: + raise HTTPException( + status_code=422, + detail=f"Too many states: {len(jobs)} > MAX_BULK_ITEMS={MAX_BULK_ITEMS}", + ) + + router = _router() + results: list[BulkItemResult] = [] + with _lock: + for state, questions, model in jobs: + try: + results.append( + BulkItemResult( + ok=True, + result=PredictResponse.model_validate( + router.predict(state, questions, model=model) + ), + ) + ) + except Exception as exc: # noqa: BLE001 - report per-item failure + results.append(BulkItemResult(ok=False, error=str(exc))) + return BulkPredictResponse(count=len(results), results=results) + + +@app.post( + "/v1/systemone", + tags=["typesafe", "predict"], response_model=TypeSafeResponse, summary="TypeSafe-compatible prediction", dependencies=[Depends(require_auth)], @@ -559,7 +683,7 @@ def typesafe_predict(request: TypeSafeRequest) -> TypeSafeResponse: return TypeSafeResponse( model=request.model, answers={key: answer.model_dump() for key, answer in validated.answers.items()}, - usage=validated.usage.model_dump(), + usage=TypeSafeUsage.model_validate(validated.usage.model_dump()), ) diff --git a/deploy/laya-deployment.yaml b/deploy/laya-deployment.yaml deleted file mode 100644 index b7a3b85..0000000 --- a/deploy/laya-deployment.yaml +++ /dev/null @@ -1,53 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - name: laya - namespace: laya-classifer - labels: - app.kubernetes.io/name: laya -spec: - replicas: 1 - strategy: - type: Recreate - selector: - matchLabels: - app.kubernetes.io/name: laya - template: - metadata: - labels: - app.kubernetes.io/name: laya - spec: - containers: - - name: laya - image: auenkr/laya-server@sha256:25b91e992824408009e9eb357f7dc879fb5ba95cab757c14c2e0209e19426396 - imagePullPolicy: Always - ports: - - name: http - containerPort: 8000 - protocol: TCP - env: - - name: PORT - value: "8000" - readinessProbe: - httpGet: - path: /healthz - port: http - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 12 - livenessProbe: - httpGet: - path: /healthz - port: http - initialDelaySeconds: 120 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 5 - resources: - requests: - cpu: "250m" - memory: 1Gi - limits: - cpu: "2" - memory: 4Gi diff --git a/openapi.json b/openapi.json index 8e3200e..e2ca6dc 100644 --- a/openapi.json +++ b/openapi.json @@ -6,80 +6,1329 @@ "version": "0.5.0" }, "paths": { - "/v1/systemone": { + "/healthz": { + "get": { + "tags": [ + "meta" + ], + "summary": "Health check", + "operationId": "healthz_healthz_get", + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HealthResponse" + } + } + } + } + } + } + }, + "/models": { + "get": { + "tags": [ + "meta" + ], + "summary": "List checkpoints", + "description": "Available checkpoints and which are currently resident.", + "operationId": "list_models_models_get", + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ModelsResponse" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/presets": { + "get": { + "tags": [ + "meta" + ], + "summary": "List built-in question presets", + "description": "Return the ready-to-use question sets (triage/email/guard/moderation/router).", + "operationId": "list_presets_presets_get", + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "additionalProperties": { + "additionalProperties": true, + "type": "object" + }, + "type": "object", + "title": "Response List Presets Presets Get" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/detect": { + "post": { + "tags": [ + "meta" + ], + "summary": "Detect script and language", + "description": "Report script, best-effort language and `is_english` for a state (what routing uses).", + "operationId": "detect_detect_post", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/DetectRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "additionalProperties": true, + "type": "object", + "title": "Response Detect Detect Post" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/email/state": { + "post": { + "tags": [ + "meta" + ], + "summary": "Build a clean email state", + "description": "Clean an email body and structure it as a state for `/predict`.", + "operationId": "email_state_endpoint_email_state_post", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/EmailStateRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "additionalProperties": true, + "type": "object", + "title": "Response Email State Endpoint Email State Post" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/predict": { "post": { "tags": [ "predict" ], - "summary": "TypeSafe-compatible prediction", - "description": "Evaluate a TypeSafe-shaped request using a Laya checkpoint.", - "operationId": "typesafe_predict_v1_systemone_post", + "summary": "Predict one state", + "description": "Run typed questions over a single state and return calibrated answers.", + "operationId": "predict_predict_post", "requestBody": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/TypeSafeRequest" + "$ref": "#/components/schemas/PredictRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/PredictResponse" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/predict/bulk": { + "post": { + "tags": [ + "predict" + ], + "summary": "Predict many states", + "description": "Run typed questions over many states.\n\nAccepts either `states` with shared `questions`/`preset`, or `items` where each\nstate can override its questions and model. Laya's public API predicts one state\nat a time (its internals can batch, but the public `Agent.predict` cannot), so\nthis loops. Per-state errors are isolated and returned inline.", + "operationId": "predict_bulk_predict_bulk_post", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/BulkPredictRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/BulkPredictResponse" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + }, + "/v1/systemone": { + "post": { + "tags": [ + "typesafe", + "predict" + ], + "summary": "TypeSafe-compatible prediction", + "description": "Evaluate a TypeSafe-shaped request using a Laya checkpoint.", + "operationId": "typesafe_predict_v1_systemone_post", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TypeSafeRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TypeSafeResponse" + } + } + } + }, + "401": { + "description": "Missing or invalid credentials" + }, + "503": { + "description": "Model not loaded yet" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + }, + "security": [ + { + "ApiKeyAuth": [] + }, + { + "BearerAuth": [] + }, + { + "BasicAuth": [] + } + ] + } + } + }, + "components": { + "schemas": { + "ActionResult": { + "properties": { + "act_probability": { + "type": "number", + "title": "Act Probability", + "description": "Probability the model acted at all." + } + }, + "type": "object", + "required": [ + "act_probability" + ], + "title": "ActionResult" + }, + "BulkItem": { + "properties": { + "state": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], + "title": "State" + }, + "questions": { + "anyOf": [ + { + "additionalProperties": { + "oneOf": [ + { + "$ref": "#/components/schemas/ChoiceQuestion" + }, + { + "$ref": "#/components/schemas/ScoreQuestion" + }, + { + "$ref": "#/components/schemas/NoulQuestion" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "choice": "#/components/schemas/ChoiceQuestion", + "noul": "#/components/schemas/NoulQuestion", + "score": "#/components/schemas/ScoreQuestion" + } + } + }, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Questions", + "description": "Overrides the request-level questions for this item." + }, + "model": { + "anyOf": [ + { + "type": "string", + "enum": [ + "english", + "multilingual", + "typed-decisions" + ] + }, + { + "type": "null" + } + ], + "title": "Model", + "description": "Overrides the request-level model for this item." + } + }, + "type": "object", + "required": [ + "state" + ], + "title": "BulkItem", + "description": "One state with optional per-item questions and model override." + }, + "BulkItemResult": { + "properties": { + "ok": { + "type": "boolean", + "title": "Ok" + }, + "result": { + "anyOf": [ + { + "$ref": "#/components/schemas/PredictResponse" + }, + { + "type": "null" + } + ] + }, + "error": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Error" + } + }, + "type": "object", + "required": [ + "ok" + ], + "title": "BulkItemResult" + }, + "BulkPredictRequest": { + "properties": { + "states": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ] + }, + "type": "array", + "minItems": 1 + }, + { + "type": "null" + } + ], + "title": "States", + "description": "States to classify with the shared `questions`/`preset`." + }, + "questions": { + "anyOf": [ + { + "additionalProperties": { + "oneOf": [ + { + "$ref": "#/components/schemas/ChoiceQuestion" + }, + { + "$ref": "#/components/schemas/ScoreQuestion" + }, + { + "$ref": "#/components/schemas/NoulQuestion" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "choice": "#/components/schemas/ChoiceQuestion", + "noul": "#/components/schemas/NoulQuestion", + "score": "#/components/schemas/ScoreQuestion" + } + } + }, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Questions", + "description": "Questions applied to each state." + }, + "preset": { + "anyOf": [ + { + "type": "string", + "enum": [ + "triage", + "email", + "guard", + "moderation", + "router" + ] + }, + { + "type": "null" + } + ], + "title": "Preset", + "description": "Preset questions for each state." + }, + "model": { + "anyOf": [ + { + "type": "string", + "enum": [ + "english", + "multilingual", + "typed-decisions" + ] + }, + { + "type": "null" + } + ], + "title": "Model", + "description": "Pin a checkpoint for all states; omit to auto-route." + }, + "items": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/BulkItem" + }, + "type": "array", + "minItems": 1 + }, + { + "type": "null" + } + ], + "title": "Items", + "description": "Per-state items, each optionally overriding questions and model." + } + }, + "type": "object", + "title": "BulkPredictRequest", + "examples": [ + { + "questions": { + "department": { + "criteria": { + "billing": "invoices, payments, refunds", + "sales": "new purchases", + "technical": "bugs and outages" + }, + "instructions": "Which team should handle this request?", + "type": "choice" + }, + "refund": { + "instructions": "Does the customer ask for money back?", + "type": "noul" + }, + "urgency": { + "criteria": [ + "not urgent", + "soon", + "critical" + ], + "instructions": "How urgent is this request?", + "type": "score" + } + }, + "states": [ + "I was billed twice. Please refund the duplicate today.", + "The app crashes on launch." + ] + }, + { + "items": [ + { + "state": "I was billed twice." + }, + { + "model": "english", + "state": "The app crashes." + } + ], + "preset": "triage" + } + ] + }, + "BulkPredictResponse": { + "properties": { + "count": { + "type": "integer", + "title": "Count" + }, + "results": { + "items": { + "$ref": "#/components/schemas/BulkItemResult" + }, + "type": "array", + "title": "Results" + } + }, + "type": "object", + "required": [ + "count", + "results" + ], + "title": "BulkPredictResponse" + }, + "ChoiceAnswer": { + "properties": { + "type": { + "type": "string", + "const": "choice", + "title": "Type" + }, + "choice": { + "type": "string", + "title": "Choice" + }, + "probabilities": { + "additionalProperties": { + "type": "number" + }, + "type": "object", + "title": "Probabilities" + }, + "confidence": { + "type": "number", + "title": "Confidence" + }, + "action": { + "$ref": "#/components/schemas/ActionResult" + } + }, + "type": "object", + "required": [ + "type", + "choice", + "probabilities", + "confidence", + "action" + ], + "title": "ChoiceAnswer" + }, + "ChoiceQuestion": { + "properties": { + "type": { + "type": "string", + "const": "choice", + "title": "Type" + }, + "instructions": { + "type": "string", + "title": "Instructions", + "description": "What to decide, phrased as a question." + }, + "criteria": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Criteria", + "description": "Option label -> description, or a plain list of labels. Descriptions may be strings or any JSON value (rendered as compact JSON).", + "examples": [ + { + "billing": "invoices, payments, refunds", + "technical": [ + "bugs", + "outages" + ] + } + ] + } + }, + "type": "object", + "required": [ + "type", + "instructions" + ], + "title": "ChoiceQuestion", + "description": "Pick one labelled option, e.g. a routing department." + }, + "DetectRequest": { + "properties": { + "state": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], + "title": "State" + } + }, + "type": "object", + "required": [ + "state" + ], + "title": "DetectRequest" + }, + "EmailStateRequest": { + "properties": { + "subject": { + "type": "string", + "title": "Subject", + "default": "" + }, + "body": { + "type": "string", + "title": "Body" + }, + "sender": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Sender" + }, + "clean": { + "type": "boolean", + "title": "Clean", + "description": "Strip quoted history, signatures, disclaimers.", + "default": true + } + }, + "type": "object", + "required": [ + "body" + ], + "title": "EmailStateRequest" + }, + "HTTPValidationError": { + "properties": { + "detail": { + "items": { + "$ref": "#/components/schemas/ValidationError" + }, + "type": "array", + "title": "Detail" + } + }, + "type": "object", + "title": "HTTPValidationError" + }, + "HealthResponse": { + "properties": { + "status": { + "type": "string", + "title": "Status" + }, + "model_loaded": { + "type": "boolean", + "title": "Model Loaded" + }, + "device": { + "type": "string", + "title": "Device" + }, + "auth_enabled": { + "type": "boolean", + "title": "Auth Enabled" + }, + "auth_methods": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Auth Methods" + }, + "models": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Models", + "description": "Checkpoints currently resident." + }, + "available_models": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Available Models" + } + }, + "type": "object", + "required": [ + "status", + "model_loaded", + "device", + "auth_enabled", + "auth_methods", + "models", + "available_models" + ], + "title": "HealthResponse" + }, + "ModelsResponse": { + "properties": { + "available": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Available" + }, + "loaded": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Loaded" + } + }, + "type": "object", + "required": [ + "available", + "loaded" + ], + "title": "ModelsResponse" + }, + "NoulAnswer": { + "properties": { + "type": { + "type": "string", + "const": "noul", + "title": "Type" + }, + "noul": { + "type": "number", + "title": "Noul" + }, + "confidence": { + "type": "number", + "title": "Confidence" + }, + "action": { + "$ref": "#/components/schemas/ActionResult" + } + }, + "type": "object", + "required": [ + "type", + "noul", + "confidence", + "action" + ], + "title": "NoulAnswer" + }, + "NoulQuestion": { + "properties": { + "type": { + "type": "string", + "const": "noul", + "title": "Type" }, - "required": true + "instructions": { + "type": "string", + "title": "Instructions" + }, + "criteria": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Criteria", + "description": "Optional `false`/`true` descriptions (any JSON value).", + "examples": [ + { + "false": "a legitimate email", + "true": "phishing or fraud" + } + ] + } }, - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/TypeSafeResponse" - } + "type": "object", + "required": [ + "type", + "instructions" + ], + "title": "NoulQuestion", + "description": "Yes/no decision without a learned neutral class (n-o-u-l)." + }, + "PredictRequest": { + "properties": { + "state": { + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" } - } + ], + "title": "State", + "description": "Text, JSON object, or conversation turns to analyse." }, - "401": { - "description": "Missing or invalid credentials" + "questions": { + "anyOf": [ + { + "additionalProperties": { + "oneOf": [ + { + "$ref": "#/components/schemas/ChoiceQuestion" + }, + { + "$ref": "#/components/schemas/ScoreQuestion" + }, + { + "$ref": "#/components/schemas/NoulQuestion" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "choice": "#/components/schemas/ChoiceQuestion", + "noul": "#/components/schemas/NoulQuestion", + "score": "#/components/schemas/ScoreQuestion" + } + } + }, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Questions", + "description": "Map of question id -> typed question definition." }, - "503": { - "description": "Model not loaded yet" + "preset": { + "anyOf": [ + { + "type": "string", + "enum": [ + "triage", + "email", + "guard", + "moderation", + "router" + ] + }, + { + "type": "null" + } + ], + "title": "Preset", + "description": "Built-in question set to use instead of `questions`." }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } + "model": { + "anyOf": [ + { + "type": "string", + "enum": [ + "english", + "multilingual", + "typed-decisions" + ] + }, + { + "type": "null" } - } + ], + "title": "Model", + "description": "Pin a checkpoint; omit to auto-route by language." } }, - "security": [ - { - "ApiKeyAuth": [] - }, + "type": "object", + "required": [ + "state" + ], + "title": "PredictRequest", + "examples": [ { - "BearerAuth": [] + "questions": { + "department": { + "criteria": { + "billing": "invoices, payments, refunds", + "sales": "new purchases", + "technical": "bugs and outages" + }, + "instructions": "Which team should handle this request?", + "type": "choice" + }, + "refund": { + "instructions": "Does the customer ask for money back?", + "type": "noul" + }, + "urgency": { + "criteria": [ + "not urgent", + "soon", + "critical" + ], + "instructions": "How urgent is this request?", + "type": "score" + } + }, + "state": "I was billed twice. Please refund the duplicate today." }, { - "BasicAuth": [] + "preset": "triage", + "state": "Hi, we were billed twice for March." } ] - } - } - }, - "components": { - "schemas": { - "HTTPValidationError": { + }, + "PredictResponse": { "properties": { - "detail": { - "items": { - "$ref": "#/components/schemas/ValidationError" + "model": { + "type": "string", + "title": "Model" + }, + "answers": { + "additionalProperties": { + "oneOf": [ + { + "$ref": "#/components/schemas/ChoiceAnswer" + }, + { + "$ref": "#/components/schemas/ScoreAnswer" + }, + { + "$ref": "#/components/schemas/NoulAnswer" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "choice": "#/components/schemas/ChoiceAnswer", + "noul": "#/components/schemas/NoulAnswer", + "score": "#/components/schemas/ScoreAnswer" + } + } + }, + "type": "object", + "title": "Answers" + }, + "usage": { + "$ref": "#/components/schemas/Usage" + }, + "routing": { + "anyOf": [ + { + "$ref": "#/components/schemas/Routing" + }, + { + "type": "null" + } + ], + "description": "How the checkpoint was chosen (auto-route or pinned)." + } + }, + "additionalProperties": true, + "type": "object", + "required": [ + "model", + "answers", + "usage" + ], + "title": "PredictResponse" + }, + "Routing": { + "properties": { + "model": { + "type": "string", + "title": "Model" + }, + "reason": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Reason" + } + }, + "additionalProperties": true, + "type": "object", + "required": [ + "model" + ], + "title": "Routing" + }, + "ScoreAnswer": { + "properties": { + "type": { + "type": "string", + "const": "score", + "title": "Type" + }, + "score": { + "type": "number", + "title": "Score", + "description": "Expected value on the 0-based criteria scale." + }, + "legend": { + "additionalProperties": { + "type": "string" + }, + "type": "object", + "title": "Legend" + }, + "probabilities": { + "additionalProperties": { + "type": "number" }, + "type": "object", + "title": "Probabilities" + }, + "confidence": { + "type": "number", + "title": "Confidence" + }, + "action": { + "$ref": "#/components/schemas/ActionResult" + } + }, + "type": "object", + "required": [ + "type", + "score", + "legend", + "probabilities", + "confidence", + "action" + ], + "title": "ScoreAnswer" + }, + "ScoreQuestion": { + "properties": { + "type": { + "type": "string", + "const": "score", + "title": "Type" + }, + "instructions": { + "type": "string", + "title": "Instructions" + }, + "criteria": { + "items": {}, "type": "array", - "title": "Detail" + "title": "Criteria", + "description": "Ordered levels, lowest first. Each may be any JSON value.", + "examples": [ + [ + "low", + { + "level": "high", + "sla_minutes": 60 + } + ] + ] } }, "type": "object", - "title": "HTTPValidationError" + "required": [ + "type", + "instructions", + "criteria" + ], + "title": "ScoreQuestion", + "description": "Rate on an ordered scale, e.g. urgency." }, "TypeSafeChoiceAnswer": { "properties": { @@ -459,6 +1708,24 @@ ], "title": "TypeSafeUsage" }, + "Usage": { + "properties": { + "input_tokens": { + "type": "integer", + "title": "Input Tokens" + }, + "output_tokens": { + "type": "integer", + "title": "Output Tokens" + } + }, + "type": "object", + "required": [ + "input_tokens", + "output_tokens" + ], + "title": "Usage" + }, "ValidationError": { "properties": { "loc": { From 0cfdfbb09b39002974f780c95729831c7b9116c6 Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:14:58 +0100 Subject: [PATCH 11/16] ci: add matrix publishing for slim and pre-baked model tags, update README --- .github/workflows/publish.yml | 66 +++++++++++++++++++++++++++-------- README.md | 17 +++++++-- 2 files changed, 65 insertions(+), 18 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 1d637d8..8b48845 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -66,46 +66,68 @@ jobs: - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - - name: Build image + - name: Build image with baked english model uses: docker/build-push-action@v6 with: context: . load: true - tags: laya-api:test - cache-from: type=gha,scope=amd64 + tags: laya-api:smoke + build-args: | + PRELOAD_MODEL=1 + MODELS=english + cache-from: type=gha,scope=amd64-english - name: Full inference smoke test run: | - docker run -d --name laya-api -p 8000:8000 -e API_KEYS=test laya-api:test + docker run -d --name laya-api -p 8000:8000 -e API_KEYS=test laya-api:smoke for i in $(seq 1 150); do if curl -sf http://localhost:8000/healthz >/dev/null 2>&1; then break; fi sleep 2 done curl -sf http://localhost:8000/healthz + curl -sf -X POST http://localhost:8000/predict \ + -H 'X-API-Key: test' -H 'Content-Type: application/json' \ + -d '{"state":"I was billed twice. Please refund.","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}' curl -sf -X POST http://localhost:8000/v1/systemone \ -H 'X-API-Key: test' -H 'Content-Type: application/json' \ -d '{"state":"I was billed twice. Please refund.","model":"laya-english","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}' docker logs laya-api docker rm -f laya-api - # Build each platform natively (no QEMU) and push by digest. + # Build each platform and model variant natively (no QEMU) and push by digest. build: needs: lint if: github.event_name != 'pull_request' - runs-on: ${{ matrix.runner }} + runs-on: ${{ matrix.platform_info.runner }} permissions: contents: read packages: write strategy: fail-fast: false matrix: - include: + platform_info: - platform: linux/amd64 runner: ubuntu-latest scope: amd64 - platform: linux/arm64 runner: ubuntu-24.04-arm scope: arm64 + variant: + - name: default + preload: "0" + models: "english" + - name: english + preload: "1" + models: "english" + - name: multilingual + preload: "1" + models: "multilingual" + - name: typed-decisions + preload: "1" + models: "typed-decisions" + - name: all + preload: "1" + models: "english,multilingual,typed-decisions" steps: - uses: actions/checkout@v4 @@ -124,10 +146,13 @@ jobs: uses: docker/build-push-action@v6 with: context: . - platforms: ${{ matrix.platform }} + platforms: ${{ matrix.platform_info.platform }} + build-args: | + PRELOAD_MODEL=${{ matrix.variant.preload }} + MODELS=${{ matrix.variant.models }} outputs: type=image,name=${{ env.IMAGE }},push-by-digest=true,name-canonical=true,push=true - cache-from: type=gha,scope=${{ matrix.scope }} - cache-to: type=gha,mode=max,scope=${{ matrix.scope }} + cache-from: type=gha,scope=${{ matrix.platform_info.scope }}-${{ matrix.variant.name }} + cache-to: type=gha,mode=max,scope=${{ matrix.platform_info.scope }}-${{ matrix.variant.name }} - name: Export digest run: | @@ -138,24 +163,33 @@ jobs: - name: Upload digest uses: actions/upload-artifact@v4 with: - name: digests-${{ matrix.scope }} + name: digests-${{ matrix.variant.name }}-${{ matrix.platform_info.scope }} path: /tmp/digests/* if-no-files-found: error retention-days: 1 - # Combine the per-platform digests into one multi-arch manifest list. + # Combine per-platform digests into multi-arch manifest lists for each variant. merge: needs: build runs-on: ubuntu-latest permissions: contents: read packages: write + strategy: + fail-fast: false + matrix: + variant: + - name: default + - name: english + - name: multilingual + - name: typed-decisions + - name: all steps: - name: Download digests uses: actions/download-artifact@v4 with: path: /tmp/digests - pattern: digests-* + pattern: digests-${{ matrix.variant.name }}-* merge-multiple: true - name: Set up Docker Buildx @@ -174,10 +208,12 @@ jobs: with: images: ${{ env.IMAGE }} tags: | - type=raw,value=latest + type=raw,value=${{ matrix.variant.name == 'default' && 'latest' || matrix.variant.name }},enable=${{ github.ref == format('refs/heads/{0}', github.event.repository.default_branch) }} type=ref,event=branch type=semver,pattern={{version}} type=semver,pattern={{major}}.{{minor}} + flavor: | + suffix=${{ matrix.variant.name == 'default' && '' || format('-{0}', matrix.variant.name) }} - name: Create manifest list and push working-directory: /tmp/digests @@ -186,4 +222,4 @@ jobs: $(printf '${{ env.IMAGE }}@sha256:%s ' *) - name: Inspect image - run: docker buildx imagetools inspect ${{ env.IMAGE }}:${{ steps.meta.outputs.version }} + run: docker buildx imagetools inspect $(jq -r '.tags[0]' <<< "$DOCKER_METADATA_OUTPUT_JSON") diff --git a/README.md b/README.md index 610cfc4..a14045e 100644 --- a/README.md +++ b/README.md @@ -49,11 +49,22 @@ docker pull ghcr.io/chneau/laya docker run -d -p 8000:8000 -e API_KEYS=key1 -v hf-cache:/data/hf ghcr.io/chneau/laya ``` -First start downloads the checkpoint (~1 GB) into the `hf-cache` volume. Bake it -into the image for instant/offline startup with `PRELOAD_MODEL=1`: +### 🏷️ Docker Image Tags & Model Variants + +Multi-architecture images (`linux/amd64` and `linux/arm64`) are published to GitHub Container Registry under several tags: + +| Image Tag | Preloaded Models | Image Size | Description | +| :--- | :--- | :--- | :--- | +| `ghcr.io/chneau/laya:latest` (or `v0.5.0`) | None (Dynamic) | ~300 MB | **Slim / Default**: Small image size. Downloads model on first run into `/data/hf`. | +| `ghcr.io/chneau/laya:english` | `english` | ~1.3 GB | **Instant Startup (English)**: Pre-baked English checkpoint, offline-ready. | +| `ghcr.io/chneau/laya:multilingual` | `multilingual` | ~1.8 GB | **Instant Startup (Multilingual)**: Pre-baked multilingual checkpoint. | +| `ghcr.io/chneau/laya:typed-decisions` | `typed-decisions` | ~1.3 GB | **Instant Startup (Typed Decisions)**: Pre-baked typed decisions checkpoint. | +| `ghcr.io/chneau/laya:all` | All 3 models | ~3.5 GB | **Full Bundle**: All checkpoints pre-baked for zero-latency multi-model routing. | + +#### Running a Pre-baked Image (Instant Startup & Air-gapped / Offline) ```bash -make build-model MODEL=english +docker run -d -p 8000:8000 -e API_KEYS=key1 ghcr.io/chneau/laya:english ``` ### Check Health From 476fb2269c5ce2d152e03b43c9c31d090d24dce5 Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:18:01 +0100 Subject: [PATCH 12/16] fix: remove unused _PRESET_NAMES constant --- app/main.py | 1 - 1 file changed, 1 deletion(-) diff --git a/app/main.py b/app/main.py index f5a9023..53a8520 100644 --- a/app/main.py +++ b/app/main.py @@ -88,7 +88,6 @@ async def lifespan(app: FastAPI): State = str | dict[str, Any] | list[Any] CriteriaValue = Any # str, number, bool, list or dict; rendered as compact JSON -_PRESET_NAMES = ("triage", "email", "guard", "moderation", "router") PresetName = Literal["triage", "email", "guard", "moderation", "router"] _presets_cache: dict[str, dict[str, Any]] = {} From 7b5d920a679b83ba50e8adf7e9d7504c23c24bcc Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:23:50 +0100 Subject: [PATCH 13/16] refactor: unify Question models and rename wrappers to SystemOne --- README.md | 4 +- app/main.py | 85 +++++----------- openapi.json | 283 +++++++++++++-------------------------------------- 3 files changed, 98 insertions(+), 274 deletions(-) diff --git a/README.md b/README.md index a14045e..8d1b921 100644 --- a/README.md +++ b/README.md @@ -93,7 +93,7 @@ curl http://localhost:8000/healthz | `POST` | `/email/state` | yes* | Clean + structure an email as a state | | `POST` | `/predict` | yes* | Typed questions over one state | | `POST` | `/predict/bulk` | yes* | Same questions over many states | -| `POST` | `/v1/systemone` | yes* | TypeSafe-compatible typed prediction | +| `POST` | `/v1/systemone` | yes* | SystemOne / TypeSafe compatible prediction | \* Enforced only when `API_KEYS` and/or `BASIC_AUTH` is set. Any of these works: @@ -150,7 +150,7 @@ with per-state errors isolated as `{"ok": false, "error": "..."}`. --- -### TypeSafe Compatible Prediction (`POST /v1/systemone`) +### SystemOne / TypeSafe Compatible Prediction (`POST /v1/systemone`) ```bash curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Content-Type: application/json' -d '{ diff --git a/app/main.py b/app/main.py index 53a8520..76c806e 100644 --- a/app/main.py +++ b/app/main.py @@ -37,7 +37,7 @@ AVAILABLE_MODELS = ("english", "multilingual", "typed-decisions") ModelName = Literal["english", "multilingual", "typed-decisions"] -TypeSafeModelName = Literal[ +SystemOneModelName = Literal[ "laya-english", "laya-multilingual", "laya-typed-decisions", @@ -86,6 +86,7 @@ async def lifespan(app: FastAPI): ) State = str | dict[str, Any] | list[Any] +InstructionValue = str | dict[str, Any] | list[Any] CriteriaValue = Any # str, number, bool, list or dict; rendered as compact JSON PresetName = Literal["triage", "email", "guard", "moderation", "router"] @@ -121,7 +122,9 @@ class ChoiceQuestion(BaseModel): """Pick one labelled option, e.g. a routing department.""" type: Literal["choice"] - instructions: str = Field(..., description="What to decide, phrased as a question.") + instructions: InstructionValue = Field( + ..., description="What to decide, phrased as a question." + ) criteria: dict[str, CriteriaValue] | list[CriteriaValue] | None = Field( default=None, description=( @@ -136,7 +139,7 @@ class ScoreQuestion(BaseModel): """Rate on an ordered scale, e.g. urgency.""" type: Literal["score"] - instructions: str + instructions: InstructionValue criteria: list[CriteriaValue] = Field( ..., description="Ordered levels, lowest first. Each may be any JSON value.", @@ -148,7 +151,7 @@ class NoulQuestion(BaseModel): """Yes/no decision without a learned neutral class (n-o-u-l).""" type: Literal["noul"] - instructions: str + instructions: InstructionValue criteria: dict[str, CriteriaValue] | None = Field( default=None, description="Optional `false`/`true` descriptions (any JSON value).", @@ -333,52 +336,19 @@ class PredictResponse(BaseModel): ) -TypeSafeInstruction = str | dict[str, Any] | list[Any] - - -class TypeSafeChoiceQuestion(BaseModel): - type: Literal["choice"] - instructions: TypeSafeInstruction - criteria: dict[str, str | dict[str, Any] | list[Any] | None] - - -class TypeSafeScoreQuestion(BaseModel): - type: Literal["score"] - instructions: TypeSafeInstruction - criteria: list[str | dict[str, Any] | list[Any]] - - -class TypeSafeNoulQuestion(BaseModel): - type: Literal["noul"] - instructions: TypeSafeInstruction - criteria: dict[str, str | dict[str, Any] | list[Any] | None] | None = None - - -TypeSafeQuestion = Annotated[ - TypeSafeChoiceQuestion | TypeSafeScoreQuestion | TypeSafeNoulQuestion, - Field(discriminator="type"), -] - - -class TypeSafeRequest(BaseModel): - state: State - model: TypeSafeModelName - questions: dict[str, TypeSafeQuestion] - - -class TypeSafeNoulAnswer(BaseModel): +class SystemOneNoulAnswer(BaseModel): type: Literal["noul"] noul: float -class TypeSafeChoiceAnswer(BaseModel): +class SystemOneChoiceAnswer(BaseModel): type: Literal["choice"] choice: str probabilities: dict[str, float] confidence: float -class TypeSafeScoreAnswer(BaseModel): +class SystemOneScoreAnswer(BaseModel): type: Literal["score"] score: float legend: dict[str, str] @@ -386,21 +356,22 @@ class TypeSafeScoreAnswer(BaseModel): confidence: float -TypeSafeAnswer = Annotated[ - TypeSafeNoulAnswer | TypeSafeChoiceAnswer | TypeSafeScoreAnswer, +SystemOneAnswer = Annotated[ + SystemOneNoulAnswer | SystemOneChoiceAnswer | SystemOneScoreAnswer, Field(discriminator="type"), ] -class TypeSafeUsage(BaseModel): - input_tokens: int - output_tokens: int +class SystemOneRequest(BaseModel): + state: State + model: SystemOneModelName + questions: dict[str, Question] -class TypeSafeResponse(BaseModel): +class SystemOneResponse(BaseModel): model: str - answers: dict[str, TypeSafeAnswer] - usage: TypeSafeUsage + answers: dict[str, SystemOneAnswer] + usage: Usage class BulkItemResult(BaseModel): @@ -505,7 +476,7 @@ def _resolve( return _dump(questions or {}) -_TYPESAFE_MODELS: dict[TypeSafeModelName, ModelName] = { +_SYSTEMONE_MODELS: dict[SystemOneModelName, ModelName] = { "laya-english": "english", "laya-multilingual": "multilingual", "laya-typed-decisions": "typed-decisions", @@ -663,26 +634,26 @@ def predict_bulk(request: BulkPredictRequest) -> BulkPredictResponse: @app.post( "/v1/systemone", - tags=["typesafe", "predict"], - response_model=TypeSafeResponse, - summary="TypeSafe-compatible prediction", + tags=["systemone", "predict"], + response_model=SystemOneResponse, + summary="SystemOne / TypeSafe-compatible prediction", dependencies=[Depends(require_auth)], responses=_AUTH_RESPONSES, ) -def typesafe_predict(request: TypeSafeRequest) -> TypeSafeResponse: - """Evaluate a TypeSafe-shaped request using a Laya checkpoint.""" +def systemone_predict(request: SystemOneRequest) -> SystemOneResponse: + """Evaluate a SystemOne / TypeSafe-shaped request using a Laya checkpoint.""" router = _router() with _lock: result = router.predict( request.state, _dump(request.questions), - model=_TYPESAFE_MODELS[request.model], + model=_SYSTEMONE_MODELS[request.model], ) validated = PredictResponse.model_validate(result) - return TypeSafeResponse( + return SystemOneResponse( model=request.model, answers={key: answer.model_dump() for key, answer in validated.answers.items()}, - usage=TypeSafeUsage.model_validate(validated.usage.model_dump()), + usage=validated.usage, ) diff --git a/openapi.json b/openapi.json index e2ca6dc..e7a1ad8 100644 --- a/openapi.json +++ b/openapi.json @@ -373,17 +373,17 @@ "/v1/systemone": { "post": { "tags": [ - "typesafe", + "systemone", "predict" ], - "summary": "TypeSafe-compatible prediction", - "description": "Evaluate a TypeSafe-shaped request using a Laya checkpoint.", - "operationId": "typesafe_predict_v1_systemone_post", + "summary": "SystemOne / TypeSafe-compatible prediction", + "description": "Evaluate a SystemOne / TypeSafe-shaped request using a Laya checkpoint.", + "operationId": "systemone_predict_v1_systemone_post", "requestBody": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/TypeSafeRequest" + "$ref": "#/components/schemas/SystemOneRequest" } } }, @@ -395,7 +395,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/TypeSafeResponse" + "$ref": "#/components/schemas/SystemOneResponse" } } } @@ -785,7 +785,19 @@ "title": "Type" }, "instructions": { - "type": "string", + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], "title": "Instructions", "description": "What to decide, phrased as a question." }, @@ -1010,7 +1022,19 @@ "title": "Type" }, "instructions": { - "type": "string", + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], "title": "Instructions" }, "criteria": { @@ -1302,7 +1326,19 @@ "title": "Type" }, "instructions": { - "type": "string", + "anyOf": [ + { + "type": "string" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "items": {}, + "type": "array" + } + ], "title": "Instructions" }, "criteria": { @@ -1330,7 +1366,7 @@ "title": "ScoreQuestion", "description": "Rate on an ordered scale, e.g. urgency." }, - "TypeSafeChoiceAnswer": { + "SystemOneChoiceAnswer": { "properties": { "type": { "type": "string", @@ -1360,63 +1396,9 @@ "probabilities", "confidence" ], - "title": "TypeSafeChoiceAnswer" - }, - "TypeSafeChoiceQuestion": { - "properties": { - "type": { - "type": "string", - "const": "choice", - "title": "Type" - }, - "instructions": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ], - "title": "Instructions" - }, - "criteria": { - "additionalProperties": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - }, - { - "type": "null" - } - ] - }, - "type": "object", - "title": "Criteria" - } - }, - "type": "object", - "required": [ - "type", - "instructions", - "criteria" - ], - "title": "TypeSafeChoiceQuestion" + "title": "SystemOneChoiceAnswer" }, - "TypeSafeNoulAnswer": { + "SystemOneNoulAnswer": { "properties": { "type": { "type": "string", @@ -1433,69 +1415,9 @@ "type", "noul" ], - "title": "TypeSafeNoulAnswer" + "title": "SystemOneNoulAnswer" }, - "TypeSafeNoulQuestion": { - "properties": { - "type": { - "type": "string", - "const": "noul", - "title": "Type" - }, - "instructions": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ], - "title": "Instructions" - }, - "criteria": { - "anyOf": [ - { - "additionalProperties": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - }, - { - "type": "null" - } - ] - }, - "type": "object" - }, - { - "type": "null" - } - ], - "title": "Criteria" - } - }, - "type": "object", - "required": [ - "type", - "instructions" - ], - "title": "TypeSafeNoulQuestion" - }, - "TypeSafeRequest": { + "SystemOneRequest": { "properties": { "state": { "anyOf": [ @@ -1526,21 +1448,21 @@ "additionalProperties": { "oneOf": [ { - "$ref": "#/components/schemas/TypeSafeChoiceQuestion" + "$ref": "#/components/schemas/ChoiceQuestion" }, { - "$ref": "#/components/schemas/TypeSafeScoreQuestion" + "$ref": "#/components/schemas/ScoreQuestion" }, { - "$ref": "#/components/schemas/TypeSafeNoulQuestion" + "$ref": "#/components/schemas/NoulQuestion" } ], "discriminator": { "propertyName": "type", "mapping": { - "choice": "#/components/schemas/TypeSafeChoiceQuestion", - "noul": "#/components/schemas/TypeSafeNoulQuestion", - "score": "#/components/schemas/TypeSafeScoreQuestion" + "choice": "#/components/schemas/ChoiceQuestion", + "noul": "#/components/schemas/NoulQuestion", + "score": "#/components/schemas/ScoreQuestion" } } }, @@ -1554,9 +1476,9 @@ "model", "questions" ], - "title": "TypeSafeRequest" + "title": "SystemOneRequest" }, - "TypeSafeResponse": { + "SystemOneResponse": { "properties": { "model": { "type": "string", @@ -1566,21 +1488,21 @@ "additionalProperties": { "oneOf": [ { - "$ref": "#/components/schemas/TypeSafeNoulAnswer" + "$ref": "#/components/schemas/SystemOneNoulAnswer" }, { - "$ref": "#/components/schemas/TypeSafeChoiceAnswer" + "$ref": "#/components/schemas/SystemOneChoiceAnswer" }, { - "$ref": "#/components/schemas/TypeSafeScoreAnswer" + "$ref": "#/components/schemas/SystemOneScoreAnswer" } ], "discriminator": { "propertyName": "type", "mapping": { - "choice": "#/components/schemas/TypeSafeChoiceAnswer", - "noul": "#/components/schemas/TypeSafeNoulAnswer", - "score": "#/components/schemas/TypeSafeScoreAnswer" + "choice": "#/components/schemas/SystemOneChoiceAnswer", + "noul": "#/components/schemas/SystemOneNoulAnswer", + "score": "#/components/schemas/SystemOneScoreAnswer" } } }, @@ -1588,7 +1510,7 @@ "title": "Answers" }, "usage": { - "$ref": "#/components/schemas/TypeSafeUsage" + "$ref": "#/components/schemas/Usage" } }, "type": "object", @@ -1597,9 +1519,9 @@ "answers", "usage" ], - "title": "TypeSafeResponse" + "title": "SystemOneResponse" }, - "TypeSafeScoreAnswer": { + "SystemOneScoreAnswer": { "properties": { "type": { "type": "string", @@ -1637,76 +1559,7 @@ "probabilities", "confidence" ], - "title": "TypeSafeScoreAnswer" - }, - "TypeSafeScoreQuestion": { - "properties": { - "type": { - "type": "string", - "const": "score", - "title": "Type" - }, - "instructions": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ], - "title": "Instructions" - }, - "criteria": { - "items": { - "anyOf": [ - { - "type": "string" - }, - { - "additionalProperties": true, - "type": "object" - }, - { - "items": {}, - "type": "array" - } - ] - }, - "type": "array", - "title": "Criteria" - } - }, - "type": "object", - "required": [ - "type", - "instructions", - "criteria" - ], - "title": "TypeSafeScoreQuestion" - }, - "TypeSafeUsage": { - "properties": { - "input_tokens": { - "type": "integer", - "title": "Input Tokens" - }, - "output_tokens": { - "type": "integer", - "title": "Output Tokens" - } - }, - "type": "object", - "required": [ - "input_tokens", - "output_tokens" - ], - "title": "TypeSafeUsage" + "title": "SystemOneScoreAnswer" }, "Usage": { "properties": { From 12ff86e3029ea4f5080a40fb3f77afc1d02472f4 Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:30:42 +0100 Subject: [PATCH 14/16] feat: full OpenRouter SystemOne spec conformance with flexible model mapping and aliases --- app/main.py | 47 ++++++++++++++++++++++++++++++++++++++++++----- openapi.json | 48 ++++++++++++++++++++++++++++++++++++++++-------- 2 files changed, 82 insertions(+), 13 deletions(-) diff --git a/app/main.py b/app/main.py index 76c806e..14b40d8 100644 --- a/app/main.py +++ b/app/main.py @@ -363,13 +363,24 @@ class SystemOneScoreAnswer(BaseModel): class SystemOneRequest(BaseModel): - state: State - model: SystemOneModelName - questions: dict[str, Question] + model_config = ConfigDict(extra="ignore") + + state: State = Field(..., description="The content to evaluate: a string, dict, or list.") + model: str = Field( + ..., + description="System One model ID, e.g. laya-english, laya-multilingual, typed-decisions.", + ) + questions: dict[str, Question] = Field(..., description="Typed questions map.") + session_id: str | None = Field(default=None, description="Optional session identifier.") + user: str | None = Field(default=None, description="Optional user identifier.") class SystemOneResponse(BaseModel): + model_config = ConfigDict(extra="allow") + + id: str = Field(default_factory=lambda: f"gen-laya-{secrets.token_hex(12)}") model: str + provider: str = "Laya" answers: dict[str, SystemOneAnswer] usage: Usage @@ -476,13 +487,31 @@ def _resolve( return _dump(questions or {}) -_SYSTEMONE_MODELS: dict[SystemOneModelName, ModelName] = { +_SYSTEMONE_MODELS: dict[str, ModelName] = { + "english": "english", "laya-english": "english", + "multilingual": "multilingual", "laya-multilingual": "multilingual", + "typed-decisions": "typed-decisions", "laya-typed-decisions": "typed-decisions", } +def _resolve_systemone_model(model_name: str) -> ModelName: + name = model_name.lower().strip() + if name.startswith("typesafe/"): + name = name.removeprefix("typesafe/") + if name.startswith("laya/"): + name = name.removeprefix("laya/") + if name in _SYSTEMONE_MODELS: + return _SYSTEMONE_MODELS[name] + if "multi" in name: + return "multilingual" + if "decision" in name: + return "typed-decisions" + return "english" + + _AUTH_RESPONSES: dict[int | str, dict[str, Any]] = { 401: {"description": "Missing or invalid credentials"}, 503: {"description": "Model not loaded yet"}, @@ -640,14 +669,22 @@ def predict_bulk(request: BulkPredictRequest) -> BulkPredictResponse: dependencies=[Depends(require_auth)], responses=_AUTH_RESPONSES, ) +@app.post( + "/systemone", + include_in_schema=False, + response_model=SystemOneResponse, + dependencies=[Depends(require_auth)], + responses=_AUTH_RESPONSES, +) def systemone_predict(request: SystemOneRequest) -> SystemOneResponse: """Evaluate a SystemOne / TypeSafe-shaped request using a Laya checkpoint.""" router = _router() + model = _resolve_systemone_model(request.model) with _lock: result = router.predict( request.state, _dump(request.questions), - model=_SYSTEMONE_MODELS[request.model], + model=model, ) validated = PredictResponse.model_validate(result) return SystemOneResponse( diff --git a/openapi.json b/openapi.json index e7a1ad8..6a868f5 100644 --- a/openapi.json +++ b/openapi.json @@ -1433,16 +1433,13 @@ "type": "array" } ], - "title": "State" + "title": "State", + "description": "The content to evaluate: a string, dict, or list." }, "model": { "type": "string", - "enum": [ - "laya-english", - "laya-multilingual", - "laya-typed-decisions" - ], - "title": "Model" + "title": "Model", + "description": "System One model ID, e.g. laya-english, laya-multilingual, typed-decisions." }, "questions": { "additionalProperties": { @@ -1467,7 +1464,32 @@ } }, "type": "object", - "title": "Questions" + "title": "Questions", + "description": "Typed questions map." + }, + "session_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Session Id", + "description": "Optional session identifier." + }, + "user": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "User", + "description": "Optional user identifier." } }, "type": "object", @@ -1480,10 +1502,19 @@ }, "SystemOneResponse": { "properties": { + "id": { + "type": "string", + "title": "Id" + }, "model": { "type": "string", "title": "Model" }, + "provider": { + "type": "string", + "title": "Provider", + "default": "Laya" + }, "answers": { "additionalProperties": { "oneOf": [ @@ -1513,6 +1544,7 @@ "$ref": "#/components/schemas/Usage" } }, + "additionalProperties": true, "type": "object", "required": [ "model", From 391a2dcbd25a2681c3ccde6fc2d7073755c2c949 Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:36:09 +0100 Subject: [PATCH 15/16] feat: make MAX_BULK_ITEMS unlimited by default --- README.md | 2 +- app/main.py | 5 +++-- docker-compose.yml | 2 +- 3 files changed, 5 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 8d1b921..22f88ff 100644 --- a/README.md +++ b/README.md @@ -187,7 +187,7 @@ Supported model names for TypeSafe requests: | --- | --- | --- | | `API_KEYS` | *(empty)* | Comma-separated keys; empty disables API-key auth. | | `BASIC_AUTH` | *(empty)* | Comma-separated `user:password` pairs. | -| `MAX_BULK_ITEMS` | `256` | Max states per `/predict/bulk`. | +| `MAX_BULK_ITEMS` | *(empty / unlimited)* | Optional limit on states per `/predict/bulk` (unlimited by default). | | `PORT` | `8000` | HTTP port (host and container). | | `MODELS` | `english` | Checkpoints to preload: `english`, `multilingual`, `typed-decisions`. | | `MODEL_ID` | `convaiinnovations/laya` | Optional repo override (mirror/local path). | diff --git a/app/main.py b/app/main.py index 14b40d8..9385422 100644 --- a/app/main.py +++ b/app/main.py @@ -26,7 +26,8 @@ BASIC_AUTH.append((_user.strip(), _password.strip())) AUTH_ENABLED = bool(API_KEYS or BASIC_AUTH) -MAX_BULK_ITEMS = int(os.environ.get("MAX_BULK_ITEMS", "256")) +_max_bulk_env = os.environ.get("MAX_BULK_ITEMS", "").strip() +MAX_BULK_ITEMS: int | None = int(_max_bulk_env) if _max_bulk_env and _max_bulk_env != "0" else None # Checkpoints to keep resident at startup (comma-separated). MODELS = [m.strip() for m in os.environ.get("MODELS", "english").split(",") if m.strip()] @@ -637,7 +638,7 @@ def predict_bulk(request: BulkPredictRequest) -> BulkPredictResponse: questions = _resolve(request.questions, request.preset) jobs = [(state, questions, request.model) for state in request.states or []] - if len(jobs) > MAX_BULK_ITEMS: + if MAX_BULK_ITEMS is not None and len(jobs) > MAX_BULK_ITEMS: raise HTTPException( status_code=422, detail=f"Too many states: {len(jobs)} > MAX_BULK_ITEMS={MAX_BULK_ITEMS}", diff --git a/docker-compose.yml b/docker-compose.yml index 631a7be..aae3d87 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -12,7 +12,7 @@ services: PORT: ${PORT:-8000} API_KEYS: ${API_KEYS:-change-me} BASIC_AUTH: ${BASIC_AUTH:-} - MAX_BULK_ITEMS: ${MAX_BULK_ITEMS:-256} + MAX_BULK_ITEMS: ${MAX_BULK_ITEMS:-} MODELS: ${MODELS:-english} MODEL_ID: ${MODEL_ID:-convaiinnovations/laya} MODEL_SUBFOLDER: ${MODEL_SUBFOLDER:-} From e07a654ddf25bcf3001746ba9c5a9cff880e6ab1 Mon Sep 17 00:00:00 2001 From: chneau Date: Thu, 24 Sep 2026 22:39:50 +0100 Subject: [PATCH 16/16] chore: bump version to 0.6.0 --- README.md | 2 +- app/main.py | 2 +- openapi.json | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 22f88ff..12f9426 100644 --- a/README.md +++ b/README.md @@ -55,7 +55,7 @@ Multi-architecture images (`linux/amd64` and `linux/arm64`) are published to Git | Image Tag | Preloaded Models | Image Size | Description | | :--- | :--- | :--- | :--- | -| `ghcr.io/chneau/laya:latest` (or `v0.5.0`) | None (Dynamic) | ~300 MB | **Slim / Default**: Small image size. Downloads model on first run into `/data/hf`. | +| `ghcr.io/chneau/laya:latest` (or `v0.6.0`) | None (Dynamic) | ~300 MB | **Slim / Default**: Small image size. Downloads model on first run into `/data/hf`. | | `ghcr.io/chneau/laya:english` | `english` | ~1.3 GB | **Instant Startup (English)**: Pre-baked English checkpoint, offline-ready. | | `ghcr.io/chneau/laya:multilingual` | `multilingual` | ~1.8 GB | **Instant Startup (Multilingual)**: Pre-baked multilingual checkpoint. | | `ghcr.io/chneau/laya:typed-decisions` | `typed-decisions` | ~1.3 GB | **Instant Startup (Typed Decisions)**: Pre-baked typed decisions checkpoint. | diff --git a/app/main.py b/app/main.py index 9385422..3df929d 100644 --- a/app/main.py +++ b/app/main.py @@ -71,7 +71,7 @@ async def lifespan(app: FastAPI): app = FastAPI( title="Laya API", - version="0.5.0", + version="0.6.0", lifespan=lifespan, description=( "Run Laya typed decisions (choice / score / noul) over text, JSON objects " diff --git a/openapi.json b/openapi.json index 6a868f5..ee24a08 100644 --- a/openapi.json +++ b/openapi.json @@ -3,7 +3,7 @@ "info": { "title": "Laya API", "description": "Run Laya typed decisions (choice / score / noul) over text, JSON objects or conversation turns, using your own questions or a built-in preset. Requests are auto-routed to the best checkpoint, or pinned with `model`.\n\nAuthenticate with an API key (`X-API-Key` or `Authorization: Bearer`) or HTTP Basic, whichever is configured on the server.", - "version": "0.5.0" + "version": "0.6.0" }, "paths": { "/healthz": {