Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
70 changes: 53 additions & 17 deletions .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@ jobs:

- name: Validate OpenAPI schema
run: |
uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/predict' in s['paths'] and '/predict/bulk' in s['paths']; print('openapi ok:', list(s['paths']))"
uv run --no-project --with 'fastapi>=0.115' --with 'pydantic>=2' python -c "import sys; sys.path.insert(0, '.'); from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' in s['paths']; print('openapi ok:', list(s['paths']))"

# Pull requests: build the image once and check it boots.
test:
Expand All @@ -54,7 +54,7 @@ jobs:

- name: Container smoke test
run: |
docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/predict' in s['paths']; print('container ok')"
docker run --rm --entrypoint python laya-api:test -c "from app.main import app; s = app.openapi(); assert '/v1/systemone' in s['paths'] and '/predict' in s['paths']; print('container ok')"

# Nightly / manual / release: real model download + inference.
smoke:
Expand All @@ -66,17 +66,20 @@ jobs:
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3

- name: Build image
- name: Build image with baked english model
uses: docker/build-push-action@v6
with:
context: .
load: true
tags: laya-api:test
cache-from: type=gha,scope=amd64
tags: laya-api:smoke
build-args: |
PRELOAD_MODEL=1
MODELS=english
cache-from: type=gha,scope=amd64-english

- name: Full inference smoke test
run: |
docker run -d --name laya-api -p 8000:8000 -e API_KEYS=test laya-api:test
docker run -d --name laya-api -p 8000:8000 -e API_KEYS=test laya-api:smoke
for i in $(seq 1 150); do
if curl -sf http://localhost:8000/healthz >/dev/null 2>&1; then break; fi
sleep 2
Expand All @@ -85,27 +88,46 @@ jobs:
curl -sf -X POST http://localhost:8000/predict \
-H 'X-API-Key: test' -H 'Content-Type: application/json' \
-d '{"state":"I was billed twice. Please refund.","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}'
curl -sf -X POST http://localhost:8000/v1/systemone \
-H 'X-API-Key: test' -H 'Content-Type: application/json' \
-d '{"state":"I was billed twice. Please refund.","model":"laya-english","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for money back?"}}}'
docker logs laya-api
docker rm -f laya-api

# Build each platform natively (no QEMU) and push by digest.
# Build each platform and model variant natively (no QEMU) and push by digest.
build:
needs: lint
if: github.event_name != 'pull_request'
runs-on: ${{ matrix.runner }}
runs-on: ${{ matrix.platform_info.runner }}
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
include:
platform_info:
- platform: linux/amd64
runner: ubuntu-latest
scope: amd64
- platform: linux/arm64
runner: ubuntu-24.04-arm
scope: arm64
variant:
- name: default
preload: "0"
models: "english"
- name: english
preload: "1"
models: "english"
- name: multilingual
preload: "1"
models: "multilingual"
- name: typed-decisions
preload: "1"
models: "typed-decisions"
- name: all
preload: "1"
models: "english,multilingual,typed-decisions"
steps:
- uses: actions/checkout@v4

Expand All @@ -124,10 +146,13 @@ jobs:
uses: docker/build-push-action@v6
with:
context: .
platforms: ${{ matrix.platform }}
platforms: ${{ matrix.platform_info.platform }}
build-args: |
PRELOAD_MODEL=${{ matrix.variant.preload }}
MODELS=${{ matrix.variant.models }}
outputs: type=image,name=${{ env.IMAGE }},push-by-digest=true,name-canonical=true,push=true
cache-from: type=gha,scope=${{ matrix.scope }}
cache-to: type=gha,mode=max,scope=${{ matrix.scope }}
cache-from: type=gha,scope=${{ matrix.platform_info.scope }}-${{ matrix.variant.name }}
cache-to: type=gha,mode=max,scope=${{ matrix.platform_info.scope }}-${{ matrix.variant.name }}

- name: Export digest
run: |
Expand All @@ -138,24 +163,33 @@ jobs:
- name: Upload digest
uses: actions/upload-artifact@v4
with:
name: digests-${{ matrix.scope }}
name: digests-${{ matrix.variant.name }}-${{ matrix.platform_info.scope }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1

# Combine the per-platform digests into one multi-arch manifest list.
# Combine per-platform digests into multi-arch manifest lists for each variant.
merge:
needs: build
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
variant:
- name: default
- name: english
- name: multilingual
- name: typed-decisions
- name: all
steps:
- name: Download digests
uses: actions/download-artifact@v4
with:
path: /tmp/digests
pattern: digests-*
pattern: digests-${{ matrix.variant.name }}-*
merge-multiple: true

- name: Set up Docker Buildx
Expand All @@ -174,10 +208,12 @@ jobs:
with:
images: ${{ env.IMAGE }}
tags: |
type=raw,value=latest
type=raw,value=${{ matrix.variant.name == 'default' && 'latest' || matrix.variant.name }},enable=${{ github.ref == format('refs/heads/{0}', github.event.repository.default_branch) }}
type=ref,event=branch
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
flavor: |
suffix=${{ matrix.variant.name == 'default' && '' || format('-{0}', matrix.variant.name) }}

- name: Create manifest list and push
working-directory: /tmp/digests
Expand All @@ -186,4 +222,4 @@ jobs:
$(printf '${{ env.IMAGE }}@sha256:%s ' *)

- name: Inspect image
run: docker buildx imagetools inspect ${{ env.IMAGE }}:${{ steps.meta.outputs.version }}
run: docker buildx imagetools inspect $(jq -r '.tags[0]' <<< "$DOCKER_METADATA_OUTPUT_JSON")
12 changes: 7 additions & 5 deletions Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -23,20 +23,22 @@ RUN --mount=type=cache,target=/root/.cache/uv \

COPY --chown=appuser:appuser app ./app

# Prepare the cache before downloading so the model layer is not duplicated by
# a later recursive chown.
RUN mkdir -p /data/hf && chown -R appuser:appuser /app /data/hf

USER appuser

# Optional: bake the checkpoint(s) into the image for instant/offline startup.
# Build with --build-arg PRELOAD_MODEL=1 (adds ~1 GB, needs network at build).
ARG PRELOAD_MODEL=0
ARG MODELS=english
ENV MODELS=${MODELS}
RUN if [ "$PRELOAD_MODEL" = "1" ]; then \
MODELS="$MODELS" uv run --no-dev python -c \
"import os; from laya import Router; Router().preload([m.strip() for m in os.environ['MODELS'].split(',') if m.strip()])"; \
fi

# Own the app and the HF cache mount so a fresh named volume inherits appuser.
RUN mkdir -p /data/hf && chown -R appuser:appuser /app /data/hf

USER appuser

EXPOSE 8000

HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=5 \
Expand Down
6 changes: 5 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
PORT ?= 8000

.PHONY: help run format check fix openapi up down build logs
.PHONY: help run format check fix openapi up down build build-model logs

help:
@echo "make run - run the API locally (uvicorn --reload on $(PORT))"
Expand All @@ -11,6 +11,7 @@ help:
@echo "make up - docker compose up --build (detached)"
@echo "make down - docker compose down"
@echo "make build - docker build the image"
@echo "make build-model MODEL=english - build an image with one model baked in"
@echo "make logs - follow container logs"

run:
Expand All @@ -37,5 +38,8 @@ down:
build:
docker build -t laya-api:latest .

build-model:
docker build --build-arg PRELOAD_MODEL=1 --build-arg MODELS=$(MODEL) -t laya-api:$(MODEL) .

logs:
docker compose logs -f
59 changes: 55 additions & 4 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
Dockerized [Laya](https://huggingface.co/convaiinnovations/laya) prediction
service: loads one or more checkpoints behind a router, then serves typed
decisions (choice / score / noul) over HTTP, auto-routed by language or pinned
with `model`.
with `model`. Also supports TypeSafe API compatibility via `/v1/systemone`.

Built and published for `linux/amd64` and `linux/arm64`.

Expand All @@ -18,6 +18,8 @@ Built and published for `linux/amd64` and `linux/arm64`.
`typed-decisions`, auto-selected by language or pinned per request.
- 🎯 **Typed Decisions**: `choice` / `score` / `noul` questions with calibrated
probabilities, confidence and action probability.
- 🤝 **TypeSafe API Compatibility**: drop-in `/v1/systemone` endpoint compatible
with TypeSafe request/response schemas.
- 🔐 **Timing-Safe Auth**: API keys (`X-API-Key` / `Authorization: Bearer`) and
HTTP Basic, compared in constant time.
- 📑 **Interactive OpenAPI Docs**: Swagger UI (`/docs`), ReDoc (`/redoc`) and the
Expand Down Expand Up @@ -47,8 +49,23 @@ docker pull ghcr.io/chneau/laya
docker run -d -p 8000:8000 -e API_KEYS=key1 -v hf-cache:/data/hf ghcr.io/chneau/laya
```

First start downloads the checkpoint (~1 GB) into the `hf-cache` volume. Bake it
into the image for instant/offline startup with `PRELOAD_MODEL=1`.
### 🏷️ Docker Image Tags & Model Variants

Multi-architecture images (`linux/amd64` and `linux/arm64`) are published to GitHub Container Registry under several tags:

| Image Tag | Preloaded Models | Image Size | Description |
| :--- | :--- | :--- | :--- |
| `ghcr.io/chneau/laya:latest` (or `v0.6.0`) | None (Dynamic) | ~300 MB | **Slim / Default**: Small image size. Downloads model on first run into `/data/hf`. |
| `ghcr.io/chneau/laya:english` | `english` | ~1.3 GB | **Instant Startup (English)**: Pre-baked English checkpoint, offline-ready. |
| `ghcr.io/chneau/laya:multilingual` | `multilingual` | ~1.8 GB | **Instant Startup (Multilingual)**: Pre-baked multilingual checkpoint. |
| `ghcr.io/chneau/laya:typed-decisions` | `typed-decisions` | ~1.3 GB | **Instant Startup (Typed Decisions)**: Pre-baked typed decisions checkpoint. |
| `ghcr.io/chneau/laya:all` | All 3 models | ~3.5 GB | **Full Bundle**: All checkpoints pre-baked for zero-latency multi-model routing. |

#### Running a Pre-baked Image (Instant Startup & Air-gapped / Offline)

```bash
docker run -d -p 8000:8000 -e API_KEYS=key1 ghcr.io/chneau/laya:english
```

### Check Health

Expand Down Expand Up @@ -76,6 +93,7 @@ curl http://localhost:8000/healthz
| `POST` | `/email/state` | yes* | Clean + structure an email as a state |
| `POST` | `/predict` | yes* | Typed questions over one state |
| `POST` | `/predict/bulk` | yes* | Same questions over many states |
| `POST` | `/v1/systemone` | yes* | SystemOne / TypeSafe compatible prediction |

\* Enforced only when `API_KEYS` and/or `BASIC_AUTH` is set. Any of these works:

Expand All @@ -89,6 +107,8 @@ curl http://localhost:8000/healthz

## 🧠 Predicting

### Standard Prediction (`POST /predict`)

```bash
curl -X POST localhost:8000/predict -H 'X-API-Key: key1' -H 'Content-Type: application/json' -d '{
"state": "I was billed twice. Please refund the duplicate today.",
Expand Down Expand Up @@ -130,13 +150,44 @@ with per-state errors isolated as `{"ok": false, "error": "..."}`.

---

### SystemOne / TypeSafe Compatible Prediction (`POST /v1/systemone`)

```bash
curl -X POST localhost:8000/v1/systemone -H 'Authorization: Bearer key1' -H 'Content-Type: application/json' -d '{
"state": "I was billed twice. Please refund the duplicate today.",
"model": "laya-english",
"questions": {
"department": {"type": "choice", "instructions": "Which team?", "criteria": {"billing": "refunds", "technical": "bugs", "sales": "purchases"}},
"urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["not urgent", "soon", "critical"]},
"refund": {"type": "noul", "instructions": "Does the customer ask for money back?"}
}
}'
```

```json
{
"model": "laya-english",
"answers": {
"department": {"type": "choice", "choice": "billing", "probabilities": {"billing": 0.96, "technical": 0.02, "sales": 0.02}, "confidence": 0.82},
"urgency": {"type": "score", "score": 1.36, "legend": {"0": "not urgent", "1": "soon", "2": "critical"}, "confidence": 0.09},
"refund": {"type": "noul", "noul": 0.82}
},
"usage": {"input_tokens": 132, "output_tokens": 0}
}
```

Supported model names for TypeSafe requests:
`laya-english`, `laya-multilingual`, `laya-typed-decisions`.

---

## ⚙️ Configuration & Environment Variables

| Variable | Default | Description |
| --- | --- | --- |
| `API_KEYS` | *(empty)* | Comma-separated keys; empty disables API-key auth. |
| `BASIC_AUTH` | *(empty)* | Comma-separated `user:password` pairs. |
| `MAX_BULK_ITEMS` | `256` | Max states per `/predict/bulk`. |
| `MAX_BULK_ITEMS` | *(empty / unlimited)* | Optional limit on states per `/predict/bulk` (unlimited by default). |
| `PORT` | `8000` | HTTP port (host and container). |
| `MODELS` | `english` | Checkpoints to preload: `english`, `multilingual`, `typed-decisions`. |
| `MODEL_ID` | `convaiinnovations/laya` | Optional repo override (mirror/local path). |
Expand Down
Loading
Loading