Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,8 @@ One SIE cluster runs the inference behind a whole agent. Each task is a handful
| **Document to markdown** | PDFs, Office files, and scans become clean markdown. | [`lightonocr`](packages/sie_server/models/lightonai__LightOnOCR-2-1B.yaml), [`glm-ocr`](packages/sie_server/models/zai-org__GLM-OCR.yaml), [`mineru`](packages/sie_server/models/opendatalab__MinerU2.5-Pro-2604-1.2B.yaml), [`paddleocr-vl`](packages/sie_server/models/PaddlePaddle__PaddleOCR-VL-1.5.yaml), [`docling`](packages/sie_server/models/docling.yaml) |
| **Structured output** | Schema-valid JSON, extracted or generated. | [`gliner2`](packages/sie_server/models/fastino__gliner2-large-v1.yaml), [`nuner-zero`](packages/sie_server/models/numind__NuNER_Zero.yaml), [`qwen3.8-27b`](packages/sie_server/models/Qwen__Qwen3.8-27B-FP8.yaml), [`qwen3.6-27b`](packages/sie_server/models/Qwen__Qwen3.6-27B.yaml) |
| **Decide** | Choice, yes/no, and score answers with probabilities to typed questions about a text or JSON state. | [`laya`](packages/sie_server/models/convaiinnovations__laya.yaml), [`laya-multilingual`](packages/sie_server/models/convaiinnovations__laya-multilingual.yaml), [`laya-typed-decisions`](packages/sie_server/models/convaiinnovations__laya-typed-decisions.yaml) |
| **Guard content** | A Yes/No safety verdict, with the decision threshold tunable in the model config. | [`granite-guardian-2b`](packages/sie_server/models/ibm-granite__granite-guardian-3.0-2b.yaml) |
| **Classify** | Zero-shot labels, with several label groups answered in one call. The instruct models also follow a task instruction and few-shot examples. | [`gliclass-large-v3`](packages/sie_server/models/knowledgator__gliclass-large-v3.0.yaml), [`gliclass-instruct-large`](packages/sie_server/models/knowledgator__gliclass-instruct-large-v1.0.yaml), [`gliclass-multilang-mini`](packages/sie_server/models/knowledgator__gliclass-multilang-mini.yaml) |
| **Guard content** | A safety verdict: Yes/No with the threshold set in the model config, or safe/unsafe and policy-label scores with the threshold chosen per request. | [`granite-guardian-2b`](packages/sie_server/models/ibm-granite__granite-guardian-3.0-2b.yaml), [`opir-multitask-large`](packages/sie_server/models/knowledgator__opir-multitask-large-v1.0.yaml), [`opir-edge`](packages/sie_server/models/knowledgator__opir-edge-v1.0.yaml) |
| **Run the agent loop** | Plan steps and call tools with an open LLM, streaming included. | [`qwen3.8-27b`](packages/sie_server/models/Qwen__Qwen3.8-27B-FP8.yaml), [`qwen3.6-27b`](packages/sie_server/models/Qwen__Qwen3.6-27B.yaml) |
| **Translate** | Text between 400+ languages. | [`madlad400-3b-mt`](packages/sie_server/models/google__madlad400-3b-mt.yaml) |
| **See images** | Caption, detect objects, and answer questions about images. | [`florence-2`](packages/sie_server/models/microsoft__Florence-2-large.yaml), [`owlv2`](packages/sie_server/models/google__owlv2-base-patch16-ensemble.yaml), [`grounding-dino`](packages/sie_server/models/IDEA-Research__grounding-dino-base.yaml) |
Expand Down
27 changes: 27 additions & 0 deletions packages/sie_gateway/src/handlers/proxy.rs
Original file line number Diff line number Diff line change
Expand Up @@ -374,6 +374,10 @@ const RESOURCE_EXHAUSTED_RETRY_AFTER: &str = RetryAfter::DEFAULT.resource_exhaus
const LORA_LOADING_ERROR_CODE: &str = "LORA_LOADING";
const LORA_LOADING_RETRY_AFTER: &str = RetryAfter::DEFAULT.lora_loading;
const INVALID_INPUT_ERROR_CODE: &str = "INVALID_INPUT";
/// Worker-side input exceeds the model's context window (for example a label
/// set that does not fit). Caller-fixable, so it maps to 400 like
/// ``INVALID_INPUT``; see ``sie_server.adapters.errors.InputTooLongError``.
const INPUT_TOO_LONG_ERROR_CODE: &str = "INPUT_TOO_LONG";
const PAYLOAD_TOO_LARGE_ERROR_CODE: &str = err_code::PAYLOAD_TOO_LARGE;

/// Fallback `max_tokens` applied to a chat-completions request that
Expand Down Expand Up @@ -9002,6 +9006,7 @@ fn unanimous_terminal_client_error(
let first = errors.first()?.error_code.as_deref()?;
let (status, canonical) = match first {
INVALID_INPUT_ERROR_CODE => (StatusCode::BAD_REQUEST, INVALID_INPUT_ERROR_CODE),
INPUT_TOO_LONG_ERROR_CODE => (StatusCode::BAD_REQUEST, INPUT_TOO_LONG_ERROR_CODE),
PAYLOAD_TOO_LARGE_ERROR_CODE => {
(StatusCode::PAYLOAD_TOO_LARGE, PAYLOAD_TOO_LARGE_ERROR_CODE)
}
Expand Down Expand Up @@ -18468,6 +18473,7 @@ mod tests {
fn test_unanimous_terminal_client_errors_map_to_400_and_413() {
for (code, status) in [
(INVALID_INPUT_ERROR_CODE, StatusCode::BAD_REQUEST),
(INPUT_TOO_LONG_ERROR_CODE, StatusCode::BAD_REQUEST),
(PAYLOAD_TOO_LARGE_ERROR_CODE, StatusCode::PAYLOAD_TOO_LARGE),
] {
let first = _err_result(Some(code), "rejected 1");
Expand All @@ -18485,6 +18491,27 @@ mod tests {
unanimous_terminal_client_error(&[&invalid, &oversized]),
None
);
let too_long = _err_result(Some(INPUT_TOO_LONG_ERROR_CODE), "labels do not fit");
let failed = _err_result(Some("inference_error"), "backend failure");
assert_eq!(unanimous_terminal_client_error(&[&too_long, &failed]), None);
}

#[tokio::test]
async fn test_input_too_long_is_a_native_400_with_its_code() {
let too_long = _err_result(Some(INPUT_TOO_LONG_ERROR_CODE), "labels do not fit");
let (status, code) = unanimous_terminal_client_error(&[&too_long]).unwrap();
let response = build_terminal_client_error_response(status, code, "labels do not fit");
assert_eq!(response.status(), StatusCode::BAD_REQUEST);
assert_eq!(
response.headers().get("x-sie-error-code").unwrap(),
INPUT_TOO_LONG_ERROR_CODE
);
let body = axum::body::to_bytes(response.into_body(), 16 * 1024)
.await
.unwrap();
let value: serde_json::Value = serde_json::from_slice(&body).unwrap();
assert_eq!(value["detail"]["code"], INPUT_TOO_LONG_ERROR_CODE);
assert_eq!(value["detail"]["message"], "labels do not fit");
}

#[test]
Expand Down
67 changes: 67 additions & 0 deletions packages/sie_sdk/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,73 @@ for entry in scores["scores"]:
print(entry["item_id"], entry["score"])
```

## Zero-shot classification

GLiClass models (`knowledgator/gliclass-*` and the `knowledgator/opir-*`
guardrail models) score text against labels passed with each request. Every
label comes back in `classifications`, sorted by score. By default the scores
form one distribution that sums to 1; `options={"classification_type":
"multi-label"}` scores each label independently instead.

```python
ticket = Item(text="I was charged twice and support has not replied in three days.")

result = client.extract(
"knowledgator/gliclass-instruct-large-v1.0",
ticket,
labels=["billing", "bug report", "feature request"],
instruction="Classify the support ticket by its main topic.",
)
print(result["classifications"][0]["label"]) # billing
```

`instruction` is the model's task prompt; the `gliclass-instruct-*` and
`opir-*` models are trained to follow one. Few-shot examples go in
`options={"examples": [{"text": ..., "labels": [...]}]}`, with labels taken
from the request's label set. They can move scores a long way, so check them
on your own data. The model reads examples after the document, so a long
document can push them out of the model's window; send
`options={"overflow_policy": "truncate_text"}` to shorten the document instead.

To answer several questions in one call, pass named `label_groups` instead of
`labels`. Scores are normalized within each group, and `data` holds one answer
per group:

```python
result = client.extract(
"knowledgator/gliclass-instruct-large-v1.0",
ticket,
instruction="Triage the support ticket.",
options={
"label_groups": {
"topic": ["billing", "bug report", "feature request"],
"urgency": ["low", "medium", "high"],
"needs_human": ["yes", "no"],
}
},
)
urgency = result["data"]["urgency"]
print(urgency["choice"], urgency["confidence"]) # e.g. medium 0.78
print(urgency["probabilities"]) # {"low": ..., "medium": ..., "high": ...}
```

Each group answers `{"type": "choice", "choice", "probabilities",
"confidence"}`, where `confidence` is `1 - entropy / log(number of labels)`.
With `"classification_type": "multi-label"` a group answers `{"labels",
"probabilities"}`: every label scored independently, and `labels` lists those
at or above `options.threshold` (0.5 when no threshold is set).
`classifications` lists the same scores under `group.label` names. In a
grouped request, an example's labels may be written as `"urgency.high"` or as
`{"urgency": "high"}`.

Usage counts each item's document tokens plus the tokens of the instruction
and example texts sent with it, since the model encodes them for every item.
Label names are not counted. With an instruction or examples, each item's
count is capped at the model window minus the label prompt, unless the
document count alone is already higher. An item whose document pushes the
labels out of the window comes back with an `INPUT_TOO_LONG` error in its
`error` field, and the other items still succeed.

## Generation prompts and guard verdicts

`generate` and `stream_generate` treat text-only prompts as raw continuation
Expand Down
4 changes: 3 additions & 1 deletion packages/sie_sdk/src/sie_sdk/types.py
Original file line number Diff line number Diff line change
Expand Up @@ -511,7 +511,9 @@ class ExtractResult(TypedDict, total=False):
relations: List of extracted relation triples.
classifications: List of classification results.
objects: List of detected objects with bounding boxes.
data: Additional structured extraction data (if output_schema was provided).
data: Structured extraction data: schema-driven results when
output_schema was provided, document parses, or one answer per
group when GLiClass options.label_groups was provided.
error: Stable per-item failure when extraction did not complete.
request: Request-scoped id, metered usage, and settled debit when
supplied by the gateway.
Expand Down
19 changes: 19 additions & 0 deletions packages/sie_server/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,25 @@ removes matched stop sequences consistently from returned text, completion-token
usage, and logprobs. Long completions delay the first visible text; existing
request timeouts still apply. Other adapters are unaffected.

### GLiClass usage

For GLiClass classification, `usage.input_tokens` counts each item's document
tokens plus the instruction and few-shot example texts sent with the request,
because the model encodes that text again for every item. Label names,
including the labels attached to examples, are not counted. With an
instruction or examples, an item's count is capped at the model's
`max_sequence_length` minus the label prompt, unless the document count alone
is already higher. An item refused because its document pushes the labels out
of the window returns a per-item `INPUT_TOO_LONG` error and counts nothing. A
request that sends no instruction or examples is counted exactly as before.

The instruction and each example text may be at most 2,048 characters, and
together with the example labels at most 8,192 characters. Up to 32 examples
are accepted, and they must leave room for the document in the model window.
Label names are refused when their total length exceeds 16 characters per
token of the window (8,192 characters for a 512-token model), more than any
label prompt can fit.

## Configuration

`sie-server` reads its config from `SIE_*` environment variables (Pydantic
Expand Down
5 changes: 3 additions & 2 deletions packages/sie_server/bundles/default.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -79,8 +79,9 @@ deps:
gliner2: '>=1.3.1,<2'
# glirel
glirel: '>=1.0,<2'
# gliclass
gliclass: '>=0.1,<1'
# gliclass (0.1.17: cross-attention scorer for the multilingual checkpoints;
# 0.1.18+ needs transformers 5)
gliclass: '>=0.1.17,<1'
# gliner/glirel/gliclass shared dep
loguru: '>=0.7,<1'
# donut, florence2
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-base-v3.0
hf_id: knowledgator/gliclass-base-v3.0
hf_revision: 77a70e6cd52e602ed18184ef37d18bdd3741e3d5
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-edge-v3.0
hf_id: knowledgator/gliclass-edge-v3.0
hf_revision: df03993a2ed98e5e4a0d2dd7efbbd105abe874cf
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-instruct-base-v1.0
hf_id: knowledgator/gliclass-instruct-base-v1.0
hf_revision: 4f6a108b08a5537f395521d19b5073e197923dd3
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-instruct-edge-v1.0
hf_id: knowledgator/gliclass-instruct-edge-v1.0
hf_revision: 727be8a417f6a7718e591b025e07054c146d8139
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-instruct-large-v1.0
hf_id: knowledgator/gliclass-instruct-large-v1.0
hf_revision: 825e5478c1bf4bffbf297690517097ccbdb2e006
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-multilang-edge
hf_id: knowledgator/gliclass-multilang-edge
hf_revision: d16c08ef70547514081952104e6fc3d190d9ee39
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
sie_id: knowledgator/gliclass-multilang-mini
hf_id: knowledgator/gliclass-multilang-mini
hf_revision: 0bd888b6c3ef9fca5f0a9d407bddfbbc7623486b
inputs:
text: true
image: false
audio: false
video: false
tasks:
encode: null
score: null
extract: {}
max_sequence_length: 512
profiles:
default:
max_batch_tokens: 16384
compute_precision: null
adapter_path: sie_server.adapters.gliclass:GLiClassAdapter
adapter_options:
loadtime:
classification_type: single-label
runtime:
threshold: 0.0
multi-label:
extends: default
adapter_options:
runtime:
classification_type: multi-label
threshold: 0.0
Loading
Loading