teren de test TypeSafe: docs offline + script de proba

Separat de produsele ROA. docs/ = documentatia oficiala descarcata ca Markdown
(111 pagini), reluabila cu update_docs.sh. typesafe_test.py face un apel cu cate
o intrebare din fiecare tip (choice/noul/score) pe o linie de factura de furnizor.
Cheia API se ia din TYPESAFE_API_KEY, nu se versioneaza.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01KHLUSsKP99G6ebv2fFUKQV
This commit is contained in:
2026-09-17 21:47:19 +03:00
commit 2d012a969c
116 changed files with 25116 additions and 0 deletions

507
docs/primitives/advanced.md Normal file
View File

@@ -0,0 +1,507 @@
> ## Documentation Index
> Fetch the complete documentation index at: https://docs.typesafe.ai/llms.txt
> Use this file to discover all available pages before exploring further.
# Advanced: structure
> Instructions, Choice options, Score levels, and Noul criteria all accept JSON structure.
export function TypesafeExample({example, display, title}) {
const keyStrUriSafe = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+-$";
function compressToEncodedURIComponent(input) {
if (input == null) return "";
return _compress(input, 6, function (a) {
return keyStrUriSafe.charAt(a);
});
}
function _compress(uncompressed, bitsPerChar, getCharFromInt) {
if (uncompressed == null) return "";
var i, value, context_dictionary = {}, context_dictionaryToCreate = {}, context_c = "", context_wc = "", context_w = "", context_enlargeIn = 2, context_dictSize = 3, context_numBits = 2, context_data = [], context_data_val = 0, context_data_position = 0, ii;
for (ii = 0; ii < uncompressed.length; ii += 1) {
context_c = uncompressed.charAt(ii);
if (!Object.prototype.hasOwnProperty.call(context_dictionary, context_c)) {
context_dictionary[context_c] = context_dictSize++;
context_dictionaryToCreate[context_c] = true;
}
context_wc = context_w + context_c;
if (Object.prototype.hasOwnProperty.call(context_dictionary, context_wc)) {
context_w = context_wc;
} else {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
context_dictionary[context_wc] = context_dictSize++;
context_w = String(context_c);
}
}
if (context_w !== "") {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
}
value = 2;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
while (true) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data.push(getCharFromInt(context_data_val));
break;
} else context_data_position++;
}
return context_data.join("");
}
function buildHref(ex) {
const documentText = ex.state === undefined ? "" : typeof ex.state === "string" ? ex.state : JSON.stringify(ex.state, null, 2);
return "https://console.typesafe.ai/decode#share/" + compressToEncodedURIComponent(JSON.stringify({
apiVersion: "v1",
documentText,
promptsText: JSON.stringify(ex.questions, null, 2),
selectedModels: ex.selectedModels
}));
}
const displayedExample = display === "questions" ? example.questions : example.state === undefined ? {
questions: example.questions
} : {
state: example.state,
questions: example.questions
};
const code = JSON.stringify(displayedExample, null, 2);
const href = buildHref(example);
return <div style={{
margin: "1.25rem 0"
}}>
<CodeBlock language="json" filename={title ?? "request"}>
{code}
</CodeBlock>
<div className="pb-8">
<a href={href} target="_blank" rel="noreferrer" className="text-primary">
Try it in the Playground →
</a>
</div>
</div>;
}
System One models are trained to understand structure.
## Where structure is allowed
Every one of these fields is an [`EntryType`](/sdk/javascript/api/type-aliases/EntryType).
| Field | Applies to | Accepted shape |
| --------------------------------------- | ------------------- | -------------------------------------- |
| `instructions` | Choice, Score, Noul | `string`, `object`, `array`, or `null` |
| `criteria` values (option descriptions) | Choice | `string`, `object`, `array`, or `null` |
| `criteria` entries (level descriptions) | Score | `string`, `object`, `array`, or `null` |
| `criteria.true` and `criteria.false` | Noul | `string`, `object`, `array`, or `null` |
## When to structure a question
* **When it helps with clarity.** When a question has multiple parts, putting them in the form of JSON helps with clarity because the keys are labeled.
* **When question needs supporting data.** A schema, a taxonomy, or a database row is already JSON. Use the JSON entirely or pass in the relevant subfields instead of serializing them into a string template.
## Structured instructions
One `field` object describes the field being checked, and each question refers to it by key. The same shape drives a Noul that verifies a value, a Choice that picks one from candidates, and two Scores that place a value on a scale.
<TypesafeExample
display="request"
example={{
state: {
source_text:
'Invoice #4471 issued March 3, 2026 to Beaver Dam Logistics for $12,840.00, net 30.',
},
selectedModels: ['jev-latest'],
questions: {
invoice_number_is_correct: {
type: 'noul',
instructions: {
field: {
name: 'invoice_number',
type: 'string',
description: 'The identifier printed on the invoice.',
},
extracted_value: '4471',
question: 'Does `extracted_value` match the `field` as it appears in `source_text`?',
},
},
customer_name: {
type: 'choice',
instructions: {
field: {
name: 'customer_name',
type: 'string',
description: 'The organization the invoice was issued to.',
},
question: 'Which option is the value of `field` in `source_text`?',
},
criteria: {
'Beaver Logistics': null,
'Dam Logistics': null,
'Beaver Dam Logistics': null,
'Beaver': null,
'Dam': null,
},
},
amount_due: {
type: 'score',
instructions: {
field: {
name: 'amount_due',
type: 'number',
unit: 'USD',
description: 'The total the invoice asks to be paid.',
},
question: 'How large is the `field` value in `source_text`?',
},
criteria: [
'Under $1,000',
'$1,000 to $10,000',
'$10,000 to $100,000',
'$100,000 to $1,000,000',
'Over $1,000,000',
],
},
payment_terms: {
type: 'score',
instructions: {
field: {
name: 'payment_terms',
type: 'integer',
unit: 'days',
description: 'Days allowed for payment, from terms such as "net 30".',
},
question: 'How many days does the `field` in `source_text` allow for payment?',
},
criteria: [
'Due on receipt',
'Net 10',
'Net 30',
'Net 60',
'Net 90',
],
},
},
}}
/>
In code, you could loop over the potential records and build one of these questions per field, all sent in a single call. The [SDE cascade cookbook](/cookbooks/sde_cascade) does something similar to this.
Arrays work too. Use one when the instruction is a list of things to check or to compare:
```json theme={null}
"instructions": {
"question": "Does the claimed sender identity conflict with the sending domain?",
"compare": ["ticket.sender.display_name", "ticket.sender.email"],
"focus": "Compare the named organization with the email domain."
}
```
## Structured Choice options
A Choice option description can be a structured object as well.
### JSON rubric for boundary clarification
<TypesafeExample
display="request"
example={{
state:
'I ordered the standing desk two weeks ago and tracking still says label created. Was I even charged?',
selectedModels: ['jev-latest'],
questions: {
department: {
type: 'choice',
instructions: {
question: 'Which team should handle this message?',
focus: "Classify the customer's primary request, not every topic mentioned.",
},
criteria: {
billing: {
what: 'Charges, invoices, refunds, or subscriptions',
not_for: 'Order tracking or account access',
examples: ['I was charged twice', 'Where is my refund?'],
},
orders: {
what: 'Order status, delivery, cancellation, or returns',
not_for: 'Charges or account access',
examples: ['Where is my package?', 'Cancel my order'],
},
account: {
what: 'Login, password, profile, or security',
not_for: 'Charges or delivery',
examples: ["I can't log in", 'Change my email'],
},
},
},
},
}}
/>
The example tells the model what each option does and does *not* cover. It sharpens the boundary between options.
### Walking a taxonomy
To classify into a deep taxonomy, ask one Choice per level and walk the tree in code. At each step the options are the children of the current node, and each option's value is the child's tree. Doing so lets the model see what lives under a branch before committing to it, which matters when the item belongs to a leaf whose name is not obvious from the branch name alone.
Here the state is a product listing and the first question picks a top-level department.
<TypesafeExample
display="request"
example={{
state:
"32oz plastic bottle with a flip straw lid. Fits most bike cages.",
selectedModels: ['jev-latest'],
questions: {
department: {
type: 'choice',
instructions: 'Which top-level department does this product belong to?',
criteria: {
'Sporting Goods': {
Cycling: ['Bike Bottles & Cages', 'Bike Lights', 'Helmets'],
Fitness: ['Yoga Mats', 'Resistance Bands'],
Outdoor: ['Tents', 'Sleeping Bags', 'Hydration Packs'],
},
'Home & Kitchen': {
Drinkware: ['Water Bottles', 'Travel Mugs', 'Tumblers'],
Cookware: ['Pots & Pans', 'Bakeware'],
},
'Baby & Toddler': ['Sippy Cups', 'Bottle Warmers', 'Bibs'],
},
},
},
}}
/>
The bottle plausibly fits under two departments. Showing the subtrees lets the model see that both `Sporting Goods > Cycling > Bike Bottles & Cages` and `Home & Kitchen > Drinkware > Water Bottles` exist, and weigh the listing's emphasis on bike cages against everyday drinkware. The `probabilities` on this answer tell you whether the split is close enough to explore both branches.
Once a department is chosen, ask the next Choice with that department's children as the options and their subtrees as the values, and repeat until you reach a leaf. In code this could be a loop over a nested dict, where each question's `criteria` is simply the current node. The [Hierarchical Classification cookbook](/cookbooks/hierarchical_classification) shows an example of a similar walk of the tree, including a beam search that keeps several candidate paths alive when the probabilities are close.
<Note>
Subtrees can get large. If a branch is too large, trim the value to its direct children and a sample of leaves.
</Note>
## Structured Score levels
Each entry in a Score `criteria` array can be an object.
<TypesafeExample
display="request"
example={{
state:
'Fixed the null check in the payment handler. Also refactored the retry loop while I was in there, and bumped the SDK version since the old one had that timeout bug.',
selectedModels: ['jev-latest'],
questions: {
pr_scope: {
type: 'score',
instructions: {
question: 'How focused is this pull request description on a single change?',
note: 'Judge the number of independent changes, not the size of any one change.',
},
criteria: [
{
summary: 'One change, clearly stated',
signals: ['A single fix or feature', 'Nothing described as "also" or "while I was in there"'],
},
{
summary: 'One main change plus a small related tweak',
signals: ['A primary change and one minor adjacent edit', 'The tweak supports the main change'],
},
{
summary: 'Several independent changes bundled together',
signals: ['Two or more unrelated fixes or features', 'Changes that could each be their own PR'],
},
],
},
},
}}
/>
## Structured Noul criteria
Noul `criteria` is optional, and when the yes/no boundary is subtle, structured `true` and `false` descriptions let you pin it down with a definition and examples on each side.
<TypesafeExample
display="request"
example={{
state: {
sender: { display_name: 'Beaver Dam Builders Ltd.', email: 'donotreply@payroll.example' },
message:
'Your Q3 bonus is ready. Reply with your login password so we can verify your identity and release the funds.',
},
selectedModels: ['jev-latest'],
questions: {
requests_credentials: {
type: 'noul',
instructions: {
question: 'Does the `message` ask the recipient to disclose a sensitive credential?',
inspect: 'message',
focus: 'Look for a request to send the credential itself, not a request to change or reset it.',
},
criteria: {
true: {
what: 'Asks the recipient to reply with, type, or send a password, PIN, one-time code, or other security sensitive answer',
examples: ['Reply with your password', 'Send us the 6-digit code you just received'],
},
false: {
what: 'No sensitive credential is requested',
examples: ['Reset your password from the settings page', 'Your statement is ready'],
},
},
},
},
}}
/>

663
docs/primitives/choice.md Normal file
View File

@@ -0,0 +1,663 @@
> ## Documentation Index
> Fetch the complete documentation index at: https://docs.typesafe.ai/llms.txt
> Use this file to discover all available pages before exploring further.
# Choice
> A Choice is a System One question type for selecting one option from a defined set. The answer includes the selected option, a probability for each option, and confidence.
export function TypesafeExample({example, display, title}) {
const keyStrUriSafe = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+-$";
function compressToEncodedURIComponent(input) {
if (input == null) return "";
return _compress(input, 6, function (a) {
return keyStrUriSafe.charAt(a);
});
}
function _compress(uncompressed, bitsPerChar, getCharFromInt) {
if (uncompressed == null) return "";
var i, value, context_dictionary = {}, context_dictionaryToCreate = {}, context_c = "", context_wc = "", context_w = "", context_enlargeIn = 2, context_dictSize = 3, context_numBits = 2, context_data = [], context_data_val = 0, context_data_position = 0, ii;
for (ii = 0; ii < uncompressed.length; ii += 1) {
context_c = uncompressed.charAt(ii);
if (!Object.prototype.hasOwnProperty.call(context_dictionary, context_c)) {
context_dictionary[context_c] = context_dictSize++;
context_dictionaryToCreate[context_c] = true;
}
context_wc = context_w + context_c;
if (Object.prototype.hasOwnProperty.call(context_dictionary, context_wc)) {
context_w = context_wc;
} else {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
context_dictionary[context_wc] = context_dictSize++;
context_w = String(context_c);
}
}
if (context_w !== "") {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
}
value = 2;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
while (true) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data.push(getCharFromInt(context_data_val));
break;
} else context_data_position++;
}
return context_data.join("");
}
function buildHref(ex) {
const documentText = ex.state === undefined ? "" : typeof ex.state === "string" ? ex.state : JSON.stringify(ex.state, null, 2);
return "https://console.typesafe.ai/decode#share/" + compressToEncodedURIComponent(JSON.stringify({
apiVersion: "v1",
documentText,
promptsText: JSON.stringify(ex.questions, null, 2),
selectedModels: ex.selectedModels
}));
}
const displayedExample = display === "questions" ? example.questions : example.state === undefined ? {
questions: example.questions
} : {
state: example.state,
questions: example.questions
};
const code = JSON.stringify(displayedExample, null, 2);
const href = buildHref(example);
return <div style={{
margin: "1.25rem 0"
}}>
<CodeBlock language="json" filename={title ?? "request"}>
{code}
</CodeBlock>
<div className="pb-8">
<a href={href} target="_blank" rel="noreferrer" className="text-primary">
Try it in the Playground →
</a>
</div>
</div>;
}
Use a Choice when the answer is one of a fixed set of options. For example, which team handles a ticket, which category a product belongs to, or which language a code snippet is written in. If the answer is a position on a spectrum, use a [Score](/primitives/score). If it's a yes or no, use a [Noul](/primitives/noul). [Choose a question type](/primitives#choose-a-question-type) compares all three.
A Choice answer is the selected option in `choice`. The model also returns a probability for every option in `probabilities`, and a `confidence` value for the selected option.
Example questions:
```
"What programming language is this code written in"
→ options: python, javascript, typescript, go, rust, other
"What type of meeting is this based on the title and description"
→ options: standup, planning, retrospective, one on one, brainstorm, none of the above
"Which product category does this item belong to"
→ options: electronics, clothing, home garden, food and beverage
```
## Request structure
The POST request body to the [TypeSafe API](/api) has a specific structure. The top level has three fields: `state`, the content to evaluate; `model`; and `questions`, a map from question ids you choose to question objects. Each Choice question has the following fields:
* `type`: Always `"choice"`.
* `instructions`: The question the model answers.
* `criteria`: The answer options, as a map. Each key is an option name and each value is a description of that option.
Below is a request where the state is a support ticket from an online shoe store and the question is which team should handle it:
<TypesafeExample
display="request"
example={{
state: 'My running shoes arrived in the wrong size. Can I swap them for a size 10?',
selectedModels: ['jev-latest'],
questions: {
department: {
type: 'choice',
instructions: 'Which team should handle this?',
criteria: {
returns: 'Exchanges, refunds, wrong or damaged items',
shipping: 'Delivery status, delays, lost packages',
billing: 'Charges, invoices, payment problems',
},
},
},
}}
/>
You choose the question id, `department` in this case. The answer is returned under the same id. The model never sees the question id. The option names and their descriptions are both sent to the model, so write descriptions that separate the options from each other.
Our [client SDKs](/sdk) provide typed questions. In Python, the same question is a `Choice`:
```python theme={null}
from typesafe_sdk import Choice, TypeSafeClient
with TypeSafeClient() as client:
response = client.system_one(
state="My running shoes arrived in the wrong size. Can I swap them for a size 10?",
questions={
"department": Choice(
instructions="Which team should handle this?",
criteria={
"returns": "Exchanges, refunds, wrong or damaged items",
"shipping": "Delivery status, delays, lost packages",
"billing": "Charges, invoices, payment problems",
},
),
},
)
print(response.answers["department"].choice)
```
Use the `system_one` method or the `https://api.typesafe.ai/v1/systemone` endpoint to call a System One model. The `model` field selects which model handles the request. [How to build with TypeSafe](/concepts/how-to-build-with-system-one) covers where in your code to call it.
Use one of our [client SDKs](/sdk) or call the [HTTP API](/api) directly. If a coding agent is writing the integration for you, install the [TypeSafe agent skill](/agent-skill#installation) first so it knows the request and response shapes.
<Note>
`instructions` and each entry in `criteria` can be a string, an object, or an array. Start with a string. Use an object when a description needs several kinds of guidance, such as what an option covers, what it doesn't cover, and some examples. See [Structured instructions and criteria](#structured-instructions-and-criteria) below and the [API reference](/api#param-instructions-1).
</Note>
## Response structure
The response has one entry in `answers` per question, under the ids from the request. This is the response to the example request above:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"department": {
"type": "choice",
"choice": "returns",
"confidence": 1.0,
"probabilities": {
"shipping": 0.0,
"returns": 1.0,
"billing": 0.0
}
}
},
"usage": {
"input_tokens": 330,
"output_tokens": 34
}
}
```
Besides `type`, each Choice answer has three values:
* `choice`: The option with the highest probability.
* `probabilities`: The full probability distribution across every option. The sum of all values is 1.
* [`confidence`](/confidence): A number from 0 to 1 computed from how `probabilities` is spread. A flat shape, with probability spread across several options, means low confidence. A single peak on one option means high confidence.
This ticket is an easy one, so all of the probability is on `returns` and confidence is 1.0. A ticket that mentions a wrong size and a missing refund would split probability between `returns` and `billing`, and confidence would drop.
## Good practice: ask more than one question per call
Ask every Choice question your code might need in a single request rather than one request per question. Questions are evaluated in parallel. Adding questions barely changes the response time, and the code can ignore answers it doesn't need. Extra questions still cost tokens. [Ask multiple questions together](/primitives#ask-multiple-questions-together) explains this in full; the next section shows five Choice questions in one call.
The same logic applies to the options inside a single Choice question. A Choice question accepts up to 255 options, and adding options costs a few tokens each, so give the model the full list of teams, categories, or products rather than a shortlist. Add an `other` or `none of the above` option when the list might not cover every input, so the model can say none of the others fit.
To classify documents through a deep hierarchy or large taxonomy, chain Choice questions level by level. The [Hierarchical Classification cookbook](/cookbooks/hierarchical_classification) shows how to run a beam search over Choice probabilities, keeping the best `K` candidate paths at each level instead of committing to a single greedy path.
## A more complex example
The basic example above routes a ticket to a team. A bigger support system might also need the return reason, the delivery problem, what the customer wants, and the customer's tone.
The request below asks five Choice questions about a ticket that is more ambiguous than the first: it involves three teams and doesn't say what the customer wants.
<TypesafeExample
display="request"
example={{
state: 'Shoes arrived two weeks late and in the wrong size. Also I see two charges on my card. What are you going to do about this?',
selectedModels: ['jev-latest'],
questions: {
department: {
type: 'choice',
instructions: 'Which team should handle this?',
criteria: {
returns: 'Exchanges, refunds, wrong or damaged items',
shipping: 'Delivery status, delays, lost packages',
billing: 'Charges, invoices, payment problems',
},
},
return_reason: {
type: 'choice',
instructions: 'If the customer wants to return something, why?',
criteria: {
wrong_size: "The item doesn't fit",
wrong_item: 'A different product was delivered',
damaged: 'The item arrived broken or faulty',
changed_mind: 'The item is fine, the customer no longer wants it',
other: 'A return reason that fits none of the above',
},
},
shipping_issue: {
type: 'choice',
instructions: 'If this is a shipping problem, which kind is it?',
criteria: {
not_delivered: 'The package never arrived',
delayed: 'The package is late but still on its way',
wrong_address: 'The package went to the wrong place',
damaged_in_transit: 'The package arrived damaged',
other: 'A shipping problem that fits none of the above',
},
},
requested_resolution: {
type: 'choice',
instructions: 'What does the customer want to happen?',
criteria: {
exchange: 'Swap the item for a different one',
refund: 'Money back',
replacement: 'The same item sent again',
information: 'Just an answer, no action needed',
},
},
tone: {
type: 'choice',
instructions: "What is the customer's tone?",
criteria: {
calm: null,
frustrated: null,
angry: null,
},
},
},
}}
/>
Two of these Choice questions are speculative: `return_reason` only matters if the `department` is `returns`, and `shipping_issue` only matters if it's `shipping`. The `tone` question uses `null` descriptions because the option names are clear on their own.
The TypeSafe response:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"department": {
"type": "choice",
"choice": "returns",
"confidence": 0.39,
"probabilities": {
"shipping": 0.02,
"billing": 0.38,
"returns": 0.6
}
},
"return_reason": {
"type": "choice",
"choice": "wrong_size",
"confidence": 1.0,
"probabilities": {
"wrong_size": 1.0,
"wrong_item": 0.0,
"other": 0.0,
"changed_mind": 0.0,
"damaged": 0.0
}
},
"shipping_issue": {
"type": "choice",
"choice": "delayed",
"confidence": 0.53,
"probabilities": {
"delayed": 0.63,
"other": 0.37,
"damaged_in_transit": 0.0,
"not_delivered": 0.0,
"wrong_address": 0.0
}
},
"requested_resolution": {
"type": "choice",
"choice": "exchange",
"confidence": 0.16,
"probabilities": {
"information": 0.1,
"exchange": 0.37,
"replacement": 0.24,
"refund": 0.29
}
},
"tone": {
"type": "choice",
"choice": "frustrated",
"confidence": 0.88,
"probabilities": {
"angry": 0.08,
"frustrated": 0.92,
"calm": 0.0
}
}
},
"usage": {
"input_tokens": 588,
"output_tokens": 212
}
}
```
Each question is answered on its own against the ticket:
* The `department` answer is `returns` with a 0.60 probability, but `billing` has 0.38 probability because of the double charge. This lowers the confidence to 0.39. The top option is clear enough to act on, but the second option is not noise.
* The `return_reason` is `wrong_size` with a confidence of 1.0, which is expected because it says this clearly in the ticket.
* The `shipping_issue` answer is split between `delayed` and `other`. It's a speculative question and `department` didn't come back as shipping, so it can be ignored by the code, as shown in the example code snippet below.
* The `requested_resolution` confidence is 0.16 because of the flat probability distribution of the answers. This is because the customer didn't say what they want.
* The `tone` answer is `frustrated` with a probability of 0.92 and a confidence of 0.88.
The example code below reads the answers it needs, ignores the rest, and treats a low-confidence answer as a reason to ask rather than act:
```python theme={null}
from typesafe_sdk import Choice, TypeSafeClient
TRIAGE_QUESTIONS = {
"department": Choice(
instructions="Which team should handle this?",
criteria={
"returns": "Exchanges, refunds, wrong or damaged items",
"shipping": "Delivery status, delays, lost packages",
"billing": "Charges, invoices, payment problems",
},
),
"return_reason": Choice(
instructions="If the customer wants to return something, why?",
criteria={
"wrong_size": "The item doesn't fit",
"wrong_item": "A different product was delivered",
"damaged": "The item arrived broken or faulty",
"changed_mind": "The item is fine, the customer no longer wants it",
"other": "A return reason that fits none of the above",
},
),
"shipping_issue": Choice(
instructions="If this is a shipping problem, which kind is it?",
criteria={
"not_delivered": "The package never arrived",
"delayed": "The package is late but still on its way",
"wrong_address": "The package went to the wrong place",
"damaged_in_transit": "The package arrived damaged",
"other": "A shipping problem that fits none of the above",
},
),
"requested_resolution": Choice(
instructions="What does the customer want to happen?",
criteria={
"exchange": "Swap the item for a different one",
"refund": "Money back",
"replacement": "The same item sent again",
"information": "Just an answer, no action needed",
},
),
"tone": Choice(
instructions="What is the customer's tone?",
criteria={"calm": None, "frustrated": None, "angry": None},
),
}
def triage(ticket: str) -> None:
with TypeSafeClient() as client:
response = client.system_one(
state=ticket,
questions=TRIAGE_QUESTIONS,
)
answers = response.answers
department = answers["department"]
if department.confidence < 0.3:
# Not clear which team to send to. Let a person decide.
send_to_manual_triage(ticket)
return
if department.choice == "returns":
# return_reason answer is only used here
assign(ticket, team="returns", issue=answers["return_reason"].choice)
elif department.choice == "shipping":
# shipping_issue answer is only used here
assign(ticket, team="shipping", issue=answers["shipping_issue"].choice)
else:
assign(ticket, team="billing")
# A second team with a real share of the probability gets a copy
for team, probability in department.probabilities.items():
if team != department.choice and probability > 0.25:
notify(ticket, team=team)
resolution = answers["requested_resolution"]
if resolution.confidence < 0.5:
# The customer hasn't said what they want. Ask, don't guess.
ask_customer_what_they_want(ticket)
elif resolution.choice == "refund":
flag_for_refund_approval(ticket)
if answers["tone"].choice == "angry":
flag_for_senior_agent(ticket)
```
For the ticket above, this assigns the ticket to the returns team with issue `wrong_size`, sends the billing team a copy, and asks the customer what they want. The code does not use the `shipping_issue` answer.
One request, five answers, and the routing logic is ordinary `if` statements. If you later need to know the customer's language, or which product the ticket is about, add another Choice question to `TRIAGE_QUESTIONS`; the request count stays at one.
The [smart home assistant demo](/demos/smart-home) evaluates every user request against a long list of Choice questions in one call: the request category, the room, the device, and the action. Most of those questions are irrelevant to any one request and the code ignores them.
## Structured instructions and criteria
Start with a one-line description per option. When two options are similar and the model keeps confusing them, describe each one with an object instead of a string. Give it fields for what the option covers, what belongs to a neighboring option instead, and a few example inputs.
The two answer options below, return\_policy and return\_status, are easy to confuse. A ticket about either one can mention returns and refunds, so each option says what it is not for.
<TypesafeExample
display="request"
example={{
state: 'I sent the shoes back a week ago. When do I get my money?',
selectedModels: ['jev-latest'],
questions: {
return_topic: {
type: 'choice',
instructions: {
question: 'Which returns topic is the customer asking about?',
focus: 'Classify the information the customer wants.',
},
criteria: {
return_policy: {
what: 'Whether and how an item can be returned',
not_for: 'Progress of a return already sent',
examples: [
"Can I return shoes I've worn once?",
'How long do I have to return an order?',
],
},
return_status: {
what: 'Progress of a return already sent',
not_for: 'Whether and how an item can be returned',
examples: [
'Has my return arrived yet?',
'When will my refund be paid?',
],
},
},
},
},
}}
/>
The response is `return_status` at confidence 1.0:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"return_topic": {
"type": "choice",
"choice": "return_status",
"confidence": 1.0,
"probabilities": {
"return_policy": 0.0,
"return_status": 1.0
}
}
},
"usage": {
"input_tokens": 407,
"output_tokens": 32
}
}
```
The field names `question`, `focus`, `what`, `not_for`, and `examples` are not part of the API, and none are reserved. You choose them, the same way you choose option names. The model sees the names along with the values, so use short names that label what follows.

320
docs/primitives/noul.md Normal file
View File

@@ -0,0 +1,320 @@
> ## Documentation Index
> Fetch the complete documentation index at: https://docs.typesafe.ai/llms.txt
> Use this file to discover all available pages before exploring further.
# Noul
> A Noul question asks the model to evaluate a yes/no question and return the probability that the answer is yes.
export function TypesafeExample({example, display, title}) {
const keyStrUriSafe = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+-$";
function compressToEncodedURIComponent(input) {
if (input == null) return "";
return _compress(input, 6, function (a) {
return keyStrUriSafe.charAt(a);
});
}
function _compress(uncompressed, bitsPerChar, getCharFromInt) {
if (uncompressed == null) return "";
var i, value, context_dictionary = {}, context_dictionaryToCreate = {}, context_c = "", context_wc = "", context_w = "", context_enlargeIn = 2, context_dictSize = 3, context_numBits = 2, context_data = [], context_data_val = 0, context_data_position = 0, ii;
for (ii = 0; ii < uncompressed.length; ii += 1) {
context_c = uncompressed.charAt(ii);
if (!Object.prototype.hasOwnProperty.call(context_dictionary, context_c)) {
context_dictionary[context_c] = context_dictSize++;
context_dictionaryToCreate[context_c] = true;
}
context_wc = context_w + context_c;
if (Object.prototype.hasOwnProperty.call(context_dictionary, context_wc)) {
context_w = context_wc;
} else {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
context_dictionary[context_wc] = context_dictSize++;
context_w = String(context_c);
}
}
if (context_w !== "") {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
}
value = 2;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
while (true) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data.push(getCharFromInt(context_data_val));
break;
} else context_data_position++;
}
return context_data.join("");
}
function buildHref(ex) {
const documentText = ex.state === undefined ? "" : typeof ex.state === "string" ? ex.state : JSON.stringify(ex.state, null, 2);
return "https://console.typesafe.ai/decode#share/" + compressToEncodedURIComponent(JSON.stringify({
apiVersion: "v1",
documentText,
promptsText: JSON.stringify(ex.questions, null, 2),
selectedModels: ex.selectedModels
}));
}
const displayedExample = display === "questions" ? example.questions : example.state === undefined ? {
questions: example.questions
} : {
state: example.state,
questions: example.questions
};
const code = JSON.stringify(displayedExample, null, 2);
const href = buildHref(example);
return <div style={{
margin: "1.25rem 0"
}}>
<CodeBlock language="json" filename={title ?? "request"}>
{code}
</CodeBlock>
<div className="pb-8">
<a href={href} target="_blank" rel="noreferrer" className="text-primary">
Try it in the Playground →
</a>
</div>
</div>;
}
Use a Noul when the answer is yes or no. For example, does this message ask for a refund, does this resume mention distributed systems, does this comment contain personal data. If the answer is one of several options, use a [Choice](/primitives/choice). If it's a position on a spectrum, use a [Score](/primitives/score). [Choose a question type](/primitives#choose-a-question-type) compares all three.
A Noul answer is a single number, `noul`, the probability that the answer is yes.
## Writing a Noul question
A Noul question evaluates a single yes/no question (or statement). It is defined by its `instructions`: the yes/no question to evaluate. It's good practice to phrase it so a high probability means "yes", so that the returned answer is unambiguous in its meaning.
You can optionally add `criteria` with `true` and `false` descriptions to clarify what each outcome means, which can be helpful when the question itself has more nuance to explain. Try your Noul question prompts with and without criteria to see which works better in your use-case.
## Request
| Field | Required | Description |
| -------------- | -------- | ---------------------------------------------------------------------------- |
| `type` | Yes | Must be `"noul"`. |
| `instructions` | Yes | The yes/no question or statement to evaluate. |
| `criteria` | No | Optional `{ true, false }` descriptions clarifying what a yes and a no mean. |
<TypesafeExample
display="request"
example={{
state: 'I have asked three times now. Can I please just talk to a real person?',
selectedModels: ['jev-latest'],
questions: {
is_human_escalation: {
type: 'noul',
instructions: 'Is the customer asking for a human agent?',
},
is_repeat_contact: {
type: 'noul',
instructions: 'Has the customer contacted support about this before?',
criteria: {
true: 'Mentions a prior attempt, ticket, or that they have asked before',
false: 'No sign of any previous contact',
},
},
},
}}
/>
## Response
```json theme={null}
{
"model": "jev-latest",
"answers": {
"is_human_escalation": {
"type": "noul",
"noul": 0.99
},
"is_repeat_contact": {
"type": "noul",
"noul": 0.93
}
},
"usage": {
"input_tokens": 360,
"output_tokens": 39
}
}
```
`noul` ranges from 0 to 1, representing the probability that the answer is **yes**. Most often you will threshold it into a boolean when your code needs a hard decision.
## Noul does not return a separate confidence value
A value near 1 means a strong yes. A value near 0 means a strong no. A value near 0.5 gives yes and no similar probability.
For "Is the candidate strong in Python?", define what "strong" means. An unclear definition makes the probability hard to interpret. A value of 0.5 does not mean medium skill. Use a [Score](/primitives/score) to measure skill along defined levels. [Choose a question type](/primitives#choose-a-question-type) explains the distinction.
## Example questions
```
"Is the customer requesting a refund?"
"Does this resume mention experience with distributed systems?"
"Does the message contain personally identifiable information?"
"Does the room have a minifridge?"
```
## Tips and advanced usage
* **Phrasing.** Beyond a plain question, you can phrase the instruction as a statement for the model to evaluate for truthfulness. For "the customer is requesting a refund", a value near 1 means the statement is true. Try both phrasings with your own data to see what works best.
* **Optional `criteria`.** The instruction is enough for most Noul questions, but when the boundary between yes and no is subtle, pass `criteria` with `true` and `false` descriptions to pin down what each outcome means — as shown in the request example above.

733
docs/primitives/score.md Normal file
View File

@@ -0,0 +1,733 @@
> ## Documentation Index
> Fetch the complete documentation index at: https://docs.typesafe.ai/llms.txt
> Use this file to discover all available pages before exploring further.
# Score
> A Score is a System One question type for rating content against ordered, descriptive levels. The answer includes a score, a probability for each level, and confidence.
export function TypesafeExample({example, display, title}) {
const keyStrUriSafe = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+-$";
function compressToEncodedURIComponent(input) {
if (input == null) return "";
return _compress(input, 6, function (a) {
return keyStrUriSafe.charAt(a);
});
}
function _compress(uncompressed, bitsPerChar, getCharFromInt) {
if (uncompressed == null) return "";
var i, value, context_dictionary = {}, context_dictionaryToCreate = {}, context_c = "", context_wc = "", context_w = "", context_enlargeIn = 2, context_dictSize = 3, context_numBits = 2, context_data = [], context_data_val = 0, context_data_position = 0, ii;
for (ii = 0; ii < uncompressed.length; ii += 1) {
context_c = uncompressed.charAt(ii);
if (!Object.prototype.hasOwnProperty.call(context_dictionary, context_c)) {
context_dictionary[context_c] = context_dictSize++;
context_dictionaryToCreate[context_c] = true;
}
context_wc = context_w + context_c;
if (Object.prototype.hasOwnProperty.call(context_dictionary, context_wc)) {
context_w = context_wc;
} else {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
context_dictionary[context_wc] = context_dictSize++;
context_w = String(context_c);
}
}
if (context_w !== "") {
if (Object.prototype.hasOwnProperty.call(context_dictionaryToCreate, context_w)) {
if (context_w.charCodeAt(0) < 256) {
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
}
value = context_w.charCodeAt(0);
for (i = 0; i < 8; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
} else {
value = 1;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = 0;
}
value = context_w.charCodeAt(0);
for (i = 0; i < 16; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
delete context_dictionaryToCreate[context_w];
} else {
value = context_dictionary[context_w];
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
}
context_enlargeIn--;
if (context_enlargeIn == 0) {
context_enlargeIn = Math.pow(2, context_numBits);
context_numBits++;
}
}
value = 2;
for (i = 0; i < context_numBits; i++) {
context_data_val = context_data_val << 1 | value & 1;
if (context_data_position == bitsPerChar - 1) {
context_data_position = 0;
context_data.push(getCharFromInt(context_data_val));
context_data_val = 0;
} else {
context_data_position++;
}
value = value >> 1;
}
while (true) {
context_data_val = context_data_val << 1;
if (context_data_position == bitsPerChar - 1) {
context_data.push(getCharFromInt(context_data_val));
break;
} else context_data_position++;
}
return context_data.join("");
}
function buildHref(ex) {
const documentText = ex.state === undefined ? "" : typeof ex.state === "string" ? ex.state : JSON.stringify(ex.state, null, 2);
return "https://console.typesafe.ai/decode#share/" + compressToEncodedURIComponent(JSON.stringify({
apiVersion: "v1",
documentText,
promptsText: JSON.stringify(ex.questions, null, 2),
selectedModels: ex.selectedModels
}));
}
const displayedExample = display === "questions" ? example.questions : example.state === undefined ? {
questions: example.questions
} : {
state: example.state,
questions: example.questions
};
const code = JSON.stringify(displayedExample, null, 2);
const href = buildHref(example);
return <div style={{
margin: "1.25rem 0"
}}>
<CodeBlock language="json" filename={title ?? "request"}>
{code}
</CodeBlock>
<div className="pb-8">
<a href={href} target="_blank" rel="noreferrer" className="text-primary">
Try it in the Playground →
</a>
</div>
</div>;
}
Use a Score when the answer is a position on a spectrum you can describe in steps. For example, how severe a bug is, how happy a customer is, or how much Python experience a candidate has. If the answer is one of a fixed set of options with no order between them, use a [Choice](/primitives/choice). If it's a yes or no, use a [Noul](/primitives/noul). [Choose a question type](/primitives#choose-a-question-type) compares all three.
A Score answer is a position along your levels in `score`, which can fall between two levels. The model also returns a probability for every level in `probabilities`, and a `confidence` value for the answer.
Example Score questions:
```
"How severe is the bug being reported?"
→ 0: Cosmetic; no impact to functionality
→ 1: Broken or degraded feature, but workaround exists
→ 2: Blocking issue; no workaround exists
"How formal is this outfit based on the description"
→ 0: gym clothes
→ 1: casual
→ 2: business casual
→ 3: formal
→ 4: black tie
"How relevant is this candidate's experience to the job posting"
→ 0: completely unrelated
→ 1: adjacent field
→ 2: some direct experience
→ 3: deep, direct experience
```
The numbers in front of each step are positions, explained under [Levels](#levels).
## Request structure
The POST request body to the [TypeSafe API](/api) has the same three top-level fields as any other question type: `state`, which is the content to evaluate; `model`; and `questions`. Each Score question has the following fields:
* `type`: Always `"score"`.
* `instructions`: The question the model answers. What it's rating.
* `criteria`: An ordered array of level descriptions, from the low end of the scale to the high end. Needs at least two levels and takes up to 10.
Below is a request where the state is a bug report and the question is how severe the bug is:
<TypesafeExample
display="request"
example={{
state: 'The export button crashes the settings page in Safari. It works in Chrome, but a few of our customers only use Safari.',
selectedModels: ['jev-latest'],
questions: {
bug_severity: {
type: 'score',
instructions: 'How severe is the reported issue?',
criteria: [
'Cosmetic; no impact to functionality',
'Broken or degraded feature, but workaround exists',
'Blocking issue; no workaround exists',
],
},
},
}}
/>
You choose the question id, `bug_severity` in this case. This id is not sent to the model. The answer is returned under the same id.
### Levels
Each entry in `criteria` is a level: one point on the spectrum of possible answers, described in words. A level's number is its position in the `criteria` array, starting at 0, so the three entries above are levels 0, 1 and 2. The order of the array is the numbering.
The model gets the descriptions and nothing else, and each level is judged on its own against the state.
The `score` in the response is a position on the levels spectrum. For a three-level scale it runs from 0 to 2, and it can land between two levels.
Our [client SDKs](/sdk) provide typed questions. In Python, the same question is a `Score`:
```python theme={null}
from typesafe_sdk import Score, TypeSafeClient
with TypeSafeClient() as client:
response = client.system_one(
state="The export button crashes the settings page in Safari. It works in Chrome, but a few of our customers only use Safari.",
questions={
"bug_severity": Score(
instructions="How severe is the reported issue?",
criteria=[
"Cosmetic; no impact to functionality",
"Broken or degraded feature, but workaround exists",
"Blocking issue; no workaround exists",
],
),
},
)
print(response.answers["bug_severity"].score)
```
Use the `system_one` method or the `https://api.typesafe.ai/v1/systemone` endpoint to call a System One model. The `model` field selects which model handles the request. [How to build with TypeSafe](/concepts/how-to-build-with-system-one) covers where in your code to call it.
Use one of our [client SDKs](/sdk) or call the [TypeSafe API](/api) directly. If a coding agent is writing the integration for you, install the [TypeSafe agent skill](/agent-skill#installation) first so it knows the request and response shapes.
<Note>
`instructions` and each level in `criteria` can be a string, an object, or an array. Start with strings. Use an object when a level needs a description plus a few example situations. See [Structured level descriptions](#structured-level-descriptions) below and the [API reference](/api#param-instructions-2).
</Note>
## Response structure
The response has one entry in `answers` per question, under the ids from the request. This is the response to the example request above:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"bug_severity": {
"type": "score",
"score": 1.3,
"confidence": 0.54,
"legend": {
"0": "Cosmetic; no impact to functionality",
"1": "Broken or degraded feature, but workaround exists",
"2": "Blocking issue; no workaround exists"
},
"probabilities": {
"0": 0.0,
"1": 0.7,
"2": 0.3
}
}
},
"usage": {
"input_tokens": 332,
"output_tokens": 18
}
}
```
Each Score answer has five values:
* `type`: The type of TypeSafe question.
* `probabilities`: The probability of each level, keyed by level number as a string. The sum of all values is 1.
* `score`: The position on the level number line, from 0 to the top level number, which is 2 here. It's each level number multiplied by its probability, added up: 0 x 0.0 + 1 x 0.70 + 2 x 0.30 = 1.30.
* `legend`: Each level number mapped back to its description.
* [`confidence`](/confidence): A number from 0 to 1 computed from how `probabilities` is spread. A single peak on one level means high confidence. Probability spread over several levels means low confidence.
A score of 1.30 means mostly level 1 with some weight on level 2. That matches the report: the export is broken, and switching to Chrome is a workaround for most customers, but not for the ones who only use Safari. The model puts 0.70 on "workaround exists" and 0.30 on "no workaround", and confidence is 0.54 because it's split.
Using the Python SDK, `ScoreAnswer` has `score`, `confidence`, `probabilities`, and `legend` as typed fields. The SDK keys `probabilities` and `legend` by integer level rather than by string.
## Reading a Score
Let's look at how the score changes with different inputs. For example, using the question and its levels from the request above:
```
"How severe is the reported issue?"
→ 0: Cosmetic; no impact to functionality
→ 1: Broken or degraded feature, but workaround exists
→ 2: Blocking issue; no workaround exists
```
We can see how different bug reports change the score:
<table>
<thead>
<tr>
<th colSpan={3} />
<th colSpan={3} style={{ textAlign: 'left' }}><code>probabilities</code></th>
</tr>
<tr>
<th style={{ width: '44%' }}>State</th>
<th style={{ width: '12%', whiteSpace: 'nowrap' }}><code>score</code></th>
<th style={{ width: '16%', whiteSpace: 'nowrap' }}><code>confidence</code></th>
<th style={{ width: '9%', whiteSpace: 'nowrap' }}>Level 0</th>
<th style={{ width: '9%', whiteSpace: 'nowrap' }}>Level 1</th>
<th style={{ width: '10%', whiteSpace: 'nowrap' }}>Level 2</th>
</tr>
</thead>
<tbody>
<tr>
<td>The export button is misaligned by a few pixels on the settings page.</td>
<td>0.0</td><td>1.0</td><td>1.0</td><td>0.0</td><td>0.0</td>
</tr>
<tr>
<td>The PDF export button does nothing when clicked. I can still export to CSV and convert it myself, but that takes ages.</td>
<td>1.0</td><td>1.0</td><td>0.0</td><td>1.0</td><td>0.0</td>
</tr>
<tr>
<td>Export to PDF fails with a spinner that never finishes. Some of our team say CSV export still works for them, others say it fails too.</td>
<td>1.12</td><td>0.81</td><td>0.0</td><td>0.88</td><td>0.12</td>
</tr>
<tr>
<td>The export button crashes the settings page in Safari. It works in Chrome, but a few of our customers only use Safari.</td>
<td>1.3</td><td>0.54</td><td>0.0</td><td>0.7</td><td>0.3</td>
</tr>
<tr>
<td>Nobody on our team can log in since this morning. We get a 500 error on every attempt.</td>
<td>2.0</td><td>1.0</td><td>0.0</td><td>0.0</td><td>1.0</td>
</tr>
</tbody>
</table>
In these examples, confidence 1.0 means the returned distribution puts all its probability on one level. This describes the model's answer, not a guarantee that the answer is correct.
The score is a probability-weighted mean of the level numbers. In the third and fourth examples, probability is split between levels 1 and 2. More weight on level 2 raises the score. It does not measure the fraction of customers without a workaround.
Different distributions can produce the same score. A score of 1.0 can mean all probability is on level 1, or half is on each of levels 0 and 2. Read `probabilities` and `confidence` alongside the score to distinguish these cases.
A fractional score is a position. You can use it to rank reports by severity, or round it to the nearest level when your code needs one outcome. Our [entity alignment cookbook](/cookbooks/entity_alignment) shows an example of rounding to the nearest level to make a decision.
Low confidence on a Score usually means one of three things. The levels overlap for this state, the question is measuring more than one thing, or the state doesn't say enough to place it. Our [Confidence](/confidence) docs cover how to use it in your code.
## Writing good levels
Describe situations, not degrees. "Broken or degraded feature, but workaround exists" gives the model something to match the state against. "Moderately severe" doesn't. Concrete descriptions can help the model distinguish levels. Check the answers against known examples; higher confidence alone does not show that a description is better.
Every level is evaluated separately. The model doesn't see a level's number or its neighbours, so "worse than the previous level" means nothing to it, and numbers in the descriptions or the instructions don't help. Here is what happens when the levels are only numbers, on the misaligned-button report from the table above:
```
instructions: "Rate severity from 0 to 2, where 2 is worst"
criteria: ["0", "1", "2"]
→ score 0.57, confidence 0.35, probabilities 0: 0.43, 1: 0.57, 2: 0.0
```
The same report with the three descriptive levels scores 0.0 at confidence 1.0. With numbers only, the model has nothing to match against and splits the probability between 0 and 1.
Use as many levels as you can describe distinctly, up to 10. Three is fine. Don't add levels you can't describe distinctly.
Keep each Score question to one dimension. If a description says "punctual and smart and experienced", the question is measuring three things, and an input that is high on one and low on another can't be placed. Confidence drops and the score means less. Split it into one Score question per thing and combine them in code, as the next section shows.
If the top of your scale has a rare extreme case you need to act on differently, give it its own level. A sentiment scale that ends at "very angry" can add "abusive or threatening". Without that level, both messages may receive a score near the top. The score alone may not distinguish them.
If there is no in-between at all, and the answer is one of a few discrete categories, use a [Choice](/primitives/choice) instead, or split the question into several [Noul](/primitives/noul) questions. It's important to test your levels against your own data. Two wordings of the same scale can behave differently on your data.
## Splitting a complex judgment into several Score questions
A complex judgment, one that depends on several things, is best split into one Score question per thing. You can then combine the Scores returned from TypeSafe in your code to make the judgment. Some Score questions may matter more than others, so give each Score question a weight for its relative importance. The weights are yours. When the combined result doesn't match what your team would decide, change them in code and run again. Send the Score questions in one request. They are evaluated in parallel. Adding questions barely changes the response time and costs a few extra question tokens; see [Ask multiple questions together](/primitives#ask-multiple-questions-together).
The request below is the spinner ticket from the table above with some more context. It asks three Score questions: how severe the bug is, how frustrated the customer is, and how much the report gives an engineer to work with.
<TypesafeExample
display="request"
example={{
state: 'Export to PDF fails with a spinner that never finishes. Some of our team say CSV export still works for them, others say it fails too. This is the third time I\'m writing in and honestly I\'m done. Steps: open any report, click Export, choose PDF. Chrome 128 on macOS.',
selectedModels: ['jev-latest'],
questions: {
severity: {
type: 'score',
instructions: 'How severe is the reported issue?',
criteria: [
'Cosmetic; no impact to functionality',
'Broken or degraded feature, but workaround exists',
'Blocking issue; no workaround exists',
],
},
frustration: {
type: 'score',
instructions: 'How frustrated is the customer?',
criteria: [
'Calm, just stating facts',
'Frustrated but civil',
'Very angry, strong language or threatening to leave',
],
},
report_quality: {
type: 'score',
instructions: 'How much does the report give an engineer to work with?',
criteria: [
'No detail; just says something is broken',
'Names the feature but no steps or environment',
'Steps to reproduce or environment, but not both',
'Steps to reproduce and environment',
],
},
},
}}
/>
TypeSafe's response:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"severity": {
"type": "score",
"score": 1.24,
"confidence": 0.63,
"legend": {
"0": "Cosmetic; no impact to functionality",
"1": "Broken or degraded feature, but workaround exists",
"2": "Blocking issue; no workaround exists"
},
"probabilities": {
"0": 0.0,
"1": 0.76,
"2": 0.24
}
},
"frustration": {
"type": "score",
"score": 1.45,
"confidence": 0.33,
"legend": {
"0": "Calm, just stating facts",
"1": "Frustrated but civil",
"2": "Very angry, strong language or threatening to leave"
},
"probabilities": {
"0": 0.0,
"1": 0.55,
"2": 0.45
}
},
"report_quality": {
"type": "score",
"score": 3.0,
"confidence": 1.0,
"legend": {
"0": "No detail; just says something is broken",
"1": "Names the feature but no steps or environment",
"2": "Steps to reproduce or environment, but not both",
"3": "Steps to reproduce and environment"
},
"probabilities": {
"0": 0.0,
"1": 0.0,
"2": 0.0,
"3": 1.0
}
}
},
"usage": {
"input_tokens": 468,
"output_tokens": 43
}
}
```
Each question is answered on its own against the ticket and given a score:
* `severity` is 1.24 at confidence 0.63. Same reading as the opening example: the export is broken and some have a workaround.
* `frustration` is 1.45 at confidence 0.33. The wording is civil, but "third time" and "I'm done" shift the score toward the top level, so the model splits 0.55 and 0.45 between "frustrated but civil" and "very angry". For this ticket the two levels overlap, which explains the low confidence.
* `report_quality` is 3.0 at confidence 1.0. The steps and browser version are both stated.
The three scales have different lengths, so before combining them, normalize each score. A four-level scale returns 0 to 3 and a three-level scale returns 0 to 2, so a top score on one is bigger than a top score on the other. Divide each score by its top level number, `len(criteria) - 1`, to put every score on 0 to 1. Then the weights mean what they say: 0.6 on severity and 0.3 on frustration makes severity count twice as much.
The TypeSafe Python SDK code below asks the three questions, normalizes each score, and combines them using an example priority calculation:
```python theme={null}
from typesafe_sdk import Score, TypeSafeClient
TRIAGE_QUESTIONS = {
"severity": Score(
instructions="How severe is the reported issue?",
criteria=[
"Cosmetic; no impact to functionality",
"Broken or degraded feature, but workaround exists",
"Blocking issue; no workaround exists",
],
),
"frustration": Score(
instructions="How frustrated is the customer?",
criteria=[
"Calm, just stating facts",
"Frustrated but civil",
"Very angry, strong language or threatening to leave",
],
),
"report_quality": Score(
instructions="How much does the report give an engineer to work with?",
criteria=[
"No detail; just says something is broken",
"Names the feature but no steps or environment",
"Steps to reproduce or environment, but not both",
"Steps to reproduce and environment",
],
),
}
def normalized(answers, question_id: str) -> float:
"""Put a score on 0 to 1 by dividing by its top level number."""
top_level = len(TRIAGE_QUESTIONS[question_id].criteria) - 1
return answers[question_id].score / top_level
def priority(ticket: str) -> float:
with TypeSafeClient() as client:
response = client.system_one(
state=ticket,
questions=TRIAGE_QUESTIONS,
)
answers = response.answers
severity = normalized(answers, "severity")
frustration = normalized(answers, "frustration")
report_quality = normalized(answers, "report_quality")
# A detailed report helps an engineer investigate, so it raises priority a little.
return 0.6 * severity + 0.3 * frustration + 0.1 * report_quality
```
For the example response above, the normalized scores are 0.62 for severity, 0.725 for frustration, and 1.0 for report quality. The priority is `0.6 × 0.62 + 0.3 × 0.725 + 0.1 × 1.0 = 0.6895`, which rounds to `0.69`.
The weights live in your code, so you can see exactly how the number is made and change it when the ranking doesn't match what your team would do. If you later need more Score questions, add them to `TRIAGE_QUESTIONS`. The request count stays at one. This technique of breaking a complex judgment into separate Scores and then combining them with weights in your code is called the [Composite scoring](/patterns/composite-scoring) pattern.
## Structured level descriptions
Start with a basic text description for each level. When the model keeps scoring between two neighbouring levels on inputs you think are clear, give each level an object instead of a string, with a field for what the level covers and a field with a few example situations. Use the same field names on every level so the model can compare like with like.
The request below is the spinner ticket that we used earlier, but with examples on each level:
<TypesafeExample
display="request"
example={{
state: 'Export to PDF fails with a spinner that never finishes. Some of our team say CSV export still works for them, others say it fails too.',
selectedModels: ['jev-latest'],
questions: {
bug_severity: {
type: 'score',
instructions: 'How severe is the reported issue?',
criteria: [
{
what: 'Cosmetic; no impact to functionality',
examples: ['typo in a label', 'misaligned icon'],
},
{
what: 'Broken or degraded feature, but workaround exists',
examples: ['export fails in one browser but works in another'],
},
{
what: 'Blocking issue; no workaround exists',
examples: ['cannot log in', 'data loss'],
},
],
},
},
}}
/>
The response:
```json theme={null}
{
"model": "jev-latest",
"answers": {
"bug_severity": {
"type": "score",
"score": 1.06,
"confidence": 0.91,
"legend": {
"0": {
"what": "Cosmetic; no impact to functionality",
"examples": [
"typo in a label",
"misaligned icon"
]
},
"1": {
"what": "Broken or degraded feature, but workaround exists",
"examples": [
"export fails in one browser but works in another"
]
},
"2": {
"what": "Blocking issue; no workaround exists",
"examples": [
"cannot log in",
"data loss"
]
}
},
"probabilities": {
"0": 0.0,
"1": 0.94,
"2": 0.06
}
}
},
"usage": {
"input_tokens": 379,
"output_tokens": 18
}
}
```
With plain strings this ticket scored 1.12 with a confidence of 0.81. With examples it scores 1.06 at 0.91 confidence.
Examples steer the model, and they only help when they look like your real inputs. The table below is the opening Safari report with three different sets of level objects:
| Level description | `score` | `confidence` |
| ------------------------------------------------------------------------------------------------------------ | ------- | ------------ |
| plain string: no object with examples | 1.30 | 0.54 |
| Added examples array with useful example: "export fails in one browser but works in another" | 1.07 | 0.90 |
| Added examples array with example unrelated to browsers: "search fails, but browsing categories still works" | 1.28 | 0.57 |
In this comparison, the matching example concentrates more probability on one level. The unrelated example changes the result only slightly compared with plain strings. Higher confidence does not establish which answer is correct. Choose examples with known expected levels, then test the revised descriptions on separate inputs before keeping them.