Roll a fair die
Open in playground →Language models favour options by where they appear in the list. Averaging over option orders brings a fair die back to about one in six for every face.
Challenges
- Every face is equally likely, but a language model favours options by where they sit in the list. Asked as listed, it gives "one" 0.93.
- With
"permutations": truethe options are asked in different orders and the probabilities averaged, which brings every face to between 0.14 and 0.23, near one in six.
import os
from typellm import TypeLLMClient
client = TypeLLMClient(api_key=os.environ["TYPELLM_API_KEY"])
response = client.generate(
context="A single roll of a fair six-sided die.",
questions={
"as_listed": {
"type": "string",
"enum": [
"one",
"two",
"three",
"four",
"five",
"six"
],
"instructions": "What number will come up on this roll?",
"return_probabilities": True
},
"averaged": {
"type": "string",
"enum": [
"one",
"two",
"three",
"four",
"five",
"six"
],
"instructions": "What number will come up on this roll?",
"return_probabilities": True,
"permutations": True
}
},
)
print(response)Generation(
result={
'as_listed': {
'value': 'one',
'probabilities': {
'one': 0.928,
'two': 0.003,
'three': 0.032,
'four': 0.013,
'five': 0.003,
'six': 0.022,
},
},
'averaged': {
'value': 'one',
'probabilities': {
'one': 0.227,
'two': 0.138,
'three': 0.169,
'four': 0.159,
'five': 0.15,
'six': 0.157,
},
},
},
thinking={},
usage=Usage(input_tokens=120, thinking_tokens=0),
)