Xunzhuo commited on
Commit
1866c44
·
1 Parent(s): 01bfda4

fix(space): align SystemOne validation and forced Tetris moves

Browse files

Signed-off-by: Xunzhuo <Xunzhuo@users.noreply.huggingface.co>

SYSTEM_ONE_MAPPING.md CHANGED
@@ -12,7 +12,7 @@ Studio exposes a **SystemOne-format Decision service** with official Python SDK
12
  | `instructions` text/object/array | Required native question text, with the same deterministic JSON rule for structured content. |
13
  | Choice `criteria` map | Each option name is semantic: null description → name; otherwise name + `: ` + full description. Object insertion order is retained. External native candidate IDs equal the names, but the names reach the model only through this explicit semantic text. |
14
  | Score `criteria` array | Ordered native levels with IDs and values 0…K−1. Returned score is the probability-weighted level index. |
15
- | Noul `criteria.false/true` | Native false/true descriptions. If absent, the original native default no/yes descriptions are used. |
16
 
17
  The public single-state endpoint requires exactly `model`, `state`, and `questions`, with a 256 KiB JSON body. The separate batch endpoint requires exactly `model`, `states`, and `questions`, with a 2 MiB JSON body and at most 1,024 states, questions, and total decisions. Both accept 2–255 Choice options, 2–10 Score levels, and up to 16 MiB of expanded context/question input. The direct runtime uses a per-model configurable physical microbatch size rather than a fixed eight rows. The complete-input limit is 1,024 tokens for Kai/Lex and 16,384 for Eos/Sol/Nox/Lux **per question**, including state and all candidate descriptions. Unsupported fields and models are rejected.
18
 
 
12
  | `instructions` text/object/array | Required native question text, with the same deterministic JSON rule for structured content. |
13
  | Choice `criteria` map | Each option name is semantic: null description → name; otherwise name + `: ` + full description. Object insertion order is retained. External native candidate IDs equal the names, but the names reach the model only through this explicit semantic text. |
14
  | Score `criteria` array | Ordered native levels with IDs and values 0…K−1. Returned score is the probability-weighted level index. |
15
+ | Noul `criteria.false/true` | Native false/true descriptions. An omitted or `null` description uses the original native default no/yes description. |
16
 
17
  The public single-state endpoint requires exactly `model`, `state`, and `questions`, with a 256 KiB JSON body. The separate batch endpoint requires exactly `model`, `states`, and `questions`, with a 2 MiB JSON body and at most 1,024 states, questions, and total decisions. Both accept 2–255 Choice options, 2–10 Score levels, and up to 16 MiB of expanded context/question input. The direct runtime uses a per-model configurable physical microbatch size rather than a fixed eight rows. The complete-input limit is 1,024 tokens for Kai/Lex and 16,384 for Eos/Sol/Nox/Lux **per question**, including state and all candidate descriptions. Unsupported fields and models are rejected.
18
 
contract.py CHANGED
@@ -19,9 +19,11 @@ def content(value, label):
19
  raise ValueError(f"{label} must be text, an object, or an array")
20
 
21
 
22
- def identifier(value, label):
23
- if not isinstance(value, str) or not value.strip() or len(value) > 128:
24
- raise ValueError(f"{label} must be a nonempty string of at most 128 characters")
 
 
25
  return value
26
 
27
 
@@ -65,7 +67,7 @@ def _single_records(body, *, model=MODEL):
65
  if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
66
  raise ValueError(f"{qid}: Noul criteria accept false and true only")
67
  for key in ("false", "true"):
68
- if key in criteria:
69
  q[key + "_criterion"] = content(criteria[key], qid + ".criteria." + key)
70
  records.append({"id": f"studio:{index}", "state_text": state, "question": q})
71
  return records
@@ -132,7 +134,7 @@ def contexts(body, *, model=MODEL):
132
  for item in states:
133
  if not isinstance(item, dict) or set(item) != {"id", "state"}:
134
  raise ValueError("Each context must contain exactly id and state")
135
- cid = identifier(item["id"], "Context ID")
136
  if cid in seen:
137
  raise ValueError("Context IDs must be unique")
138
  seen.add(cid)
 
19
  raise ValueError(f"{label} must be text, an object, or an array")
20
 
21
 
22
+ def identifier(value, label, *, max_length=None):
23
+ if not isinstance(value, str) or not value.strip():
24
+ raise ValueError(f"{label} must be a nonempty string")
25
+ if max_length is not None and len(value) > max_length:
26
+ raise ValueError(f"{label} must be at most {max_length} characters")
27
  return value
28
 
29
 
 
67
  if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
68
  raise ValueError(f"{qid}: Noul criteria accept false and true only")
69
  for key in ("false", "true"):
70
+ if criteria.get(key) is not None:
71
  q[key + "_criterion"] = content(criteria[key], qid + ".criteria." + key)
72
  records.append({"id": f"studio:{index}", "state_text": state, "question": q})
73
  return records
 
134
  for item in states:
135
  if not isinstance(item, dict) or set(item) != {"id", "state"}:
136
  raise ValueError("Each context must contain exactly id and state")
137
+ cid = identifier(item["id"], "Context ID", max_length=128)
138
  if cid in seen:
139
  raise ValueError("Context IDs must be unique")
140
  seen.add(cid)
static/contract.js CHANGED
@@ -3,7 +3,10 @@
3
  const object = v => v !== null && typeof v === 'object' && !Array.isArray(v);
4
  const keys = (value, allowed, label) => { if (!object(value) || Object.keys(value).some(k => !allowed.includes(k))) throw Error(`${label}: unsupported fields or invalid object.`); };
5
  const content = (v,label) => { if (typeof v === 'string') { if(!v.trim()) throw Error(`${label} must not be empty.`); } else if(!object(v) && !Array.isArray(v)) throw Error(`${label} must be text, an object, or an array.`); };
6
- const id = (v,label) => { if(typeof v!=='string'||!v.trim()||v.length>128)throw Error(`${label} needs 1–128 characters.`); };
 
 
 
7
  function finiteJSON(value) {
8
  if(typeof value==='number'&&!Number.isFinite(value))throw Error('JSON numbers must be finite.');
9
  if(value && typeof value==='object')Object.values(value).forEach(finiteJSON);
@@ -16,7 +19,7 @@ export function validateRequest(body, expectedModel) {
16
  if(Object.hasOwn(body,'states')){
17
  if(!Array.isArray(body.states)||body.states.length<1)throw Error('Provide at least one context.');
18
  const seen=new Set();
19
- for(const row of body.states){keys(row,['id','state'],'Context');id(row.id,'Context ID');if(seen.has(row.id))throw Error('Context IDs must be unique.');seen.add(row.id);content(row.state,`Context ${row.id}`);}
20
  }else content(body.state,'State');
21
  if(!object(body.questions)||Object.keys(body.questions).length<1)throw Error('Provide at least one named question.');
22
  for(const[qid,q]of Object.entries(body.questions)) {
@@ -30,7 +33,8 @@ export function validateRequest(body, expectedModel) {
30
  if(!Array.isArray(q.criteria)||q.criteria.length<2||q.criteria.length>10)throw Error(`${qid}: Score needs 2–10 ordered levels.`);
31
  q.criteria.forEach(v=>content(v,`${qid} level`));
32
  } else if(q.criteria!==undefined&&q.criteria!==null) {
33
- keys(q.criteria,['false','true'],`${qid} criteria`);Object.values(q.criteria).forEach(v=>content(v,`${qid} criterion`));
 
34
  }
35
  }
36
  const requestLimit=Object.hasOwn(body,'states')?2*1024*1024:256*1024;
 
3
  const object = v => v !== null && typeof v === 'object' && !Array.isArray(v);
4
  const keys = (value, allowed, label) => { if (!object(value) || Object.keys(value).some(k => !allowed.includes(k))) throw Error(`${label}: unsupported fields or invalid object.`); };
5
  const content = (v,label) => { if (typeof v === 'string') { if(!v.trim()) throw Error(`${label} must not be empty.`); } else if(!object(v) && !Array.isArray(v)) throw Error(`${label} must be text, an object, or an array.`); };
6
+ const id = (v,label,maxLength) => {
7
+ if(typeof v!=='string'||!v.trim())throw Error(`${label} needs a nonempty string.`);
8
+ if(maxLength!==undefined&&v.length>maxLength)throw Error(`${label} needs at most ${maxLength} characters.`);
9
+ };
10
  function finiteJSON(value) {
11
  if(typeof value==='number'&&!Number.isFinite(value))throw Error('JSON numbers must be finite.');
12
  if(value && typeof value==='object')Object.values(value).forEach(finiteJSON);
 
19
  if(Object.hasOwn(body,'states')){
20
  if(!Array.isArray(body.states)||body.states.length<1)throw Error('Provide at least one context.');
21
  const seen=new Set();
22
+ for(const row of body.states){keys(row,['id','state'],'Context');id(row.id,'Context ID',128);if(seen.has(row.id))throw Error('Context IDs must be unique.');seen.add(row.id);content(row.state,`Context ${row.id}`);}
23
  }else content(body.state,'State');
24
  if(!object(body.questions)||Object.keys(body.questions).length<1)throw Error('Provide at least one named question.');
25
  for(const[qid,q]of Object.entries(body.questions)) {
 
33
  if(!Array.isArray(q.criteria)||q.criteria.length<2||q.criteria.length>10)throw Error(`${qid}: Score needs 2–10 ordered levels.`);
34
  q.criteria.forEach(v=>content(v,`${qid} level`));
35
  } else if(q.criteria!==undefined&&q.criteria!==null) {
36
+ keys(q.criteria,['false','true'],`${qid} criteria`);
37
+ Object.values(q.criteria).forEach(v=>{if(v!==null)content(v,`${qid} criterion`);});
38
  }
39
  }
40
  const requestLimit=Object.hasOwn(body,'states')?2*1024*1024:256*1024;
tests/studio_contract.test.mjs CHANGED
@@ -36,7 +36,7 @@ const answers = {
36
  urgency: {
37
  type: 'score',
38
  score: 1.2,
39
- confidence: 0.3,
40
  legend: { 0: 'Routine', 1: 'Soon', 2: 'Now' },
41
  probabilities: { 0: 0.2, 1: 0.4, 2: 0.4 },
42
  },
@@ -75,6 +75,26 @@ test('Choice and Score require instructions and at least two criteria', () => {
75
  }
76
  });
77
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
78
  test('browser request bytes match the single and batch API routes', () => {
79
  const utf8 = new TextEncoder();
80
  const questions = { check: { type: 'noul', instructions: 'Check this input.' } };
 
36
  urgency: {
37
  type: 'score',
38
  score: 1.2,
39
+ confidence: 0.16,
40
  legend: { 0: 'Routine', 1: 'Soon', 2: 'Now' },
41
  probabilities: { 0: 0.2, 1: 0.4, 2: 0.4 },
42
  },
 
75
  }
76
  });
77
 
78
+ test('runtime-compatible nullable Noul criteria and long question labels are accepted', () => {
79
+ const long = 'q'.repeat(129);
80
+ const payload = {
81
+ model: request.model,
82
+ state: 'A request',
83
+ questions: {
84
+ [long]: { type: 'noul', instructions: 'Check this.', criteria: { true: null, false: 'No' } },
85
+ category: { type: 'choice', instructions: 'Choose.', criteria: { [long]: null, other: null } },
86
+ },
87
+ };
88
+ assert.equal(validateRequest(payload, request.model), payload);
89
+ assert.throws(() => validateRequest({
90
+ ...payload,
91
+ questions: { check: { type: 'noul', instructions: 'Check.', criteria: { true: '' } } },
92
+ }, request.model));
93
+ const batch = { ...payload, states: [{ id: 'x'.repeat(129), state: 'A request' }] };
94
+ delete batch.state;
95
+ assert.throws(() => validateRequest(batch, request.model));
96
+ });
97
+
98
  test('browser request bytes match the single and batch API routes', () => {
99
  const utf8 = new TextEncoder();
100
  const questions = { check: { type: 'noul', instructions: 'Check this input.' } };
tests/test_public_api_contract.py CHANGED
@@ -135,6 +135,30 @@ class PublicAPIContractTests(unittest.TestCase):
135
  with self.subTest(noul_instructions=instructions), self.assertRaises(ValueError):
136
  to_records({"model": MODEL, "state": "A request", "questions": {"decision": question}})
137
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
138
 
139
  if __name__ == "__main__":
140
  unittest.main()
 
135
  with self.subTest(noul_instructions=instructions), self.assertRaises(ValueError):
136
  to_records({"model": MODEL, "state": "A request", "questions": {"decision": question}})
137
 
138
+ def test_runtime_compatible_nullable_noul_criteria_and_long_question_names(self):
139
+ long_name = "q" * 129
140
+ payload = {
141
+ "model": CANONICAL,
142
+ "state": "A request",
143
+ "questions": {
144
+ long_name: {
145
+ "type": "noul",
146
+ "instructions": "Does this need review?",
147
+ "criteria": {"true": None, "false": "No review needed"},
148
+ },
149
+ "category": {
150
+ "type": "choice",
151
+ "instructions": "Pick a category.",
152
+ "criteria": {"x" * 129: None, "other": None},
153
+ },
154
+ },
155
+ }
156
+ internal = dict(payload, model=MODEL)
157
+ records = to_records(internal, model=MODEL)
158
+ self.assertNotIn("true_criterion", records[0]["question"])
159
+ self.assertEqual(records[0]["question"]["false_criterion"], "No review needed")
160
+ self.assertEqual(records[1]["question"]["options"][0]["id"], "x" * 129)
161
+
162
 
163
  if __name__ == "__main__":
164
  unittest.main()
tests/test_tetris_arena.py CHANGED
@@ -1,6 +1,7 @@
1
  import asyncio
2
  import json
3
  import unittest
 
4
 
5
  from tetris_arena import (
6
  LOCAL_MODELS,
@@ -578,12 +579,19 @@ class TetrisArenaTests(unittest.IsolatedAsyncioTestCase):
578
  if not legal_placements:
579
  continue
580
  placements = shortlist_placements(legal_placements, next_piece="T")
 
 
 
 
 
 
 
581
  payload, state = build_decision_request(
582
  board, piece, "T", placements, LOCAL_MODELS[4].request_model
583
  )
584
  question = payload["questions"]["placement"]
585
  criteria = question["criteria"]
586
- self.assertGreaterEqual(len(criteria), 1)
587
  self.assertTrue(question["instructions"].strip())
588
  largest_choice_count = max(largest_choice_count, len(criteria))
589
  self.assertEqual(set(criteria), {placement.id for placement in placements})
@@ -610,6 +618,30 @@ class TetrisArenaTests(unittest.IsolatedAsyncioTestCase):
610
  self.assertLessEqual(len(encoded.encode("utf-8")), 1800)
611
  self.assertEqual(largest_choice_count, 5)
612
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
613
  async def test_avoidable_top_out_is_shortlisted_only_after_safe_moves_and_overridden(self):
614
  game = Game()
615
  for row in game.board:
 
1
  import asyncio
2
  import json
3
  import unittest
4
+ from unittest.mock import patch
5
 
6
  from tetris_arena import (
7
  LOCAL_MODELS,
 
579
  if not legal_placements:
580
  continue
581
  placements = shortlist_placements(legal_placements, next_piece="T")
582
+ if len(placements) == 1:
583
+ with self.assertRaises(ValueError):
584
+ build_decision_request(
585
+ board, piece, "T", placements,
586
+ LOCAL_MODELS[4].request_model,
587
+ )
588
+ continue
589
  payload, state = build_decision_request(
590
  board, piece, "T", placements, LOCAL_MODELS[4].request_model
591
  )
592
  question = payload["questions"]["placement"]
593
  criteria = question["criteria"]
594
+ self.assertGreaterEqual(len(criteria), 2)
595
  self.assertTrue(question["instructions"].strip())
596
  largest_choice_count = max(largest_choice_count, len(criteria))
597
  self.assertEqual(set(criteria), {placement.id for placement in placements})
 
618
  self.assertLessEqual(len(encoded.encode("utf-8")), 1800)
619
  self.assertEqual(largest_choice_count, 5)
620
 
621
+ async def test_single_legal_placement_is_applied_without_invalid_choice_call(self):
622
+ adapter = RecordingAdapter()
623
+ original = shortlist_placements
624
+
625
+ def one_placement(placements, **kwargs):
626
+ return original(placements, **kwargs)[:1]
627
+
628
+ with patch("tetris_arena.shortlist_placements", side_effect=one_placement):
629
+ race = await self.manager(adapter).create({
630
+ "left": LEFT.id,
631
+ "right": RIGHT.id,
632
+ "seed": 42,
633
+ "mode": "steps",
634
+ "max_steps": 1,
635
+ })
636
+ result = await asyncio.wait_for(race.wait(), 1)
637
+
638
+ self.assertEqual(adapter.calls, [])
639
+ self.assertEqual(result["status"], "finished")
640
+ self.assertEqual(result["results"]["left"]["pieces"], 1)
641
+ self.assertEqual(result["results"]["right"]["pieces"], 1)
642
+ self.assertEqual(race.traces["left"][0]["requested_choice"],
643
+ race.traces["left"][0]["applied_choice"])
644
+
645
  async def test_avoidable_top_out_is_shortlisted_only_after_safe_moves_and_overridden(self):
646
  game = Game()
647
  for row in game.board:
tetris_arena.py CHANGED
@@ -566,25 +566,30 @@ def _orientation(piece: str, rotation: int) -> str:
566
  return ("up", "right", "down", "left")[rotation % 4]
567
 
568
 
569
- def build_decision_request(
570
- game: Game,
571
- piece: str,
572
- next_piece: str | None,
573
- placements: tuple[Placement, ...],
574
- model: str | None,
575
- ) -> tuple[dict[str, Any], str]:
576
- if not placements:
577
- raise ValueError("at least one placement is required")
578
  stats = _board_stats(game.board)
579
  board = "/".join(game.rows())
580
  board = "".join("#" if cell != "." else "." for cell in board)
581
- state = "; ".join((
582
  "Tetris", f"piece={piece}", f"next={next_piece or '?'}",
583
  f"score={game.score}", f"lines={game.lines}", f"pieces={game.pieces}",
584
  f"holes={stats.holes}", f"agg={stats.aggregate_height}",
585
  f"max={stats.max_height}", f"bump={stats.bumpiness}",
586
  f"heights={','.join(map(str, stats.heights))}", f"board={board}",
587
  ))
 
 
 
 
 
 
 
 
 
 
 
 
 
588
  ordered = sorted(placements, key=lambda option: option.id)
589
  offset = game.pieces % len(ordered)
590
  neutral_order = ordered[offset:] + ordered[:offset]
@@ -1372,21 +1377,30 @@ class RaceSession:
1372
  placements = await asyncio.to_thread(
1373
  shortlist_placements, legal_placements, next_piece=next_piece
1374
  )
1375
- request, state = build_decision_request(
1376
- game,
1377
- piece,
1378
- next_piece,
1379
- placements,
1380
- competitor.request_model,
1381
- )
 
 
 
 
 
 
1382
  before = game.public()
1383
  legal_choices = frozenset(placement.id for placement in placements)
1384
  try:
1385
- outcome = await self.adapter.decide(
1386
- competitor,
1387
- request,
1388
- legal_choices,
1389
- )
 
 
 
1390
  except ArenaUpstreamError as exc:
1391
  status = "error"
1392
  reason = "provider_error"
 
566
  return ("up", "right", "down", "left")[rotation % 4]
567
 
568
 
569
+ def _decision_state(game: Game, piece: str, next_piece: str | None) -> str:
 
 
 
 
 
 
 
 
570
  stats = _board_stats(game.board)
571
  board = "/".join(game.rows())
572
  board = "".join("#" if cell != "." else "." for cell in board)
573
+ return "; ".join((
574
  "Tetris", f"piece={piece}", f"next={next_piece or '?'}",
575
  f"score={game.score}", f"lines={game.lines}", f"pieces={game.pieces}",
576
  f"holes={stats.holes}", f"agg={stats.aggregate_height}",
577
  f"max={stats.max_height}", f"bump={stats.bumpiness}",
578
  f"heights={','.join(map(str, stats.heights))}", f"board={board}",
579
  ))
580
+
581
+
582
+ def build_decision_request(
583
+ game: Game,
584
+ piece: str,
585
+ next_piece: str | None,
586
+ placements: tuple[Placement, ...],
587
+ model: str | None,
588
+ ) -> tuple[dict[str, Any], str]:
589
+ if not 2 <= len(placements) <= 255:
590
+ raise ValueError("a Choice decision requires 2–255 placements")
591
+ state = _decision_state(game, piece, next_piece)
592
+ stats = _board_stats(game.board)
593
  ordered = sorted(placements, key=lambda option: option.id)
594
  offset = game.pieces % len(ordered)
595
  neutral_order = ordered[offset:] + ordered[:offset]
 
1377
  placements = await asyncio.to_thread(
1378
  shortlist_placements, legal_placements, next_piece=next_piece
1379
  )
1380
+ request = None
1381
+ if len(placements) == 1:
1382
+ # A single legal move is not a Choice decision. Apply it
1383
+ # directly so strict providers do not reject the turn.
1384
+ state = _decision_state(game, piece, next_piece)
1385
+ else:
1386
+ request, state = build_decision_request(
1387
+ game,
1388
+ piece,
1389
+ next_piece,
1390
+ placements,
1391
+ competitor.request_model,
1392
+ )
1393
  before = game.public()
1394
  legal_choices = frozenset(placement.id for placement in placements)
1395
  try:
1396
+ if request is None:
1397
+ outcome = DecisionOutcome(
1398
+ choice=placements[0].id, request_ms=0.0
1399
+ )
1400
+ else:
1401
+ outcome = await self.adapter.decide(
1402
+ competitor, request, legal_choices
1403
+ )
1404
  except ArenaUpstreamError as exc:
1405
  status = "error"
1406
  reason = "provider_error"