Michael Stattelman commited on
Commit
d8c255d
·
1 Parent(s): 996a2db

Version updates

Browse files
Files changed (6) hide show
  1. README.md +4 -4
  2. app/demos.py +2 -2
  3. app/main.py +1 -1
  4. app/static/app.css +5 -0
  5. app/static/app.js +34 -1
  6. tests/test_demos.py +2 -0
README.md CHANGED
@@ -81,7 +81,7 @@ cd <your-space>
81
  git lfs install
82
  # copy the contents of this DecisionLab folder into it (not the folder itself), then:
83
  git add .
84
- git commit -m "DecisionLab 2.3.0"
85
  git push
86
  ```
87
 
@@ -122,7 +122,7 @@ The plain HTTP API works on the Space too, at `https://<your-user>-<your-space>.
122
  ## Using the lab
123
 
124
  1. **Setup** shows each model's status, size, device, load time and how it defines confidence.
125
- 2. **Decision**: pick one of the 47 demos (tabs group them into triage, routing and planning, loop control, guardrails, and agent security (nano set)) or write your own state and questions. Nothing is executed; every demo is a decision only. Questions use the Laya/Jev JSON shape:
126
  ```json
127
  {"team": {"type": "choice", "instructions": "Which team?", "criteria": {"bug": "Something is broken", "sales": "Pricing"}},
128
  "urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["Can wait", "Today", "Right now"]},
@@ -135,7 +135,7 @@ The plain HTTP API works on the Space too, at `https://<your-user>-<your-space>.
135
 
136
  ## Demos
137
 
138
- 47 demos in 5 groups. The tab numbers match the table. Demos 25–47 are the **Agent security (nano set)** group: the states and questions of the Athr_Agent_Sec nano demo set (`app/demos_nano.json`), with no reference answers, so they show each model's answers, confidence and agreement but are not scored for matches.
139
 
140
  | # | Group | Demo | What it tests |
141
  |---|---|---|---|
@@ -242,7 +242,7 @@ DecisionLab/
242
  │ ├── security.py # body limit and security headers (plain ASGI)
243
  │ ├── validation.py # checks /api/decide question sets, sizes and model keys (no web framework)
244
  │ ├── demos.py # groups, the 47 demos (24 scored + 23 nano), reference answers and high-stakes flags
245
- │ ├── demos_nano.json # Agent security (nano set): states and questions only
246
  │ └── static/ # index.html, app.css, app.js, logo.png
247
  └── tests/ # unittest suite, copied into the image
248
  ```
 
81
  git lfs install
82
  # copy the contents of this DecisionLab folder into it (not the folder itself), then:
83
  git add .
84
+ git commit -m "DecisionLab 2.3.1"
85
  git push
86
  ```
87
 
 
122
  ## Using the lab
123
 
124
  1. **Setup** shows each model's status, size, device, load time and how it defines confidence.
125
+ 2. **Decision**: pick one of the 47 demos (tabs group them into triage, routing and planning, loop control, guardrails, and agent security) or write your own state and questions. Nothing is executed; every demo is a decision only. Questions use the Laya/Jev JSON shape:
126
  ```json
127
  {"team": {"type": "choice", "instructions": "Which team?", "criteria": {"bug": "Something is broken", "sales": "Pricing"}},
128
  "urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["Can wait", "Today", "Right now"]},
 
135
 
136
  ## Demos
137
 
138
+ 47 demos in 5 groups. The tab numbers match the table. Demos 25–47 are the **Agent security** group: the states and questions of the Athr_Agent_Sec nano demo set (`app/demos_nano.json`), with no reference answers, so they show each model's answers, confidence and agreement but are not scored. A scoreboard run of only these demos shows no Agentic Use Score (there is nothing to score it against), just what each model would act on, its speed and the agreement.
139
 
140
  | # | Group | Demo | What it tests |
141
  |---|---|---|---|
 
242
  │ ├── security.py # body limit and security headers (plain ASGI)
243
  │ ├── validation.py # checks /api/decide question sets, sizes and model keys (no web framework)
244
  │ ├── demos.py # groups, the 47 demos (24 scored + 23 nano), reference answers and high-stakes flags
245
+ │ ├── demos_nano.json # Agent security: states and questions only
246
  │ └── static/ # index.html, app.css, app.js, logo.png
247
  └── tests/ # unittest suite, copied into the image
248
  ```
app/demos.py CHANGED
@@ -31,7 +31,7 @@ GROUPS = [
31
  "blurb": "Inside the agent loop: stop, retry, change course, or verify the work before reporting it."},
32
  {"id": "guardrails", "family": "Agent decisions", "label": "Guardrails",
33
  "blurb": "Irreversible actions, prompt injection, fraud, data leaving the company, secrets and permissions."},
34
- {"id": "agent_security", "family": "Agent decisions", "label": "Agent security (nano set)",
35
  "blurb": "Drift, report auditing, action risk, speech consistency and prompt-injection screening, from the "
36
  "Athr_Agent_Sec nano demo set. No reference answers: compare the models' answers, confidence and agreement."},
37
  ]
@@ -410,7 +410,7 @@ DEMOS = [
410
  },
411
  ]
412
 
413
- # Agent security (nano set): only the states and questions of decisionlab_nano_demos.json (operator ruling
414
  # 2026-09-30). No reference answers or stakes, so these demos are not scored for matches; every model still answers
415
  # them, with confidence, agreement and speed.
416
  _NANO = json.loads((Path(__file__).with_name("demos_nano.json")).read_text(encoding="utf-8"))["demos"]
 
31
  "blurb": "Inside the agent loop: stop, retry, change course, or verify the work before reporting it."},
32
  {"id": "guardrails", "family": "Agent decisions", "label": "Guardrails",
33
  "blurb": "Irreversible actions, prompt injection, fraud, data leaving the company, secrets and permissions."},
34
+ {"id": "agent_security", "family": "Agent decisions", "label": "Agent security",
35
  "blurb": "Drift, report auditing, action risk, speech consistency and prompt-injection screening, from the "
36
  "Athr_Agent_Sec nano demo set. No reference answers: compare the models' answers, confidence and agreement."},
37
  ]
 
410
  },
411
  ]
412
 
413
+ # Agent security: only the states and questions of decisionlab_nano_demos.json (operator ruling
414
  # 2026-09-30). No reference answers or stakes, so these demos are not scored for matches; every model still answers
415
  # them, with confidence, agreement and speed.
416
  _NANO = json.loads((Path(__file__).with_name("demos_nano.json")).read_text(encoding="utf-8"))["demos"]
app/main.py CHANGED
@@ -27,7 +27,7 @@ MAX_BODY_BYTES = int(os.getenv("MAX_BODY_BYTES", str(256 * 1024)))
27
  DECIDE_SLOTS = Gate(int(os.getenv("MAX_PENDING_DECIDES", "4"))) # DL-SA-004
28
 
29
  # DL-SA-007: no /docs, /redoc or /openapi.json
30
- VERSION = "2.3.0"
31
  PAGE = stamp_assets((STATIC / "index.html").read_text(encoding="utf-8"), VERSION)
32
 
33
 
 
27
  DECIDE_SLOTS = Gate(int(os.getenv("MAX_PENDING_DECIDES", "4"))) # DL-SA-004
28
 
29
  # DL-SA-007: no /docs, /redoc or /openapi.json
30
+ VERSION = "2.3.1"
31
  PAGE = stamp_assets((STATIC / "index.html").read_text(encoding="utf-8"), VERSION)
32
 
33
 
app/static/app.css CHANGED
@@ -416,3 +416,8 @@ input[type="range"] { height: 28px; }
416
  .repo a { padding: 12px 0; }
417
  .search input { font-size: 16px; }
418
  }
 
 
 
 
 
 
416
  .repo a { padding: 12px 0; }
417
  .search input { font-size: 16px; }
418
  }
419
+
420
+ /* ---------- runs with no reference answers: no score, a plain note instead */
421
+ .unscored-note { grid-column: 1 / -1; border: 1px dashed var(--line); border-radius: var(--radius-m); padding: 18px 22px; }
422
+ .unscored-note h3 { font: 500 22px var(--display); margin: 0 0 6px; }
423
+ .unscored-note p { margin: 0; color: var(--graphite); max-width: 70ch; }
app/static/app.js CHANGED
@@ -467,7 +467,7 @@ function groupBreakdown(rows, thr) {
467
  const w = best(Object.fromEntries(MODELS.map((m) => [m.key, L[m.key].rate])));
468
  return `<tr><td>${esc(groupOf(id)?.label ?? id)}<small>${esc(groupOf(id)?.family ?? "")}</small></td>${MODELS.map((m) => {
469
  const x = L[m.key];
470
- if (!x.n) return `<td><b>–</b><small>no reference answers</small></td>`; // e.g. the nano set: states and questions only
471
  return `<td class="${w.has(m.key) ? "win" : ""}"><b>${x.hits} / ${x.n}</b><small>${x.cw} confident mistake${x.cw === 1 ? "" : "s"}</small></td>`;
472
  }).join("")}</tr>`;
473
  }).join("");
@@ -484,8 +484,41 @@ function renderScoreHead() {
484
  $("score-head").innerHTML = `<th scope="col">Demo</th><th scope="col">Question</th><th scope="col">Reference</th>${modelHeads()}<th scope="col">Agree</th>`;
485
  }
486
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
487
  function renderSummary() {
488
  const { rows, times } = board;
 
489
  const thr = threshold();
490
  const S = Object.fromEntries(MODELS.map((m) => [m.key, modelStats(rows, m.key, times[m.key], thr)]));
491
  const ratio = (x, y) => (y ? x / y : null);
 
467
  const w = best(Object.fromEntries(MODELS.map((m) => [m.key, L[m.key].rate])));
468
  return `<tr><td>${esc(groupOf(id)?.label ?? id)}<small>${esc(groupOf(id)?.family ?? "")}</small></td>${MODELS.map((m) => {
469
  const x = L[m.key];
470
+ if (!x.n) return `<td><b>–</b><small>no reference answers</small></td>`; // e.g. Agent security: states and questions only
471
  return `<td class="${w.has(m.key) ? "win" : ""}"><b>${x.hits} / ${x.n}</b><small>${x.cw} confident mistake${x.cw === 1 ? "" : "s"}</small></td>`;
472
  }).join("")}</tr>`;
473
  }).join("");
 
484
  $("score-head").innerHTML = `<th scope="col">Demo</th><th scope="col">Question</th><th scope="col">Reference</th>${modelHeads()}<th scope="col">Agree</th>`;
485
  }
486
 
487
+ // A run with no reference answers (e.g. only the Agent security demos) cannot be scored: the Agentic Use Score
488
+ // would fall back to the same defaults for every model. Show what CAN be compared instead.
489
+ function renderUnscoredSummary() {
490
+ const { rows, times } = board;
491
+ const thr = threshold();
492
+ const card = (m) => {
493
+ const recs = rows.map((r) => r.ans[m.key]).filter(Boolean);
494
+ const acts = recs.filter((rec) => conf(rec) >= thr).length;
495
+ const med = median(times[m.key] || []);
496
+ const facts = [
497
+ ["Questions answered", `${recs.length} / ${rows.length}`],
498
+ ["Median model time", med != null ? `${Math.round(med)} ms` : "–"],
499
+ [`Acts on (conf ≥ ${thr.toFixed(2)})`, `${acts} / ${recs.length}`],
500
+ ["Defers", `${recs.length - acts} / ${recs.length}`],
501
+ ];
502
+ return `<div class="sum-card side-${esc(m.side)}">
503
+ <h3>${esc(m.name)}</h3>
504
+ <dl>${facts.map(([k, v]) => `<div><dt>${k}</dt><dd>${v}</dd></div>`).join("")}</dl>
505
+ </div>`;
506
+ };
507
+ const answered = rows.filter((r) => MODELS.every((m) => r.ans[m.key]));
508
+ const allAgree = answered.filter((r) => new Set(MODELS.map((m) => r.ans[m.key].choice)).size === 1).length;
509
+ const sum = $("summary");
510
+ sum.hidden = false;
511
+ sum.innerHTML = `<div class="unscored-note">
512
+ <h3>Not scored</h3>
513
+ <p>The Agentic Use Score compares each answer with a reference answer, and the questions in this run have none. So there is no score or ranking here; compare the models below by what they would act on at your threshold, their speed, and how often they agree.</p>
514
+ </div>`
515
+ + MODELS.map(card).join("")
516
+ + `<p class="sum-both">All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.</p>`;
517
+ }
518
+
519
  function renderSummary() {
520
  const { rows, times } = board;
521
+ if (!rows.some((r) => r.ref !== undefined)) return renderUnscoredSummary();
522
  const thr = threshold();
523
  const S = Object.fromEntries(MODELS.map((m) => [m.key, modelStats(rows, m.key, times[m.key], thr)]));
524
  const ratio = (x, y) => (y ? x / y : null);
tests/test_demos.py CHANGED
@@ -47,6 +47,8 @@ class DemoSetTest(unittest.TestCase):
47
  """Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added."""
48
  nano = [d for d in DEMOS if d["group"] == "agent_security"]
49
  self.assertEqual(len(nano), 23)
 
 
50
  self.assertEqual(sum(len(d["questions"]) for d in nano), 34)
51
  for d in nano:
52
  with self.subTest(demo=d["id"]):
 
47
  """Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added."""
48
  nano = [d for d in DEMOS if d["group"] == "agent_security"]
49
  self.assertEqual(len(nano), 23)
50
+ label = next(g["label"] for g in GROUPS if g["id"] == "agent_security")
51
+ self.assertEqual(label, "Agent security") # operator, 2026-10-01: no "(nano set)"
52
  self.assertEqual(sum(len(d["questions"]) for d in nano), 34)
53
  for d in nano:
54
  with self.subTest(demo=d["id"]):