Spaces:
Running on Zero
Running on Zero
Michael Stattelman commited on
Commit ·
d8c255d
1
Parent(s): 996a2db
Version updates
Browse files- README.md +4 -4
- app/demos.py +2 -2
- app/main.py +1 -1
- app/static/app.css +5 -0
- app/static/app.js +34 -1
- tests/test_demos.py +2 -0
README.md
CHANGED
|
@@ -81,7 +81,7 @@ cd <your-space>
|
|
| 81 |
git lfs install
|
| 82 |
# copy the contents of this DecisionLab folder into it (not the folder itself), then:
|
| 83 |
git add .
|
| 84 |
-
git commit -m "DecisionLab 2.3.
|
| 85 |
git push
|
| 86 |
```
|
| 87 |
|
|
@@ -122,7 +122,7 @@ The plain HTTP API works on the Space too, at `https://<your-user>-<your-space>.
|
|
| 122 |
## Using the lab
|
| 123 |
|
| 124 |
1. **Setup** shows each model's status, size, device, load time and how it defines confidence.
|
| 125 |
-
2. **Decision**: pick one of the 47 demos (tabs group them into triage, routing and planning, loop control, guardrails, and agent security
|
| 126 |
```json
|
| 127 |
{"team": {"type": "choice", "instructions": "Which team?", "criteria": {"bug": "Something is broken", "sales": "Pricing"}},
|
| 128 |
"urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["Can wait", "Today", "Right now"]},
|
|
@@ -135,7 +135,7 @@ The plain HTTP API works on the Space too, at `https://<your-user>-<your-space>.
|
|
| 135 |
|
| 136 |
## Demos
|
| 137 |
|
| 138 |
-
47 demos in 5 groups. The tab numbers match the table. Demos 25–47 are the **Agent security
|
| 139 |
|
| 140 |
| # | Group | Demo | What it tests |
|
| 141 |
|---|---|---|---|
|
|
@@ -242,7 +242,7 @@ DecisionLab/
|
|
| 242 |
│ ├── security.py # body limit and security headers (plain ASGI)
|
| 243 |
│ ├── validation.py # checks /api/decide question sets, sizes and model keys (no web framework)
|
| 244 |
│ ├── demos.py # groups, the 47 demos (24 scored + 23 nano), reference answers and high-stakes flags
|
| 245 |
-
│ ├── demos_nano.json # Agent security
|
| 246 |
│ └── static/ # index.html, app.css, app.js, logo.png
|
| 247 |
└── tests/ # unittest suite, copied into the image
|
| 248 |
```
|
|
|
|
| 81 |
git lfs install
|
| 82 |
# copy the contents of this DecisionLab folder into it (not the folder itself), then:
|
| 83 |
git add .
|
| 84 |
+
git commit -m "DecisionLab 2.3.1"
|
| 85 |
git push
|
| 86 |
```
|
| 87 |
|
|
|
|
| 122 |
## Using the lab
|
| 123 |
|
| 124 |
1. **Setup** shows each model's status, size, device, load time and how it defines confidence.
|
| 125 |
+
2. **Decision**: pick one of the 47 demos (tabs group them into triage, routing and planning, loop control, guardrails, and agent security) or write your own state and questions. Nothing is executed; every demo is a decision only. Questions use the Laya/Jev JSON shape:
|
| 126 |
```json
|
| 127 |
{"team": {"type": "choice", "instructions": "Which team?", "criteria": {"bug": "Something is broken", "sales": "Pricing"}},
|
| 128 |
"urgency": {"type": "score", "instructions": "How urgent?", "criteria": ["Can wait", "Today", "Right now"]},
|
|
|
|
| 135 |
|
| 136 |
## Demos
|
| 137 |
|
| 138 |
+
47 demos in 5 groups. The tab numbers match the table. Demos 25–47 are the **Agent security** group: the states and questions of the Athr_Agent_Sec nano demo set (`app/demos_nano.json`), with no reference answers, so they show each model's answers, confidence and agreement but are not scored. A scoreboard run of only these demos shows no Agentic Use Score (there is nothing to score it against), just what each model would act on, its speed and the agreement.
|
| 139 |
|
| 140 |
| # | Group | Demo | What it tests |
|
| 141 |
|---|---|---|---|
|
|
|
|
| 242 |
│ ├── security.py # body limit and security headers (plain ASGI)
|
| 243 |
│ ├── validation.py # checks /api/decide question sets, sizes and model keys (no web framework)
|
| 244 |
│ ├── demos.py # groups, the 47 demos (24 scored + 23 nano), reference answers and high-stakes flags
|
| 245 |
+
│ ├── demos_nano.json # Agent security: states and questions only
|
| 246 |
│ └── static/ # index.html, app.css, app.js, logo.png
|
| 247 |
└── tests/ # unittest suite, copied into the image
|
| 248 |
```
|
app/demos.py
CHANGED
|
@@ -31,7 +31,7 @@ GROUPS = [
|
|
| 31 |
"blurb": "Inside the agent loop: stop, retry, change course, or verify the work before reporting it."},
|
| 32 |
{"id": "guardrails", "family": "Agent decisions", "label": "Guardrails",
|
| 33 |
"blurb": "Irreversible actions, prompt injection, fraud, data leaving the company, secrets and permissions."},
|
| 34 |
-
{"id": "agent_security", "family": "Agent decisions", "label": "Agent security
|
| 35 |
"blurb": "Drift, report auditing, action risk, speech consistency and prompt-injection screening, from the "
|
| 36 |
"Athr_Agent_Sec nano demo set. No reference answers: compare the models' answers, confidence and agreement."},
|
| 37 |
]
|
|
@@ -410,7 +410,7 @@ DEMOS = [
|
|
| 410 |
},
|
| 411 |
]
|
| 412 |
|
| 413 |
-
# Agent security
|
| 414 |
# 2026-09-30). No reference answers or stakes, so these demos are not scored for matches; every model still answers
|
| 415 |
# them, with confidence, agreement and speed.
|
| 416 |
_NANO = json.loads((Path(__file__).with_name("demos_nano.json")).read_text(encoding="utf-8"))["demos"]
|
|
|
|
| 31 |
"blurb": "Inside the agent loop: stop, retry, change course, or verify the work before reporting it."},
|
| 32 |
{"id": "guardrails", "family": "Agent decisions", "label": "Guardrails",
|
| 33 |
"blurb": "Irreversible actions, prompt injection, fraud, data leaving the company, secrets and permissions."},
|
| 34 |
+
{"id": "agent_security", "family": "Agent decisions", "label": "Agent security",
|
| 35 |
"blurb": "Drift, report auditing, action risk, speech consistency and prompt-injection screening, from the "
|
| 36 |
"Athr_Agent_Sec nano demo set. No reference answers: compare the models' answers, confidence and agreement."},
|
| 37 |
]
|
|
|
|
| 410 |
},
|
| 411 |
]
|
| 412 |
|
| 413 |
+
# Agent security: only the states and questions of decisionlab_nano_demos.json (operator ruling
|
| 414 |
# 2026-09-30). No reference answers or stakes, so these demos are not scored for matches; every model still answers
|
| 415 |
# them, with confidence, agreement and speed.
|
| 416 |
_NANO = json.loads((Path(__file__).with_name("demos_nano.json")).read_text(encoding="utf-8"))["demos"]
|
app/main.py
CHANGED
|
@@ -27,7 +27,7 @@ MAX_BODY_BYTES = int(os.getenv("MAX_BODY_BYTES", str(256 * 1024)))
|
|
| 27 |
DECIDE_SLOTS = Gate(int(os.getenv("MAX_PENDING_DECIDES", "4"))) # DL-SA-004
|
| 28 |
|
| 29 |
# DL-SA-007: no /docs, /redoc or /openapi.json
|
| 30 |
-
VERSION = "2.3.
|
| 31 |
PAGE = stamp_assets((STATIC / "index.html").read_text(encoding="utf-8"), VERSION)
|
| 32 |
|
| 33 |
|
|
|
|
| 27 |
DECIDE_SLOTS = Gate(int(os.getenv("MAX_PENDING_DECIDES", "4"))) # DL-SA-004
|
| 28 |
|
| 29 |
# DL-SA-007: no /docs, /redoc or /openapi.json
|
| 30 |
+
VERSION = "2.3.1"
|
| 31 |
PAGE = stamp_assets((STATIC / "index.html").read_text(encoding="utf-8"), VERSION)
|
| 32 |
|
| 33 |
|
app/static/app.css
CHANGED
|
@@ -416,3 +416,8 @@ input[type="range"] { height: 28px; }
|
|
| 416 |
.repo a { padding: 12px 0; }
|
| 417 |
.search input { font-size: 16px; }
|
| 418 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 416 |
.repo a { padding: 12px 0; }
|
| 417 |
.search input { font-size: 16px; }
|
| 418 |
}
|
| 419 |
+
|
| 420 |
+
/* ---------- runs with no reference answers: no score, a plain note instead */
|
| 421 |
+
.unscored-note { grid-column: 1 / -1; border: 1px dashed var(--line); border-radius: var(--radius-m); padding: 18px 22px; }
|
| 422 |
+
.unscored-note h3 { font: 500 22px var(--display); margin: 0 0 6px; }
|
| 423 |
+
.unscored-note p { margin: 0; color: var(--graphite); max-width: 70ch; }
|
app/static/app.js
CHANGED
|
@@ -467,7 +467,7 @@ function groupBreakdown(rows, thr) {
|
|
| 467 |
const w = best(Object.fromEntries(MODELS.map((m) => [m.key, L[m.key].rate])));
|
| 468 |
return `<tr><td>${esc(groupOf(id)?.label ?? id)}<small>${esc(groupOf(id)?.family ?? "")}</small></td>${MODELS.map((m) => {
|
| 469 |
const x = L[m.key];
|
| 470 |
-
if (!x.n) return `<td><b>–</b><small>no reference answers</small></td>`; // e.g.
|
| 471 |
return `<td class="${w.has(m.key) ? "win" : ""}"><b>${x.hits} / ${x.n}</b><small>${x.cw} confident mistake${x.cw === 1 ? "" : "s"}</small></td>`;
|
| 472 |
}).join("")}</tr>`;
|
| 473 |
}).join("");
|
|
@@ -484,8 +484,41 @@ function renderScoreHead() {
|
|
| 484 |
$("score-head").innerHTML = `<th scope="col">Demo</th><th scope="col">Question</th><th scope="col">Reference</th>${modelHeads()}<th scope="col">Agree</th>`;
|
| 485 |
}
|
| 486 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 487 |
function renderSummary() {
|
| 488 |
const { rows, times } = board;
|
|
|
|
| 489 |
const thr = threshold();
|
| 490 |
const S = Object.fromEntries(MODELS.map((m) => [m.key, modelStats(rows, m.key, times[m.key], thr)]));
|
| 491 |
const ratio = (x, y) => (y ? x / y : null);
|
|
|
|
| 467 |
const w = best(Object.fromEntries(MODELS.map((m) => [m.key, L[m.key].rate])));
|
| 468 |
return `<tr><td>${esc(groupOf(id)?.label ?? id)}<small>${esc(groupOf(id)?.family ?? "")}</small></td>${MODELS.map((m) => {
|
| 469 |
const x = L[m.key];
|
| 470 |
+
if (!x.n) return `<td><b>–</b><small>no reference answers</small></td>`; // e.g. Agent security: states and questions only
|
| 471 |
return `<td class="${w.has(m.key) ? "win" : ""}"><b>${x.hits} / ${x.n}</b><small>${x.cw} confident mistake${x.cw === 1 ? "" : "s"}</small></td>`;
|
| 472 |
}).join("")}</tr>`;
|
| 473 |
}).join("");
|
|
|
|
| 484 |
$("score-head").innerHTML = `<th scope="col">Demo</th><th scope="col">Question</th><th scope="col">Reference</th>${modelHeads()}<th scope="col">Agree</th>`;
|
| 485 |
}
|
| 486 |
|
| 487 |
+
// A run with no reference answers (e.g. only the Agent security demos) cannot be scored: the Agentic Use Score
|
| 488 |
+
// would fall back to the same defaults for every model. Show what CAN be compared instead.
|
| 489 |
+
function renderUnscoredSummary() {
|
| 490 |
+
const { rows, times } = board;
|
| 491 |
+
const thr = threshold();
|
| 492 |
+
const card = (m) => {
|
| 493 |
+
const recs = rows.map((r) => r.ans[m.key]).filter(Boolean);
|
| 494 |
+
const acts = recs.filter((rec) => conf(rec) >= thr).length;
|
| 495 |
+
const med = median(times[m.key] || []);
|
| 496 |
+
const facts = [
|
| 497 |
+
["Questions answered", `${recs.length} / ${rows.length}`],
|
| 498 |
+
["Median model time", med != null ? `${Math.round(med)} ms` : "–"],
|
| 499 |
+
[`Acts on (conf ≥ ${thr.toFixed(2)})`, `${acts} / ${recs.length}`],
|
| 500 |
+
["Defers", `${recs.length - acts} / ${recs.length}`],
|
| 501 |
+
];
|
| 502 |
+
return `<div class="sum-card side-${esc(m.side)}">
|
| 503 |
+
<h3>${esc(m.name)}</h3>
|
| 504 |
+
<dl>${facts.map(([k, v]) => `<div><dt>${k}</dt><dd>${v}</dd></div>`).join("")}</dl>
|
| 505 |
+
</div>`;
|
| 506 |
+
};
|
| 507 |
+
const answered = rows.filter((r) => MODELS.every((m) => r.ans[m.key]));
|
| 508 |
+
const allAgree = answered.filter((r) => new Set(MODELS.map((m) => r.ans[m.key].choice)).size === 1).length;
|
| 509 |
+
const sum = $("summary");
|
| 510 |
+
sum.hidden = false;
|
| 511 |
+
sum.innerHTML = `<div class="unscored-note">
|
| 512 |
+
<h3>Not scored</h3>
|
| 513 |
+
<p>The Agentic Use Score compares each answer with a reference answer, and the questions in this run have none. So there is no score or ranking here; compare the models below by what they would act on at your threshold, their speed, and how often they agree.</p>
|
| 514 |
+
</div>`
|
| 515 |
+
+ MODELS.map(card).join("")
|
| 516 |
+
+ `<p class="sum-both">All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.</p>`;
|
| 517 |
+
}
|
| 518 |
+
|
| 519 |
function renderSummary() {
|
| 520 |
const { rows, times } = board;
|
| 521 |
+
if (!rows.some((r) => r.ref !== undefined)) return renderUnscoredSummary();
|
| 522 |
const thr = threshold();
|
| 523 |
const S = Object.fromEntries(MODELS.map((m) => [m.key, modelStats(rows, m.key, times[m.key], thr)]));
|
| 524 |
const ratio = (x, y) => (y ? x / y : null);
|
tests/test_demos.py
CHANGED
|
@@ -47,6 +47,8 @@ class DemoSetTest(unittest.TestCase):
|
|
| 47 |
"""Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added."""
|
| 48 |
nano = [d for d in DEMOS if d["group"] == "agent_security"]
|
| 49 |
self.assertEqual(len(nano), 23)
|
|
|
|
|
|
|
| 50 |
self.assertEqual(sum(len(d["questions"]) for d in nano), 34)
|
| 51 |
for d in nano:
|
| 52 |
with self.subTest(demo=d["id"]):
|
|
|
|
| 47 |
"""Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added."""
|
| 48 |
nano = [d for d in DEMOS if d["group"] == "agent_security"]
|
| 49 |
self.assertEqual(len(nano), 23)
|
| 50 |
+
label = next(g["label"] for g in GROUPS if g["id"] == "agent_security")
|
| 51 |
+
self.assertEqual(label, "Agent security") # operator, 2026-10-01: no "(nano set)"
|
| 52 |
self.assertEqual(sum(len(d["questions"]) for d in nano), 34)
|
| 53 |
for d in nano:
|
| 54 |
with self.subTest(demo=d["id"]):
|