REPL when skill_create needs user input

This commit is contained in:
Denton Social
2026-09-24 05:46:23 -05:00
parent e92b3e7228
commit 63922c1fad
15 changed files with 545 additions and 16 deletions
+9 -3
View File
@@ -19,14 +19,20 @@ supplied options; an LLM is used only for generation and self-assessment.
``` ```
cli.py argparse: run (REPL / --script), dream, skills, status, relabel, cli.py argparse: run (REPL / --script), dream, skills, status, relabel,
dashboard dashboard
scheduler.py gate -> choice(tau) -> score -> queue; preempt + requeue scheduler.py gate -> choice(tau) -> score -> queue; preempt + requeue;
a skill run paused for input (needs_input) keeps `current`
busy; `answer` routes straight to the pending run, bypassing
gate/score/navigation
queue.py urgency max-heap (desc weight, FIFO seq), age pulls toward 1.0 queue.py urgency max-heap (desc weight, FIFO seq), age pulls toward 1.0
skills.py tree + registry (email.compose, response.reject, tracking.check), skills.py tree + registry (email.compose, response.reject, tracking.check),
navigation = SemIf choices per level (logged), create_category navigation = SemIf choices per level (logged), create_category
and create_skill author + register stubs via the decision model and create_skill author + register stubs via the decision model
in generation mode; SkillBodyStore + materialize_skill persist in generation mode; SkillBodyStore + materialize_skill persist
and hot-load runnable skill bodies from data/skills/ and hot-load runnable skill bodies from data/skills/;
skill.py loop: observe -> predict -> act -> observe -> assess (LLM) ActionResult.needs_input pauses a run for human input
skill.py loop: observe -> predict -> act -> observe -> assess (LLM);
a run paused for input is resumed by re-invoking act with the
answer on request.user_input (predict is never re-run)
engine.py SemIfEngine -> semif_phase1.llamacpp_backend (lazy import) engine.py SemIfEngine -> semif_phase1.llamacpp_backend (lazy import)
codegen.py CodegenClient (OpenAI-compatible) writes runnable skill bodies codegen.py CodegenClient (OpenAI-compatible) writes runnable skill bodies
against SKILL.md; parse/validate (compile + predict/act) against SKILL.md; parse/validate (compile + predict/act)
+11 -2
View File
@@ -55,10 +55,12 @@ def act(ctx, request, prediction) -> ActionResult:
`ctx.config` (the agent config dict). `ctx.config` (the agent config dict).
- `request` is the `Request` being handled. - `request` is the `Request` being handled.
- `Prediction(text: str, decisions: list)` and - `Prediction(text: str, decisions: list)` and
`ActionResult(action_log: str, new_state: str)` are imported from `ActionResult(action_log: str, new_state: str, needs_input: str | None = None)`
are imported from
`semif_agent.skills`; return those exact types. `decisions` carries any `semif_agent.skills`; return those exact types. `decisions` carries any
`(DecisionRequest, DecisionResult)` pairs made during predict so they are `(DecisionRequest, DecisionResult)` pairs made during predict so they are
logged as training rows. logged as training rows. `needs_input` carries a question for the human; see
the rules below.
### Rules (hard requirements) ### Rules (hard requirements)
@@ -70,6 +72,13 @@ def act(ctx, request, prediction) -> ActionResult:
- **Never swallow the request.** If the skill cannot act, return an - **Never swallow the request.** If the skill cannot act, return an
`ActionResult` with a short `action_log` explaining why and set `new_state` `ActionResult` with a short `action_log` explaining why and set `new_state`
back to `request.text`. back to `request.text`.
- **Request input when data is missing.** If a required piece of data is not
in the request or in local files, do not fail silently: return an
`ActionResult(action_log="...", new_state=request.text, needs_input="<question>")`.
The run pauses and the human is asked. The answer arrives on
`request.user_input` and `act` is called again with the *same* prediction —
check `request.user_input` on the resume pass to finish the run (or ask again
if it is still insufficient).
- **Write files under configured data dirs only** (e.g. `ctx.config["drafts"]`), - **Write files under configured data dirs only** (e.g. `ctx.config["drafts"]`),
never anywhere else on disk. never anywhere else on disk.
- **Fail fast on budget.** Keep the work small; do not loop or retry in code. - **Fail fast on budget.** Keep the work small; do not loop or retry in code.
+4
View File
@@ -121,6 +121,10 @@ def repl(scheduler: Scheduler, config: dict) -> None:
continue continue
status, detail = scheduler.submit(line) status, detail = scheduler.submit(line)
print(f"[{status}] {detail}") print(f"[{status}] {detail}")
while scheduler.pending is not None:
answer = input(f"{scheduler.pending.question} ")
status, detail = scheduler.answer(answer)
print(f"[{status}] {detail}")
def scripted(scheduler: Scheduler, path: str) -> None: def scripted(scheduler: Scheduler, path: str) -> None:
+18 -2
View File
@@ -3,8 +3,8 @@
A pure-stdlib HTTP server on localhost serving a Redux-DevTools-style A pure-stdlib HTTP server on localhost serving a Redux-DevTools-style
inspector over the agent's decision flow. Reads the decision log inspector over the agent's decision flow. Reads the decision log
(`decisions.jsonl`) plus the run lifecycle trace (`runs.jsonl`), exposes the (`decisions.jsonl`) plus the run lifecycle trace (`runs.jsonl`), exposes the
static skill tree, the dream cost report, and two write endpoints: submit a static skill tree, the dream cost report, and three write endpoints: submit a
request and relabel a decision (human override). request, answer a run paused for input, and relabel a decision (human override).
The scheduler's engine and LLM are built lazily, so the dashboard runs on the The scheduler's engine and LLM are built lazily, so the dashboard runs on the
thin dev box in replay mode (reads logs; submit degrades to a JSON error) and thin dev box in replay mode (reads logs; submit degrades to a JSON error) and
@@ -98,12 +98,18 @@ def build_status(scheduler: Scheduler) -> dict:
if scheduler.current if scheduler.current
else None else None
) )
pending = (
{"skill": scheduler.pending.skill.name, "question": scheduler.pending.question}
if scheduler.pending
else None
)
queue = [ queue = [
{"id": request.id, "weight": weight, "text": request.text[:80]} {"id": request.id, "weight": weight, "text": request.text[:80]}
for weight, request in scheduler.queue.items() for weight, request in scheduler.queue.items()
] ]
return { return {
"current": current, "current": current,
"pending": pending,
"queue": queue, "queue": queue,
"tau": scheduler.tau, "tau": scheduler.tau,
"queue_max": scheduler.queue.max_size, "queue_max": scheduler.queue.max_size,
@@ -217,6 +223,16 @@ class DashboardHandler(BaseHTTPRequestHandler):
return return
self._send(200, {"ok": ok}) self._send(200, {"ok": ok})
return return
if path == "/api/answer":
with self.lock:
try:
body = self._read_json()
status, detail = self.scheduler.answer(str(body.get("text", "")))
except Exception as exc:
self._send_error(500, str(exc))
return
self._send(200, {"status": status, "detail": detail})
return
self._send_error(404, "no such endpoint") self._send_error(404, "no such endpoint")
def log_message(self, format, *args): def log_message(self, format, *args):
+1
View File
@@ -71,6 +71,7 @@ class Request:
priority: float = 0.5 priority: float = 0.5
reentries: int = 0 reentries: int = 0
resume: dict[str, Any] = field(default_factory=dict) resume: dict[str, Any] = field(default_factory=dict)
user_input: str | None = None
def copy_for_requeue(self) -> "Request": def copy_for_requeue(self) -> "Request":
return Request( return Request(
+81 -3
View File
@@ -21,6 +21,7 @@ from .skills import (
CategoryRegistry, CategoryRegistry,
CreateCategory, CreateCategory,
CreateSkill, CreateSkill,
Prediction,
Skill, Skill,
SkillBodyStore, SkillBodyStore,
build_skills, build_skills,
@@ -53,13 +54,29 @@ class Process:
weight: float weight: float
@dataclass
class PendingRun:
"""A skill run paused awaiting human input.
`prediction` is kept so resume re-invokes only `act` (predict is not
re-run, avoiding duplicate SemIf sub-decisions); `question` is what the
run asked the human.
"""
request: Request
skill: Skill
prediction: Prediction
question: str
@dataclass @dataclass
class DispatchResult: class DispatchResult:
kind: str # ran | create_category | create_skill | error kind: str # ran | create_category | create_skill | needs_input | error
summary: str summary: str
skill: str | None = None skill: str | None = None
decisions_logged: int = 0 decisions_logged: int = 0
body_written: bool = False body_written: bool = False
needs_input: str | None = None
class Scheduler: class Scheduler:
@@ -95,6 +112,7 @@ class Scheduler:
self.ctx = ActionContext(engine=self.engine, config=config) self.ctx = ActionContext(engine=self.engine, config=config)
self.runner = SkillRunner(self.ctx, self.llm, self.log) self.runner = SkillRunner(self.ctx, self.llm, self.log)
self.current: Process | None = None self.current: Process | None = None
self.pending: PendingRun | None = None
# ---- decision templates (all real SemIf, all logged) ---- # ---- decision templates (all real SemIf, all logged) ----
@@ -164,18 +182,26 @@ class Scheduler:
try: try:
outcome = self._dispatch(request) outcome = self._dispatch(request)
finally: finally:
self.current = None if self.pending is None:
self.current = None
if outcome.kind == "needs_input":
return "needs_input", outcome.summary
self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary) self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary)
return "running", f"[{label}] {outcome.summary}" return "running", f"[{label}] {outcome.summary}"
interrupt = self._choice(request, self.current) interrupt = self._choice(request, self.current)
if interrupt: if interrupt:
if self.pending is not None:
self.trace.append("pending_abandoned", self.pending.request.id)
self.pending = None
previous = self.current previous = self.current
previous.request.resume["from_skill"] = previous.skill previous.request.resume["from_skill"] = previous.skill
self.queue.push(previous.request, previous.weight) self.queue.push(previous.request, previous.weight)
self.current = Process(request=request, skill="(scheduling)", weight=1.0) self.current = Process(request=request, skill="(scheduling)", weight=1.0)
self.trace.append("preempted", request.id, preempted=previous.skill) self.trace.append("preempted", request.id, preempted=previous.skill)
outcome = self._dispatch(request) outcome = self._dispatch(request)
if outcome.kind == "needs_input":
return "preempted", f"interrupted {previous.skill}; {outcome.summary}"
self.current = None self.current = None
self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary) self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary)
return "preempted", f"interrupted {previous.skill}; {outcome.summary}" return "preempted", f"interrupted {previous.skill}; {outcome.summary}"
@@ -190,9 +216,15 @@ class Scheduler:
def busy(self, text: str, skill: str = "(driving)") -> None: def busy(self, text: str, skill: str = "(driving)") -> None:
"""Set a fake in-progress process so the choice/score path is exercised.""" """Set a fake in-progress process so the choice/score path is exercised."""
if self.pending is not None:
self.trace.append("pending_abandoned", self.pending.request.id)
self.pending = None
self.current = Process(request=Request(text, source="busy"), skill=skill, weight=1.0) self.current = Process(request=Request(text, source="busy"), skill=skill, weight=1.0)
def idle(self) -> None: def idle(self) -> None:
if self.pending is not None:
self.trace.append("pending_abandoned", self.pending.request.id)
self.pending = None
self.current = None self.current = None
def run_queue(self) -> list[tuple[str, str]]: def run_queue(self) -> list[tuple[str, str]]:
@@ -212,7 +244,11 @@ class Scheduler:
self.trace.append("error", request.id, phase="dispatch", message=str(exc)) self.trace.append("error", request.id, phase="dispatch", message=str(exc))
outcome = DispatchResult(kind="error", summary=f"dispatch failed: {exc}") outcome = DispatchResult(kind="error", summary=f"dispatch failed: {exc}")
finally: finally:
self.current = None if self.pending is None:
self.current = None
if outcome.kind == "needs_input":
results.append(("needs_input", f"[{request.id}] {outcome.summary}"))
continue
self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary) self.trace.append("ran", request.id, skill=outcome.skill, summary=outcome.summary)
results.append(("ran", f"[{request.id}] {outcome.summary}")) results.append(("ran", f"[{request.id}] {outcome.summary}"))
return results return results
@@ -249,11 +285,51 @@ class Scheduler:
def _run_skill(self, skill: Skill, request: Request) -> DispatchResult: def _run_skill(self, skill: Skill, request: Request) -> DispatchResult:
outcome = self.runner.run(skill, request) outcome = self.runner.run(skill, request)
return self._finish_run(skill, request, outcome)
def answer(self, text: str) -> tuple[str, str]:
"""Feed the human's answer to a run paused for input.
Routed directly to the pending run — no gate, score, or navigation —
and the run resumes by re-invoking only `act` with the same prediction.
"""
if self.pending is None:
return "error", "no run is waiting for input"
pending = self.pending
self.pending = None
pending.request.user_input = text
self.trace.append("answered", pending.request.id, text=text)
outcome = self.runner.resume(pending.skill, pending.request, pending.prediction)
result = self._finish_run(pending.skill, pending.request, outcome)
if result.kind == "needs_input":
return "needs_input", result.summary
self.current = None
self.trace.append("ran", pending.request.id, skill=result.skill, summary=result.summary)
return "ran", f"[resumed] {result.summary}"
def _finish_run(self, skill: Skill, request: Request, outcome) -> DispatchResult:
if outcome.error: if outcome.error:
self.trace.append( self.trace.append(
"error", request.id, skill=skill.name, message=outcome.error "error", request.id, skill=skill.name, message=outcome.error
) )
return DispatchResult(kind="error", summary=f"skill error: {outcome.error}") return DispatchResult(kind="error", summary=f"skill error: {outcome.error}")
if outcome.needs_input:
self.pending = PendingRun(
request=request,
skill=skill,
prediction=outcome.prediction,
question=outcome.needs_input,
)
self.current = Process(request=request, skill=skill.name, weight=0.5)
self.trace.append(
"needs_input", request.id, skill=skill.name, question=outcome.needs_input
)
return DispatchResult(
kind="needs_input",
summary=outcome.needs_input,
skill=skill.name,
needs_input=outcome.needs_input,
)
self.trace.append( self.trace.append(
"assessed", "assessed",
request.id, request.id,
@@ -413,6 +489,8 @@ class Scheduler:
lines = [] lines = []
current = f"{self.current.skill} ({self.current.request.id})" if self.current else "idle" current = f"{self.current.skill} ({self.current.request.id})" if self.current else "idle"
lines.append(f"current: {current}") lines.append(f"current: {current}")
if self.pending is not None:
lines.append(f"awaiting input: {self.pending.question}")
lines.append(f"queue: {len(self.queue)} pending") lines.append(f"queue: {len(self.queue)} pending")
for weight, request in self.queue.items(): for weight, request in self.queue.items():
lines.append(f" {request.id} w={weight:.2f} {request.text[:60]}") lines.append(f" {request.id} w={weight:.2f} {request.text[:60]}")
+61 -2
View File
@@ -24,6 +24,8 @@ class RunResult:
updated_request: str | None = None updated_request: str | None = None
decisions_logged: int = 0 decisions_logged: int = 0
error: str | None = None error: str | None = None
needs_input: str | None = None
prediction: Prediction | None = None
class SkillRunner: class SkillRunner:
@@ -37,8 +39,6 @@ class SkillRunner:
try: try:
prediction = skill.predict(self.ctx, request) if skill.predict else Prediction(text="") prediction = skill.predict(self.ctx, request) if skill.predict else Prediction(text="")
action = skill.act(self.ctx, request, prediction) action = skill.act(self.ctx, request, prediction)
observed = action.new_state
assessment: Assessment = self.llm.assess(skill.name, baseline, action.action_log)
except Exception as exc: except Exception as exc:
return RunResult( return RunResult(
skill=skill.name, skill=skill.name,
@@ -48,6 +48,64 @@ class SkillRunner:
new_state=baseline, new_state=baseline,
error=str(exc), error=str(exc),
) )
if action.needs_input:
return RunResult(
skill=skill.name,
success=False,
summary="",
action_log=action.action_log,
new_state=baseline,
needs_input=action.needs_input,
prediction=prediction,
)
return self._finish(skill, request, prediction, action)
def resume(self, skill: Skill, request: Request, prediction: Prediction) -> RunResult:
"""Re-invoke act with the human's answer (on request.user_input) and finish.
predict is not re-run: its SemIf sub-decisions were already made and are
logged here, at completion, so their run_ok label reflects the outcome.
"""
baseline = request.text
try:
action = skill.act(self.ctx, request, prediction)
except Exception as exc:
return RunResult(
skill=skill.name,
success=False,
summary="",
action_log="",
new_state=baseline,
error=str(exc),
)
if action.needs_input:
return RunResult(
skill=skill.name,
success=False,
summary="",
action_log=action.action_log,
new_state=baseline,
needs_input=action.needs_input,
prediction=prediction,
)
return self._finish(skill, request, prediction, action)
def _finish(
self, skill: Skill, request: Request, prediction: Prediction, action
) -> RunResult:
baseline = request.text
observed = action.new_state
try:
assessment: Assessment = self.llm.assess(skill.name, baseline, action.action_log)
except Exception as exc:
return RunResult(
skill=skill.name,
success=False,
summary="",
action_log="",
new_state=observed,
error=str(exc),
)
run_ok = assessment.success run_ok = assessment.success
decisions = getattr(prediction, "decisions", []) decisions = getattr(prediction, "decisions", [])
@@ -71,4 +129,5 @@ class SkillRunner:
new_state=observed, new_state=observed,
updated_request=assessment.updated_request, updated_request=assessment.updated_request,
decisions_logged=len(decisions), decisions_logged=len(decisions),
prediction=prediction,
) )
+1
View File
@@ -31,6 +31,7 @@ from .trace import TraceLog
class ActionResult: class ActionResult:
action_log: str action_log: str
new_state: str new_state: str
needs_input: str | None = None
@dataclass @dataclass
+39 -1
View File
@@ -5,7 +5,7 @@ const $ = (sel) => document.querySelector(sel);
const state = { const state = {
runs: [], runs: [],
tree: { categories: {} }, tree: { categories: {} },
status: { current: null, queue: [], tau: 0.6 }, status: { current: null, pending: null, queue: [], tau: 0.6 },
dream: {}, dream: {},
selectedRunId: null, selectedRunId: null,
selectedDecisionId: null, selectedDecisionId: null,
@@ -126,6 +126,13 @@ function eventRow(evt) {
} else if (evt.kind === "skill_created") { } else if (evt.kind === "skill_created") {
div.textContent = `created ${evt.skill}${evt.written ? " (body written)" : " (stub)"}`; div.textContent = `created ${evt.skill}${evt.written ? " (body written)" : " (stub)"}`;
div.title = evt.body || evt.description || ""; div.title = evt.body || evt.description || "";
} else if (evt.kind === "needs_input") {
div.textContent = "needs input";
div.title = evt.question || "";
} else if (evt.kind === "answered") {
div.textContent = `answered: ${evt.text || ""}`;
} else if (evt.kind === "pending_abandoned") {
div.textContent = "pending input abandoned";
} }
return div; return div;
} }
@@ -362,6 +369,13 @@ function eventNode(evt) {
node.appendChild(p); node.appendChild(p);
} }
node.classList.add(evt.written ? "ok" : "stub"); node.classList.add(evt.written ? "ok" : "stub");
} else if (evt.kind === "needs_input") {
node.classList.add("needs-input");
body.textContent = `awaiting input: ${evt.question || ""}`;
} else if (evt.kind === "answered") {
body.textContent = `answered: ${evt.text || ""}`;
} else if (evt.kind === "pending_abandoned") {
body.textContent = "pending input abandoned";
} else { } else {
body.textContent = evt.summary || evt.text || ""; body.textContent = evt.summary || evt.text || "";
} }
@@ -470,6 +484,12 @@ function renderStatus() {
.map((item) => `${item.id} w=${item.weight.toFixed(2)} ${item.text}`) .map((item) => `${item.id} w=${item.weight.toFixed(2)} ${item.text}`)
.join("\n") || "queue empty"; .join("\n") || "queue empty";
el.appendChild(q); el.appendChild(q);
const pending = state.status.pending;
$("#answer-form").classList.toggle("hidden", !pending);
if (pending) {
$("#answer-input").placeholder = `${pending.skill}: ${pending.question}`;
}
} }
function stat(k, v) { function stat(k, v) {
@@ -544,6 +564,24 @@ $("#submit-form").addEventListener("submit", async (e) => {
await refreshAll(); await refreshAll();
}); });
$("#answer-form").addEventListener("submit", async (e) => {
e.preventDefault();
const text = $("#answer-input").value.trim();
if (!text) return;
try {
const res = await getJSON("/api/answer", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ text }),
});
flash(`[${res.status}] ${res.detail}`);
} catch (err) {
flash(`answer failed: ${err.message}`);
}
$("#answer-input").value = "";
await refreshAll();
});
$("#refresh-btn").addEventListener("click", async () => { $("#refresh-btn").addEventListener("click", async () => {
try { try {
await refreshAll(); await refreshAll();
+5
View File
@@ -18,6 +18,11 @@
placeholder="type a request, e.g. 'send my girlfriend an email that I'm running late'"> placeholder="type a request, e.g. 'send my girlfriend an email that I'm running late'">
<button type="submit">submit</button> <button type="submit">submit</button>
</form> </form>
<form id="answer-form" class="hidden">
<input id="answer-input" type="text" autocomplete="off" spellcheck="false"
placeholder="answer the pending question">
<button type="submit">answer</button>
</form>
<button id="refresh-btn" type="button" title="reload trace">refresh</button> <button id="refresh-btn" type="button" title="reload trace">refresh</button>
</section> </section>
+61 -1
View File
@@ -17,7 +17,10 @@ from semif_agent.decisions import Request
from semif_agent.dream import dream from semif_agent.dream import dream
from semif_agent.engine import EngineUnavailable from semif_agent.engine import EngineUnavailable
from semif_agent.skills import ( from semif_agent.skills import (
ActionResult,
CategoryDraft, CategoryDraft,
Prediction,
Skill,
SkillBodyStore, SkillBodyStore,
SkillDraft, SkillDraft,
build_skills, build_skills,
@@ -237,4 +240,61 @@ def test_create_category_chain_runs_new_skill(tmp_path):
created = next(e for e in rows if e["kind"] == "skill_created") created = next(e for e in rows if e["kind"] == "skill_created")
assert created["written"] is True, "codegen must produce a runnable body" assert created["written"] is True, "codegen must produce a runnable body"
assessed = next(e for e in rows if e["kind"] == "assessed") assessed = next(e for e in rows if e["kind"] == "assessed")
assert assessed["skill"] == created["skill"], "the created skill must run" assert assessed["skill"] == created["skill"], "the created skill must run"
def test_skill_pauses_for_input_and_resumes(tmp_path):
"""A run paused for input keeps `current` busy, then `answer` resumes it.
Uses the real scheduler (real engine + real LLM assessment on the resumed
run). The skill itself is injected, not authored, so the flow is
deterministic: pause -> answer -> resume -> assessed.
"""
config = load_config()
require_real(config)
config["log"] = str(tmp_path / "decisions.jsonl")
config["trace"] = str(tmp_path / "runs.jsonl")
scheduler, config = build_scheduler(config)
seen = []
def predict(ctx, request):
return Prediction(text="", decisions=[])
def act(ctx, request, prediction):
if request.user_input:
seen.append(request.user_input)
return ActionResult(
action_log=f"resumed with {request.user_input}",
new_state=f"done {request.user_input}",
)
return ActionResult(
action_log="need a tracking number",
new_state=request.text,
needs_input="What's the tracking number?",
)
skill = Skill(
name="track.manual",
category="tracking",
description="Resolve a tracking number with the human.",
predict=predict,
act=act,
)
result = scheduler._run_skill(skill, Request("track my package manually"))
print(f"[{result.kind}] {result.summary}")
assert result.kind == "needs_input"
assert scheduler.pending is not None
assert scheduler.current is not None
status, detail = scheduler.answer("AB123")
print(f"[{status}] {detail}")
assert status == "ran"
assert seen == ["AB123"]
assert scheduler.pending is None
assert scheduler.current is None
rows = scheduler.trace.read()
kinds = [e["kind"] for e in rows]
assert "needs_input" in kinds and "answered" in kinds and "assessed" in kinds
+5 -1
View File
@@ -39,4 +39,8 @@ def test_request_requeue_preserves_state():
assert updated.id == original.id assert updated.id == original.id
assert updated.priority == original.priority assert updated.priority == original.priority
assert updated.resume["from_skill"] == "email.compose" assert updated.resume["from_skill"] == "email.compose"
assert updated.reentries == original.reentries + 1 assert updated.reentries == original.reentries + 1
def test_request_user_input_defaults_none():
assert Request(text="t").user_input is None
+67 -1
View File
@@ -12,11 +12,12 @@ import urllib.request
from http.server import ThreadingHTTPServer from http.server import ThreadingHTTPServer
from semif_agent.dashboard import DashboardHandler from semif_agent.dashboard import DashboardHandler
from semif_agent.decisions import DecisionRequest, DecisionResult, Option from semif_agent.decisions import DecisionRequest, DecisionResult, Option, Request
from semif_agent.engine import EngineConfig, SemIfEngine from semif_agent.engine import EngineConfig, SemIfEngine
from semif_agent.llm import LLMClient from semif_agent.llm import LLMClient
from semif_agent.log import DecisionLog from semif_agent.log import DecisionLog
from semif_agent.scheduler import Scheduler from semif_agent.scheduler import Scheduler
from semif_agent.skills import ActionResult, Prediction, Skill
from semif_agent.trace import TraceLog from semif_agent.trace import TraceLog
@@ -182,5 +183,70 @@ def test_skill_writing_and_created_events_in_payload(tmp_path):
created = next(e for e in run["events"] if e["kind"] == "skill_created") created = next(e for e in run["events"] if e["kind"] == "skill_created")
assert created["written"] is True assert created["written"] is True
assert created["body"] == "data/skills/tracking/track_live.py" assert created["body"] == "data/skills/tracking/track_live.py"
finally:
server.close()
def need_input_skill(seen):
def predict(ctx, request):
return Prediction(text="", decisions=[])
def act(ctx, request, prediction):
if request.user_input:
seen.append(request.user_input)
return ActionResult(action_log="ok", new_state=f"done {request.user_input}")
return ActionResult(
action_log="ask",
new_state=request.text,
needs_input="What's the tracking number?",
)
return Skill(
name="track.manual",
category="tracking",
description="Resolve a tracking number with the human.",
predict=predict,
act=act,
)
def test_answer_without_pending_returns_error_json(tmp_path):
scheduler = build_scheduler(tmp_path)
server = Server(scheduler)
try:
status, payload = server.post("/api/answer", {"text": "hello"})
assert status == 200
assert payload["status"] == "error"
assert "waiting for input" in payload["detail"]
finally:
server.close()
def test_status_includes_pending(tmp_path):
scheduler = build_scheduler(tmp_path)
scheduler._run_skill(need_input_skill([]), Request("track my package"))
server = Server(scheduler)
try:
status, payload = server.get("/api/status")
assert status == 200
assert payload["pending"]["skill"] == "track.manual"
assert payload["pending"]["question"] == "What's the tracking number?"
finally:
server.close()
def test_answer_roundtrip_via_api(tmp_path):
scheduler = build_scheduler(tmp_path)
seen = []
scheduler._run_skill(need_input_skill(seen), Request("track my package"))
server = Server(scheduler)
try:
status, payload = server.post("/api/answer", {"text": "AB123"})
assert status == 200
assert payload["status"] == "ran"
assert seen == ["AB123"]
status, payload = server.get("/api/status")
assert payload["pending"] is None
finally: finally:
server.close() server.close()
+177
View File
@@ -0,0 +1,177 @@
"""Pure-stdlib tests for the runtime user-input channel.
A skill can pause its run by returning ActionResult(..., needs_input="<q>");
the scheduler keeps the run pending, and `answer` resumes it by re-invoking
only `act` with the human's answer on request.user_input. No mocking: the
scheduler uses the lazy engine (never loaded) and a real-but-unreachable LLM
endpoint, so assessments degrade to failure — which is fine for these tests.
"""
from semif_agent.decisions import DecisionRequest, DecisionResult, Option, Request
from semif_agent.engine import EngineConfig, SemIfEngine
from semif_agent.llm import LLMClient
from semif_agent.log import DecisionLog
from semif_agent.scheduler import Scheduler
from semif_agent.skills import ActionResult, Prediction, Skill
from semif_agent.trace import TraceLog
def build_scheduler(tmp_path):
log = DecisionLog(str(tmp_path / "decisions.jsonl"))
trace = TraceLog(str(tmp_path / "runs.jsonl"))
engine = SemIfEngine(EngineConfig())
llm = LLMClient(base_url="http://localhost:1/v1", model="test")
return Scheduler(
engine=engine,
llm=llm,
log=log,
config={"skills": {}},
trace=trace,
)
def need_input_skill(seen):
def predict(ctx, request):
return Prediction(text="", decisions=[])
def act(ctx, request, prediction):
if request.user_input:
seen.append(request.user_input)
return ActionResult(
action_log=f"got {request.user_input}", new_state=f"done {request.user_input}"
)
return ActionResult(
action_log="need a tracking number",
new_state=request.text,
needs_input="What's the tracking number?",
)
return Skill(
name="track.manual",
category="tracking",
description="Resolve a tracking number with the human.",
predict=predict,
act=act,
)
def test_skill_pause_and_answer(tmp_path):
scheduler = build_scheduler(tmp_path)
seen = []
request = Request("track my package manually")
result = scheduler._run_skill(need_input_skill(seen), request)
assert result.kind == "needs_input"
assert result.needs_input == "What's the tracking number?"
assert scheduler.pending is not None
assert scheduler.pending.question == "What's the tracking number?"
assert scheduler.current is not None
assert scheduler.current.skill == "track.manual"
assert scheduler.log.read() == [], "predict decisions must be deferred until completion"
status, detail = scheduler.answer("AB123")
assert status == "ran"
assert seen == ["AB123"]
assert scheduler.pending is None
assert scheduler.current is None
kinds = [e["kind"] for e in scheduler.trace.read()]
assert "needs_input" in kinds
assert "answered" in kinds
assert "assessed" in kinds
assert "ran" in kinds
def test_answer_without_pending_is_error(tmp_path):
scheduler = build_scheduler(tmp_path)
status, detail = scheduler.answer("hello")
assert status == "error"
assert "waiting for input" in detail
def test_predict_decisions_logged_on_completion(tmp_path):
decision = DecisionRequest(
state="s", question="which?", options=[Option("a", "A."), Option("b", "B.")]
)
result = DecisionResult(
request=decision, option_ids=["a", "b"], probabilities=[0.3, 0.7]
)
def predict(ctx, request):
return Prediction(text="", decisions=[(decision, result)])
def act(ctx, request, prediction):
if request.user_input:
return ActionResult(action_log="ok", new_state="done")
return ActionResult(
action_log="ask", new_state=request.text, needs_input="confirm?"
)
skill = Skill(name="t.x", category="t", description="", predict=predict, act=act)
scheduler = build_scheduler(tmp_path)
scheduler._run_skill(skill, Request("x"))
assert scheduler.log.read() == []
scheduler.answer("yes")
rows = scheduler.log.read()
assert len(rows) == 1
assert rows[0]["extra"]["phase"] == "predict"
assert rows[0]["extra"]["run_ok"] is False
def test_busy_abandons_pending(tmp_path):
scheduler = build_scheduler(tmp_path)
request = Request("track")
scheduler._run_skill(need_input_skill([]), request)
assert scheduler.pending is not None
scheduler.busy("driving on the freeway", skill="driving")
assert scheduler.pending is None
assert any(e["kind"] == "pending_abandoned" for e in scheduler.trace.read())
def test_idle_abandons_pending(tmp_path):
scheduler = build_scheduler(tmp_path)
scheduler._run_skill(need_input_skill([]), Request("track"))
assert scheduler.pending is not None
scheduler.idle()
assert scheduler.pending is None
assert scheduler.current is None
def test_resume_can_ask_again(tmp_path):
seen = []
def predict(ctx, request):
return Prediction(text="", decisions=[])
def act(ctx, request, prediction):
if request.user_input == "AB123":
return ActionResult(action_log="done", new_state="resolved")
if request.user_input:
seen.append(request.user_input)
return ActionResult(
action_log="wrong format",
new_state=request.text,
needs_input="That wasn't a valid tracking number. Try again?",
)
return ActionResult(
action_log="ask", new_state=request.text, needs_input="Tracking number?"
)
skill = Skill(name="t.x", category="t", description="", predict=predict, act=act)
scheduler = build_scheduler(tmp_path)
scheduler._run_skill(skill, Request("track"))
status, detail = scheduler.answer("XYZ")
assert status == "needs_input"
assert seen == ["XYZ"]
assert scheduler.pending is not None
status, detail = scheduler.answer("AB123")
assert status == "ran"
assert scheduler.pending is None
assert scheduler.current is None
+5
View File
@@ -13,6 +13,7 @@ from semif_agent.llm import LLMClient
from semif_agent.log import DecisionLog from semif_agent.log import DecisionLog
from semif_agent.scheduler import Scheduler from semif_agent.scheduler import Scheduler
from semif_agent.skills import ( from semif_agent.skills import (
ActionResult,
CategoryDraft, CategoryDraft,
CategoryRegistry, CategoryRegistry,
CreateCategory, CreateCategory,
@@ -31,6 +32,10 @@ from semif_agent.skills import (
from semif_agent.trace import TraceLog from semif_agent.trace import TraceLog
def test_action_result_needs_input_defaults_none():
assert ActionResult("log", "state").needs_input is None
def test_build_category_prompt_contains_request_and_tree(): def test_build_category_prompt_contains_request_and_tree():
tree = build_tree(build_skills({"skills": {}})) tree = build_tree(build_skills({"skills": {}}))
messages = build_category_prompt(Request("tracking for my drone delivery"), tree) messages = build_category_prompt(Request("tracking for my drone delivery"), tree)