Plan a text-extraction pipeline from your own tools
Send a representative sample of the structured text you need to parse — quiz banks,
invoice lines, form exports, log lines, email threads, OCR output — plus what you want
extracted, and get back one JSON object: an honest verdict between
regex-first, hybrid and llm-first, a reading of the
format with a consistency percentage, the target fields mapped to where they come from, the
actual regex patterns with flavor, flags, capture groups and expected match counts, ordered
cleaner rules, the mechanical confidence checks that decide which records escalate to a
model, an LLM stage sized to the cheapest tier that works, the pipeline and the edge cases.
Everything this app does goes through the SkillSafe App API — plain JSON over HTTPS
— so you can re-plan whenever an upstream export changes, compile the returned patterns
in CI and assert their match counts against a fixture, or gate a data pipeline on the
verdict. Every code step below is shown in cURL, Python, JavaScript, Go, Java, Ruby, PHP and
C#; pick a language once and the whole page follows.
Basics
Base URL: https://api.skillsafe.ai/v1/app-api, app slug
parse-planner. Every request sends
Authorization: Bearer <token> and JSON bodies with
Content-Type: application/json. Responses are wrapped in an envelope:
{"data": …} on success, {"error": {"code", "message"}} on failure.
The plan is produced by the gpt-terra model. Estimates are free; runs are
metered against your credit balance. There is a single run task — one sample and goal
in, one parse plan out, no follow-up calls and no session state to carry.
| Status | Meaning |
|---|---|
401 | Missing or expired token — create a new session. |
402 | Not enough credits — top up at skillsafe.ai/account/credits. |
403 | The token isn't allowed to do this (e.g. a guest submitting a very large sample). |
404 | Unknown job or record id. |
5xx | Transient platform error — retry with backoff. |
Browsers enforce CORS for this API, so run these examples from a server, script or terminal — not from another website's frontend.
Step 0 — A tiny client
Every task below is a single HTTP call, so start with a short helper that adds the auth
header, sends JSON and unwraps the data envelope. The later steps reuse it.
export API="https://api.skillsafe.ai/v1/app-api"
export TOKEN="YOUR_TOKEN" # see step 1
# every call looks like:
# curl -s "$API/..." -H "Authorization: Bearer $TOKEN" [-d '{json}']
# jq is used below to pull fields out of the {"data": ...} envelope
import json, requests
API = "https://api.skillsafe.ai/v1/app-api"
TOKEN = "YOUR_TOKEN" # see step 1 — read it from your shell environment in real code
def api(method, path, body=None, **headers):
res = requests.request(method, API + path, json=body,
headers={"Authorization": f"Bearer {TOKEN}", **headers})
payload = res.json()
if not res.ok:
raise RuntimeError(payload.get("error", {}).get("message", res.reason))
return payload["data"]
// Node 18+ (built-in fetch)
const API = "https://api.skillsafe.ai/v1/app-api";
const TOKEN = "YOUR_TOKEN"; // see step 1 — read it from your shell environment in real code
async function api(method, path, body, extraHeaders = {}) {
const res = await fetch(API + path, {
method,
headers: { Authorization: `Bearer ${TOKEN}`, "Content-Type": "application/json", ...extraHeaders },
body: body === undefined ? undefined : JSON.stringify(body),
});
const json = await res.json();
if (!res.ok) throw new Error(json.error?.message ?? res.statusText);
return json.data;
}
package main
import (
"bytes"
"encoding/json"
"fmt"
"net/http"
"os"
)
const API = "https://api.skillsafe.ai/v1/app-api"
var token = os.Getenv("SKILLSAFE_TOKEN") // see step 1
func call(method, path string, body, out any) error {
var buf bytes.Buffer
if body != nil {
json.NewEncoder(&buf).Encode(body)
}
req, _ := http.NewRequest(method, API+path, &buf)
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", "application/json")
res, err := http.DefaultClient.Do(req)
if err != nil {
return err
}
defer res.Body.Close()
var env struct {
Data json.RawMessage `json:"data"`
Error *struct{ Message string `json:"message"` } `json:"error"`
}
json.NewDecoder(res.Body).Decode(&env)
if res.StatusCode >= 400 {
return fmt.Errorf("api %s %s: %s", method, path, env.Error.Message)
}
if out == nil {
return nil
}
return json.Unmarshal(env.Data, out)
}
// Java 17+, no dependencies. Pair with your JSON library (Jackson, Gson…)
// to read fields out of the returned envelope.
import java.net.URI;
import java.net.http.HttpClient;
import java.net.http.HttpRequest;
import java.net.http.HttpResponse;
public class SkillSafe {
static final String API = "https://api.skillsafe.ai/v1/app-api";
static final String TOKEN = System.getenv("SKILLSAFE_TOKEN"); // see step 1
static final HttpClient HTTP = HttpClient.newHttpClient();
static String api(String method, String path, String jsonBody) throws Exception {
var req = HttpRequest.newBuilder(URI.create(API + path))
.header("Authorization", "Bearer " + TOKEN)
.header("Content-Type", "application/json")
.method(method, jsonBody == null
? HttpRequest.BodyPublishers.noBody()
: HttpRequest.BodyPublishers.ofString(jsonBody))
.build();
var res = HTTP.send(req, HttpResponse.BodyHandlers.ofString());
if (res.statusCode() >= 400) throw new RuntimeException(res.body());
return res.body(); // envelope: {"data": …}
}
}
require "net/http"
require "json"
API = "https://api.skillsafe.ai/v1/app-api"
TOKEN = ENV.fetch("SKILLSAFE_TOKEN") # see step 1
def api(method, path, body = nil)
uri = URI(API + path)
req = Net::HTTP.const_get(method.capitalize).new(uri)
req["Authorization"] = "Bearer #{TOKEN}"
req["Content-Type"] = "application/json"
req.body = body.to_json if body
res = Net::HTTP.start(uri.host, uri.port, use_ssl: true) { |h| h.request(req) }
payload = JSON.parse(res.body)
raise (payload.dig("error", "message") || res.message) unless res.is_a?(Net::HTTPSuccess)
payload["data"]
end
<?php
const API = "https://api.skillsafe.ai/v1/app-api";
$TOKEN = getenv("SKILLSAFE_TOKEN"); // see step 1
function api(string $method, string $path, ?array $body = null): mixed {
global $TOKEN;
$ch = curl_init(API . $path);
curl_setopt_array($ch, [
CURLOPT_CUSTOMREQUEST => $method,
CURLOPT_RETURNTRANSFER => true,
CURLOPT_HTTPHEADER => [
"Authorization: Bearer $TOKEN",
"Content-Type: application/json",
],
CURLOPT_POSTFIELDS => $body === null ? null : json_encode($body),
]);
$payload = json_decode(curl_exec($ch), true);
$status = curl_getinfo($ch, CURLINFO_RESPONSE_CODE);
curl_close($ch);
if ($status >= 400) {
throw new Exception($payload["error"]["message"] ?? "HTTP $status");
}
return $payload["data"];
}
// .NET 8+
using System.Net.Http.Json;
using System.Text.Json;
static class SkillSafe
{
const string Api = "https://api.skillsafe.ai/v1/app-api";
static readonly HttpClient Http = new();
static SkillSafe() =>
Http.DefaultRequestHeaders.Authorization =
new("Bearer", Environment.GetEnvironmentVariable("SKILLSAFE_TOKEN")); // see step 1
public static async Task<JsonElement> ApiAsync(HttpMethod method, string path, object? body = null)
{
var req = new HttpRequestMessage(method, Api + path);
if (body != null) req.Content = JsonContent.Create(body);
var res = await Http.SendAsync(req);
var json = await res.Content.ReadFromJsonAsync<JsonElement>();
if (!res.IsSuccessStatusCode)
throw new Exception(json.GetProperty("error").GetProperty("message").GetString());
return json.GetProperty("data");
}
}
Step 1 — Get a token
A guest token lets you check balances and estimate costs for free. For metered planning runs
billed to your own account, use your personal token: open the
token page, sign in with SkillSafe, and press
Copy shell export — it puts export SKILLSAFE_TOKEN="…" on your
clipboard, which every example below reads. Treat the token like a password: it can spend
your credits. For fully headless scripts, POST /guest mints a guest token with
no browser involved.
curl -s -X POST "$API/guest" \
-H "Content-Type: application/json" \
-d '{"slug":"parse-planner"}' | jq -r '.data.token'
token = api("POST", "/guest", {"slug": "parse-planner"})["token"]
const { token } = await api("POST", "/guest", { slug: "parse-planner" });
var guest struct{ Token string `json:"token"` }
err := call("POST", "/guest", map[string]string{"slug": "parse-planner"}, &guest)
String envelope = api("POST", "/guest", """
{"slug":"parse-planner"}""");
// token is at data.token in the returned JSON
token = api("POST", "/guest", { slug: "parse-planner" })["token"]
$token = api("POST", "/guest", ["slug" => "parse-planner"])["token"];
var guest = await SkillSafe.ApiAsync(HttpMethod.Post, "/guest",
new { slug = "parse-planner" });
var token = guest.GetProperty("token").GetString();
The app stores this browser's token under the localStorage key
skillsafe_app_token:parse-planner, on the app's own origin. The
token page reads and manages it for you — you never need
to open developer tools.
Step 2 — Check who you are and your balance
Returns subject_type ("user" or "guest"),
subject_id and your credits balance. Check this before sending a
large sample: a run is refused when the balance is under the model's minimum, and a balance
between that minimum and the full hold can cut a long plan short.
curl -s "$API/me" -H "Authorization: Bearer $TOKEN" | jq '.data'
me = api("GET", "/me")
print(me["subject_type"], me["credits"])
const me = await api("GET", "/me");
console.log(me.subject_type, me.credits);
var me struct {
SubjectType string `json:"subject_type"`
Credits int64 `json:"credits"`
}
err := call("GET", "/me", nil, &me)
String envelope = api("GET", "/me", null);
// data.subject_type, data.credits
me = api("GET", "/me")
puts "#{me["subject_type"]}: #{me["credits"]} credits"
$me = api("GET", "/me");
echo "{$me['subject_type']}: {$me['credits']} credits\n";
var me = await SkillSafe.ApiAsync(HttpMethod.Get, "/me");
Console.WriteLine($"{me.GetProperty("subject_type")}: {me.GetProperty("credits")} credits");
Step 3 — Estimate the cost
Send exactly the input you would send to /run; the response's
hold_credits is the worst-case amount reserved for a run, min_credits
the floor below which a run is refused, and model the resolved model. Nothing is
charged and no job is created, so estimating is free — useful when you are piping a long
sample in and want a ceiling before spending credits. The input is the same five-field object
the web form submits:
| Input field | Type | Notes |
|---|---|---|
sample | string, required | A representative slice of the text you need to parse, exactly as pasted: quiz questions, invoice lines, form exports, log lines, email digests, OCR output. Twenty to sixty records is plenty — include the ugly ones, because the verdict hangs on how much of the sample really follows one shape. This is the model's only evidence: no pattern, no expected_matches count and no edge case marked seen_in_sample may come from anywhere else. At least 40 characters are needed for a run. The web UI clips at 40,000 characters middle-out and inserts a line reading [sample truncated - middle omitted] where the cut is; API callers should do the same when they trim a long corpus, because that marker is what tells the model it is seeing head and tail only and must state its counts as counts over the visible sample. |
goal | string, required | What you want extracted, in your own words — this defines the target fields, and every field named here comes back in fields. Be concrete ("question number, question text, the choices, and the answer letter" beats "parse the quiz"). If the goal asks for something the sample cannot supply, the field still appears, with maps_to: "not present in sample" and the reason in notes. Minimum 8 characters; the app clips at 4,000. |
target | string, required | python | javascript | both — where the patterns will run, which decides the regex dialect. python yields patterns valid for re.compile with flag names like MULTILINE; javascript yields patterns valid for new RegExp(pattern, flags) with flags like gm; both restricts the model to syntax valid in either dialect (no possessive quantifiers, no atomic groups) and emits the flavor-appropriate spelling of named groups. Each returned pattern declares its own flavor. |
context | string, optional | Corpus volume, runtime and cost constraints, where the text comes from (an OCR pipeline, a vendor export, scraping), anything else worth knowing. It is what turns a plan into a budgeted plan — "40k invoices a month, cents not dollars" changes the LLM stage. Send "" when you have nothing to add; the app clips at 8,000. |
prescan_facts | object, optional — null is fine | The result of the free, mechanical, in-browser profile of the same sample, which knows nothing about meaning: {"stats": {"content_lines", "blocks", "consistency_pct", "delimiter"}, "flags": [{"id", "label"}]}. Flag ids are <family>:<check> — consistency:high, consistency:mixed, consistency:low, delimited:tab (also comma, pipe, semicolon), records:numbered, records:lettered, records:kv, records:blocks, noise:page-numbers, noise:ruling, noise:encoding, noise:form-feeds, noise:hyphen-wrap, sample:tiny, sample:truncated. Every flag id you send comes back exactly once in coverage_check. API callers may send null (or omit the field) and get an empty coverage_check; nothing else about the plan changes. |
retry_note | string, optional | Only set by the app's automatic reformat retry when a first reply was not valid JSON. Leave it out. |
cat > sample.txt <<'SAMPLE'
1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
SAMPLE
# trimming a long corpus yourself? mark the cut so the model knows
# it is seeing head and tail only:
# { head -c 20000 corpus.txt; echo; echo "[sample truncated - middle omitted]";
# echo; tail -c 20000 corpus.txt; } > sample.txt
jq -n --rawfile s sample.txt \
'{sample: $s,
goal: "question number, question text, the four choices, and the answer letter",
target: "both",
context: "",
prescan_facts: null}' > input.json
curl -s -X POST "$API/estimate" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-d @input.json | jq '.data | {model, hold_credits, min_credits}'
SAMPLE = """1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
"""
# trimming a long corpus yourself? mark the cut so the model knows it sees head and tail only:
# SAMPLE = corpus[:20000] + "\n[sample truncated - middle omitted]\n" + corpus[-20000:]
payload = {
"sample": SAMPLE,
"goal": "question number, question text, the four choices, and the answer letter",
"target": "both",
"context": "",
"prescan_facts": None,
}
est = api("POST", "/estimate", payload)
print(est["model"], "worst case:", est["hold_credits"], "min:", est["min_credits"])
const sample = [
"1. What is the capital of France?",
" A) Berlin B) Paris C) Rome D) Madrid",
" Answer: B",
"",
"2. Which planet is closest to the Sun?",
" A) Venus B) Mercury C) Mars D) Earth",
" Answer: B",
"",
"3. In which year did the Berlin Wall fall?",
" A) 1987 B) 1989 C) 1991 D) 1993",
" Answer: B",
].join("\n");
// trimming a long corpus yourself? mark the cut:
// corpus.slice(0, 20000) + "\n[sample truncated - middle omitted]\n" + corpus.slice(-20000)
const payload = {
sample,
goal: "question number, question text, the four choices, and the answer letter",
target: "both",
context: "",
prescan_facts: null,
};
const est = await api("POST", "/estimate", payload);
console.log(est.model, "worst case:", est.hold_credits, "min:", est.min_credits);
const sample = "1. What is the capital of France?\n" +
" A) Berlin B) Paris C) Rome D) Madrid\n" +
" Answer: B\n" +
"\n" +
"2. Which planet is closest to the Sun?\n" +
" A) Venus B) Mercury C) Mars D) Earth\n" +
" Answer: B\n" +
"\n" +
"3. In which year did the Berlin Wall fall?\n" +
" A) 1987 B) 1989 C) 1991 D) 1993\n" +
" Answer: B\n"
// trimming a long corpus yourself? join head and tail with the marker line
// "[sample truncated - middle omitted]" so the model knows what it is seeing.
payload := map[string]any{
"sample": sample,
"goal": "question number, question text, the four choices, and the answer letter",
"target": "both",
"context": "",
"prescan_facts": nil,
}
var est struct {
Model string `json:"model"`
HoldCredits int64 `json:"hold_credits"`
MinCredits int64 `json:"min_credits"`
}
err := call("POST", "/estimate", payload, &est)
String sample = """
1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
""";
// trimming a long corpus yourself? join head and tail with the marker line
// "[sample truncated - middle omitted]" so the model knows what it is seeing.
String jsonPayload = """
{"sample": %s,
"goal": "question number, question text, the four choices, and the answer letter",
"target": "both",
"context": "",
"prescan_facts": null}
""".formatted(toJsonString(sample));
String envelope = api("POST", "/estimate", jsonPayload);
// data.model, data.hold_credits and data.min_credits carry the price
SAMPLE = <<~TEXT
1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
TEXT
# trimming a long corpus yourself? mark the cut:
# corpus[0, 20_000] + "\n[sample truncated - middle omitted]\n" + corpus[-20_000..]
payload = { sample: SAMPLE,
goal: "question number, question text, the four choices, and the answer letter",
target: "both",
context: "",
prescan_facts: nil }
est = api("POST", "/estimate", payload)
puts "#{est["model"]} hold=#{est["hold_credits"]} min=#{est["min_credits"]}"
$sample = <<<'TEXT'
1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
TEXT;
// trimming a long corpus yourself? put the line
// "[sample truncated - middle omitted]" between head and tail.
$payload = [
"sample" => $sample,
"goal" => "question number, question text, the four choices, and the answer letter",
"target" => "both",
"context" => "",
"prescan_facts" => null,
];
$est = api("POST", "/estimate", $payload);
echo $est["model"], " hold=", $est["hold_credits"], " min=", $est["min_credits"], "\n";
var sample = """
1. What is the capital of France?
A) Berlin B) Paris C) Rome D) Madrid
Answer: B
2. Which planet is closest to the Sun?
A) Venus B) Mercury C) Mars D) Earth
Answer: B
3. In which year did the Berlin Wall fall?
A) 1987 B) 1989 C) 1991 D) 1993
Answer: B
""";
// trimming a long corpus yourself? join head and tail with the marker line
// "[sample truncated - middle omitted]" so the model knows what it is seeing.
var payload = new {
sample,
goal = "question number, question text, the four choices, and the answer letter",
target = "both",
context = "",
prescan_facts = (object?)null,
};
var est = await SkillSafe.ApiAsync(HttpMethod.Post, "/estimate", payload);
Console.WriteLine($"{est.GetProperty("model")} hold={est.GetProperty("hold_credits")}");
prescan_facts.flags is how you make the plan answer for things you already know
about. Send {"stats": {"content_lines": 11, "blocks": 3, "consistency_pct": 91, "delimiter": "none detected"},
"flags": [{"id": "records:numbered", "label": "3 numbered record starts"},
{"id": "noise:page-numbers", "label": "2 standalone page-number lines"}]} and every flag
id comes back in coverage_check — confirmed as having shaped the plan, or
set aside with the reason. Nothing you flag is silently dropped, which makes it the field to
assert on in a CI check. Send null if you have no scanner of your own.
hold_credits is a reservation, not the price: the settled
charged_credits after a run is usually far lower.
Step 4 — Run the planner and wait for the result
/run takes the same input as /estimate, places a credit hold and
returns a job_id. Poll /jobs/{job_id} every 1–2 seconds
until status is succeeded or failed (a run typically
takes 30–90 s, since the reply carries the patterns, the cleaner rules and the whole
pipeline). Always send an Idempotency-Key header so a network retry can't start
a second, double-charged run. The reply is in output — usually nested as
output.output, and as a JSON string, so parse defensively. The samples
below print the verdict, the fields, the patterns with their expected match counts and the
LLM stage, then save the whole object to plan.json and the patterns on their own
to patterns.json — and, where the language's regex engine allows it, check
each pattern's claimed count against the real one.
JOB_ID=$(curl -s -X POST "$API/run" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-H "Idempotency-Key: pp-$(date +%s)" \
-d @input.json | jq -r '.data.job_id')
while :; do
JOB=$(curl -s "$API/jobs/$JOB_ID" -H "Authorization: Bearer $TOKEN")
STATUS=$(echo "$JOB" | jq -r '.data.status')
[ "$STATUS" = "succeeded" ] || [ "$STATUS" = "failed" ] && break
sleep 2
done
# unwrap the reply once, then read it
echo "$JOB" | jq -r '.data.output.output' > plan.json
# the patterns are the part you wire into code — pull them out on their own
jq '.patterns' plan.json > patterns.json
jq -r '
"\(.plan_name) [\(.verdict)] — \(.format_reading.consistency_pct)% consistent",
" \(.verdict_reason)",
"",
"FIELDS",
(.fields[] | " \(.name) <- \(.maps_to)"),
"",
"PATTERNS",
(.patterns[] | " \(.id) [\(.flavor) /\(.flags)] x\(.expected_matches) \(.purpose)\n \(.regex)"),
"",
"CLEANER RULES",
(.cleaner_rules[] | " \(.order). \(.action): \(.find) -> \(.replace)"),
"",
"CONFIDENCE CHECKS",
(.confidence_checks[] | " -\(.penalty) \(.check)"),
"",
"LLM STAGE",
" \(.llm_stage.role) on \(.llm_stage.model_tier), \(.llm_stage.expected_share) of records",
" \(.llm_stage.budget_note)",
"",
"PIPELINE",
(.pipeline[] | " \(.stage) [\(.tool)] \(.detail)"),
"",
"EDGE CASES",
(.edge_cases[] | " \(if .seen_in_sample then "seen" else "possible" end): \(.case)"),
"",
"COVERAGE",
(.coverage_check[] | " \(.id): \(if .addressed then "ok" else "SET ASIDE" end) - \(.note)"),
"",
"QUICK WINS",
(.quick_wins[] | " - \(.)")' \
plan.json
# gate a pipeline on the verdict: no model calls allowed in this corpus
jq -e '.verdict == "regex-first"' plan.json > /dev/null \
|| { echo "plan needs a model stage"; exit 1; }
import re, time
job_id = api("POST", "/run", payload,
**{"Idempotency-Key": "pp-001"})["job_id"]
while True:
job = api("GET", f"/jobs/{job_id}")
if job["status"] in ("succeeded", "failed"):
break
time.sleep(1.5)
if job["status"] == "failed":
raise RuntimeError(job.get("error", "run failed"))
raw = job["output"]
if isinstance(raw, dict) and "output" in raw:
raw = raw["output"]
plan = json.loads(raw) if isinstance(raw, str) else raw
fr = plan["format_reading"]
print(f'{plan["plan_name"]} [{plan["verdict"]}] — {fr["consistency_pct"]}% consistent')
print(" ", plan["verdict_reason"])
print(" record shape:", fr["record_shape"])
for n in fr["noise"]:
print(" noise:", n)
for a in fr["assumptions"]:
print(" assumption:", a)
for f in plan["fields"]:
print(f' {f["name"]} <- {f["maps_to"]} {f["notes"]}')
for p in plan["patterns"]:
print(f' {p["id"]} [{p["flavor"]} /{p["flags"]}] x{p["expected_matches"]} {p["purpose"]}')
print(" ", p["regex"])
for g in p["groups"]:
print(f' group {g["group"]} -> {g["field"]}')
for c in plan["cleaner_rules"]:
print(f' {c["order"]}. {c["action"]}: find={c["find"]!r} replace={c["replace"]!r}')
for c in plan["confidence_checks"]:
print(f' -{c["penalty"]} {c["check"]} ({c["reason"]})')
llm = plan["llm_stage"]
print(f' llm: {llm["role"]} on {llm["model_tier"]}, {llm["expected_share"]} of records')
print(" ", llm["budget_note"])
for s in plan["pipeline"]:
print(f' {s["stage"]} [{s["tool"]}] {s["detail"]}')
for e in plan["edge_cases"]:
print(f' {"seen" if e["seen_in_sample"] else "possible"}: {e["case"]} -> {e["handling"]}')
for c in plan["coverage_check"]:
print(f' {c["id"]}: {"ok" if c["addressed"] else "SET ASIDE"} - {c["note"]}')
for w in plan["quick_wins"]:
print(" win:", w)
print(plan["summary"])
with open("plan.json", "w", encoding="utf-8") as fh:
json.dump(plan, fh, indent=2)
with open("patterns.json", "w", encoding="utf-8") as fh:
json.dump(plan["patterns"], fh, indent=2)
# the claims are checkable: compile each pattern and count matches on the same sample
FLAGS = {"MULTILINE": re.M, "M": re.M, "DOTALL": re.S, "S": re.S,
"IGNORECASE": re.I, "I": re.I, "VERBOSE": re.X, "X": re.X}
for p in plan["patterns"]:
if p["flavor"] == "javascript":
continue
flags = 0
for name in filter(None, (p["flags"] or "").split("|")):
flags |= FLAGS.get(name.strip().upper(), 0)
actual = len(re.findall(p["regex"], SAMPLE, flags))
if actual != p["expected_matches"]:
print(f' MISMATCH {p["id"]}: claimed {p["expected_matches"]}, got {actual}')
import { writeFileSync } from "node:fs";
const { job_id } = await api("POST", "/run", payload,
{ "Idempotency-Key": crypto.randomUUID() });
let job;
do {
await new Promise((r) => setTimeout(r, 1500));
job = await api("GET", `/jobs/${job_id}`);
} while (job.status !== "succeeded" && job.status !== "failed");
if (job.status === "failed") throw new Error(job.error ?? "run failed");
const raw = job.output?.output ?? job.output;
const plan = typeof raw === "string" ? JSON.parse(raw) : raw;
const fr = plan.format_reading;
console.log(`${plan.plan_name} [${plan.verdict}] - ${fr.consistency_pct}% consistent`);
console.log(" ", plan.verdict_reason);
console.log(" record shape:", fr.record_shape);
for (const n of fr.noise) console.log(" noise:", n);
for (const a of fr.assumptions) console.log(" assumption:", a);
for (const f of plan.fields) console.log(` ${f.name} <- ${f.maps_to} ${f.notes}`);
for (const p of plan.patterns) {
console.log(` ${p.id} [${p.flavor} /${p.flags}] x${p.expected_matches} ${p.purpose}`);
console.log(" ", p.regex);
for (const g of p.groups) console.log(` group ${g.group} -> ${g.field}`);
}
for (const c of plan.cleaner_rules) {
console.log(` ${c.order}. ${c.action}: ${c.find} -> ${c.replace}`);
}
for (const c of plan.confidence_checks) console.log(` -${c.penalty} ${c.check}`);
console.log(` llm: ${plan.llm_stage.role} on ${plan.llm_stage.model_tier}, ` +
`${plan.llm_stage.expected_share} of records`);
for (const s of plan.pipeline) console.log(` ${s.stage} [${s.tool}] ${s.detail}`);
for (const e of plan.edge_cases) {
console.log(` ${e.seen_in_sample ? "seen" : "possible"}: ${e.case} -> ${e.handling}`);
}
for (const c of plan.coverage_check) {
console.log(` ${c.id}: ${c.addressed ? "ok" : "SET ASIDE"} - ${c.note}`);
}
for (const w of plan.quick_wins) console.log(" win:", w);
console.log(plan.summary);
writeFileSync("plan.json", JSON.stringify(plan, null, 2));
writeFileSync("patterns.json", JSON.stringify(plan.patterns, null, 2));
// the claims are checkable: compile each pattern and count matches on the same sample
let mismatches = 0;
for (const p of plan.patterns) {
if (p.flavor === "python") continue;
const flags = (p.flags || "").includes("g") ? p.flags : (p.flags || "") + "g";
const actual = [...sample.matchAll(new RegExp(p.regex, flags))].length;
if (actual !== p.expected_matches) {
mismatches++;
console.log(` MISMATCH ${p.id}: claimed ${p.expected_matches}, got ${actual}`);
}
}
if (mismatches) throw new Error(`${mismatches} pattern(s) disagree with the plan`);
var started struct{ JobID string `json:"job_id"` }
if err := call("POST", "/run", payload, &started); err != nil {
log.Fatal(err)
}
var job struct {
Status string `json:"status"`
Error string `json:"error"`
Output json.RawMessage `json:"output"`
}
for {
if err := call("GET", "/jobs/"+started.JobID, nil, &job); err != nil {
log.Fatal(err)
}
if job.Status == "succeeded" || job.Status == "failed" {
break
}
time.Sleep(1500 * time.Millisecond)
}
// job.Output is {"output": "<json string>"} — unwrap, then unmarshal:
type Plan struct {
PlanName string `json:"plan_name"`
Verdict string `json:"verdict"`
VerdictReason string `json:"verdict_reason"`
ExecSummary string `json:"exec_summary"`
FormatReading struct {
RecordShape string `json:"record_shape"`
ConsistencyPct int `json:"consistency_pct"`
Noise []string `json:"noise"`
Assumptions []string `json:"assumptions"`
} `json:"format_reading"`
Fields []struct {
Name, Notes string
MapsTo string `json:"maps_to"`
} `json:"fields"`
Patterns []struct {
ID, Purpose, Flavor, Regex, Flags, Notes string
Groups []struct{ Group, Field string } `json:"groups"`
ExpectedMatches int `json:"expected_matches"`
} `json:"patterns"`
CleanerRules []struct {
Order int
Action, Find, Replace, Why string
} `json:"cleaner_rules"`
ConfidenceChecks []struct {
Check, Reason string
Penalty float64
} `json:"confidence_checks"`
LLMStage struct {
Role string `json:"role"`
ModelTier string `json:"model_tier"`
PromptSketch string `json:"prompt_sketch"`
ExpectedShare string `json:"expected_share"`
BudgetNote string `json:"budget_note"`
} `json:"llm_stage"`
Pipeline []struct{ Stage, Tool, Detail string } `json:"pipeline"`
EdgeCases []struct {
Case, Handling string
SeenInSample bool `json:"seen_in_sample"`
} `json:"edge_cases"`
CoverageCheck []struct {
ID, Note string
Addressed bool
} `json:"coverage_check"`
QuickWins []string `json:"quick_wins"`
Summary string `json:"summary"`
}
var wrapper struct{ Output string `json:"output"` }
json.Unmarshal(job.Output, &wrapper)
var plan Plan
json.Unmarshal([]byte(wrapper.Output), &plan)
fmt.Printf("%s [%s] - %d%% consistent\n", plan.PlanName, plan.Verdict,
plan.FormatReading.ConsistencyPct)
fmt.Println(" ", plan.VerdictReason)
for _, f := range plan.Fields {
fmt.Printf(" %s <- %s\n", f.Name, f.MapsTo)
}
for _, p := range plan.Patterns {
fmt.Printf(" %s [%s /%s] x%d %s\n %s\n",
p.ID, p.Flavor, p.Flags, p.ExpectedMatches, p.Purpose, p.Regex)
}
for _, c := range plan.CleanerRules {
fmt.Printf(" %d. %s: %s -> %s\n", c.Order, c.Action, c.Find, c.Replace)
}
for _, s := range plan.Pipeline {
fmt.Printf(" %s [%s] %s\n", s.Stage, s.Tool, s.Detail)
}
fmt.Printf(" llm: %s on %s, %s of records\n",
plan.LLMStage.Role, plan.LLMStage.ModelTier, plan.LLMStage.ExpectedShare)
for _, c := range plan.CoverageCheck {
fmt.Printf(" %s addressed=%v %s\n", c.ID, c.Addressed, c.Note)
}
os.WriteFile("plan.json", []byte(wrapper.Output), 0o644)
patterns, _ := json.MarshalIndent(plan.Patterns, "", " ")
os.WriteFile("patterns.json", patterns, 0o644)
// Go's regexp is RE2 and rejects lookaround, so verify expected_matches in the
// runtime the patterns were written for (target "python" or "javascript").
String envelope = api("POST", "/run", jsonPayload);
String jobId = /* data.job_id via your JSON library */;
while (true) {
String job = api("GET", "/jobs/" + jobId, null);
String status = /* data.status */;
if (status.equals("succeeded") || status.equals("failed")) break;
Thread.sleep(1500);
}
// The reply is at data.output.output as a JSON string — parse it again, then read
// plan_name, verdict (regex-first | hybrid | llm-first), verdict_reason, exec_summary,
// format_reading (record_shape / consistency_pct / noise[] / assumptions[]),
// fields[] (name/maps_to/notes),
// patterns[] (id/purpose/flavor/regex/flags/groups[]/expected_matches/notes),
// cleaner_rules[] (order/action/find/replace/why),
// confidence_checks[] (check/penalty/reason),
// llm_stage (role/model_tier/prompt_sketch/expected_share/budget_note),
// pipeline[] (stage/tool/detail), edge_cases[] (case/seen_in_sample/handling),
// coverage_check[] (id/addressed/note), quick_wins[] and summary.
// Keep both on disk:
// Files.writeString(Path.of("plan.json"), planJson);
// Files.writeString(Path.of("patterns.json"), patternsJson);
// Then verify the claims — java.util.regex accepts most "python" and "both"
// flavor patterns:
// var m = Pattern.compile(regex, Pattern.MULTILINE).matcher(sample);
// int actual = 0; while (m.find()) actual++;
// if (actual != expectedMatches) System.out.println("MISMATCH " + id);
started = api("POST", "/run", payload)
job = nil
loop do
job = api("GET", "/jobs/#{started["job_id"]}")
break if %w[succeeded failed].include?(job["status"])
sleep 1.5
end
raise (job["error"] || "run failed") if job["status"] == "failed"
raw = job["output"].is_a?(Hash) ? job["output"].fetch("output", job["output"]) : job["output"]
plan = raw.is_a?(String) ? JSON.parse(raw) : raw
fr = plan["format_reading"]
puts "#{plan["plan_name"]} [#{plan["verdict"]}] - #{fr["consistency_pct"]}% consistent"
puts " #{plan["verdict_reason"]}"
puts " record shape: #{fr["record_shape"]}"
fr["noise"].each { |n| puts " noise: #{n}" }
fr["assumptions"].each { |a| puts " assumption: #{a}" }
plan["fields"].each { |f| puts " #{f["name"]} <- #{f["maps_to"]} #{f["notes"]}" }
plan["patterns"].each do |p|
puts " #{p["id"]} [#{p["flavor"]} /#{p["flags"]}] x#{p["expected_matches"]} #{p["purpose"]}"
puts " #{p["regex"]}"
p["groups"].each { |g| puts " group #{g["group"]} -> #{g["field"]}" }
end
plan["cleaner_rules"].each { |c| puts " #{c["order"]}. #{c["action"]}: #{c["find"]} -> #{c["replace"]}" }
plan["confidence_checks"].each { |c| puts " -#{c["penalty"]} #{c["check"]}" }
llm = plan["llm_stage"]
puts " llm: #{llm["role"]} on #{llm["model_tier"]}, #{llm["expected_share"]} of records"
plan["pipeline"].each { |s| puts " #{s["stage"]} [#{s["tool"]}] #{s["detail"]}" }
plan["edge_cases"].each { |e| puts " #{e["seen_in_sample"] ? "seen" : "possible"}: #{e["case"]}" }
plan["coverage_check"].each { |c| puts " #{c["id"]}: #{c["addressed"] ? "ok" : "SET ASIDE"}" }
plan["quick_wins"].each { |w| puts " win: #{w}" }
puts plan["summary"]
File.write("plan.json", JSON.pretty_generate(plan))
File.write("patterns.json", JSON.pretty_generate(plan["patterns"]))
# Onigmo accepts most "python" and "both" flavor patterns — count matches yourself:
plan["patterns"].each do |p|
next if p["flavor"] == "javascript"
actual = SAMPLE.scan(Regexp.new(p["regex"], Regexp::MULTILINE)).length
warn " MISMATCH #{p["id"]}: claimed #{p["expected_matches"]}, got #{actual}" if actual != p["expected_matches"]
end
$started = api("POST", "/run", $payload);
do {
sleep(2);
$job = api("GET", "/jobs/" . $started["job_id"]);
} while (!in_array($job["status"], ["succeeded", "failed"], true));
if ($job["status"] === "failed") {
throw new Exception($job["error"] ?? "run failed");
}
$raw = is_array($job["output"]) ? ($job["output"]["output"] ?? $job["output"]) : $job["output"];
$plan = is_string($raw) ? json_decode($raw, true) : $raw;
$fr = $plan["format_reading"];
echo "{$plan['plan_name']} [{$plan['verdict']}] - {$fr['consistency_pct']}% consistent\n";
echo " {$plan['verdict_reason']}\n";
echo " record shape: {$fr['record_shape']}\n";
foreach ($fr["noise"] as $n) { echo " noise: $n\n"; }
foreach ($fr["assumptions"] as $a) { echo " assumption: $a\n"; }
foreach ($plan["fields"] as $f) {
echo " {$f['name']} <- {$f['maps_to']} {$f['notes']}\n";
}
foreach ($plan["patterns"] as $p) {
echo " {$p['id']} [{$p['flavor']} /{$p['flags']}] x{$p['expected_matches']} {$p['purpose']}\n";
echo " {$p['regex']}\n";
foreach ($p["groups"] as $g) { echo " group {$g['group']} -> {$g['field']}\n"; }
}
foreach ($plan["cleaner_rules"] as $c) {
echo " {$c['order']}. {$c['action']}: {$c['find']} -> {$c['replace']}\n";
}
foreach ($plan["confidence_checks"] as $c) { echo " -{$c['penalty']} {$c['check']}\n"; }
echo " llm: {$plan['llm_stage']['role']} on {$plan['llm_stage']['model_tier']}, " .
"{$plan['llm_stage']['expected_share']} of records\n";
foreach ($plan["pipeline"] as $s) { echo " {$s['stage']} [{$s['tool']}] {$s['detail']}\n"; }
foreach ($plan["edge_cases"] as $e) {
echo " " . ($e["seen_in_sample"] ? "seen" : "possible") . ": {$e['case']}\n";
}
foreach ($plan["coverage_check"] as $c) {
echo " {$c['id']}: " . ($c["addressed"] ? "ok" : "SET ASIDE") . " - {$c['note']}\n";
}
foreach ($plan["quick_wins"] as $w) { echo " win: $w\n"; }
echo $plan["summary"], "\n";
file_put_contents("plan.json", json_encode($plan, JSON_PRETTY_PRINT));
file_put_contents("patterns.json", json_encode($plan["patterns"], JSON_PRETTY_PRINT));
// PCRE accepts most "python" and "both" flavor patterns — count matches yourself:
foreach ($plan["patterns"] as $p) {
if ($p["flavor"] === "javascript") { continue; }
$n = preg_match_all("/" . str_replace("/", "\\/", $p["regex"]) . "/m", $sample);
if ($n !== $p["expected_matches"]) {
echo " MISMATCH {$p['id']}: claimed {$p['expected_matches']}, got $n\n";
}
}
using System.Text.RegularExpressions;
var started = await SkillSafe.ApiAsync(HttpMethod.Post, "/run", payload);
var jobId = started.GetProperty("job_id").GetString();
JsonElement job;
while (true)
{
job = await SkillSafe.ApiAsync(HttpMethod.Get, $"/jobs/{jobId}");
var status = job.GetProperty("status").GetString();
if (status is "succeeded" or "failed") break;
await Task.Delay(1500);
}
var rawText = job.GetProperty("output").GetProperty("output").GetString();
using var doc = JsonDocument.Parse(rawText!);
var plan = doc.RootElement;
var fr = plan.GetProperty("format_reading");
Console.WriteLine($"{plan.GetProperty("plan_name")} [{plan.GetProperty("verdict")}] - " +
$"{fr.GetProperty("consistency_pct")}% consistent");
Console.WriteLine($" {plan.GetProperty("verdict_reason")}");
Console.WriteLine($" record shape: {fr.GetProperty("record_shape")}");
foreach (var f in plan.GetProperty("fields").EnumerateArray())
{
Console.WriteLine($" {f.GetProperty("name")} <- {f.GetProperty("maps_to")}");
}
foreach (var p in plan.GetProperty("patterns").EnumerateArray())
{
Console.WriteLine($" {p.GetProperty("id")} [{p.GetProperty("flavor")} " +
$"/{p.GetProperty("flags")}] x{p.GetProperty("expected_matches")} " +
$"{p.GetProperty("purpose")}");
Console.WriteLine($" {p.GetProperty("regex")}");
}
foreach (var s in plan.GetProperty("pipeline").EnumerateArray())
{
Console.WriteLine($" {s.GetProperty("stage")} [{s.GetProperty("tool")}] {s.GetProperty("detail")}");
}
var llm = plan.GetProperty("llm_stage");
Console.WriteLine($" llm: {llm.GetProperty("role")} on {llm.GetProperty("model_tier")}, " +
$"{llm.GetProperty("expected_share")} of records");
foreach (var c in plan.GetProperty("coverage_check").EnumerateArray())
{
Console.WriteLine($" {c.GetProperty("id")}: {c.GetProperty("addressed")} - {c.GetProperty("note")}");
}
await File.WriteAllTextAsync("plan.json", rawText!);
await File.WriteAllTextAsync("patterns.json", plan.GetProperty("patterns").GetRawText());
// .NET regex accepts most "python" and "both" flavor patterns — check the claims:
foreach (var p in plan.GetProperty("patterns").EnumerateArray())
{
if (p.GetProperty("flavor").GetString() == "javascript") continue;
var actual = Regex.Matches(sample, p.GetProperty("regex").GetString()!,
RegexOptions.Multiline).Count;
var claimed = p.GetProperty("expected_matches").GetInt32();
if (actual != claimed)
Console.WriteLine($" MISMATCH {p.GetProperty("id")}: claimed {claimed}, got {actual}");
}
The model is asked for one JSON object and nothing else, but a stray code fence or preamble
is always possible. Strip a leading ```json fence, take the text between the
first { and the last }, and only then parse — that is what
the app does before it falls back to a retry_note reformat run.
The reply object — output schema
One JSON object, always the same shape. Every array and every key is present — empty
states are explicit ("noise": [], "cleaner_rules": []), never
omitted. The whole plan is grounded in the sample you sent: every pattern, every
expected_matches count and every edge case marked
seen_in_sample: true comes from that text alone, nothing is executed, and no
corpus beyond the sample is seen. Where the sample is too small or too truncated to settle
consistency, that is stated in format_reading.assumptions rather than guessed
confidently.
| Field | Type | Meaning |
|---|---|---|
plan_name | string | A short name for this parse plan, taken from what the text actually is — e.g. Quiz bank extraction plan. |
verdict | string | Exactly one of regex-first, hybrid, llm-first. See the table below. This is the field to gate a pipeline on. |
verdict_reason | string | Two to four sentences: the consistency evidence, what it implies, and what evidence would flip the verdict. |
exec_summary | string | One or two short paragraphs a tech lead reads: what the text is, what gets extracted, how, and roughly what it costs in model calls (possibly none). |
format_reading | object | {record_shape, consistency_pct, noise[], assumptions[]} — what one record looks like and how records are delimited; the model's own integer 0–100 reading of how much of the sample follows one shape (it may disagree with a prescan estimate, and says so in the matching coverage_check note); the non-content material present (page numbers, running headers, OCR junk), empty when clean; and anything assumed because the sample could not settle it. Read assumptions first — a wrong one invalidates the patterns built on it. |
fields | array | {name, maps_to, notes} — one entry for every field the goal asked for, at least one. maps_to says where in the record it comes from, or is literally not present in sample when the text cannot supply it, with the reason in notes. |
patterns | array | The patterns to actually run — ids RX-001, RX-002, … in sequence. Non-empty unless verdict is llm-first (where it may be empty or hold only pre-splitting helpers). Columns are listed below. |
cleaner_rules | array | {order, action, find, replace, why} — ordered, apply lowest order first. find is a regex or literal and replace may be the empty string; why says why this noise must go before parsing. Empty array when the text is clean. |
confidence_checks | array | {check, penalty, reason} — mechanical, cheap tests on one parsed record (field count, answer key in range, date parses, amount numeric, choice letters unique). Penalties sum against a 1.0 score; in hybrid plans a record scoring under about 0.95 goes to the LLM stage. |
llm_stage | object | {role, model_tier, prompt_sketch, expected_share, budget_note} — role is none, edge-cases or primary; model_tier is none when the role is none, else cheapest or balanced with a word on why; prompt_sketch is the essence of the validation or extraction prompt (empty string when the role is none) and validates one flagged record against the raw text it came from, never the whole corpus; expected_share is roughly what fraction of records reach a model ("0%", "~2%", "all"); budget_note ties share and tier to cost. |
pipeline | array | {stage, tool, detail} — the stages in order, e.g. "1. Clean"; tool is regex, code or llm. |
edge_cases | array | {case, seen_in_sample, handling} — the specific ways records break, whether that break is actually present in the sample (a boolean, not a hedge), and how the plan handles it. |
coverage_check | array | {id, addressed, note} — one entry per prescan_facts.flags id you sent, each appearing exactly once. See the semantics below. Empty when you sent prescan_facts: null. |
quick_wins | string[] | Upstream changes that would make parsing simpler — ask the exporter for TSV, keep the header row, stop paginating the dump. May be empty. |
summary | string | Two or three closing sentences: the verdict, the coverage to expect, the next step. |
The three verdict values, and the rules behind them:
| verdict | What it means |
|---|---|
regex-first | The dominant record pattern covers roughly 90% or more of the sample and the target fields sit inside it. The plan may still carry confidence checks, but llm_stage.role is none or edge-cases with an expected share under about 5%. "No model needed" is a success here, not a failure of imagination — if regex covers 95%+ of records, ship it without a model. |
hybrid | A clear repeating structure exists, but a real fraction of records (roughly 5–40%) break it: wrapped lines, OCR damage, optional fields, several formats mixed together. Regex extracts the majority; the confidence checks route the remainder to llm_stage.role = "edge-cases" with a threshold near 0.95. |
llm-first | No reliable repeating structure, or the wanted fields are semantic judgements (intent, sentiment, topic) rather than positional captures. llm_stage.role is primary, and the plan must contain cost containment: batching, the cheapest capable tier, caching, and any mechanical pre-extraction that shrinks the prompt. |
When the sample honestly sits on a boundary, the cheaper verdict is chosen and
verdict_reason names the evidence that would flip it.
Each entry in patterns:
| Column | Meaning |
|---|---|
id | Sequential RX-001, RX-002, … — the stable handle to reference from your own code and tests. |
purpose | What this pattern extracts. |
flavor | both | python | javascript — which dialect the regex string is written for, following the target you sent. both means it avoids anything the two disagree on. |
regex | The pattern body only — no surrounding slashes, and correctly escaped for a JSON string (a digit class arrives as \\d on the wire and is \d once parsed). It compiles: valid for re.compile for the Python flavor, valid for new RegExp(pattern, flags) for the JavaScript flavor. |
flags | The flags in that flavor's spelling — gm for JavaScript, MULTILINE (joined with | when there are several) for Python. |
groups | {group, field} — each capture group, by number or name, mapped to the field it fills. |
expected_matches | How many matches this pattern produces on the visible sample — counted, not estimated. The app compiles every pattern and runs it against the same sample in the browser, showing any disagreement next to this number; do the same in your own tests, as the run samples above do. |
notes | The limits of the pattern, or an empty string. |
coverage_check semantics:
| Case | What you get |
|---|---|
| Every flag id you sent | Each prescan_facts.flags id appears in coverage_check exactly once. Nothing you flagged is silently dropped, which makes this the field to assert on in a CI check. The prescan_facts.stats numbers are not reconciled here — they inform format_reading instead. |
addressed: true | The flag shaped the plan; note says how — the cleaner rule that removes that noise, the pattern that hangs off that record marker, the assumption a low consistency reading forced. |
addressed: false | The flag was deliberately set aside; note gives the reason — a mechanical check that fired but does not matter for this text (ruling lines that fall outside every capture, a delimiter guess the sample contradicts once meaning is read). |
| Nothing sent | Send prescan_facts: null, or omit the field, and coverage_check comes back empty. The rest of the reply is unaffected. |
A small, realistic result for the quiz sample above (long strings wrapped for readability):
{
"plan_name": "Quiz bank extraction plan",
"verdict": "regex-first",
"verdict_reason": "All three visible records use the same three-line shape: a numbered
question line, one choices line with four lettered options, and an
Answer line. The wanted fields are all positional and no judgement is
required, so regex covers this without a model. A wrapped question
line or a five-choice item in the unseen remainder would push this
to hybrid.",
"exec_summary": "This is a three-field-per-record quiz bank: question number and text,
four lettered choices, and an answer key letter. Two patterns and a
split get all of it, at zero model cost.
Budget: no model calls at any corpus size. Spend the effort on the
confidence checks instead, so a malformed item is reported rather than
silently dropped.",
"format_reading": {
"record_shape": "Three lines per record: 'N. question text', then a choices line with
'A) ... B) ... C) ... D) ...', then 'Answer: X'. Records are
separated by one blank line.",
"consistency_pct": 100,
"noise": [],
"assumptions": [
"All items have exactly four choices, A-D; the visible sample never varies.",
"Question text never wraps onto a second line — only three records were visible."
]
},
"fields": [
{ "name": "number", "maps_to": "leading integer of the question line",
"notes": "integer" },
{ "name": "question", "maps_to": "rest of the question line after 'N. '",
"notes": "string, trim trailing whitespace" },
{ "name": "choices", "maps_to": "the A)-D) segments of the choices line",
"notes": "list of four strings, split with RX-002" },
{ "name": "answer", "maps_to": "letter after 'Answer: '",
"notes": "single character A-D; check it appears among the choices" }
],
"patterns": [
{
"id": "RX-001",
"purpose": "One whole record: number, question text, choices line, answer letter",
"flavor": "both",
"regex": "^(\\d+)\\.\\s+(.+)\\n\\s+(A\\).+)\\n\\s+Answer:\\s*([A-D])\\s*$",
"flags": "gm",
"groups": [
{ "group": "1", "field": "number" },
{ "group": "2", "field": "question" },
{ "group": "3", "field": "choices (raw line, split with RX-002)" },
{ "group": "4", "field": "answer" }
],
"expected_matches": 3,
"notes": "Anchored per line, so a wrapped question line will not match — those
records fall out and must be counted, not ignored."
},
{
"id": "RX-002",
"purpose": "Split one choices line into its four lettered options",
"flavor": "both",
"regex": "([A-D])\\)\\s*(.+?)(?=\\s{2,}[A-D]\\)|$)",
"flags": "gm",
"groups": [
{ "group": "1", "field": "choice letter" },
{ "group": "2", "field": "choice text" }
],
"expected_matches": 12,
"notes": "Four matches per record across three records. A choice whose own text
contains two spaces followed by a capital letter and ')' would split
early."
}
],
"cleaner_rules": [],
"confidence_checks": [
{ "check": "RX-002 yields exactly four choices for the record", "penalty": 0.4,
"reason": "A different count means the item is not the assumed four-option shape." },
{ "check": "the answer letter is among the parsed choice letters", "penalty": 0.5,
"reason": "An answer key pointing at a choice that was not captured means the
choices line was mis-split." },
{ "check": "question text is non-empty after trimming", "penalty": 0.3,
"reason": "An empty question usually means the line wrapped and group 2 caught
only the tail." }
],
"llm_stage": {
"role": "none",
"model_tier": "none",
"prompt_sketch": "",
"expected_share": "0%",
"budget_note": "No model calls at all: at full pattern coverage on the visible
sample, records that fail the confidence checks belong in a reject
file for a human, not in a prompt."
},
"pipeline": [
{ "stage": "1. Split", "tool": "code",
"detail": "Split the file on blank lines into candidate records." },
{ "stage": "2. Extract", "tool": "regex",
"detail": "Apply RX-001 per record, then RX-002 to the captured choices line." },
{ "stage": "3. Score", "tool": "code",
"detail": "Run the three confidence checks; a record under 0.95 goes to
rejects.txt with its raw text." },
{ "stage": "4. Emit", "tool": "code",
"detail": "Write one JSON object per record: number, question, choices[], answer." }
],
"edge_cases": [
{ "case": "Question text wrapping onto a second line", "seen_in_sample": false,
"handling": "RX-001 will not match; the record lands in rejects.txt where the
count is visible instead of silently lost." },
{ "case": "An item with more or fewer than four choices", "seen_in_sample": false,
"handling": "RX-002's match count drives the four-choice confidence check." }
],
"coverage_check": [
{ "id": "records:numbered", "addressed": true,
"note": "RX-001 anchors on the numbered question line." },
{ "id": "consistency:high", "addressed": true,
"note": "Agreed — read as 100% on the visible sample, which is what makes this
regex-first." }
],
"quick_wins": [
"If the source system can export the bank as TSV with a header row, all of this
collapses to a split plus the answer-key check."
],
"summary": "Regex-first with two patterns and no model stage: RX-001 per record and
RX-002 for the choices. Expect full coverage on text shaped like the
sample, with the confidence checks turning any deviation into a visible
reject rather than a silent drop. Next step: run the patterns over the full
corpus and read rejects.txt before trusting the counts. …"
}
This is an AI-generated plan from a pasted sample, not a tested pipeline: it sees only the
text you sent, never your full corpus. Check format_reading.assumptions, then
verify the claims mechanically — compile every pattern, count the matches, and compare
against expected_matches, exactly as the run samples above do and as the app's
"Test all on my sample" button does in the browser. Counts stated for a truncated sample are
counts over the visible slice only.
Step 5 — Stream the plan as it is written
/run-stream takes exactly the same body as /run but answers with
server-sent events, so you can show progress instead of a spinner — useful here because
the patterns, cleaner rules and pipeline make for a long reply. This app's own progress panel
is this endpoint. Events are separated by a blank line; each has an event: line
and a data: line carrying JSON.
| Event | Payload | Meaning |
|---|---|---|
job | {job_id, status} | Sent once, when the job is accepted — show "starting". |
delta | {text} | A chunk of the reply, in order. Append it; the accumulated length is your only progress signal (the total is not known in advance). The app advances its step list by watching for the "verdict", "format_reading", "fields", "patterns", "cleaner_rules", "confidence_checks", "llm_stage", "pipeline", "edge_cases", "coverage_check" and "summary" keys as they arrive. |
done | {job_id, status, charged_credits, output} | The final, authoritative result — read the plan from output.output rather than trusting concatenated deltas, and the settled price from charged_credits. |
error | {code, message} | Replaces done when the run fails. |
# -N disables buffering so events print as they arrive
curl -N -s -X POST "$API/run-stream" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-H "Idempotency-Key: pp-$(date +%s)" \
-d @input.json
# event: job
# data: {"job_id":"job_...","status":"running"}
#
# event: delta
# data: {"text":"{\"plan_name\":\"Quiz bank"}
# ...
# event: done
# data: {"job_id":"job_...","status":"succeeded","charged_credits":418,"output":{"output":"{...}"}}
import json, requests
result = None
with requests.post(
API + "/run-stream",
headers={"Authorization": f"Bearer {TOKEN}",
"Idempotency-Key": "pp-001"},
json=payload,
stream=True,
) as r:
r.raise_for_status()
event = None
for line in r.iter_lines(decode_unicode=True):
if not line:
continue
if line.startswith("event:"):
event = line[len("event:"):].strip()
elif line.startswith("data:"):
data = json.loads(line[len("data:"):].strip())
if event == "delta":
print(".", end="", flush=True) # live progress
elif event == "done":
result = data
elif event == "error":
raise RuntimeError(data.get("message", "run failed"))
plan = json.loads(result["output"]["output"]) # authoritative
print("\ncharged:", result["charged_credits"], "-", plan["plan_name"])
print("verdict:", plan["verdict"], "-", plan["llm_stage"]["expected_share"], "to the model")
for p in plan["patterns"]:
print(f' {p["id"]} x{p["expected_matches"]} {p["regex"]}')
with open("plan.json", "w", encoding="utf-8") as fh:
json.dump(plan, fh, indent=2)
with open("patterns.json", "w", encoding="utf-8") as fh:
json.dump(plan["patterns"], fh, indent=2)
const res = await fetch(API + "/run-stream", {
method: "POST",
headers: {
Authorization: `Bearer ${TOKEN}`,
"Content-Type": "application/json",
"Idempotency-Key": crypto.randomUUID(),
},
body: JSON.stringify(payload),
});
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buf = "", done = null, deltas = 0;
for (;;) {
const chunk = await reader.read();
if (chunk.done) break;
buf += decoder.decode(chunk.value, { stream: true });
const frames = buf.split("\n\n");
buf = frames.pop();
for (const frame of frames) {
const name = /^event:\s*(.+)$/m.exec(frame)?.[1];
const body = /^data:\s*(.+)$/m.exec(frame)?.[1];
if (!name || !body) continue;
const data = JSON.parse(body);
if (name === "delta") deltas++; // live progress
if (name === "done") done = data;
if (name === "error") throw new Error(data.message ?? "run failed");
}
}
const plan = JSON.parse(done.output.output); // authoritative
console.log(`${deltas} chunks, ${done.charged_credits} credits - ` +
`${plan.plan_name} [${plan.verdict}]`);
for (const p of plan.patterns) console.log(` ${p.id} x${p.expected_matches} ${p.regex}`);
writeFileSync("plan.json", JSON.stringify(plan, null, 2));
writeFileSync("patterns.json", JSON.stringify(plan.patterns, null, 2));
body, _ := json.Marshal(payload)
req, _ := http.NewRequest("POST", API+"/run-stream", bytes.NewReader(body))
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", "application/json")
req.Header.Set("Idempotency-Key", "pp-001")
res, err := http.DefaultClient.Do(req)
if err != nil {
log.Fatal(err)
}
defer res.Body.Close()
var event string
var final map[string]any
sc := bufio.NewScanner(res.Body)
sc.Buffer(make([]byte, 0, 64*1024), 4*1024*1024)
for sc.Scan() {
line := sc.Text()
switch {
case strings.HasPrefix(line, "event:"):
event = strings.TrimSpace(strings.TrimPrefix(line, "event:"))
case strings.HasPrefix(line, "data:"):
var data map[string]any
json.Unmarshal([]byte(strings.TrimPrefix(line, "data:")), &data)
switch event {
case "delta":
fmt.Print(".") // live progress
case "done":
final = data
case "error":
log.Fatal(data["message"])
}
}
}
// final["output"].(map[string]any)["output"].(string) is the reply JSON —
// unmarshal it into the Plan struct from step 4, then write it to plan.json
// and plan.Patterns to patterns.json.
// Java 17+ — read the stream line by line instead of buffering the body.
var req = HttpRequest.newBuilder(URI.create(API + "/run-stream"))
.header("Authorization", "Bearer " + TOKEN)
.header("Content-Type", "application/json")
.header("Idempotency-Key", "pp-001")
.POST(HttpRequest.BodyPublishers.ofString(jsonPayload))
.build();
var res = HTTP.send(req, HttpResponse.BodyHandlers.ofLines());
String event = null, done = null;
for (String line : (Iterable<String>) res.body()::iterator) {
if (line.startsWith("event:")) {
event = line.substring(6).trim();
} else if (line.startsWith("data:")) {
String data = line.substring(5).trim();
if ("delta".equals(event)) System.out.print("."); // live progress
else if ("done".equals(event)) done = data;
else if ("error".equals(event)) throw new RuntimeException(data);
}
}
// parse `done`, then parse data.output.output again — it is a JSON string holding
// plan_name, verdict, verdict_reason, exec_summary, format_reading, fields[],
// patterns[], cleaner_rules[], confidence_checks[], llm_stage, pipeline[],
// edge_cases[], coverage_check[], quick_wins[] and summary.
require "net/http"
require "json"
uri = URI(API + "/run-stream")
req = Net::HTTP::Post.new(uri)
req["Authorization"] = "Bearer #{TOKEN}"
req["Content-Type"] = "application/json"
req["Idempotency-Key"] = "pp-001"
req.body = payload.to_json
event = nil
done = nil
Net::HTTP.start(uri.host, uri.port, use_ssl: true) do |http|
http.request(req) do |res|
res.read_body do |chunk|
chunk.each_line do |line|
line = line.strip
if line.start_with?("event:")
event = line.delete_prefix("event:").strip
elsif line.start_with?("data:")
data = JSON.parse(line.delete_prefix("data:").strip)
case event
when "delta" then print "." # live progress
when "done" then done = data
when "error" then raise (data["message"] || "run failed")
end
end
end
end
end
end
plan = JSON.parse(done["output"]["output"])
puts "\n#{done["charged_credits"]} credits - #{plan["plan_name"]} [#{plan["verdict"]}]"
plan["patterns"].each { |p| puts " #{p["id"]} x#{p["expected_matches"]} #{p["regex"]}" }
File.write("plan.json", JSON.pretty_generate(plan))
File.write("patterns.json", JSON.pretty_generate(plan["patterns"]))
$event = null;
$done = null;
$ch = curl_init(API . "/run-stream");
curl_setopt_array($ch, [
CURLOPT_POST => true,
CURLOPT_HTTPHEADER => [
"Authorization: Bearer $TOKEN",
"Content-Type: application/json",
"Idempotency-Key: pp-001",
],
CURLOPT_POSTFIELDS => json_encode($payload),
CURLOPT_WRITEFUNCTION => function ($ch, $chunk) use (&$event, &$done) {
foreach (explode("\n", $chunk) as $line) {
$line = trim($line);
if (str_starts_with($line, "event:")) {
$event = trim(substr($line, 6));
} elseif (str_starts_with($line, "data:")) {
$data = json_decode(trim(substr($line, 5)), true);
if ($event === "delta") { echo "."; } // live progress
elseif ($event === "done") { $done = $data; }
elseif ($event === "error") { throw new Exception($data["message"] ?? "run failed"); }
}
}
return strlen($chunk);
},
]);
curl_exec($ch);
curl_close($ch);
$plan = json_decode($done["output"]["output"], true);
echo "\n{$done['charged_credits']} credits - {$plan['plan_name']} [{$plan['verdict']}]\n";
foreach ($plan["patterns"] as $p) {
echo " {$p['id']} x{$p['expected_matches']} {$p['regex']}\n";
}
file_put_contents("plan.json", json_encode($plan, JSON_PRETTY_PRINT));
file_put_contents("patterns.json", json_encode($plan["patterns"], JSON_PRETTY_PRINT));
var req = new HttpRequestMessage(HttpMethod.Post, Api + "/run-stream") {
Content = JsonContent.Create(payload),
};
req.Headers.Add("Idempotency-Key", "pp-001");
using var res = await Http.SendAsync(req, HttpCompletionOption.ResponseHeadersRead);
using var reader = new StreamReader(await res.Content.ReadAsStreamAsync());
string? evt = null, done = null;
while (await reader.ReadLineAsync() is { } line)
{
if (line.StartsWith("event:")) evt = line[6..].Trim();
else if (line.StartsWith("data:"))
{
var data = line[5..].Trim();
if (evt == "delta") Console.Write("."); // live progress
else if (evt == "done") done = data;
else if (evt == "error") throw new Exception(data);
}
}
using var final = JsonDocument.Parse(done!);
var text = final.RootElement.GetProperty("output").GetProperty("output").GetString();
using var planDoc = JsonDocument.Parse(text!);
var plan = planDoc.RootElement;
Console.WriteLine($"{plan.GetProperty("plan_name")} [{plan.GetProperty("verdict")}]");
foreach (var p in plan.GetProperty("patterns").EnumerateArray())
Console.WriteLine($" {p.GetProperty("id")} x{p.GetProperty("expected_matches")} " +
$"{p.GetProperty("regex")}");
await File.WriteAllTextAsync("plan.json", text!);
await File.WriteAllTextAsync("patterns.json", plan.GetProperty("patterns").GetRawText());
In a browser, the native EventSource only speaks GET, and this endpoint is a
POST — read the fetch response body incrementally, as the JavaScript
sample above does. On an idempotent replay the server may answer with a plain JSON
envelope instead of an event stream; check the Content-Type before you start
parsing frames.