Review your PyTorch code from your own scripts
Send PyTorch source — one file or several, each preceded by a # file: train.py comment —
and get back one JSON object: a training-trustworthiness posture, the inventory of every
module, loop, optimizer, loss and DataLoader with its role, prioritized findings across
correctness, performance, memory, reproducibility, data and hygiene, each with a corrected
Python fragment, quick wins, and the focus areas to work through first. Everything this app
does goes through the SkillSafe App API — plain JSON over HTTPS — so you can hang
a review off any pull request that touches the training code. Wire it into whatever produces
or reviews your models: a pre-merge check on train/, a scheduled audit of the
training repo, or an editor command. Pick a language once and the whole page follows.
Basics
Base URL: https://api.skillsafe.ai/v1/app-api, app slug
torch-clinic. Every request sends
Authorization: Bearer <token> and JSON bodies with
Content-Type: application/json. Responses are wrapped in an envelope:
{"data": …} on success, {"error": {"code", "message"}} on failure.
The review itself is produced by the gpt-terra model. Estimates are free;
runs are metered against your credit balance. There is a single run task — one bundle of
PyTorch code in, one review out, no follow-up calls and no session state to carry.
| Status | Meaning |
|---|---|
401 | Missing or expired token — create a new session. |
402 | Not enough credits — top up at skillsafe.ai/account/credits. |
403 | The token isn't allowed to do this (e.g. a guest reviewing a very large script). |
404 | Unknown job or record id. |
5xx | Transient platform error — retry with backoff. |
Browsers enforce CORS for this API, so run these examples from a server, script or terminal — not from another website's frontend.
Step 0 — A tiny client
Every task below is a single HTTP call, so start with a short helper that adds the auth
header, sends JSON and unwraps the data envelope. The later steps reuse it.
export API="https://api.skillsafe.ai/v1/app-api"
export TOKEN="YOUR_TOKEN" # see step 1
# every call looks like:
# curl -s "$API/..." -H "Authorization: Bearer $TOKEN" [-d '{json}']
# jq is used below to pull fields out of the {"data": ...} envelope
import json, requests
API = "https://api.skillsafe.ai/v1/app-api"
TOKEN = "YOUR_TOKEN" # see step 1 — read it from your shell environment in real code
def api(method, path, body=None, **headers):
res = requests.request(method, API + path, json=body,
headers={"Authorization": f"Bearer {TOKEN}", **headers})
payload = res.json()
if not res.ok:
raise RuntimeError(payload.get("error", {}).get("message", res.reason))
return payload["data"]
// Node 18+ (built-in fetch)
const API = "https://api.skillsafe.ai/v1/app-api";
const TOKEN = "YOUR_TOKEN"; // see step 1 — read it from your shell environment in real code
async function api(method, path, body, extraHeaders = {}) {
const res = await fetch(API + path, {
method,
headers: { Authorization: `Bearer ${TOKEN}`, "Content-Type": "application/json", ...extraHeaders },
body: body === undefined ? undefined : JSON.stringify(body),
});
const json = await res.json();
if (!res.ok) throw new Error(json.error?.message ?? res.statusText);
return json.data;
}
package main
import (
"bytes"
"encoding/json"
"fmt"
"net/http"
"os"
)
const API = "https://api.skillsafe.ai/v1/app-api"
var token = os.Getenv("SKILLSAFE_TOKEN") // see step 1
func call(method, path string, body, out any) error {
var buf bytes.Buffer
if body != nil {
json.NewEncoder(&buf).Encode(body)
}
req, _ := http.NewRequest(method, API+path, &buf)
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", "application/json")
res, err := http.DefaultClient.Do(req)
if err != nil {
return err
}
defer res.Body.Close()
var env struct {
Data json.RawMessage `json:"data"`
Error *struct{ Message string `json:"message"` } `json:"error"`
}
json.NewDecoder(res.Body).Decode(&env)
if res.StatusCode >= 400 {
return fmt.Errorf("api %s %s: %s", method, path, env.Error.Message)
}
if out == nil {
return nil
}
return json.Unmarshal(env.Data, out)
}
// Java 17+, no dependencies. Pair with your JSON library (Jackson, Gson…)
// to read fields out of the returned envelope.
import java.net.URI;
import java.net.http.HttpClient;
import java.net.http.HttpRequest;
import java.net.http.HttpResponse;
public class SkillSafe {
static final String API = "https://api.skillsafe.ai/v1/app-api";
static final String TOKEN = System.getenv("SKILLSAFE_TOKEN"); // see step 1
static final HttpClient HTTP = HttpClient.newHttpClient();
static String api(String method, String path, String jsonBody) throws Exception {
var req = HttpRequest.newBuilder(URI.create(API + path))
.header("Authorization", "Bearer " + TOKEN)
.header("Content-Type", "application/json")
.method(method, jsonBody == null
? HttpRequest.BodyPublishers.noBody()
: HttpRequest.BodyPublishers.ofString(jsonBody))
.build();
var res = HTTP.send(req, HttpResponse.BodyHandlers.ofString());
if (res.statusCode() >= 400) throw new RuntimeException(res.body());
return res.body(); // envelope: {"data": …}
}
}
require "net/http"
require "json"
API = "https://api.skillsafe.ai/v1/app-api"
TOKEN = ENV.fetch("SKILLSAFE_TOKEN") # see step 1
def api(method, path, body = nil)
uri = URI(API + path)
req = Net::HTTP.const_get(method.capitalize).new(uri)
req["Authorization"] = "Bearer #{TOKEN}"
req["Content-Type"] = "application/json"
req.body = body.to_json if body
res = Net::HTTP.start(uri.host, uri.port, use_ssl: true) { |h| h.request(req) }
payload = JSON.parse(res.body)
raise (payload.dig("error", "message") || res.message) unless res.is_a?(Net::HTTPSuccess)
payload["data"]
end
<?php
const API = "https://api.skillsafe.ai/v1/app-api";
$TOKEN = getenv("SKILLSAFE_TOKEN"); // see step 1
function api(string $method, string $path, ?array $body = null): mixed {
global $TOKEN;
$ch = curl_init(API . $path);
curl_setopt_array($ch, [
CURLOPT_CUSTOMREQUEST => $method,
CURLOPT_RETURNTRANSFER => true,
CURLOPT_HTTPHEADER => [
"Authorization: Bearer $TOKEN",
"Content-Type: application/json",
],
CURLOPT_POSTFIELDS => $body === null ? null : json_encode($body),
]);
$payload = json_decode(curl_exec($ch), true);
$status = curl_getinfo($ch, CURLINFO_RESPONSE_CODE);
curl_close($ch);
if ($status >= 400) {
throw new Exception($payload["error"]["message"] ?? "HTTP $status");
}
return $payload["data"];
}
// .NET 8+
using System.Net.Http.Json;
using System.Text.Json;
static class SkillSafe
{
const string Api = "https://api.skillsafe.ai/v1/app-api";
static readonly HttpClient Http = new();
static SkillSafe() =>
Http.DefaultRequestHeaders.Authorization =
new("Bearer", Environment.GetEnvironmentVariable("SKILLSAFE_TOKEN")); // see step 1
public static async Task<JsonElement> ApiAsync(HttpMethod method, string path, object? body = null)
{
var req = new HttpRequestMessage(method, Api + path);
if (body != null) req.Content = JsonContent.Create(body);
var res = await Http.SendAsync(req);
var json = await res.Content.ReadFromJsonAsync<JsonElement>();
if (!res.IsSuccessStatusCode)
throw new Exception(json.GetProperty("error").GetProperty("message").GetString());
return json.GetProperty("data");
}
}
Step 1 — Get a token
A guest token lets you check balances and estimate costs for free. For metered review runs
billed to your own account, use your personal token: open the
token page, sign in with SkillSafe, and press
Copy shell export — it puts export SKILLSAFE_TOKEN="…" on your
clipboard, which every example below reads. Treat the token like a password: it can spend
your credits. For fully headless scripts, POST /guest mints a guest token with
no browser involved.
curl -s -X POST "$API/guest" \
-H "Content-Type: application/json" \
-d '{"slug":"torch-clinic"}' | jq -r '.data.token'
token = api("POST", "/guest", {"slug": "torch-clinic"})["token"]
const { token } = await api("POST", "/guest", { slug: "torch-clinic" });
var guest struct{ Token string `json:"token"` }
err := call("POST", "/guest", map[string]string{"slug": "torch-clinic"}, &guest)
String envelope = api("POST", "/guest", """
{"slug":"torch-clinic"}""");
// token is at data.token in the returned JSON
token = api("POST", "/guest", { slug: "torch-clinic" })["token"]
$token = api("POST", "/guest", ["slug" => "torch-clinic"])["token"];
var guest = await SkillSafe.ApiAsync(HttpMethod.Post, "/guest",
new { slug = "torch-clinic" });
var token = guest.GetProperty("token").GetString();
The app stores this browser's token under the localStorage key
skillsafe_app_token:torch-clinic, on the app's own origin. The
token page reads and manages it for you — you never need
to open developer tools.
Step 2 — Check who you are and your balance
Returns subject_type ("user" or "guest"),
subject_id and your credits balance. Check this before reviewing a
large bundle.
curl -s "$API/me" -H "Authorization: Bearer $TOKEN" | jq '.data'
me = api("GET", "/me")
print(me["subject_type"], me["credits"])
const me = await api("GET", "/me");
console.log(me.subject_type, me.credits);
var me struct {
SubjectType string `json:"subject_type"`
Credits int64 `json:"credits"`
}
err := call("GET", "/me", nil, &me)
String envelope = api("GET", "/me", null);
// data.subject_type, data.credits
me = api("GET", "/me")
puts "#{me["subject_type"]}: #{me["credits"]} credits"
$me = api("GET", "/me");
echo "{$me['subject_type']}: {$me['credits']} credits\n";
var me = await SkillSafe.ApiAsync(HttpMethod.Get, "/me");
Console.WriteLine($"{me.GetProperty("subject_type")}: {me.GetProperty("credits")} credits");
Step 3 — Estimate the cost
Send exactly the input you would send to /run; the response's
hold_credits is the worst-case cost. Nothing is charged and no job is created,
so estimating is free — useful when you are piping a whole training
directory in and want a ceiling before spending credits.
| Input field | Type | Notes |
|---|---|---|
code | string, required | The PyTorch source: model definitions, training and evaluation loops, dataset and DataLoader code, config. One file or several concatenated, each preceded by a # file: train.py comment. This is the model's only evidence — nothing is executed and no GPU is touched. Inputs longer than 100,000 characters are clipped middle-out, with a # [... clipped ...] comment showing where. At least 60 characters are needed for a review. |
framework | string | pytorch | lightning | mixed | unknown — changes what is idiomatic: plain PyTorch owns its own loop, so missing zero_grad/eval/no_grad are real findings; Lightning handles the loop itself, so those checks apply only to code that bypasses it. |
concern | string | general | correctness | performance | memory | reproducibility — the review emphasis. It weights the findings and the summary, but it is emphasis and not exclusivity: a high-severity finding from another category is never suppressed. |
context | string, optional | Extra context: PyTorch version, GPU model and count, dataset size, how long training runs, the symptom that brought you here, what is already handled in code you did not paste, and any gap you have already chosen to accept. Clipped at 20,000 characters. |
prescan_facts | object, optional | What a client-side scanner mechanically matched in the code: {"resources": [], "flags": []}. Each entry is {id, label}. Resource ids look like res:module/convnet or res:loop/train.py-38; flag ids are <check>:<name> — zero-grad:train.py, no-step:train.py, loss-pairing:softmax-xent, loss-accum:48, eval-mode:val_loader, no-grad-eval:val_loader, seed-missing:global, device-hardcode:33, dataloader-config:29, retain-graph:71, checkpoint-missing:global, plain-secret:token. Every flag id you send comes back in coverage_check. The web UI fills this from its own scan; API callers may omit the field or send the two empty arrays. |
retry_note | string, optional | Only set by the app's automatic reformat retry when a first reply was not valid JSON. Leave it out. |
cat > train.py <<'CODE'
# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
CODE
jq -n --rawfile code train.py \
'{code: $code,
framework: "pytorch",
concern: "general",
context: "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
prescan_facts: {resources: [], flags: []}}' > input.json
curl -s -X POST "$API/estimate" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-d @input.json | jq '.data.hold_credits'
CODE = """# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
"""
payload = {
"code": CODE,
"framework": "pytorch",
"concern": "general",
"context": "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
"prescan_facts": {"resources": [], "flags": []},
}
est = api("POST", "/estimate", payload)
print("worst case:", est.get("hold_credits", est.get("credits")), "credits")
const code = [
"# file: train.py",
"for x, y in train_loader:",
" loss = criterion(model(x), y); loss.backward(); optimizer.step()",
].join("\n");
const payload = {
code,
framework: "pytorch",
concern: "general",
context: "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
prescan_facts: { resources: [], flags: [] },
};
const est = await api("POST", "/estimate", payload);
console.log("worst case:", est.hold_credits ?? est.credits, "credits");
const code = `# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()`
payload := map[string]any{
"code": code,
"framework": "pytorch",
"concern": "general",
"context": "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
"prescan_facts": map[string]any{
"resources": []any{}, "flags": []any{},
},
}
var est struct{ HoldCredits int64 `json:"hold_credits"` }
err := call("POST", "/estimate", payload, &est)
String code = """
# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
""";
String jsonPayload = """
{"code": %s,
"framework": "pytorch",
"concern": "general",
"context": "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
"prescan_facts": {"resources": [], "flags": []}}
""".formatted(toJsonString(code));
String envelope = api("POST", "/estimate", jsonPayload);
// worst-case cost is at data.hold_credits
CODE = <<~PY
# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
PY
payload = { code: CODE,
framework: "pytorch",
concern: "general",
context: "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
prescan_facts: { resources: [], flags: [] } }
est = api("POST", "/estimate", payload)
puts "worst case: #{est["hold_credits"] || est["credits"]} credits"
$code = <<<'PY'
# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
PY;
$payload = [
"code" => $code,
"framework" => "pytorch",
"concern" => "general",
"context" => "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
"prescan_facts" => ["resources" => [], "flags" => []],
];
$est = api("POST", "/estimate", $payload);
echo "worst case: " . ($est["hold_credits"] ?? $est["credits"]) . " credits\n";
var code = """
# file: train.py
for x, y in train_loader:
loss = criterion(model(x), y); loss.backward(); optimizer.step()
""";
var payload = new {
code,
framework = "pytorch",
concern = "general",
context = "PyTorch 2.5 on one RTX 3090; val accuracy plateaus and memory creeps.",
prescan_facts = new {
resources = Array.Empty<object>(), flags = Array.Empty<object>(),
},
};
var est = await SkillSafe.ApiAsync(HttpMethod.Post, "/estimate", payload);
Console.WriteLine($"worst case: {est.GetProperty("hold_credits")} credits");
prescan_facts.flags is how you make the review answer for things you already
know about. Send {"resources": [{"id": "res:loop/train.py-2", "label": "Loop/train.py:2"}],
"flags": [{"id": "zero-grad:train.py", "label": "backward() without zero_grad() in train.py"}]} and every flag id
comes back in coverage_check — addressed by a finding, or set aside with
the reason. Nothing you flag is silently dropped, which makes it the field to assert on in a
CI check.
Step 4 — Run the review and wait for the result
/run takes the same input as /estimate, places a credit hold and
returns a job_id. Poll /jobs/{job_id} every 1–2 seconds
until status is succeeded or failed (a run typically
takes 30–90 s, since every finding carries a corrected code fragment). Always send
an Idempotency-Key header so a network retry can't start a second,
double-charged run. The review is in output — usually nested as
output.output, and as a JSON string, so parse defensively. The samples
below print the posture, the inventory, the prioritized findings and the focus areas, then
save the whole object to review.json.
JOB_ID=$(curl -s -X POST "$API/run" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-H "Idempotency-Key: tc-$(date +%s)" \
-d @input.json | jq -r '.data.job_id')
while :; do
JOB=$(curl -s "$API/jobs/$JOB_ID" -H "Authorization: Bearer $TOKEN")
STATUS=$(echo "$JOB" | jq -r '.data.status')
[ "$STATUS" = "succeeded" ] || [ "$STATUS" = "failed" ] && break
sleep 2
done
# unwrap the review once, then read it
echo "$JOB" | jq -r '.data.output.output' > review.json
jq -r '
"\(.review_name) [\(.posture)]: \(.verdict)",
"",
"INVENTORY",
(.inventory[] | " \(.kind)/\(.name) in \(.scope) - \(.role)"),
"",
"FINDINGS",
(.findings[] | " [\(.priority)] \(.id) \(.category) \(.resource): \(.problem)"),
"",
"QUICK WINS",
(.quick_wins[] | " - \(.)"),
"",
"FOCUS AREAS",
(.focus_areas[] | " \(.area) - \(.why)"),
"",
"COVERAGE",
(.coverage_check[] | " \(.id): \(if .addressed then "ok" else "SET ASIDE" end) - \(.note)")' \
review.json
# fail the pipeline on anything critical
jq -e '[.findings[] | select(.priority == "critical")] | length == 0' review.json > /dev/null \
|| { echo "critical findings present"; exit 1; }
import time
job_id = api("POST", "/run", payload,
**{"Idempotency-Key": "tc-001"})["job_id"]
while True:
job = api("GET", f"/jobs/{job_id}")
if job["status"] in ("succeeded", "failed"):
break
time.sleep(1.5)
if job["status"] == "failed":
raise RuntimeError(job.get("error", "run failed"))
raw = job["output"]
if isinstance(raw, dict) and "output" in raw:
raw = raw["output"]
review = json.loads(raw) if isinstance(raw, str) else raw
print(f'{review["review_name"]} [{review["posture"]}]: {review["verdict"]}')
for r in review["inventory"]:
print(f' {r["kind"]}/{r["name"]:<24} in={r["scope"] or "-":<16} {r["role"]}')
for f in review["findings"]:
print(f' [{f["priority"]:>8}] {f["id"]} {f["category"]} {f["resource"]}')
print(f' L:{f["likelihood"]}/S:{f["severity"]} {f["problem"]}')
print(f' fix: {f["fix"]}')
if f["snippet"]:
print(" snippet:", f["snippet"].splitlines()[0], "...")
for w in review["quick_wins"]:
print(" win:", w)
for a in review["focus_areas"]:
print(f' focus {a["area"]} {a["finding_ids"]} - {a["why"]}')
for c in review["coverage_check"]:
print(f' {c["id"]}: {"ok" if c["addressed"] else "SET ASIDE"} - {c["note"]}')
with open("review.json", "w", encoding="utf-8") as fh:
json.dump(review, fh, indent=2)
critical = [f for f in review["findings"] if f["priority"] == "critical"]
if critical:
raise SystemExit(f"{len(critical)} critical finding(s)")
import { writeFileSync } from "node:fs";
const { job_id } = await api("POST", "/run", payload,
{ "Idempotency-Key": crypto.randomUUID() });
let job;
do {
await new Promise((r) => setTimeout(r, 1500));
job = await api("GET", `/jobs/${job_id}`);
} while (job.status !== "succeeded" && job.status !== "failed");
if (job.status === "failed") throw new Error(job.error ?? "run failed");
const raw = job.output?.output ?? job.output;
const review = typeof raw === "string" ? JSON.parse(raw) : raw;
console.log(`${review.review_name} [${review.posture}]: ${review.verdict}`);
for (const r of review.inventory) {
console.log(` ${r.kind}/${r.name} (${r.scope || "-"}): ${r.role}`);
}
for (const f of review.findings) {
console.log(` [${f.priority}] ${f.id} ${f.category} ${f.resource}`);
console.log(` L:${f.likelihood}/S:${f.severity} - ${f.fix}`);
}
for (const w of review.quick_wins) console.log(` win: ${w}`);
for (const a of review.focus_areas) {
console.log(` focus ${a.area} (${a.finding_ids.join(", ")}): ${a.why}`);
}
for (const c of review.coverage_check) {
console.log(` ${c.id}: ${c.addressed ? "ok" : "SET ASIDE"} - ${c.note}`);
}
writeFileSync("review.json", JSON.stringify(review, null, 2));
const critical = review.findings.filter((f) => f.priority === "critical");
if (critical.length) process.exitCode = 1;
var started struct{ JobID string `json:"job_id"` }
if err := call("POST", "/run", payload, &started); err != nil {
log.Fatal(err)
}
var job struct {
Status string `json:"status"`
Error string `json:"error"`
Output json.RawMessage `json:"output"`
}
for {
if err := call("GET", "/jobs/"+started.JobID, nil, &job); err != nil {
log.Fatal(err)
}
if job.Status == "succeeded" || job.Status == "failed" {
break
}
time.Sleep(1500 * time.Millisecond)
}
// job.Output is {"output": "<json string>"} — unwrap, then unmarshal:
type Review struct {
ReviewName string `json:"review_name"`
Posture string `json:"posture"`
Verdict string `json:"verdict"`
ExecSummary string `json:"exec_summary"`
Assumptions []string `json:"assumptions"`
OpenQuestions []string `json:"open_questions"`
Inventory []struct {
Kind, Name, Scope, Role string
} `json:"inventory"`
Findings []struct {
ID, Category, Severity, Likelihood, Priority string
Resource, Problem, Impact, Fix, Snippet string
} `json:"findings"`
CoverageCheck []struct {
ID, Note string
Addressed bool
} `json:"coverage_check"`
QuickWins []string `json:"quick_wins"`
FocusAreas []struct {
Area, Why string
FindingIDs []string `json:"finding_ids"`
} `json:"focus_areas"`
Summary string `json:"summary"`
}
var wrapper struct{ Output string `json:"output"` }
json.Unmarshal(job.Output, &wrapper)
var review Review
json.Unmarshal([]byte(wrapper.Output), &review)
fmt.Printf("%s [%s]: %s\n", review.ReviewName, review.Posture, review.Verdict)
for _, r := range review.Inventory {
fmt.Printf(" %s/%s (%s): %s\n", r.Kind, r.Name, r.Scope, r.Role)
}
for _, f := range review.Findings {
fmt.Printf(" [%s] %s %s %s: %s\n", f.Priority, f.ID, f.Category, f.Resource, f.Problem)
}
for _, a := range review.FocusAreas {
fmt.Printf(" focus %s %v: %s\n", a.Area, a.FindingIDs, a.Why)
}
os.WriteFile("review.json", []byte(wrapper.Output), 0o644)
String envelope = api("POST", "/run", jsonPayload);
String jobId = /* data.job_id via your JSON library */;
while (true) {
String job = api("GET", "/jobs/" + jobId, null);
String status = /* data.status */;
if (status.equals("succeeded") || status.equals("failed")) break;
Thread.sleep(1500);
}
// The review is at data.output.output as a JSON string — parse it again, then read
// review_name, posture, verdict, exec_summary, assumptions[], open_questions[],
// inventory[] (kind/name/scope/role),
// findings[] (id/category/severity/likelihood/priority/resource/problem/impact/fix/snippet),
// coverage_check[] (id/addressed/note), quick_wins[],
// focus_areas[] (area/why/finding_ids[]) and summary.
// Finally keep the review on disk:
// Files.writeString(Path.of("review.json"), reviewJson);
started = api("POST", "/run", payload)
job = nil
loop do
job = api("GET", "/jobs/#{started["job_id"]}")
break if %w[succeeded failed].include?(job["status"])
sleep 1.5
end
raise (job["error"] || "run failed") if job["status"] == "failed"
raw = job["output"].is_a?(Hash) ? job["output"].fetch("output", job["output"]) : job["output"]
review = raw.is_a?(String) ? JSON.parse(raw) : raw
puts "#{review["review_name"]} [#{review["posture"]}]: #{review["verdict"]}"
review["inventory"].each { |r| puts " #{r["kind"]}/#{r["name"]} (#{r["scope"]}): #{r["role"]}" }
review["findings"].each do |f|
puts " [#{f["priority"]}] #{f["id"]} #{f["category"]} #{f["resource"]}"
puts " L:#{f["likelihood"]}/S:#{f["severity"]} - #{f["fix"]}"
end
review["quick_wins"].each { |w| puts " win: #{w}" }
review["focus_areas"].each { |a| puts " focus #{a["area"]} #{a["finding_ids"].join(", ")}" }
review["coverage_check"].each { |c| puts " #{c["id"]}: #{c["addressed"] ? "ok" : "SET ASIDE"}" }
File.write("review.json", JSON.pretty_generate(review))
exit 1 if review["findings"].any? { |f| f["priority"] == "critical" }
$started = api("POST", "/run", $payload);
do {
sleep(2);
$job = api("GET", "/jobs/" . $started["job_id"]);
} while (!in_array($job["status"], ["succeeded", "failed"]));
if ($job["status"] === "failed") {
throw new Exception($job["error"] ?? "run failed");
}
$raw = is_array($job["output"]) ? ($job["output"]["output"] ?? $job["output"]) : $job["output"];
$review = is_string($raw) ? json_decode($raw, true) : $raw;
echo "{$review['review_name']} [{$review['posture']}]: {$review['verdict']}\n";
foreach ($review["inventory"] as $r) {
echo " {$r['kind']}/{$r['name']} ({$r['scope']}): {$r['role']}\n";
}
foreach ($review["findings"] as $f) {
echo " [{$f['priority']}] {$f['id']} {$f['category']} {$f['resource']}\n";
echo " L:{$f['likelihood']}/S:{$f['severity']} - {$f['fix']}\n";
}
foreach ($review["quick_wins"] as $w) {
echo " win: $w\n";
}
foreach ($review["focus_areas"] as $a) {
echo " focus {$a['area']}: " . implode(", ", $a["finding_ids"]) . "\n";
}
foreach ($review["coverage_check"] as $c) {
echo " {$c['id']}: " . ($c["addressed"] ? "ok" : "SET ASIDE") . "\n";
}
file_put_contents("review.json", json_encode($review, JSON_PRETTY_PRINT));
var started = await SkillSafe.ApiAsync(HttpMethod.Post, "/run", payload);
var jobId = started.GetProperty("job_id").GetString();
JsonElement job;
while (true)
{
job = await SkillSafe.ApiAsync(HttpMethod.Get, $"/jobs/{jobId}");
var status = job.GetProperty("status").GetString();
if (status is "succeeded" or "failed") break;
await Task.Delay(1500);
}
var rawText = job.GetProperty("output").GetProperty("output").GetString();
using var doc = JsonDocument.Parse(rawText!);
var review = doc.RootElement;
Console.WriteLine($"{review.GetProperty("review_name")} " +
$"[{review.GetProperty("posture")}]: {review.GetProperty("verdict")}");
foreach (var r in review.GetProperty("inventory").EnumerateArray())
{
Console.WriteLine($" {r.GetProperty("kind")}/{r.GetProperty("name")}: {r.GetProperty("role")}");
}
foreach (var f in review.GetProperty("findings").EnumerateArray())
{
Console.WriteLine($" [{f.GetProperty("priority")}] {f.GetProperty("id")} " +
$"{f.GetProperty("category")} {f.GetProperty("resource")} " +
$"(L:{f.GetProperty("likelihood")}/S:{f.GetProperty("severity")})");
}
foreach (var a in review.GetProperty("focus_areas").EnumerateArray())
{
Console.WriteLine($" focus {a.GetProperty("area")}: {a.GetProperty("why")}");
}
await File.WriteAllTextAsync("review.json", rawText!);
The model is asked for one JSON object and nothing else, but a stray code fence or preamble
is always possible. Strip a leading ```json fence, take the text between the
first { and the last }, and only then parse — that is what
the app does before it falls back to a retry_note reformat run.
The review object — output schema
One JSON object, always the same shape. Every array is present, and the review is grounded in
the pasted source alone: findings cite only modules, loops, functions and files that actually
appear in code, and a construct that is simply absent (no seed, no
checkpoint, no eval-mode switch) is reported against the closest real construct or against
(missing from the script). Where the code is silent on something that changes the
verdict you get an entry in assumptions and, if it would change the ranking, in
open_questions. Expect five to fifteen findings on a typical script — a
well-built one may honestly yield two or three, and findings is never empty.
| Field | Type | Meaning |
|---|---|---|
review_name | string | A short title naming the script, taken from the code's own naming — e.g. resnet fine-tune loop — pytorch review. |
posture | string | train-ready | hardening-recommended | silent-failure-risk. See the table below. |
verdict | string | One sentence justifying the posture and naming the single most important change. |
exec_summary | string | Two or three paragraphs, separated by blank lines, on the dominant themes across the script. |
assumptions | string[] | Explicit assumptions filling gaps the code left open. Read these first — a wrong assumption invalidates the findings built on it. |
open_questions | string[] | Questions whose answers would change the ranking. |
inventory | array | {kind, name, scope, role} — every Module, Loop, Function, Optimizer, Loss, Dataset, DataLoader, Scheduler and Config the review parsed out of the code and the part it plays. scope is the file it is defined in. |
findings | array | The prioritized findings table — ids TC-001, TC-002, … in sequence, at least one entry. Columns are listed below. |
coverage_check | array | {id, addressed, note} — one entry per prescan_facts.flags id you sent, each appearing exactly once. See the semantics below. |
quick_wins | string[] | One-line changes worth doing immediately, ahead of any planning. May be empty when nothing here is a one-liner. |
focus_areas | array | {area, why, finding_ids} — what to work through first, one sentence tied to the review, and the finding ids that motivate it. Every id in finding_ids exists in findings. |
summary | string | Closing paragraph: what to fix first, and what risk remains after that. |
The three posture values:
| posture | What it means |
|---|---|
train-ready | The script holds up as written: losses paired with raw logits, modes and gradients handled, metrics trustworthy, runs reproducible. Findings still exist, but they are additions and refinements — AMP, torch.compile, profiling — not blockers. Genuinely well-built scripts land here rather than having severity manufactured for them. |
hardening-recommended | The training is sound, but named gaps should be closed before long or shared runs — no seed, no checkpointing, a starved DataLoader on an expensive GPU. |
silent-failure-risk | At least one pattern corrupts the model, the metric or the memory profile without raising an error as written: gradients accumulating across steps, a softmax feeding CrossEntropyLoss, validation running in train mode, an epoch loss holding every batch's autograd graph. |
Each entry in findings:
| Column | Meaning |
|---|---|
id | Sequential TC-001, TC-002, … — the stable handle referenced from focus_areas[].finding_ids. |
category | correctness | performance | memory | reproducibility | data | hygiene. Weighted by the concern you sent, but never restricted to it. |
severity | low | medium | high — how bad it is when it bites. |
likelihood | low | medium | high — how likely it is to bite. |
priority | critical | high | medium | low — severity by likelihood. critical is reserved for something that makes the training lie (gradients never zeroed, a metric computed in train mode, a loss double-normalized) or that reliably kills the run, so sort on this field and work top-down. This is also the field to gate a pipeline on. |
resource | The Module/name, Loop/name, Function/name or file this is about — always something that appears in code, or the literal (missing from the script) when the finding is about an absent construct. |
problem | What is wrong, in this code specifically. |
impact | What happens to the run, the metric and the team because of it. |
fix | The concrete change to make — not "add validation". |
snippet | A corrected Python fragment you can paste: the fixed block, correctly indented, matching your script's style, not the whole file. Empty string when a snippet would add nothing. Secret values are never echoed — a placeholder appears instead. |
coverage_check semantics:
| Case | What you get |
|---|---|
| Every flag id you sent | Each prescan_facts.flags id appears in coverage_check exactly once. Nothing you flagged is silently dropped, which makes this the field to assert on in a CI check. Ids in prescan_facts.resources are not reconciled here — they shape the inventory instead. |
addressed: true | The flag is covered by the review; note names the finding id that covers it. |
addressed: false | The flag was deliberately set aside; note gives the reason — a check that fired but is not a real problem for this script (a hardcoded "cuda" in a cluster-only launcher, a missing checkpoint in a two-minute smoke run). |
| Nothing sent | Omit prescan_facts, or send the two empty arrays, and coverage_check comes back empty. The rest of the review is unaffected. |
A small, realistic result for the snippet above, trimmed for length:
{
"review_name": "train.py loop — pytorch review",
"posture": "silent-failure-risk",
"verdict": "Gradients are never zeroed, so every step applies the running sum of all
previous batches; add optimizer.zero_grad() before anything else.",
"exec_summary": "A three-line training step with one decisive defect: loss.backward()
accumulates into .grad on every iteration and nothing ever clears it, so
the effective update grows without bound. The loss will appear to fall for
a while, then oscillate or diverge — with no exception raised at any
point.
The loss call itself is paired correctly — criterion receives the
model output directly — so the fix is one line and the structure can
stay.",
"assumptions": [
"The paste is the whole training step; no zero_grad happens in a helper that was not shown.",
"criterion pairs with raw logits, since no activation appears between model and loss."
],
"open_questions": [
"Does the loss curve you see oscillate after the first few hundred steps?",
"Is there a seed or checkpoint in code that was not pasted?"
],
"inventory": [
{ "kind": "Loop", "name": "train_loader loop", "scope": "train.py",
"role": "The only loop in the paste; one optimization step per batch." }
],
"findings": [
{ "id": "TC-001", "category": "correctness",
"severity": "high", "likelihood": "high", "priority": "critical",
"resource": "Loop/train_loader loop",
"problem": "loss.backward() adds into every parameter's .grad each iteration, but
optimizer.zero_grad() is never called.",
"impact": "Each step applies the sum of all gradients so far; the effective learning
rate grows every batch, training oscillates or diverges, and the final
model is unusable — silently.",
"fix": "Zero the gradients at the top of every iteration.",
"snippet": "for x, y in train_loader:\n optimizer.zero_grad(set_to_none=True)\n loss = criterion(model(x), y)\n loss.backward()\n optimizer.step()" },
{ "id": "TC-002", "category": "reproducibility",
"severity": "medium", "likelihood": "high", "priority": "medium",
"resource": "(missing from the script)",
"problem": "No torch.manual_seed (or random/numpy seeding) appears anywhere in the
paste.",
"impact": "No run can be reproduced or compared against a baseline — after the
zero_grad fix you cannot even prove the fix helped.",
"fix": "Seed torch, random and numpy once at startup, before the model is built.",
"snippet": "torch.manual_seed(1337)\nrandom.seed(1337)\nnp.random.seed(1337)" }
],
"coverage_check": [
{ "id": "zero-grad:train.py", "addressed": true, "note": "TC-001." },
{ "id": "seed-missing:global", "addressed": true, "note": "TC-002." }
],
"quick_wins": [
"Add optimizer.zero_grad(set_to_none=True) as the first line of the batch loop."
],
"focus_areas": [
{ "area": "Gradient hygiene",
"why": "The only correctness defect in this paste invalidates every optimization step.",
"finding_ids": ["TC-001"] },
{ "area": "Reproducibility",
"why": "Without a seed the before/after comparison of the fix cannot be trusted.",
"finding_ids": ["TC-002"] }
],
"summary": "Zero the gradients and this loop trains correctly; seed the run next so the
improvement is provable. After that the useful additions are a validation pass
under torch.no_grad() with the model switched to eval mode, and a checkpoint
so a long run can resume."
}
This is AI-generated review from source text, not a CI sign-off: it sees only what you
sent, never a real run, the real failure history or the rest of the codebase. Check
assumptions and open_questions before you act on the rankings,
run every snippet through your own tests and linter,
and keep a human reviewer in the loop.
Step 5 — Stream the review as it is written
/run-stream takes exactly the same body as /run but answers with
server-sent events, so you can show progress instead of a spinner — useful here
because a full findings table with corrected code makes for a long reply. This app's own
progress panel is this endpoint. Events are separated by a blank line; each has an
event: line and a data: line carrying JSON.
| Event | Payload | Meaning |
|---|---|---|
job | {job_id, status} | Sent once, when the job is accepted — show "starting". |
delta | {text} | A chunk of the reply, in order. Append it; the accumulated length is your only progress signal (the total is not known in advance). The app advances its step list by watching for the "review_name", "inventory", "findings", "coverage_check" and "focus_areas" keys as they arrive. |
done | {job_id, status, charged_credits, output} | The final, authoritative result — read the review from output.output rather than trusting concatenated deltas, and the settled price from charged_credits. |
error | {code, message} | Replaces done when the run fails. |
# -N disables buffering so events print as they arrive
curl -N -s -X POST "$API/run-stream" \
-H "Authorization: Bearer $TOKEN" -H "Content-Type: application/json" \
-H "Idempotency-Key: tc-$(date +%s)" \
-d @input.json
# event: job
# data: {"job_id":"job_...","status":"running"}
#
# event: delta
# data: {"text":"{\"review_name\":\"web"}
# ...
# event: done
# data: {"job_id":"job_...","status":"succeeded","charged_credits":612,"output":{"output":"{...}"}}
import json, requests
result = None
with requests.post(
API + "/run-stream",
headers={"Authorization": f"Bearer {TOKEN}",
"Idempotency-Key": "tc-001"},
json=payload,
stream=True,
) as r:
r.raise_for_status()
event = None
for line in r.iter_lines(decode_unicode=True):
if not line:
continue
if line.startswith("event:"):
event = line[len("event:"):].strip()
elif line.startswith("data:"):
data = json.loads(line[len("data:"):].strip())
if event == "delta":
print(".", end="", flush=True) # live progress
elif event == "done":
result = data
elif event == "error":
raise RuntimeError(data.get("message", "run failed"))
review = json.loads(result["output"]["output"]) # authoritative
print("charged:", result["charged_credits"], "-", review["review_name"])
print("posture:", review["posture"])
for f in review["findings"]:
print(f' [{f["priority"]}] {f["id"]} {f["resource"]}: {f["problem"]}')
with open("review.json", "w", encoding="utf-8") as fh:
json.dump(review, fh, indent=2)
const res = await fetch(API + "/run-stream", {
method: "POST",
headers: {
Authorization: `Bearer ${TOKEN}`,
"Content-Type": "application/json",
"Idempotency-Key": crypto.randomUUID(),
},
body: JSON.stringify(payload),
});
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buf = "", done = null;
for (;;) {
const chunk = await reader.read();
if (chunk.done) break;
buf += decoder.decode(chunk.value, { stream: true });
const frames = buf.split("\n\n");
buf = frames.pop();
for (const frame of frames) {
const name = /^event:\s*(.+)$/m.exec(frame)?.[1];
const body = /^data:\s*(.+)$/m.exec(frame)?.[1];
if (!name || !body) continue;
const data = JSON.parse(body);
if (name === "delta") process.stdout.write("."); // live progress
if (name === "done") done = data;
if (name === "error") throw new Error(data.message ?? "run failed");
}
}
const review = JSON.parse(done.output.output);
console.log(`\n${done.charged_credits} credits - ${review.review_name} [${review.posture}]`);
for (const f of review.findings) console.log(` [${f.priority}] ${f.id} ${f.resource}`);
writeFileSync("review.json", JSON.stringify(review, null, 2));
body, _ := json.Marshal(payload)
req, _ := http.NewRequest("POST", API+"/run-stream", bytes.NewReader(body))
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", "application/json")
req.Header.Set("Idempotency-Key", "tc-001")
res, err := http.DefaultClient.Do(req)
if err != nil {
log.Fatal(err)
}
defer res.Body.Close()
var event string
var final map[string]any
sc := bufio.NewScanner(res.Body)
sc.Buffer(make([]byte, 0, 64*1024), 4*1024*1024)
for sc.Scan() {
line := sc.Text()
switch {
case strings.HasPrefix(line, "event:"):
event = strings.TrimSpace(strings.TrimPrefix(line, "event:"))
case strings.HasPrefix(line, "data:"):
var data map[string]any
json.Unmarshal([]byte(strings.TrimPrefix(line, "data:")), &data)
switch event {
case "delta":
fmt.Print(".") // live progress
case "done":
final = data
case "error":
log.Fatal(data["message"])
}
}
}
// final["output"].(map[string]any)["output"].(string) is the review JSON —
// unmarshal it into the Review struct from step 4, then write it to review.json.
// Java 17+ — read the stream line by line instead of buffering the body.
var req = HttpRequest.newBuilder(URI.create(API + "/run-stream"))
.header("Authorization", "Bearer " + TOKEN)
.header("Content-Type", "application/json")
.header("Idempotency-Key", "tc-001")
.POST(HttpRequest.BodyPublishers.ofString(jsonPayload))
.build();
var res = HTTP.send(req, HttpResponse.BodyHandlers.ofLines());
String event = null, done = null;
for (String line : (Iterable<String>) res.body()::iterator) {
if (line.startsWith("event:")) {
event = line.substring(6).trim();
} else if (line.startsWith("data:")) {
String data = line.substring(5).trim();
if ("delta".equals(event)) System.out.print("."); // live progress
else if ("done".equals(event)) done = data;
else if ("error".equals(event)) throw new RuntimeException(data);
}
}
// parse `done`, then parse data.output.output again — it is a JSON string holding
// review_name, posture, verdict, inventory[], findings[], coverage_check[],
// quick_wins[], focus_areas[] and the rest.
require "net/http"
require "json"
uri = URI(API + "/run-stream")
req = Net::HTTP::Post.new(uri)
req["Authorization"] = "Bearer #{TOKEN}"
req["Content-Type"] = "application/json"
req["Idempotency-Key"] = "tc-001"
req.body = payload.to_json
event = nil
done = nil
Net::HTTP.start(uri.host, uri.port, use_ssl: true) do |http|
http.request(req) do |res|
res.read_body do |chunk|
chunk.each_line do |line|
line = line.strip
if line.start_with?("event:")
event = line.delete_prefix("event:").strip
elsif line.start_with?("data:")
data = JSON.parse(line.delete_prefix("data:").strip)
case event
when "delta" then print "." # live progress
when "done" then done = data
when "error" then raise (data["message"] || "run failed")
end
end
end
end
end
end
review = JSON.parse(done["output"]["output"])
puts "\n#{done["charged_credits"]} credits - #{review["review_name"]} [#{review["posture"]}]"
review["findings"].each { |f| puts " [#{f["priority"]}] #{f["id"]} #{f["resource"]}" }
File.write("review.json", JSON.pretty_generate(review))
$event = null;
$done = null;
$ch = curl_init(API . "/run-stream");
curl_setopt_array($ch, [
CURLOPT_POST => true,
CURLOPT_HTTPHEADER => [
"Authorization: Bearer $TOKEN",
"Content-Type: application/json",
"Idempotency-Key: tc-001",
],
CURLOPT_POSTFIELDS => json_encode($payload),
CURLOPT_WRITEFUNCTION => function ($ch, $chunk) use (&$event, &$done) {
foreach (explode("\n", $chunk) as $line) {
$line = trim($line);
if (str_starts_with($line, "event:")) {
$event = trim(substr($line, 6));
} elseif (str_starts_with($line, "data:")) {
$data = json_decode(trim(substr($line, 5)), true);
if ($event === "delta") { echo "."; } // live progress
elseif ($event === "done") { $done = $data; }
elseif ($event === "error") { throw new Exception($data["message"] ?? "run failed"); }
}
}
return strlen($chunk);
},
]);
curl_exec($ch);
curl_close($ch);
$review = json_decode($done["output"]["output"], true);
echo "\n{$done['charged_credits']} credits - {$review['review_name']} [{$review['posture']}]\n";
foreach ($review["findings"] as $f) {
echo " [{$f['priority']}] {$f['id']} {$f['resource']}\n";
}
file_put_contents("review.json", json_encode($review, JSON_PRETTY_PRINT));
var req = new HttpRequestMessage(HttpMethod.Post, Api + "/run-stream") {
Content = JsonContent.Create(payload),
};
req.Headers.Add("Idempotency-Key", "tc-001");
using var res = await Http.SendAsync(req, HttpCompletionOption.ResponseHeadersRead);
using var reader = new StreamReader(await res.Content.ReadAsStreamAsync());
string? evt = null, done = null;
while (await reader.ReadLineAsync() is { } line)
{
if (line.StartsWith("event:")) evt = line[6..].Trim();
else if (line.StartsWith("data:"))
{
var data = line[5..].Trim();
if (evt == "delta") Console.Write("."); // live progress
else if (evt == "done") done = data;
else if (evt == "error") throw new Exception(data);
}
}
using var final = JsonDocument.Parse(done!);
var text = final.RootElement.GetProperty("output").GetProperty("output").GetString();
using var reviewDoc = JsonDocument.Parse(text!);
var review = reviewDoc.RootElement;
Console.WriteLine($"{review.GetProperty("review_name")} [{review.GetProperty("posture")}]");
foreach (var f in review.GetProperty("findings").EnumerateArray())
Console.WriteLine($" [{f.GetProperty("priority")}] {f.GetProperty("id")} {f.GetProperty("resource")}");
await File.WriteAllTextAsync("review.json", text!);
In a browser, the native EventSource only speaks GET, and this endpoint is a
POST — read the fetch response body incrementally, as the JavaScript
sample above does. On an idempotent replay the server may answer with a plain JSON
envelope instead of an event stream; check the Content-Type before you start
parsing frames.