diff --git a/experiments/2026-05-12-cross-model-authoring/format-findings.md b/experiments/2026-05-12-cross-model-authoring/format-findings.md index 44366cf..d6e6c73 100644 --- a/experiments/2026-05-12-cross-model-authoring/format-findings.md +++ b/experiments/2026-05-12-cross-model-authoring/format-findings.md @@ -107,15 +107,36 @@ and the semicolons *add* a second consistency requirement Qwen does not reliably meet (the trailing-`;` slips). It is explicit, but with extra load, not redundant cue. -## Overall finding across all five formats +## Tool-calling / structured AST tested: WORST, and it confirms the rule -| format | green | structure marking | +The intuitive bet: Qwen3-Coder is trained hard on tool-calling, the API +validates the arguments, so submitting the program as canonical AST JSON via a +`submit_program` tool should be the most reliable channel. Tool-calling is +supported (verified: a dummy `add` tool returned a correct `tool_call`). But +the result is **2/8 — tied for worst.** Two failure modes, both the model's: +- `check` (L1/L3/L5/L6): valid JSON, wrong AST — e.g. `* ` given 3 args. +- **`bad-json` (L4/L7): the tool arguments were not valid JSON at all** — + completion hit the 2500-token cap. The AST JSON is so verbose (every leaf is + `{"t":"var","name":"x"}`) that a non-trivial program overruns the budget and + the JSON is truncated. IONOS does **not** constrain decoding to the schema, + so the structure burden is fully on the model — and it is the *heaviest* + burden of all: braces + brackets + keys + quotes + commas + `"t":` + discriminators, every node. + +So the "modern" structured-output channel is the worst surface here, for +exactly the reason the rule predicts: AST JSON is maximal extra bookkeeping +with no redundant cue. + +## Overall finding across all six surfaces + +| surface | green | structure marking | |---|---|---| -| **format 4 — annotated parens** | **7/8** | maximal-explicit (every paren + its depth) | -| plain Form-A — parens | 6/8 | explicit (parens) | +| **format 4 — annotated parens** | **7/8** | maximal-explicit, redundant (every paren + its depth) | +| plain Form-A — parens | 6/8 | explicit, one channel | | M-expressions `head[a; b]` | 4–5/8 | explicit, but `[]`+`;` adds a separator burden | | yamlish / indent (own parser) | 3/8 | implicit (indentation) | | YAML (library) | 2/8 | implicit + special-char collision | +| tool-calling / AST JSON | 2/8 | maximal bookkeeping, verbose, no redundancy | The result is monotone and the opposite of the starting hypothesis. The discriminator is sharper than "explicit vs implicit": it is **redundant cue vs @@ -125,8 +146,10 @@ parens carry the nesting depth redundantly, so the model cannot lose its place (format 4, best). What hurts is either (a) moving the tracking onto an implicit channel it must maintain itself — indentation (yamlish/YAML, worst) — or (b) adding a second thing to keep consistent on top of bracket balance — the -M-expr semicolons (middle). Plain parens sit in between: one explicit channel, -no redundancy, no extra burden. +M-expr semicolons (middle), or (c) maximising the bookkeeping outright — AST +JSON over the tool channel, the heaviest surface and tied for worst, even +though tool-calling is the model's home turf. Plain parens sit in between: one +explicit channel, no redundancy, no extra burden. The L7 closing-token repetition tail is a separate single-shot ceiling, not something any surface here fixes — the bracket-free surfaces fail earlier, diff --git a/experiments/2026-05-12-cross-model-authoring/qwen-tool.md b/experiments/2026-05-12-cross-model-authoring/qwen-tool.md new file mode 100644 index 0000000..fd73d3c --- /dev/null +++ b/experiments/2026-05-12-cross-model-authoring/qwen-tool.md @@ -0,0 +1,546 @@ +# Qwen format probe — tool-calling / structured AST + +**Model:** Qwen/Qwen3-Coder-Next; submit_program tool, forced; AST written directly to .ail.json. + +--- + +## L0_const — ✅ green + +tokens: prompt=1078 completion=114 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "answer", + "params": [], + "type": { + "k": "fn", + "param_modes": [], + "params": [], + "ret": { + "k": "con", + "name": "Int" + }, + "ret_mode": "own", + "effects": [] + }, + "body": { + "lit": { + "kind": "int", + "value": 42 + }, + "t": "lit" + } + } + ] +} +``` + + +## L1_arith — ❌ check + +tokens: prompt=1079 completion=170 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "triple", + "params": [ + "x" + ], + "type": { + "k": "fn", + "param_modes": [ + "own" + ], + "params": [ + { + "k": "con", + "name": "Int" + } + ], + "ret": { + "k": "con", + "name": "Int" + }, + "ret_mode": "own", + "effects": [] + }, + "body": { + "t": "app", + "fn": { + "t": "var", + "name": "*" + }, + "args": [ + { + "t": "var", + "name": "x" + }, + { + "t": "var", + "name": "x" + }, + { + "t": "var", + "name": "x" + } + ] + } + } + ] +} +``` + +**check:** +``` +error: [arity-mismatch] triple: arity mismatch for `*`: expected 2 args, got 3 +``` + +## L2_print — ✅ green + +tokens: prompt=1075 completion=280 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "main", + "params": [], + "type": { + "k": "fn", + "param_modes": [], + "params": [], + "ret": { + "k": "con", + "name": "Unit", + "effects": [ + "IO" + ] + }, + "ret_mode": "own", + "effects": [ + "IO" + ] + }, + "body": { + "lhs": { + "args": [ + { + "lit": { + "kind": "int", + "value": 7 + }, + "t": "lit" + } + ], + "fn": { + "name": "print", + "t": "var" + }, + "t": "app" + }, + "rhs": { + "args": [ + { + "lit": { + "kind": "str", + "value": "\n" + }, + "t": "lit" + } + ], + "op": "io/print_str", + "t": "do" + }, + "t": "seq" + } + } + ] +} +``` + + +## L3_adt — ❌ check + +tokens: prompt=1092 completion=263 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "type", + "name": "Light", + "ctors": [ + { + "name": "Red", + "fields": [] + }, + { + "name": "Green", + "fields": [] + } + ] + }, + { + "kind": "fn", + "name": "code", + "params": [ + "light" + ], + "type": { + "k": "fn", + "param_modes": [ + "own" + ], + "params": [ + { + "k": "con", + "name": "Light" + } + ], + "ret": { + "k": "con", + "name": "Int" + }, + "ret_mode": "own", + "effects": [] + }, + "body": { + "kind": "match", + "scrutinee": { + "name": "light", + "t": "var" + }, + "arms": [ + { + "pat": { + "p": "ctor", + "ctor": "Red", + "fields": [] + }, + "body": { + "kind": "lit", + "lit": { + "kind": "int", + "value": 0 + } + } + }, + { + "pat": { + "p": "ctor", + "ctor": "Green", + "fields": [] + }, + "body": { + "kind": "lit", + "lit": { + "kind": "int", + "value": 1 + } + } + } + ] + } + } + ] +} +``` + +**check:** +``` +Error: schema/parse error in /tmp/abl/m.ail.json: json: missing field `t` at line 82 column 3 + +Caused by: + 0: json: missing field `t` at line 82 column 3 + 1: missing field `t` at line 82 column 3 +``` + +## L4_rec — ❌ bad-json + +tokens: prompt=1104 completion=2500 + +submitted AST: +```json +{ + "module": { + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "type", + "name": "IntList", + "ctors": [ + { + "name": "Nil", + "fields": [] + }, + { + "name": "Cons", + "fields": [ + { + "k": "con", + "name": "Int" + +``` + +**bad-json:** +``` +tool arguments not valid JSON: Unterminated string starting at: line 487 column 143 (char 39567) +``` + +## L5_series_len — ❌ check + +tokens: prompt=1494 completion=763 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "s", + "params": [], + "type": { + "k": "fn", + "param_modes": [], + "params": [], + "ret": { + "k": "con", + "name": "Unit" + }, + "ret_mode": "own", + "effects": [ + "IO" + ] + }, + "body": { + "args": [ + { + "name": "s", + "t": "var" + }, + { + "kind": "float", + "bits": "4000000000000000", + "t": "lit" + } + ], + "fn": { + "name": "Series.push", + "t": "var" + }, + "t": "app" + }, + "t": "let", + "value": { + "args": [ + { + "kind": "type", + "value": { + "k": "con", + "name": "Float" + } + }, + { + "kind": "value", + "value": { + "lit": { + "kind": "int", + "value": 3 + }, + "t": "lit" + } + }, + "t" + ], + "t": "new", + "type": "Series" + } + } + ] +} +``` + +**check:** +``` +Error: schema/parse error in /tmp/abl/m.ail.json: json: missing field `lit` at line 67 column 3 + +Caused by: + 0: json: missing field `lit` at line 67 column 3 + 1: missing field `lit` at line 67 column 3 +``` + +## L6_series_at — ❌ check + +tokens: prompt=1502 completion=797 + +submitted AST: +```json +{ + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "main", + "params": [], + "type": { + "k": "fn", + "param_modes": [], + "params": [], + "ret": { + "k": "con", + "name": "Unit" + }, + "ret_mode": "own", + "effects": [ + "IO" + ] + }, + "body": { + "body": { + "body": { + "body": { + "lhs": { + "args": [ + { + "name": "s3", + "t": "var" + }, + { + "args": [ + { + "lit": { + "kind": "int", + "value": 0 + }, + "t": "lit" + }, + "fn", + "name", + "Series", + "at", + "t", + "var", + "app", + "app" + ], + "fn": { + "name": "print", + "t": "var" + }, + "t": "app" + }, + "rhs", + "args", + "lit", + "str", + "value", + "\n", + "t", + "lit", + "op", + "io/print_str", + "t", + "do", + "seq" + ], + "name": "s3", + "t": "let", + "value": { + "args": [ + { + "name": "s2", + "t": "var" + }, + { + "lit": { + "kind": "float", + "bits": "3ff000000000000" + }, + +``` + +**check:** +``` +Error: schema/parse error in /tmp/abl/m.ail.json: json: missing field `t` at line 167 column 3 + +Caused by: + 0: json: missing field `t` at line 167 column 3 + 1: missing field `t` at line 167 column 3 +``` + +## L7_sma — ❌ bad-json + +tokens: prompt=1588 completion=2500 + +submitted AST: +```json +{ + "module": { + "schema": "ailang/v0", + "name": "m", + "imports": [], + "defs": [ + { + "kind": "fn", + "name": "main", + "params": [], + "type": { + "k": "fn", + "param_modes": [], + "params": [], + "ret": { + "k": "con", + "name": "Unit" + }, + "ret_mode": "own", + "effects": [ + +``` + +**bad-json:** +``` +tool arguments not valid JSON: Expecting property name enclosed in double quotes: line 347 column 129 (char 110384) +``` + + +--- +## Ladder + +- L0_const: ✅ green +- L1_arith: ❌ check +- L2_print: ✅ green +- L3_adt: ❌ check +- L4_rec: ❌ bad-json +- L5_series_len: ❌ check +- L6_series_at: ❌ check +- L7_sma: ❌ bad-json \ No newline at end of file diff --git a/experiments/2026-05-12-cross-model-authoring/qwen_tool.py b/experiments/2026-05-12-cross-model-authoring/qwen_tool.py new file mode 100644 index 0000000..7639d16 --- /dev/null +++ b/experiments/2026-05-12-cross-model-authoring/qwen_tool.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Format probe — tool-calling / structured AST. Instead of writing surface +syntax (where bracket-tracking is the wall), Qwen submits the program as the +canonical AILang AST JSON via a `submit_program` tool. Hypothesis: the model +produces structured JSON through the tool channel (which it is trained hard on, +and which the API validates) more reliably than free-text syntax. No converter +risk: the tool argument IS the .ail.json, written and `ail check`ed directly. + +Same ablation ladder. Reads $IONOS_API_TOKEN (never printed). Live IONOS. +""" +from __future__ import annotations +import json, os, re, subprocess, urllib.request +from pathlib import Path + +ENDPOINT="https://openai.inference.de-txl.ionos.com/v1/chat/completions" +MODEL="Qwen/Qwen3-Coder-Next" +REPO=Path(__file__).resolve().parents[2] +AIL=os.environ.get("AIL_BIN", str(REPO/"target/debug/ail")) +DOC=REPO/"experiments/2026-05-12-cross-model-authoring/qwen-tool.md" + +def ail_ast(src): + p=Path("/tmp/abl/_fs.ail"); p.write_text(src+"\n") + return subprocess.run([AIL,"parse",str(p)],capture_output=True,text=True).stdout.strip() + +A="""(module demo_a + (fn double + (type (fn-type (params (own (con Int))) (ret (own (con Int))))) + (params x) (body (app + x x))) + (fn main + (type (fn-type (params) (ret (own (con Unit))) (effects IO))) + (params) (body (seq (app print (app double 21)) (do io/print_str "\\n")))))""" +B="""(module demo_b + (data IntList (ctor Nil) (ctor Cons (con Int) (con IntList))) + (fn sum + (type (fn-type (params (own (con IntList))) (ret (own (con Int))))) + (params xs) (body (match xs (case (pat-ctor Nil) 0) (case (pat-ctor Cons h t) (app + h (app sum t)))))) + (fn main + (type (fn-type (params) (ret (own (con Unit))) (effects IO))) + (params) (body (let xs (term-ctor IntList Cons 1 (term-ctor IntList Cons 2 (term-ctor IntList Nil))) + (seq (app print (app sum xs)) (do io/print_str "\\n"))))))""" +S="""(module demo_s + (fn main + (type (fn-type (params) (ret (own (con Unit))) (effects IO))) + (params) (body (let s (new Series (con Float) 3) + (let s2 (app Series.push s 1.0) + (let s3 (app Series.push s2 2.0) + (seq (app print (app Series.len s3)) (do io/print_str "\\n"))))))))""" + +TOOL={"type":"function","function":{ + "name":"submit_program", + "description":"Submit a complete AILang program as its canonical AST JSON.", + "parameters":{"type":"object","properties":{ + "module":{"type":"object","description":"The full module AST: an object with keys schema, name, imports, defs."}}, + "required":["module"]}}} + +def sys_base(extra=""): + return ("You author AILang programs as a canonical Abstract Syntax Tree in JSON, " + "submitted with the submit_program tool (the `module` argument is the full AST: " + "an object with keys schema, name, imports, defs). The AST node forms are shown " + "by these complete, valid example programs.\n\n" + "Example 1 AST:\n"+ail_ast(A)+"\n\nExample 2 AST:\n"+ail_ast(B)+"\n"+extra+ + "\nAlways call submit_program with the full module AST.") +SYS=sys_base() +SYS_SERIES=sys_base("\nExample 3 AST (uses the Series library — `new` term with type-arg " + "Float and capacity; Series.push/Series.at/Series.len/Series.total_count called via " + "app on a `var` named e.g. \"Series.push\"):\n"+ail_ast(S)+"\n") + +SMA_OUT="3.0\n5.33333\n5.66667\n5.33333\n" +SMA_TASK=("Write a program named m computing a simple moving average, window 3, over the " + "float stream 1.0, 5.0, 3.0, 8.0, 6.0, 2.0 using the Series library: create a Float " + "Series lookback 3, push each value in turn, and after each push once at least 3 " + "values have been pushed, print the average of the most recent 3 (their sum divided " + "by 3.0) on its own line. Expected stdout: "+repr(SMA_OUT)) + +STAGES=[ + ("L0_const",SYS,"Write a program named m with a function answer taking no parameters that returns the Int 42.",None), + ("L1_arith",SYS,"Write a program named m with a function triple taking one Int parameter, returning it multiplied by 3.",None), + ("L2_print",SYS,"Write a program named m whose main prints the Int 7 on its own line.","7\n"), + ("L3_adt",SYS,"Write a program named m with an ADT Light (ctors Red, Green) and a function code returning Int 0 for Red and 1 for Green via match.",None), + ("L4_rec",SYS,"Write a program named m with an IntList ADT (Nil/Cons), a function length counting elements by recursion, and a main that prints the length of the list [1,2,3] on its own line.","3\n"), + ("L5_series_len",SYS_SERIES,"Write a program named m using the Series library: create a Float Series lookback 3, push 1.0 then 2.0 then 3.0, then print its Series.len on its own line.","3\n"), + ("L6_series_at",SYS_SERIES,"Write a program named m using the Series library: create a Float Series lookback 3, push 1.0 then 2.0 then 3.0, then print (Series.at on index 0, the newest) on its own line.","3.0\n"), + ("L7_sma",SYS_SERIES,SMA_TASK,SMA_OUT), +] + +def call(system,task): + body=json.dumps({"model":MODEL,"temperature":0.2,"max_tokens":2500, + "messages":[{"role":"system","content":system},{"role":"user","content":task}], + "tools":[TOOL],"tool_choice":{"type":"function","function":{"name":"submit_program"}}}).encode() + req=urllib.request.Request(ENDPOINT,data=body,headers={ + "Authorization":f"Bearer {os.environ['IONOS_API_TOKEN']}","Content-Type":"application/json"}) + with urllib.request.urlopen(req,timeout=120) as r: d=json.loads(r.read()) + return d["choices"][0]["message"], d.get("usage",{}) + +def evaluate(msg,expected): + tc=msg.get("tool_calls") + if not tc: return "no-tool-call",f"model returned content instead: {str(msg.get('content'))[:200]}","" + try: args=json.loads(tc[0]["function"]["arguments"]) + except Exception as e: return "bad-json",f"tool arguments not valid JSON: {e}",tc[0]["function"]["arguments"][:400] + module=args.get("module",args) + if not isinstance(module,dict): return "bad-shape",f"module not an object: {type(module).__name__}",json.dumps(args)[:400] + module.setdefault("schema","ailang/v0"); module.setdefault("imports",[]) + name=module.get("name","m") + f=Path(f"/tmp/abl/{name}.ail.json"); f.parent.mkdir(exist_ok=True) + f.write_text(json.dumps(module,indent=2)+"\n") + pretty=json.dumps(module,indent=1) + chk=subprocess.run([AIL,"check",str(f)],capture_output=True,text=True) + if chk.returncode!=0: return "check",(chk.stderr or chk.stdout).strip(),pretty + if expected is None: return "green","(check-only)",pretty + run=subprocess.run([AIL,"run",str(f)],capture_output=True,text=True) + if run.returncode!=0: return "run",(run.stderr or run.stdout).strip(),pretty + if run.stdout!=expected: return "output",f"got {run.stdout!r}, expected {expected!r}",pretty + return "green",run.stdout,pretty + +def main(): + log=["# Qwen format probe — tool-calling / structured AST\n", + f"**Model:** {MODEL}; submit_program tool, forced; AST written directly to .ail.json.\n","---\n"] + summary=[] + for sid,sysmsg,task,exp in STAGES: + msg,usage=call(sysmsg,task) + stage,fb,pretty=evaluate(msg,exp) + ok=stage=="green"; mark="✅ green" if ok else f"❌ {stage}" + summary.append((sid,mark)); print(f"{sid}: {mark}") + log+=[f"## {sid} — {mark}", + f"\ntokens: prompt={usage.get('prompt_tokens','?')} completion={usage.get('completion_tokens','?')}\n", + "submitted AST:\n```json\n"+pretty[:1400]+"\n```\n", + ("" if ok else f"**{stage}:**\n```\n{fb[:600]}\n```\n")] + log+=["\n---\n## Ladder\n"]+[f"- {s}: {m}" for s,m in summary] + DOC.write_text("\n".join(log)); print("doc →",DOC) + +if __name__=="__main__": + main()