{
  "benchmark": "Benchmark 1",
  "title": "ERPNext collections-to-payment assistant",
  "publishedAt": "2026-09-27",
  "orderMeaning": "Runs are listed in publication order. The order is not a numerical ranking.",
  "runs": [
    {
      "modelId": "qwen-next-fp8",
      "model": "Qwen 3.8 Flash Next FP8",
      "configuration": "FP8 quantization, xhigh reasoning, agent harness",
      "thinkingLevel": "xhigh",
      "powerDraw": "Not measured during this run.",
      "status": "Live chain verified; operator access blocked",
      "completed": "The complete read, allocation, review, and assistant integration path, including a patched assistant image that ran in a disposable container.",
      "didRight": [
        "Reused existing permission-checked reads and the native payment-entry path.",
        "Kept confirmation before every write and, in reviewer diagnostics, the real Qwen model called ERPNext and returned an exact product count."
      ],
      "wentWrong": [
        "A normal operator can see the imm-business preset but receives Model not found because access to its base model was not granted.",
        "Needed two neutral completion nudges to stop polishing and report; some internal coverage claims could not be reconstructed from the artifact alone."
      ],
      "unitTestQuality": "Strong layered coverage: pure Decimal allocation tests, ERP integration checks, frontend unit and browser checks, and a real patched-container harness. One weakness is that the 100-reference allocation cap is asserted as expected behaviour instead of being rejected or surfaced to the operator.",
      "backendBugs": "A reviewed edge case remains: oldest-first allocation silently considers at most 100 references. An account with more references can retain unallocated money even while later invoices remain outstanding.",
      "verification": "Backend 45/45; frontend green with 244 tests; deterministic assistant checks 4/4; patched image built and the reviewer reproduced the real-container harness. A normal operator prompt failed before inference with Model not found. With a temporary reviewer-only role elevation, the real Qwen model called ERPNext’s product_count tool and returned an exact answer; no implementation code was changed.",
      "humanHelp": "Two completion nudges; no code or technical steering.",
      "audit": {
        "modelUse": "Yes — real Qwen",
        "liveModelEndToEnd": "Conditional — diagnostic access",
        "continueInterventions": 2,
        "uiComplete": "Yes",
        "uiBackendIntegration": "Yes — diagnostic path",
        "servedRequests": "Yes — after access elevation",
        "implementationUnitTests": 30,
        "testQuality": "Strong"
      },
      "runnableUiCapture": true
    },
    {
      "modelId": "mimo-v2-6-flash",
      "model": "MiMo V2.6 Flash",
      "configuration": "Local Flash variant, default model reasoning, agent harness",
      "thinkingLevel": "Default model setting; no explicit effort level was selected.",
      "powerDraw": "Not measured during this run.",
      "status": "Reviewer-reproduced after configuration repair",
      "completed": "The collections feature, real central-ui embed, model response, and ERPNext tool call, reproduced after a reviewer configuration repair.",
      "didRight": [
        "Implemented balance lookup, the 90-day bill window, oldest-first partial allocation, confirmation, and the native payment path.",
        "Built a real assistant image whose OAuth, central-ui embed, handshake, chat-only mode, responsive layout, model response, and ERPNext tool call were reproduced."
      ],
      "wentWrong": [
        "As provisioned, an operator could see the imm-business preset but not its base model, so the first real prompt failed with Model not found.",
        "A fresh OpenWebUI volume could not be provisioned because the inherited setup passed an unawaited password-hash coroutine to SQLite."
      ],
      "unitTestQuality": "Good Decimal allocation edge cases, real ERP integration coverage, and broad frontend checks. The original assistant suite was mocked, so it missed the base-model access and fresh-provisioning failures that the reviewer found in real containers.",
      "backendBugs": "No defect was confirmed in the reviewed allocation or native payment path. The confirmed failures were integration and deployment defects: a missing base-model access grant and a broken fresh-volume provisioning path.",
      "verification": "Backend unit 41/41; real ERP assistant integration 12/12; frontend green with 237 tests and one nonfatal warning; mocked assistant checks 10/10. The reviewer built and ran the real image, reproduced the access failure, applied a configuration-only grant, then observed a model response and a successful ERPNext find_records call.",
      "humanHelp": "No extra implementation turns. After the run, the reviewer made one configuration-only base-model access repair to reproduce the flow; no MiMo code or logic changed.",
      "audit": {
        "modelUse": "Yes",
        "liveModelEndToEnd": "Yes — after config repair",
        "continueInterventions": 0,
        "uiComplete": "Yes",
        "uiBackendIntegration": "Yes — observed live",
        "servedRequests": "Yes — model + ERP tool",
        "implementationUnitTests": 29,
        "testQuality": "Good"
      },
      "runnableUiCapture": true
    },
    {
      "modelId": "deepseek-v4-flash",
      "model": "DeepSeek V4 Flash 0731",
      "configuration": "Local Flash variant, max reasoning, agent harness",
      "thinkingLevel": "max",
      "powerDraw": "Not measured during this run.",
      "status": "Incomplete at the fair cutoff",
      "completed": "A central-ui assistant drawer and expanded view, an OpenWebUI chat patch, a top-level sign-in handoff, and an initial message bridge.",
      "didRight": [
        "Kept one persistent iframe so a conversation could survive drawer close and expansion, and kept ERPNext sign-in outside the iframe.",
        "Added seven focused unit cases for parent-side message source/origin validation and cache-invalidation mapping."
      ],
      "wentWrong": [
        "The child accepted a parent origin claimed in a message and sent its initial handshake to *, while message listeners were not cleaned up.",
        "The loading/auth state could report success for an error page, the embedded surface was not a verified real chat-only mode, and no live model-to-ERPNext tool flow was demonstrated."
      ],
      "unitTestQuality": "Partial. Seven new bridge unit cases cover parent-side source/origin rejection and mutation-to-cache mapping. They encode one invalid query prefix and do not cover child-side origin trust, wildcard handshakes, listener cleanup, authentication return timing, malformed URLs, real embed chrome, model access, or a live ERPNext tool flow.",
      "backendBugs": "The pending-review lookup limits events to 25 before filtering for OpenWebUI conversations. Newer manual review events can therefore hide an older eligible assistant review. The pure money helper also lacks an ERP-rounding boundary test.",
      "verification": "At the fair cutoff, the frontend check and assistant-shell browser test were reported green, but the assistant checks were deterministic or mocked and no live local-model plus ERPNext tool flow had run. Direct inspection confirmed the origin-trust, listener, auth-state, embed, and deployment gaps. All later technically guided fixes and live diagnostics are excluded.",
      "humanHelp": "Two neutral completion and verification prompts are included. Three later technical finding rounds, and every resulting change, are excluded from this result.",
      "audit": {
        "modelUse": "Configured; not live-proven",
        "liveModelEndToEnd": "No",
        "continueInterventions": 2,
        "uiComplete": "Partial",
        "uiBackendIntegration": "Implemented; not proven",
        "servedRequests": "No",
        "implementationUnitTests": 7,
        "testQuality": "Partial"
      },
      "runnableUiCapture": false
    },
    {
      "modelId": "qwen-next-nvfp4",
      "model": "Qwen 3.8 Flash Next NVFP4",
      "configuration": "NVFP4 quantization, xhigh reasoning, agent harness",
      "thinkingLevel": "xhigh",
      "powerDraw": "Not measured during this run.",
      "status": "Partial: chat bridge missing",
      "completed": "A defensive server-side allocation implementation with extensive backend tests.",
      "didRight": [
        "Passed 60 backend checks and 249 frontend tests.",
        "Included oldest-first allocation and explicit over-allocation rejection."
      ],
      "wentWrong": [
        "The chat bridge was absent, so the assistant UI could not reach the server implementation.",
        "The browser suite required unavailable deployment secrets and could not be independently rerun."
      ],
      "unitTestQuality": "Detailed Decimal money tests cover invalid inputs, ordering, ties, remainders, non-mutation, and invariants, with additional permission and confirmation contracts. Several integration claims are source-text assertions, and the missing chat bridge left the browser path untested.",
      "backendBugs": "No backend defect was confirmed in the reviewed allocation path. The blocking defect was integration-level: the assistant UI had no chat bridge to reach the server implementation.",
      "verification": "Backend 60/60 and frontend green; browser evidence was not independently runnable; direct inspection confirmed the missing chat bridge.",
      "humanHelp": "One completion nudge; no technical steering.",
      "audit": {
        "modelUse": "Configured; not reached",
        "liveModelEndToEnd": "No",
        "continueInterventions": 1,
        "uiComplete": "No",
        "uiBackendIntegration": "No — bridge missing",
        "servedRequests": "No",
        "implementationUnitTests": 60,
        "testQuality": "Good"
      },
      "runnableUiCapture": false
    },
    {
      "modelId": "qwen-27b-bf16",
      "model": "Qwen 3.8 27B BF16",
      "configuration": "Unquantized BF16 weights, agent harness",
      "thinkingLevel": "xhigh",
      "powerDraw": "Not measured during this run.",
      "status": "Incomplete",
      "completed": "A backend slice whose available unit checks passed, plus partial permission and payment-entry work.",
      "didRight": [
        "Passed 40 backend checks.",
        "Kept permission and payment-entry safeguards in the implementation."
      ],
      "wentWrong": [
        "The frontend failed TypeScript and the assistant image did not build.",
        "The session needed twelve extra turns and ended with an incomplete working tree before browser verification."
      ],
      "unitTestQuality": "The pure allocation tests cover ordering, partial payments, advances, determinism, and basic invariants. They do not cover retry idempotency or the mutation of stored task values, and the frontend and artifact never reached a green gate.",
      "backendBugs": "Automatic allocation mutates the request values after the idempotency comparison. Retrying the same request ID with the original oldest-first intent can be rejected as different values.",
      "verification": "Backend 40/40; frontend typecheck failed; artifact build failed; browser work was not reached; repository remained dirty.",
      "humanHelp": "Twelve turns: permission, harness, and resume unblocks, one security-boundary reminder, and one instruction to stop looping and finish.",
      "audit": {
        "modelUse": "Attempted; not reached",
        "liveModelEndToEnd": "No",
        "continueInterventions": 12,
        "uiComplete": "No",
        "uiBackendIntegration": "No",
        "servedRequests": "No",
        "implementationUnitTests": 17,
        "testQuality": "Partial"
      },
      "runnableUiCapture": false
    },
    {
      "modelId": "gemma-4-31b-bf16",
      "model": "Gemma 4 31B BF16",
      "configuration": "Unquantized BF16 weights, default model reasoning, agent harness",
      "thinkingLevel": "Default model setting; no explicit effort level was selected.",
      "powerDraw": "Not measured during this run.",
      "status": "Incomplete",
      "completed": "A narrow backend slice whose available unit checks passed.",
      "didRight": [
        "Passed 31 backend checks.",
        "Requested no human assistance."
      ],
      "wentWrong": [
        "The frontend had lint and parser failures, no assistant integration was completed, and no commit was made.",
        "The required write path was absent."
      ],
      "unitTestQuality": "No feature-specific unit test was added for the new automatic-allocation path. The passing backend count belongs to the pre-existing suite and does not verify Gemma’s new logic.",
      "backendBugs": "Automatic allocation silently overwrites explicit allocation rows instead of rejecting the ambiguity, and it mutates request values in a way that breaks same-request idempotent retries.",
      "verification": "Backend 31/31; frontend failed with ten errors and three warnings; no browser or assistant artifact existed; repository remained dirty.",
      "humanHelp": "No extra turns, but the run stopped early.",
      "audit": {
        "modelUse": "Not reached",
        "liveModelEndToEnd": "No",
        "continueInterventions": 0,
        "uiComplete": "No",
        "uiBackendIntegration": "No",
        "servedRequests": "No",
        "implementationUnitTests": 0,
        "testQuality": "Poor"
      },
      "runnableUiCapture": false
    }
  ]
}
