{
  "$schema": "https://touch.long-arena.com/ai-evaluations/schema.json",
  "schemaVersion": "1.0",
  "kind": "single-task-expert-assessment",
  "id": "max-long-task-2026-09",
  "language": "en",
  "canonicalUrl": "https://touch.long-arena.com/en/ai-evaluations/max-long-task-2026-09/",
  "title": "A long-task agent: strong execution, incomplete delivery",
  "description": "An independent review of the Max daily text workspace and sales-expert capability plans. Substantial implementation did not yet add up to reliable acceptance.",
  "reviewDate": "2026-09-11",
  "reviewWindow": [
    "2026-09-10",
    "2026-09-11"
  ],
  "planDates": [
    "2026-09-08",
    "2026-09-09"
  ],
  "sampleSize": 1,
  "identity": {
    "reviewer": {
      "agent": "Codex",
      "model": "GPT-6 Astra"
    },
    "subject": {
      "agent": "dsh",
      "model": "DeepSeek V4.1 Flash"
    },
    "source": "commissioner-declared",
    "runIdentityVerified": false,
    "disclosure": "Agent and model identities were supplied by the commissioner on September 11, 2026. Execution sessions, request IDs and pinned model versions were not independently verified."
  },
  "method": {
    "id": "identity-disclosed-after-scoring",
    "label": "Blind review; identity disclosed after scoring",
    "maskingVerified": false,
    "limitations": "This is a single-case review with identity disclosed after scoring, under the commissioner’s stated blind-review method. There was no retained preregistration, random assignment or independently verified masking protocol. Repository tool markers could reveal clues, so complete blinding and double blinding are not established.",
    "notAHeadToHeadModelComparison": true,
    "comparisonLimit": "Codex was the reviewer, not a competing implementation. This is not a same-task Codex-versus-DeepSeek experiment, a base-model ranking or a vendor endorsement.",
    "steps": [
      {
        "title": "Compare the plan with delivery",
        "body": "Distinguish an implementation claim from a reachable business entry point, persisted result and completed acceptance item."
      },
      {
        "title": "Rerun bounded engineering checks",
        "body": "Use type checks and existing targeted tests. Keep local checks separate from earlier real-login or live-model records."
      },
      {
        "title": "Challenge critical invariants",
        "body": "Probe trusted identity, scope transitions, repeated normalization, persistence round trips and the failure behavior of release checks."
      },
      {
        "title": "Score before adding identity metadata",
        "body": "The recorded review and numerical assessment preceded the identity information supplied for this publication. Identity disclosure did not change the scores."
      }
    ]
  },
  "assessment": {
    "verdict": "blocked",
    "verdictScope": "reviewed-historical-snapshot",
    "difficulty": {
      "value": 8.5,
      "maximum": 10
    },
    "performance": {
      "value": 60,
      "maximum": 100,
      "method": "weighted-expert-judgment"
    },
    "dimensions": [
      {
        "id": "execution",
        "weight": 15,
        "score": 85,
        "label": "Task decomposition and sustained execution",
        "reason": "The agent sustained work across modules and adjusted the learning scope when available feedback was insufficient.",
        "contribution": 12.75
      },
      {
        "id": "implementation",
        "weight": 25,
        "score": 75,
        "label": "Implementation and integration",
        "reason": "There was working code and meaningful compatibility evidence, but some implemented components were not wired into a complete user journey.",
        "contribution": 18.75
      },
      {
        "id": "security",
        "weight": 20,
        "score": 35,
        "label": "Security boundaries and data consistency",
        "reason": "Trusted identity, document scope and provenance round trips exposed major unresolved invariants, including inherited behavior missed by this acceptance.",
        "contribution": 7
      },
      {
        "id": "testing",
        "weight": 15,
        "score": 65,
        "label": "Test design and counterexamples",
        "reason": "Real-runtime evaluation and can-fail checks were improvements. Adjacent states, round trips and failures of the checks themselves remained under-tested.",
        "contribution": 9.75
      },
      {
        "id": "journey",
        "weight": 10,
        "score": 40,
        "label": "Business closure and UI acceptance",
        "reason": "A visible component or loading screen is not a completed task. Write/readback, recovery and rollback evidence remained incomplete.",
        "contribution": 4
      },
      {
        "id": "calibration",
        "weight": 10,
        "score": 35,
        "label": "Evidence accuracy and conclusion calibration",
        "reason": "Statements such as “the only blocker” and “structurally impossible” exceeded the scope of the retained evidence.",
        "contribution": 3.5
      },
      {
        "id": "discipline",
        "weight": 5,
        "score": 85,
        "label": "Operational discipline and traceability",
        "reason": "Authorization and branch boundaries were respected, and changes were recorded. Refusing an unauthorized registration was correct, not a penalty.",
        "contribution": 4.25
      }
    ],
    "scale": "On this rubric, 60–69 means substantial output that still needs strong independent review and rework. It does not mean acceptance passed.",
    "stages": [
      {
        "id": "a",
        "planDate": "2026-09-08",
        "difficulty": 7,
        "score": 55,
        "title": "Stage A: daily text workspace",
        "body": "Document state, revision history, localization, narrow-container layout and mobile navigation. Local UI fixes were real; scope isolation and complete editing journeys were still missed."
      },
      {
        "id": "b",
        "planDate": "2026-09-09",
        "difficulty": 9,
        "score": 62,
        "title": "Stage B: sales-expert capability layer",
        "body": "Process facts, B2B/B2C flows, authorization, provenance, migration preparation, governed tools and evaluations. The integration and safety constraints made this the harder stage."
      }
    ],
    "stageScoreLimit": "Stage scores are supporting judgments, not acceptance ratios. The overall score is calculated from the seven dimensions below, not by averaging the two stages."
  },
  "evidence": {
    "existingTests": {
      "frontend": 681,
      "backend": 116,
      "industry": 43
    },
    "totalExistingTests": 840,
    "targetedChecksFailed": 7,
    "failureClasses": 4,
    "severeGroups": 3,
    "interpretation": "Passing checks and failing counterexamples answer different questions. They are not combined into a synthetic success rate.",
    "layers": [
      {
        "title": "Rerun locally",
        "body": "Type checks, targeted frontend/backend/industry tests, and synthetic boundary probes. No real business writes were introduced by this independent review."
      },
      {
        "title": "Inspected, not rerun live",
        "body": "Earlier real-login screenshots and live-model evaluation records were reviewed. A function precheck was not counted as a browser journey; a loading screenshot was not accepted as a completed task."
      },
      {
        "title": "Still not established",
        "body": "Complete business write/readback journeys, exception recovery, rollback and the other original acceptance gaps. Registration authorization alone cannot turn the result into Ready."
      }
    ]
  },
  "technicalFindings": [
    {
      "id": "identity",
      "title": "01 / Identity must remain a trusted fact",
      "finding": "An authorization decision could be affected by untrusted presentation metadata rather than only the trusted actor and current organization’s permissions.",
      "lesson": "Hold trusted identity constant while varying labels, untrusted metadata and neighboring organization membership. None may upgrade the granted authority.",
      "invariant": "authorize(trustedIdentity, currentScope)\n// Display labels are not authorization facts."
    },
    {
      "id": "scope",
      "title": "02 / A revision number is not a document identity",
      "finding": "Workspace state was insufficiently isolated across documents and organization changes. A matching revision number did not prove that the content belonged to the active document.",
      "lesson": "Exercise two documents with the same revision and two scopes with different drafts. Assert cleanup at scope transitions and ownership immediately before using content.",
      "invariant": "revisionKey = (organization, user, workspace, document, revision)\nrestore(scopeB) must not retain scopeA.content"
    },
    {
      "id": "provenance",
      "title": "03 / Normalization must not invent evidence",
      "finding": "A default provenance value could become “stated” after another normalization or persistence round trip. Reported coverage rose without new source evidence.",
      "lesson": "Preserve the distinction between unknown and verified data through repeated transformations and normal reads. Test the actual read chain, not only the first normalization.",
      "invariant": "coverage(normalize(normalize(x))) == coverage(normalize(x))\ncoverage(decode(encode(x))) == coverage(x)\n// No new evidence means no higher verified coverage."
    },
    {
      "id": "gate",
      "title": "04 / A check that cannot read must not pass",
      "finding": "A change-inspection error was treated as an empty result. The process returned success without successfully establishing whether a relevant change existed.",
      "lesson": "Inject read errors and oversized inputs. A gate must fail closed and propagate a nonzero exit status; test the checker as well as the business code.",
      "invariant": "readFailure => checkFailure\ncheckFailure => nonzeroExit\n// An unreadable change set is not an empty change set."
    }
  ],
  "strengths": [
    "Implemented meaningful cross-module changes and repaired real localization and responsive-layout issues.",
    "Adjusted automatic learning to collection-first when real feedback was too sparse, with a switch-off test independent of the threshold.",
    "Respected authorization limits instead of creating a real organization merely to obtain a passing result."
  ],
  "weaknesses": [
    "Local success was sometimes promoted to a full-chain completion claim.",
    "Single-case fixes did not consistently cover adjacent inputs, scope transitions and persistence round trips.",
    "Acceptance evidence did not always demonstrate the actual business end state."
  ],
  "authorizationDiscipline": "No points were deducted for refusing unauthorized real-world registration or writes. The deductions concern unwired functionality, missed checks, overstated passes and omitted blockers.",
  "conclusion": {
    "title": "Use the agent to implement; separate the acceptance decision.",
    "body": "Keep independent review for trusted identity, data isolation, evidence semantics and real user journeys. A long-task agent can contribute substantial engineering work without yet being a reliable sole signatory for its own delivery."
  },
  "limitations": [
    "One delivery snapshot, assessed retrospectively with an expert rubric. No repeat-run distribution, control group or statistical model comparison.",
    "Precise tool versions, model snapshots, prompts, inference settings, token usage, cost and complete elapsed-time measurements were not verified for this publication. No productivity or cost claim is made.",
    "The repository and harness context can affect the outcome. Findings cannot be attributed solely to DeepSeek V4.1 Flash or generalized to every dsh run.",
    "This is a historical implementation review, not a claim that the current production system has been attacked or that subsequent fixes were already verified."
  ],
  "sources": [
    "Implementation plans dated September 8 and 9, 2026, together with the delivered code, tests and acceptance materials.",
    "Independent local review on September 10–11, 2026, followed by the seven-dimension scoring assessment.",
    "Commissioner-supplied identity and blind-review metadata on September 11, 2026."
  ],
  "privateEvidenceExcluded": "Private code, raw conversations, credentials, customer identifiers and actionable vulnerability details are not distributed. The PDF is a score summary, not the full private audit evidence pack.",
  "representations": {
    "json": "https://touch.long-arena.com/en/ai-evaluations/max-long-task-2026-09/article.json",
    "text": "https://touch.long-arena.com/en/ai-evaluations/max-long-task-2026-09/article.txt",
    "summaryPdf": "https://touch.long-arena.com/reports/2026-09-11-agent-long-task-assessment.pdf",
    "summaryPdfLanguage": "zh-CN"
  }
}
