{
  "schemaVersion": "dxrg-trading-agent-benchmark-audit-registry/v1",
  "version": "1.4.0",
  "title": "DXRG Trading-Agent Benchmark Audit Registry",
  "description": "Eight review units applied to three registered trading-agent benchmarks, with three frozen primary-source audits and exact disclosure locators.",
  "publishedDate": "2026-08-12",
  "modifiedDate": "2026-08-24",
  "publisher": "DX Research Group",
  "canonicalPage": "https://www.dxrg.ai/blogs/how-to-evaluate-an-ai-trading-agent",
  "methodology": "Register one benchmark version and one intended claim. Review primary documentation against each unit, preserve exact locators, record disclosed and undisclosed fields separately, and issue a scoped finding only after all required evidence has been checked. Result fields remain null until that source review is complete.",
  "evidenceBoundary": "A completed row records what a frozen primary source discloses. It is not an independent rerun, result validation, benchmark ranking, endorsement, or rejection; unrun rows carry no finding.",
  "versionHistory": [
    {
      "version": "1.0.0",
      "releasedAt": "2026-08-12",
      "changeClass": "METHOD_AND_REGISTRY_PUBLICATION",
      "summary": "Published the eight-unit source-disclosure method, registered three benchmark reviews, and left every audit result UNRUN."
    },
    {
      "version": "1.1.0",
      "releasedAt": "2026-08-15",
      "changeClass": "SOURCE_AUDIT_ADDED",
      "summary": "Added the frozen Agent Market Arena v2 source-disclosure audit with exact locators and scoped findings."
    },
    {
      "version": "1.2.0",
      "releasedAt": "2026-08-18",
      "changeClass": "SOURCE_AUDIT_ADDED",
      "summary": "Added the frozen KTD-Fin v1 source-disclosure audit with exact locators and scoped findings."
    },
    {
      "version": "1.3.0",
      "releasedAt": "2026-08-21",
      "changeClass": "SOURCE_AUDIT_ADDED",
      "summary": "Added the frozen CLQT v1 source-disclosure audit and completed the three-benchmark review set."
    },
    {
      "version": "1.4.0",
      "releasedAt": "2026-08-24",
      "changeClass": "PROVENANCE_METADATA_ADDED",
      "summary": "Added this version history, the correction policy, and the public correction log without changing an audit finding."
    }
  ],
  "correctionPolicy": {
    "route": "mailto:poof@dxrg.ai",
    "method": "Append a dated correction record that names the affected registry version, benchmark, fields changed, reason, and replacement text. Preserve the prior version in this history and issue a new registry version.",
    "materialChangeRule": "A change to a source version, locator, disclosure state, finding, verdict, or evidence boundary is material and requires a correction record."
  },
  "corrections": [],
  "requiredAuditHeader": [
    "benchmark title and immutable version",
    "primary source and exact locator",
    "claim the benchmark result is being used to support",
    "reviewer and review date",
    "disclosure state for every audit unit",
    "scoped finding and unresolved limitations"
  ],
  "disclosureStates": [
    "DISCLOSED",
    "PARTIAL",
    "UNDISCLOSED",
    "NOT_APPLICABLE",
    "UNRUN"
  ],
  "sources": [
    {
      "id": "nist-ai-800-2",
      "title": "Practices for Automated Benchmark Evaluations of Language Models",
      "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.800-2.ipd.pdf",
      "version": "Initial Public Draft, January 2026",
      "locator": "Introduction; Practices 1.1, 1.2, 2.1, 3.1, 3.2, and 3.3",
      "use": "Government draft practices for defining the measurement target, selecting a fitting benchmark, fixing the protocol, quantifying uncertainty, sharing details, and qualifying claims.",
      "boundary": "An initial public draft for automated language-model and agent benchmarks, not a final standard or a trading-specific prescription."
    },
    {
      "id": "nist-ai-800-3",
      "title": "Expanding the AI Evaluation Toolbox with Statistical Models",
      "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.800-3.pdf",
      "version": "February 2026",
      "locator": "Abstract; Sections 1, 3, 4, and 6",
      "use": "Government measurement research distinguishing fixed-benchmark accuracy from generalized accuracy and examining uncertainty estimates.",
      "boundary": "General AI benchmark measurement research; its statistical examples do not establish trading-agent performance or external validity."
    },
    {
      "id": "agentic-trading-survey",
      "title": "Agentic Trading: When LLM Agents Meet Financial Markets",
      "url": "https://arxiv.org/abs/2605.19337",
      "version": "v2",
      "locator": "Abstract; protocol-coded evidence map; reproducibility audit; reporting checklist",
      "use": "Trading-specific primary survey with an audit-oriented coding scheme for temporal splits, costs, execution semantics, universe handling, and reproducible artifacts.",
      "boundary": "A survey preprint and coded literature snapshot, not a completed DXRG audit or a pooled estimate of trading skill."
    },
    {
      "id": "agent-market-arena",
      "title": "When Agents Trade: Live Multi-Market Trading Benchmark for LLM Agents",
      "url": "https://arxiv.org/abs/2510.11695v2",
      "version": "arXiv:2510.11695v2, revised October 30, 2025",
      "locator": "Abstract; Sections 1, 3.1, 3.2, 3.3, 4, and 5; Table 1; Appendices A, B, and C; Table 2",
      "use": "Primary source for a live multi-market benchmark that varies both agent framework and model backbone inside one evaluation framework.",
      "boundary": "One benchmark with its own markets, agents, data, and operating period; its findings do not transfer to DXRG systems."
    },
    {
      "id": "ktd-fin",
      "title": "From Knowing to Doing: A Memory-Controlled Benchmark for LLM Trading Agents on Stock Markets",
      "url": "https://arxiv.org/abs/2605.28359v1",
      "version": "arXiv:2605.28359v1, submitted May 27, 2026",
      "locator": "Abstract; Sections 1, 3.1-3.6, 4.1-4.5, 5.1-5.3, and 6; Tables 1-5; Limitations; Appendices B-D and G-K; Tables 6-12",
      "use": "Primary source for evaluating historical-market memory controls and separating market, style, and stock-selection components in its benchmark.",
      "boundary": "A historical equity benchmark and attribution method; it does not test deployed runtime memory provenance or validate another harness."
    },
    {
      "id": "clqt",
      "title": "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents",
      "url": "https://arxiv.org/abs/2606.29771",
      "version": "v1",
      "locator": "Abstract; closed-loop protocol; cost model; DecisionRound audit trail; repeated-run design",
      "use": "Primary source for a temporally gated portfolio-agent benchmark with cost modeling, process diagnostics, and reconstructable decision records.",
      "boundary": "A separate benchmark, market design, scoring method, and result set; it does not validate DXRG methods or establish cross-market transfer."
    }
  ],
  "claimPlan": [
    {
      "claimId": "benchmark-fit-precedes-score",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "A benchmark audit should define the intended claim and test whether the benchmark measures the construct needed for that claim before interpreting its score.",
      "sourceIds": [
        "nist-ai-800-2"
      ],
      "boundary": "DXRG audit recommendation informed by an initial government draft."
    },
    {
      "claimId": "fixed-score-and-generalization-differ",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Performance on a fixed benchmark and performance on similar unseen tasks are different measurement targets.",
      "sourceIds": [
        "nist-ai-800-3"
      ],
      "boundary": "General measurement distinction; no trading-agent generalization result is implied."
    },
    {
      "claimId": "trading-protocols-need-auditable-fields",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Trading-agent benchmark reviews should make temporal splits, costs, execution semantics, universe handling, and artifacts inspectable.",
      "sourceIds": [
        "agentic-trading-survey"
      ],
      "boundary": "Audit-field recommendation informed by a survey preprint rather than a DXRG result."
    },
    {
      "claimId": "benchmark-designs-test-different-mechanisms",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Live multi-market comparison, historical-memory control, and closed-loop process diagnosis answer different evaluation questions.",
      "sourceIds": [
        "agent-market-arena",
        "ktd-fin",
        "clqt"
      ],
      "boundary": "Cross-source design distinction, not a ranking or pooled comparison of the cited benchmarks."
    }
  ],
  "auditUnits": [
    {
      "auditId": "construct-and-intended-use",
      "reviewQuestion": "What exact property or outcome does the benchmark claim to measure, and what downstream claim will this review use it to support?",
      "requiredEvidence": [
        "stated evaluation objective",
        "measurement construct",
        "intended use of results",
        "scope and excluded uses"
      ],
      "failureState": "The score is interpretable only as an observation inside the benchmark, without the broader claimed meaning.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2"
      ],
      "boundary": "Construct-fit review only."
    },
    {
      "auditId": "evaluation-subject-and-versions",
      "reviewQuestion": "Does the record identify the evaluated model, harness, prompts, memory, tools, policies, and execution code at stable versions?",
      "requiredEvidence": [
        "model and model version",
        "agent framework and configuration",
        "prompt, tool, and memory versions",
        "policy and execution versions"
      ],
      "failureState": "Model effects, harness effects, and configuration effects remain confounded.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agent-market-arena",
        "clqt"
      ],
      "boundary": "System-identification and attribution review only."
    },
    {
      "auditId": "protocol-and-tool-environment",
      "reviewQuestion": "Are task inputs, tool access, submission rules, feedback, action schema, abstention path, and execution environment fixed and disclosed?",
      "requiredEvidence": [
        "task and state inputs",
        "tool and data access",
        "submission and feedback rules",
        "typed action and abstention behavior"
      ],
      "failureState": "Differences in access or protocol may explain differences in measured behavior.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "agent-market-arena"
      ],
      "boundary": "Protocol comparability review only."
    },
    {
      "auditId": "point-in-time-and-contamination",
      "reviewQuestion": "Can the reviewer reconstruct what data was available at each decision and how future information, revisions, survivorship, and model-memory leakage were handled?",
      "requiredEvidence": [
        "decision and availability timestamps",
        "split and cutoff rules",
        "universe and revision policy",
        "historical-memory controls"
      ],
      "failureState": "The result cannot distinguish supplied-state reasoning from look-ahead or remembered history.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agentic-trading-survey",
        "ktd-fin"
      ],
      "boundary": "Temporal-integrity review rather than evidence that one masking design eliminates every leakage path."
    },
    {
      "auditId": "execution-and-costs",
      "reviewQuestion": "Are decision latency, validation, order handling, fills, retries, fees, spread, slippage, impact, financing, and settlement represented at the level required by the claim?",
      "requiredEvidence": [
        "decision-to-submission clock",
        "fill and partial-fill rules",
        "cost and financing model",
        "retry, acknowledgement, and settlement handling"
      ],
      "failureState": "The reported outcome may depend on an execution shortcut or omitted cost.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agentic-trading-survey",
        "clqt"
      ],
      "boundary": "Execution-realism review scaled to the stated evidence class."
    },
    {
      "auditId": "baselines-and-attribution",
      "reviewQuestion": "Do matched baselines and ablations separate model, harness, market, style, exposure, and execution effects from the reported outcome?",
      "requiredEvidence": [
        "same-information baselines",
        "same-cost comparators",
        "component ablations",
        "risk and return attribution"
      ],
      "failureState": "Outcome differences cannot be assigned to the agent capability named in the claim.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agent-market-arena",
        "ktd-fin",
        "clqt"
      ],
      "boundary": "Attribution review rather than an alpha or profitability test."
    },
    {
      "auditId": "repetition-and-uncertainty",
      "reviewQuestion": "Are repeated runs, sample construction, dispersion, worst cases, uncertainty, and the population to which the estimate applies reported?",
      "requiredEvidence": [
        "run and item counts",
        "sampling and seed procedure",
        "dispersion and uncertainty estimate",
        "fixed-benchmark versus generalized target"
      ],
      "failureState": "A point estimate may be mistaken for a stable or general result.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "nist-ai-800-3",
        "clqt"
      ],
      "boundary": "Statistical-support review without prescribing one estimator for every benchmark."
    },
    {
      "auditId": "artifacts-trace-and-corrections",
      "reviewQuestion": "Can an independent reviewer inspect the code, configurations, data lineage, run outputs, decision trace, exclusions, and correction history needed to reconstruct the claim?",
      "requiredEvidence": [
        "immutable code and configuration references",
        "data lineage and benchmark version",
        "run outputs and representative traces",
        "limitations and correction route"
      ],
      "failureState": "The claim remains difficult to reproduce, diagnose, or compare across revisions.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "agentic-trading-survey",
        "clqt"
      ],
      "boundary": "Artifact and traceability review; public release remains bounded by privacy, security, and data rights."
    }
  ],
  "benchmarkAudits": [
    {
      "benchmarkId": "agent-market-arena",
      "benchmarkTitle": "Agent Market Arena",
      "sourceId": "agent-market-arena",
      "auditState": "REVIEWED",
      "reviewedSource": {
        "immutableVersion": "arXiv:2510.11695v2",
        "sourceUrl": "https://arxiv.org/abs/2510.11695v2",
        "pdfUrl": "https://arxiv.org/pdf/2510.11695v2",
        "submittedAt": "2025-10-13",
        "revisedAt": "2025-10-30",
        "reviewedAt": "2026-08-15",
        "pageCount": 12,
        "supplementaryLiveInterface": "https://ama.thefin.ai/live",
        "supplementaryBoundary": "The interface is continuously updated and was not treated as an immutable source for this audit."
      },
      "intendedClaim": "Whether the frozen paper supports claims of fair and reproducible live multi-market comparison, realistic trading performance, and attribution between agent architecture and model backbone.",
      "reviewMethod": {
        "passOne": "Inventory every claimed objective, evaluated subject, input, action, date, metric, baseline, artifact, and limitation in the frozen paper.",
        "passTwo": "Reopen each required audit field at an exact section, table, or appendix locator; record disclosed and missing fields separately; then issue only a scoped source-disclosure finding."
      },
      "unitResults": [
        {
          "auditId": "construct-and-intended-use",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 1, Introduction",
            "Section 5, RQ1-RQ4",
            "Section 6, Conclusion"
          ],
          "finding": "The paper states that AMA evaluates live trading behavior across agent frameworks, model backbones, markets, profitability, adaptability, and risk. It does not define excluded downstream uses or bound its broader reproducibility and generalization language to the four assets and two-month paper window.",
          "verdict": "CONSTRUCT_STATED_SCOPE_LIMITS_PARTIAL",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "evaluation-subject-and-versions",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 3.2, Agent Execution Protocol",
            "Section 4, LLM selection",
            "Appendix B, System Prompt Definition",
            "Table 2"
          ],
          "finding": "The paper names four agent frameworks, five model families, and selected DeepFund parameters. It does not freeze agent code and configuration commits, exact provider model snapshots, complete per-agent prompts, memory implementations, policy versions, or execution code.",
          "verdict": "ATTRIBUTION_LIMITED_BY_VERSIONING",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "protocol-and-tool-environment",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, Market Intelligence Stream",
            "Section 3.2, Inputs, Outputs, and Prompts and Hyperparameters",
            "Appendix A",
            "Appendix B",
            "Table 2"
          ],
          "finding": "The source reports common daily inputs, BUY/SELL/HOLD outputs, synchronous decisions, and fixed generation settings. The exact decision time, complete input schema, starting capital value, per-agent tool access, and full configurations are absent. HOLD keeps the current position in Section 3.2 but goes flat in Appendix B, leaving the action rule internally inconsistent.",
          "verdict": "PROTOCOL_PARTIALLY_RECONSTRUCTABLE",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "point-in-time-and-contamination",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, Market Intelligence Stream and Quality control",
            "Section 4, Agent initialization and evaluation",
            "Appendix A"
          ],
          "finding": "The source gives initialization and evaluation dates, names input-source families, and reports a sampled news-quality review that observed one-to-two-day API latency. It does not provide per-item availability timestamps, cutoff enforcement, revision policy, universe construction, survivorship handling, or controls for historical knowledge in model and agent memory.",
          "verdict": "TEMPORAL_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "execution-and-costs",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.2, Outputs",
            "Section 3.3, Performance Analytics Interface",
            "Table 1",
            "Appendix C, Overview Dashboard"
          ],
          "finding": "The paper describes simulated portfolio updates from daily directional signals and a next-day return formula; Appendix C says the evolving dashboard can show returns with and without fees. The frozen source does not specify the price timestamp, order type, fill and partial-fill rules, fee schedule, spread, slippage, impact, financing, latency, validation, retry acknowledgement, or settlement treatment behind Table 1.",
          "verdict": "EXECUTION_REALISM_NOT_ESTABLISHED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "baselines-and-attribution",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Table 1",
            "Section 5.1, Performance of Agents in Live Trading",
            "Section 5.2, Which Matters More: Agent Architecture or LLM Backbone?",
            "Figure 2"
          ],
          "finding": "The study includes Buy and Hold and crosses listed agent frameworks with model families. It does not report same-information and same-cost matched comparators, component ablations, or exposure and execution attribution sufficient to isolate agent architecture as the cause of observed return differences.",
          "verdict": "CAUSAL_ATTRIBUTION_NOT_ESTABLISHED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "repetition-and-uncertainty",
          "state": "REVIEWED",
          "disclosure": "UNDISCLOSED",
          "locators": [
            "Section 4, Agent initialization and evaluation",
            "Table 1",
            "Figure 2",
            "Section 5"
          ],
          "finding": "The frozen paper reports one realized August-to-September 2025 market path. It does not report repeated stochastic runs, seeds, sampling variation, confidence intervals, sensitivity analysis, or an uncertainty target for similar unseen periods. Figure 2 shows ranges across agent or model choices, not repeated-run uncertainty.",
          "verdict": "STABILITY_AND_GENERALIZATION_UNESTIMATED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "artifacts-trace-and-corrections",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 1, AMA evaluation pipeline link",
            "Appendices A, B, and C",
            "Table 2",
            "Figures 5 and 6"
          ],
          "finding": "The arXiv source package, appendix material, parameter table, and evolving dashboard expose part of the method and aggregate output. The frozen record does not identify an immutable code repository, versioned dataset export, complete configurations, raw run outputs, representative decision traces, exclusion ledger, or correction history needed for independent reconstruction.",
          "verdict": "INDEPENDENT_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-15"
        }
      ],
      "scopedFinding": "The frozen v2 paper supports describing a reported two-month, four-asset study of daily simulated directional signals across named agent frameworks and model families. The reviewed source does not fully establish realistic execution profitability, causal agent-versus-model attribution, repeated-run stability, independent reconstruction, or transfer beyond that window.",
      "verdict": "REPORTED_STUDY_WITH_MATERIAL_DISCLOSURE_LIMITS",
      "reviewedAt": "2026-08-15",
      "boundary": "DXRG reviewed source disclosure only. We did not rerun the benchmark, validate its inputs or outputs, reproduce its metrics, rank it against another benchmark, or assess an agent for deployment."
    },
    {
      "benchmarkId": "ktd-fin",
      "benchmarkTitle": "KTD-Fin",
      "sourceId": "ktd-fin",
      "auditState": "REVIEWED",
      "reviewedSource": {
        "immutableVersion": "arXiv:2605.28359v1",
        "sourceUrl": "https://arxiv.org/abs/2605.28359v1",
        "pdfUrl": "https://arxiv.org/pdf/2605.28359v1",
        "sourcePackageUrl": "https://export.arxiv.org/e-print/2605.28359v1",
        "submittedAt": "2026-05-27",
        "reviewedAt": "2026-08-18",
        "pageCount": 20,
        "pdfSha256": "dbbe88939fb2774303e11d7972b5f7493858fcd0e07a1a021e8f66657dfb2627",
        "sourcePackageSha256": "16ec645cf91785bb17a1cab5838f7195bf5622e12690ed59b7c012e9148cfed2"
      },
      "intendedClaim": "Whether the frozen paper supports claims of leakage-controlled and attribution-aware trading-agent evaluation, reproducible simulated performance, and transferable stock-selection evidence.",
      "reviewMethod": {
        "passOne": "Inventory every claimed objective, evaluated subject, input, action, date, metric, baseline, artifact, and limitation in the frozen paper and arXiv source package.",
        "passTwo": "Reopen each required audit field at an exact section, table, or appendix locator; record disclosed and missing fields separately; then issue only a scoped source-disclosure finding."
      },
      "unitResults": [
        {
          "auditId": "construct-and-intended-use",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 1, Introduction",
            "Section 6, Conclusion",
            "Limitations"
          ],
          "finding": "The paper defines two measurement targets: reduce historical-identifier memory exposure through data-side masking and separate common, style, and stock-selection components through cross-sectional attribution. It bounds the implementation to price-only CSI300 simulation, but its language about a reproducible and transferable template does not define excluded downstream uses or establish that the reported results generalize beyond the studied market, models, and window.",
          "verdict": "CONSTRUCT_STATED_TRANSFER_SCOPE_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "evaluation-subject-and-versions",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 4.1, Models",
            "Table 4",
            "Appendix B, Tool interface specification",
            "Appendix C, Prompts and a worked masked trace",
            "Appendix I, Table 10"
          ],
          "finding": "The paper names the ten evaluated model labels, supplies decision-mode prompts, summarizes six tools and the action schema, and reports headline harness settings. It does not freeze provider model snapshots, code commits, complete tool-return schemas, the executable harness, data adapters, policy implementation, or per-model configurations needed to isolate a model effect from version and runtime effects.",
          "verdict": "SUBJECT_NAMED_VERSIONING_INCOMPLETE",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "protocol-and-tool-environment",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Sections 3.1-3.5",
            "Table 1",
            "Appendix B, Table 6",
            "Appendix C, worked masked trace",
            "Appendix D, Action schema",
            "Appendix I, Table 10"
          ],
          "finding": "The source describes price-only state, three decision modes, six read-only research tools, structured feedback, a typed JSON action, schema retries, and a hold fallback. The full data and tool contracts, provider-adapter behavior, every submission constraint, and the executable environment remain unavailable, so an independent reviewer cannot reconstruct the protocol from the frozen source alone.",
          "verdict": "PROTOCOL_SUBSTANTIVE_RECONSTRUCTION_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "point-in-time-and-contamination",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.3, Four-level masking protocol",
            "Section 4.1, Evaluation windows",
            "Table 2",
            "Section 4.3, De-anonymization probe",
            "Appendix C, worked masked trace",
            "Appendix K, Table 12"
          ],
          "finding": "The paper gives evaluation windows, uses T-1 price features, masks tickers and calendar identifiers across prompts and tools, and reports a 200-probe de-anonymization test for each of ten attacker models. The mask controls identifier recognition rather than all future-information paths. The source does not disclose the price-data vendor and revision history, point-in-time CSI300 constituent construction, survivorship treatment, per-model knowledge cutoffs for nine models, or raw probe samples and failures needed to reconstruct the contamination boundary.",
          "verdict": "MASK_DISCLOSED_TEMPORAL_LINEAGE_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "execution-and-costs",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, Execution engine",
            "Section 3.5, Action schema",
            "Appendix H, price-limit rules",
            "Appendix I, Table 10"
          ],
          "finding": "The simulator reports CNY 1,000,000 initial cash, next-day-open execution, long-only positions, T+1 sell availability, board-specific price limits, 5 bps buy cost, 15 bps sell cost, a CNY 5 minimum fee, schema retries, and a hold fallback. It does not specify partial fills, spread and market impact, liquidity capacity, financing, venue acknowledgement, settlement reconciliation, or latency between decision and executable order at the level needed for live-performance claims.",
          "verdict": "SIMULATED_EXECUTION_DISCLOSED_LIVE_REALISM_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "baselines-and-attribution",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 4.1, Baselines",
            "Sections 4.2 and 4.5",
            "Tables 3-5",
            "Appendices E, F, and H",
            "Tables 7-9"
          ],
          "finding": "The study includes CSI300 buy-and-hold, eighteen Qlib factor models using the same trading layer, mask and decision-mode contrasts on an anchor model, and a nine-factor cross-sectional return decomposition. The manuscript is internally inconsistent about whether baseline training begins in 2008 or 2015, does not expose fitted models or full hyperparameters, and treats regression residual as a closer proxy for selection ability rather than independently validated alpha. These limits constrain causal and economic attribution.",
          "verdict": "BASELINES_AND_ATTRIBUTION_SUBSTANTIVE_LIMITS_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "repetition-and-uncertainty",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 4.1, Models and Protocol",
            "Tables 3-5",
            "Limitations",
            "Appendix J, Table 11",
            "Appendix K, Table 12"
          ],
          "finding": "The anchor grid uses five seeds, the headline long-window cells use three seeds, additional mask sensitivity uses three seeds, multi-seed cells use medians and IQRs, paired contrasts use Wilcoxon tests, and probe cells report Wilson intervals. The headline leaderboard and attribution table omit per-seed values and uncertainty, seed identities and sampling procedures are absent, model coverage differs across ablations, and one long CSI300 window cannot establish stability in a broader target population.",
          "verdict": "MULTI_SEED_DESIGN_OUTPUT_UNCERTAINTY_PARTIAL",
          "reviewedAt": "2026-08-18"
        },
        {
          "auditId": "artifacts-trace-and-corrections",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 1, Contributions",
            "Appendix B, Table 6",
            "Appendix C, worked masked trace",
            "Appendix D, Action schema",
            "Appendices G-I, Tables 9-10",
            "arXiv v1 source package"
          ],
          "finding": "The paper and arXiv manuscript source provide prompts, a tool summary, a representative masked decision trace, metric definitions, baseline tables, and headline harness settings. Although the paper says the environment, scripts, protocols, and evaluation dataset are publicly released, the frozen record provides no repository or artifact locator for them. It also lacks immutable code and data versions, data lineage, raw run outputs, full decision traces, an exclusion ledger, and a correction history required for independent reconstruction.",
          "verdict": "MANUSCRIPT_ARTIFACTS_PRESENT_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-18"
        }
      ],
      "scopedFinding": "The frozen v1 paper supports describing a reported price-only CSI300 simulation from January 1, 2024 through April 10, 2026 with identifier-and-calendar masking, next-day-open execution, explicit fees and T+1 constraints, multi-seed comparisons, a de-anonymization probe, and Barra-style attribution. The reviewed source does not fully establish complete contamination control, point-in-time data reconstruction, provider-snapshot attribution, independently reproducible results, live execution performance, or transfer beyond that market and window.",
      "verdict": "REPORTED_MASKED_SIMULATION_WITH_MATERIAL_DISCLOSURE_LIMITS",
      "reviewedAt": "2026-08-18",
      "boundary": "DXRG reviewed source disclosure only. We did not rerun the benchmark, validate its data or results, reproduce its metrics, certify the masking protocol, rank it against another benchmark, or assess an agent for deployment."
    },
    {
      "benchmarkId": "clqt",
      "benchmarkTitle": "CLQT",
      "sourceId": "clqt",
      "auditState": "REVIEWED",
      "reviewedSource": {
        "immutableVersion": "arXiv:2606.29771v1",
        "sourceUrl": "https://arxiv.org/abs/2606.29771v1",
        "pdfUrl": "https://arxiv.org/pdf/2606.29771v1",
        "sourcePackageUrl": "https://export.arxiv.org/e-print/2606.29771v1",
        "submittedAt": "2026-06-29",
        "reviewedAt": "2026-08-21",
        "pageCount": 50,
        "pdfSha256": "6ee29ff2fc339035fd739d2273856332cdf83d2e3e8557969b660e0a590fdf12",
        "sourcePackageSha256": "4f1115d3a88cd8372bc511fb1eb704469c3be2fcdae5701a7e31a10db088709b"
      },
      "intendedClaim": "Whether the frozen paper supports claims of point-in-time, cost-aware, reconstructable diagnosis of portfolio-agent behavior across a historical campaign and a broker paper track, with stable attribution beyond one cohort and regime.",
      "reviewMethod": {
        "passOne": "Inventory every claimed objective, evaluated subject, input, action, date, metric, baseline, artifact, and limitation in the frozen paper and arXiv source package.",
        "passTwo": "Reopen each required audit field at an exact section, table, or appendix locator; record disclosed and missing fields separately; then issue only a scoped source-disclosure finding."
      },
      "unitResults": [
        {
          "auditId": "construct-and-intended-use",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 1, Introduction and Contributions",
            "Sections 6.3-6.4, Metric Dictionary and APM-CS",
            "Sections 8.1 and 8.7, Discussion and Limitations",
            "Section 9, Conclusion"
          ],
          "finding": "The paper defines CLQT as a diagnostic instrument for separating outcome from five process-capability axes rather than as a return leaderboard, and it states a consistent-dominance bar across axes and sub-periods. The source also makes broader claims about portable agent capabilities and instrument consistency from one rising-market backtest cohort and a different short paper-trading cohort, while acknowledging that cross-regime ranking, broader repeated runs, and same-model transfer remain future work.",
          "verdict": "DIAGNOSTIC_CONSTRUCT_STATED_VALIDITY_SCOPE_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "evaluation-subject-and-versions",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Sections 3.1-3.2, Architecture and Skill Modes",
            "Section 5.5, Live Track",
            "Section 7.2, Setup",
            "Appendix C, Model Registry"
          ],
          "finding": "The paper names backtest and paper-track model IDs, modes, effort settings, output and context budgets, cohort dates, mandate structure, and major runtime components. Some model IDs use dated snapshots, while several cutoffs are undisclosed or approximate; exact skill files, prompts, tool assignments, code commits, provider routing, broker integration, and complete configurations are available only on request or absent from the frozen package.",
          "verdict": "SUBJECT_AND_COHORT_NAMED_RUNTIME_VERSIONING_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "protocol-and-tool-environment",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Sections 3.1-3.8, System Design",
            "Section 5.3, Rebalancing vs. Re-Optimization",
            "Section 6.1, Ablation Dimensions",
            "Appendices B and D"
          ],
          "finding": "The source describes a five-stage loop, structured and autonomous modes, a DecisionRound schema, 19 typed in-process tools, memory, target allocation, constraint projection, risk review, no-trade fields, and three rebalance modes. It does not publish the executable environment, exact skill and prompt files, complete schemas, tool-return contracts, provider adapters, retry rules, or configuration files needed to reconstruct the protocol independently.",
          "verdict": "CLOSED_LOOP_PROTOCOL_DESCRIBED_RECONSTRUCTION_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "point-in-time-and-contamination",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Table 2, Data Sources",
            "Sections 4.1-4.3, TimeGate and Contamination",
            "Sections 5.1 and 5.5, Backtest and Paper Tracks",
            "Section 7.2, Cutoff Safety",
            "Appendix C, Model Registry"
          ],
          "finding": "The paper states a uniform as-of TimeGate, an intraday cutoff, point-in-time S&P 100 membership, per-model cutoff filtering, a June 16, 2025 to June 12, 2026 backtest, and a June 15-26, 2026 paper track. The canary and future-shift probes were not exercised; provider snapshots, data revisions, release timestamps, universe lineage, the exclusion manifest, and raw TimeGate evidence are not in the frozen package. Qwen uses a one-month cutoff grace, and several paper-track model cutoffs are undisclosed or inferred from release timing.",
          "verdict": "TEMPORAL_GATES_DESCRIBED_LINEAGE_AND_PROBES_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "execution-and-costs",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.4, Cost-Aware Execution",
            "Section 5.2, Broker Paper-Trading Architecture",
            "Section 5.5, Two-Week Campaign",
            "Sections 7.7 and 8.5, Cost Ablation and Auditability",
            "Appendix A, Cost Tiers"
          ],
          "finding": "The paper discloses spread, commission, slippage, borrow, impact tiers, financing accrual, target-weight locking, paper-order submission, fill polling, cash and position reconciliation, and daily flattening. The HIGH tier is explicitly miscalibrated at about 527 bps per round, while the retained LOW tier is calibrated against broker paper fills. The broker, order types, partial-fill and rejection details, latency distributions, raw acknowledgements, and settlement records are unavailable, and paper fills do not establish live economic execution.",
          "verdict": "COST_AND_PAPER_LIFECYCLE_DISCLOSED_LIVE_REALISM_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "baselines-and-attribution",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Sections 6.1-6.4, Ablations, Baselines, and Scorecard",
            "Sections 7.3-7.7, Campaign and Ablation Results",
            "Section 7.9, Passive Comparison",
            "Sections 8.1 and 8.7, Discussion and Limitations"
          ],
          "finding": "The study reports eight passive or algorithmic comparators, structured-versus-autonomous contrasts, single- and multi-module ablations, cost-tier sensitivity, pre- and post-constraint fields, and five capability axes. Attribution remains limited by one anchor model for module ablations, a single rising regime, cohort-relative composite scores, mode confounding for Acuity and Discipline, an investment-target component that was not ablated alone, and a primary coherence judge that also appears in the paper-track cohort.",
          "verdict": "MULTI_AXIS_ATTRIBUTION_DESIGN_WITH_MATERIAL_CONFOUNDS_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "repetition-and-uncertainty",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 5.5, Two-Week Campaign",
            "Sections 7 and 7.2, Experiments and Setup",
            "Section 7.7, Repeated-Run Ablations",
            "Section 8.7, Limitations",
            "Section 8.8, Future Work"
          ],
          "finding": "The historical campaign repeats runs for three structurally reliable models, reports ranges for repeated module and cluster ablations, and uses a plus-or-minus 0.42 Sharpe noise band. Haiku and qwen are single-run or partial, cost-tier variants are single-run, the paper track has nine valid days without repeated paths, seed identities and sampling procedures are absent, and confidence intervals plus rank stability remain planned rather than reported.",
          "verdict": "REPEATED_SUBSETS_WITH_GENERAL_UNCERTAINTY_PARTIAL",
          "reviewedAt": "2026-08-21"
        },
        {
          "auditId": "artifacts-trace-and-corrections",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, DecisionRound Schema",
            "Section 4.4, Tamper-Evident Audit Trail",
            "Section 6.5, Behavioral Interpretability",
            "Appendix D, Audit-Trail Field Map",
            "Appendix E, Round-Pinned Anecdotes",
            "arXiv v1 source package"
          ],
          "finding": "The paper defines hash-linked DecisionRound fields, maps metrics to stored fields, names verification and table-building scripts, includes round-pinned anecdotes, and supplies the manuscript and figures in its arXiv package. The package contains no executable code, immutable configurations, raw data lineage, run stores, hash manifests, judge prompts, exclusion manifest, broker records, or correction history, so a reader cannot recompute the hashes or reconstruct the reported metrics independently.",
          "verdict": "TRACE_SCHEMA_DISCLOSED_ARTIFACT_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-21"
        }
      ],
      "scopedFinding": "The frozen v1 paper supports describing a reported 26-round, S&P-100-derived historical campaign across five model families and two skill modes, with cost tiers, eight baselines, repeated-run ablations, a hash-linked DecisionRound design, and a separate nine-day broker paper track. The reviewed source does not fully establish point-in-time data lineage, executable reconstruction, broad statistical stability, same-model live transfer, live economic execution, or generalization beyond the reported cohorts and rising-market window.",
      "verdict": "REPORTED_CLOSED_LOOP_CAMPAIGN_WITH_MATERIAL_DISCLOSURE_LIMITS",
      "reviewedAt": "2026-08-21",
      "boundary": "DXRG reviewed source disclosure only. We did not rerun CLQT, validate its data or results, recompute its hash chains, reproduce its paper fills or metrics, rank it against another benchmark, or assess an agent for deployment."
    }
  ]
}
