{
  "schemaVersion": "dxrg-trading-agent-benchmark-audit-registry/v1",
  "version": "1.1.0",
  "title": "DXRG Trading-Agent Benchmark Audit Registry",
  "description": "Eight review units applied to three registered trading-agent benchmarks, with one frozen primary-source audit and two entries kept explicitly unrun.",
  "publishedDate": "2026-08-12",
  "modifiedDate": "2026-08-15",
  "publisher": "DX Research Group",
  "canonicalPage": "https://www.dxrg.ai/blogs/how-to-evaluate-an-ai-trading-agent",
  "methodology": "Register one benchmark version and one intended claim. Review primary documentation against each unit, preserve exact locators, record disclosed and undisclosed fields separately, and issue a scoped finding only after all required evidence has been checked. Result fields remain null until that source review is complete.",
  "evidenceBoundary": "A completed row records what a frozen primary source discloses. It is not an independent rerun, result validation, benchmark ranking, endorsement, or rejection; unrun rows carry no finding.",
  "requiredAuditHeader": [
    "benchmark title and immutable version",
    "primary source and exact locator",
    "claim the benchmark result is being used to support",
    "reviewer and review date",
    "disclosure state for every audit unit",
    "scoped finding and unresolved limitations"
  ],
  "disclosureStates": [
    "DISCLOSED",
    "PARTIAL",
    "UNDISCLOSED",
    "NOT_APPLICABLE",
    "UNRUN"
  ],
  "sources": [
    {
      "id": "nist-ai-800-2",
      "title": "Practices for Automated Benchmark Evaluations of Language Models",
      "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.800-2.ipd.pdf",
      "version": "Initial Public Draft, January 2026",
      "locator": "Introduction; Practices 1.1, 1.2, 2.1, 3.1, 3.2, and 3.3",
      "use": "Government draft practices for defining the measurement target, selecting a fitting benchmark, fixing the protocol, quantifying uncertainty, sharing details, and qualifying claims.",
      "boundary": "An initial public draft for automated language-model and agent benchmarks, not a final standard or a trading-specific prescription."
    },
    {
      "id": "nist-ai-800-3",
      "title": "Expanding the AI Evaluation Toolbox with Statistical Models",
      "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.800-3.pdf",
      "version": "February 2026",
      "locator": "Abstract; Sections 1, 3, 4, and 6",
      "use": "Government measurement research distinguishing fixed-benchmark accuracy from generalized accuracy and examining uncertainty estimates.",
      "boundary": "General AI benchmark measurement research; its statistical examples do not establish trading-agent performance or external validity."
    },
    {
      "id": "agentic-trading-survey",
      "title": "Agentic Trading: When LLM Agents Meet Financial Markets",
      "url": "https://arxiv.org/abs/2605.19337",
      "version": "v2",
      "locator": "Abstract; protocol-coded evidence map; reproducibility audit; reporting checklist",
      "use": "Trading-specific primary survey with an audit-oriented coding scheme for temporal splits, costs, execution semantics, universe handling, and reproducible artifacts.",
      "boundary": "A survey preprint and coded literature snapshot, not a completed DXRG audit or a pooled estimate of trading skill."
    },
    {
      "id": "agent-market-arena",
      "title": "When Agents Trade: Live Multi-Market Trading Benchmark for LLM Agents",
      "url": "https://arxiv.org/abs/2510.11695v2",
      "version": "arXiv:2510.11695v2, revised October 30, 2025",
      "locator": "Abstract; Sections 1, 3.1, 3.2, 3.3, 4, and 5; Table 1; Appendices A, B, and C; Table 2",
      "use": "Primary source for a live multi-market benchmark that varies both agent framework and model backbone inside one evaluation framework.",
      "boundary": "One benchmark with its own markets, agents, data, and operating period; its findings do not transfer to DXRG systems."
    },
    {
      "id": "ktd-fin",
      "title": "From Knowing to Doing: A Memory-Controlled Benchmark for LLM Trading Agents on Stock Markets",
      "url": "https://arxiv.org/abs/2605.28359",
      "version": "v1",
      "locator": "Abstract; masking protocol; decision modes; performance attribution framework",
      "use": "Primary source for evaluating historical-market memory controls and separating market, style, and stock-selection components in its benchmark.",
      "boundary": "A historical equity benchmark and attribution method; it does not test deployed runtime memory provenance or validate another harness."
    },
    {
      "id": "clqt",
      "title": "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents",
      "url": "https://arxiv.org/abs/2606.29771",
      "version": "v1",
      "locator": "Abstract; closed-loop protocol; cost model; DecisionRound audit trail; repeated-run design",
      "use": "Primary source for a temporally gated portfolio-agent benchmark with cost modeling, process diagnostics, and reconstructable decision records.",
      "boundary": "A separate benchmark, market design, scoring method, and result set; it does not validate DXRG methods or establish cross-market transfer."
    }
  ],
  "claimPlan": [
    {
      "claimId": "benchmark-fit-precedes-score",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "A benchmark audit should define the intended claim and test whether the benchmark measures the construct needed for that claim before interpreting its score.",
      "sourceIds": [
        "nist-ai-800-2"
      ],
      "boundary": "DXRG audit recommendation informed by an initial government draft."
    },
    {
      "claimId": "fixed-score-and-generalization-differ",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Performance on a fixed benchmark and performance on similar unseen tasks are different measurement targets.",
      "sourceIds": [
        "nist-ai-800-3"
      ],
      "boundary": "General measurement distinction; no trading-agent generalization result is implied."
    },
    {
      "claimId": "trading-protocols-need-auditable-fields",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Trading-agent benchmark reviews should make temporal splits, costs, execution semantics, universe handling, and artifacts inspectable.",
      "sourceIds": [
        "agentic-trading-survey"
      ],
      "boundary": "Audit-field recommendation informed by a survey preprint rather than a DXRG result."
    },
    {
      "claimId": "benchmark-designs-test-different-mechanisms",
      "status": "SOURCE_PLAN_ONLY",
      "eligibleWording": "Live multi-market comparison, historical-memory control, and closed-loop process diagnosis answer different evaluation questions.",
      "sourceIds": [
        "agent-market-arena",
        "ktd-fin",
        "clqt"
      ],
      "boundary": "Cross-source design distinction, not a ranking or pooled comparison of the cited benchmarks."
    }
  ],
  "auditUnits": [
    {
      "auditId": "construct-and-intended-use",
      "reviewQuestion": "What exact property or outcome does the benchmark claim to measure, and what downstream claim will this review use it to support?",
      "requiredEvidence": [
        "stated evaluation objective",
        "measurement construct",
        "intended use of results",
        "scope and excluded uses"
      ],
      "failureState": "The score is interpretable only as an observation inside the benchmark, without the broader claimed meaning.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2"
      ],
      "boundary": "Construct-fit review only."
    },
    {
      "auditId": "evaluation-subject-and-versions",
      "reviewQuestion": "Does the record identify the evaluated model, harness, prompts, memory, tools, policies, and execution code at stable versions?",
      "requiredEvidence": [
        "model and model version",
        "agent framework and configuration",
        "prompt, tool, and memory versions",
        "policy and execution versions"
      ],
      "failureState": "Model effects, harness effects, and configuration effects remain confounded.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agent-market-arena",
        "clqt"
      ],
      "boundary": "System-identification and attribution review only."
    },
    {
      "auditId": "protocol-and-tool-environment",
      "reviewQuestion": "Are task inputs, tool access, submission rules, feedback, action schema, abstention path, and execution environment fixed and disclosed?",
      "requiredEvidence": [
        "task and state inputs",
        "tool and data access",
        "submission and feedback rules",
        "typed action and abstention behavior"
      ],
      "failureState": "Differences in access or protocol may explain differences in measured behavior.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "agent-market-arena"
      ],
      "boundary": "Protocol comparability review only."
    },
    {
      "auditId": "point-in-time-and-contamination",
      "reviewQuestion": "Can the reviewer reconstruct what data was available at each decision and how future information, revisions, survivorship, and model-memory leakage were handled?",
      "requiredEvidence": [
        "decision and availability timestamps",
        "split and cutoff rules",
        "universe and revision policy",
        "historical-memory controls"
      ],
      "failureState": "The result cannot distinguish supplied-state reasoning from look-ahead or remembered history.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agentic-trading-survey",
        "ktd-fin"
      ],
      "boundary": "Temporal-integrity review rather than evidence that one masking design eliminates every leakage path."
    },
    {
      "auditId": "execution-and-costs",
      "reviewQuestion": "Are decision latency, validation, order handling, fills, retries, fees, spread, slippage, impact, financing, and settlement represented at the level required by the claim?",
      "requiredEvidence": [
        "decision-to-submission clock",
        "fill and partial-fill rules",
        "cost and financing model",
        "retry, acknowledgement, and settlement handling"
      ],
      "failureState": "The reported outcome may depend on an execution shortcut or omitted cost.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agentic-trading-survey",
        "clqt"
      ],
      "boundary": "Execution-realism review scaled to the stated evidence class."
    },
    {
      "auditId": "baselines-and-attribution",
      "reviewQuestion": "Do matched baselines and ablations separate model, harness, market, style, exposure, and execution effects from the reported outcome?",
      "requiredEvidence": [
        "same-information baselines",
        "same-cost comparators",
        "component ablations",
        "risk and return attribution"
      ],
      "failureState": "Outcome differences cannot be assigned to the agent capability named in the claim.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "agent-market-arena",
        "ktd-fin",
        "clqt"
      ],
      "boundary": "Attribution review rather than an alpha or profitability test."
    },
    {
      "auditId": "repetition-and-uncertainty",
      "reviewQuestion": "Are repeated runs, sample construction, dispersion, worst cases, uncertainty, and the population to which the estimate applies reported?",
      "requiredEvidence": [
        "run and item counts",
        "sampling and seed procedure",
        "dispersion and uncertainty estimate",
        "fixed-benchmark versus generalized target"
      ],
      "failureState": "A point estimate may be mistaken for a stable or general result.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "nist-ai-800-3",
        "clqt"
      ],
      "boundary": "Statistical-support review without prescribing one estimator for every benchmark."
    },
    {
      "auditId": "artifacts-trace-and-corrections",
      "reviewQuestion": "Can an independent reviewer inspect the code, configurations, data lineage, run outputs, decision trace, exclusions, and correction history needed to reconstruct the claim?",
      "requiredEvidence": [
        "immutable code and configuration references",
        "data lineage and benchmark version",
        "run outputs and representative traces",
        "limitations and correction route"
      ],
      "failureState": "The claim remains difficult to reproduce, diagnose, or compare across revisions.",
      "result": {
        "state": "UNRUN",
        "disclosure": null,
        "finding": null,
        "reviewedAt": null
      },
      "sourceIds": [
        "nist-ai-800-2",
        "agentic-trading-survey",
        "clqt"
      ],
      "boundary": "Artifact and traceability review; public release remains bounded by privacy, security, and data rights."
    }
  ],
  "benchmarkAudits": [
    {
      "benchmarkId": "agent-market-arena",
      "benchmarkTitle": "Agent Market Arena",
      "sourceId": "agent-market-arena",
      "auditState": "REVIEWED",
      "reviewedSource": {
        "immutableVersion": "arXiv:2510.11695v2",
        "sourceUrl": "https://arxiv.org/abs/2510.11695v2",
        "pdfUrl": "https://arxiv.org/pdf/2510.11695v2",
        "submittedAt": "2025-10-13",
        "revisedAt": "2025-10-30",
        "reviewedAt": "2026-08-15",
        "pageCount": 12,
        "supplementaryLiveInterface": "https://ama.thefin.ai/live",
        "supplementaryBoundary": "The interface is continuously updated and was not treated as an immutable source for this audit."
      },
      "intendedClaim": "Whether the frozen paper supports claims of fair and reproducible live multi-market comparison, realistic trading performance, and attribution between agent architecture and model backbone.",
      "reviewMethod": {
        "passOne": "Inventory every claimed objective, evaluated subject, input, action, date, metric, baseline, artifact, and limitation in the frozen paper.",
        "passTwo": "Reopen each required audit field at an exact section, table, or appendix locator; record disclosed and missing fields separately; then issue only a scoped source-disclosure finding."
      },
      "unitResults": [
        {
          "auditId": "construct-and-intended-use",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 1, Introduction",
            "Section 5, RQ1-RQ4",
            "Section 6, Conclusion"
          ],
          "finding": "The paper states that AMA evaluates live trading behavior across agent frameworks, model backbones, markets, profitability, adaptability, and risk. It does not define excluded downstream uses or bound its broader reproducibility and generalization language to the four assets and two-month paper window.",
          "verdict": "CONSTRUCT_STATED_SCOPE_LIMITS_PARTIAL",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "evaluation-subject-and-versions",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Abstract",
            "Section 3.2, Agent Execution Protocol",
            "Section 4, LLM selection",
            "Appendix B, System Prompt Definition",
            "Table 2"
          ],
          "finding": "The paper names four agent frameworks, five model families, and selected DeepFund parameters. It does not freeze agent code and configuration commits, exact provider model snapshots, complete per-agent prompts, memory implementations, policy versions, or execution code.",
          "verdict": "ATTRIBUTION_LIMITED_BY_VERSIONING",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "protocol-and-tool-environment",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, Market Intelligence Stream",
            "Section 3.2, Inputs, Outputs, and Prompts and Hyperparameters",
            "Appendix A",
            "Appendix B",
            "Table 2"
          ],
          "finding": "The source reports common daily inputs, BUY/SELL/HOLD outputs, synchronous decisions, and fixed generation settings. The exact decision time, complete input schema, starting capital value, per-agent tool access, and full configurations are absent. HOLD keeps the current position in Section 3.2 but goes flat in Appendix B, leaving the action rule internally inconsistent.",
          "verdict": "PROTOCOL_PARTIALLY_RECONSTRUCTABLE",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "point-in-time-and-contamination",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.1, Market Intelligence Stream and Quality control",
            "Section 4, Agent initialization and evaluation",
            "Appendix A"
          ],
          "finding": "The source gives initialization and evaluation dates, names input-source families, and reports a sampled news-quality review that observed one-to-two-day API latency. It does not provide per-item availability timestamps, cutoff enforcement, revision policy, universe construction, survivorship handling, or controls for historical knowledge in model and agent memory.",
          "verdict": "TEMPORAL_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "execution-and-costs",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 3.2, Outputs",
            "Section 3.3, Performance Analytics Interface",
            "Table 1",
            "Appendix C, Overview Dashboard"
          ],
          "finding": "The paper describes simulated portfolio updates from daily directional signals and a next-day return formula; Appendix C says the evolving dashboard can show returns with and without fees. The frozen source does not specify the price timestamp, order type, fill and partial-fill rules, fee schedule, spread, slippage, impact, financing, latency, validation, retry acknowledgement, or settlement treatment behind Table 1.",
          "verdict": "EXECUTION_REALISM_NOT_ESTABLISHED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "baselines-and-attribution",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Table 1",
            "Section 5.1, Performance of Agents in Live Trading",
            "Section 5.2, Which Matters More: Agent Architecture or LLM Backbone?",
            "Figure 2"
          ],
          "finding": "The study includes Buy and Hold and crosses listed agent frameworks with model families. It does not report same-information and same-cost matched comparators, component ablations, or exposure and execution attribution sufficient to isolate agent architecture as the cause of observed return differences.",
          "verdict": "CAUSAL_ATTRIBUTION_NOT_ESTABLISHED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "repetition-and-uncertainty",
          "state": "REVIEWED",
          "disclosure": "UNDISCLOSED",
          "locators": [
            "Section 4, Agent initialization and evaluation",
            "Table 1",
            "Figure 2",
            "Section 5"
          ],
          "finding": "The frozen paper reports one realized August-to-September 2025 market path. It does not report repeated stochastic runs, seeds, sampling variation, confidence intervals, sensitivity analysis, or an uncertainty target for similar unseen periods. Figure 2 shows ranges across agent or model choices, not repeated-run uncertainty.",
          "verdict": "STABILITY_AND_GENERALIZATION_UNESTIMATED",
          "reviewedAt": "2026-08-15"
        },
        {
          "auditId": "artifacts-trace-and-corrections",
          "state": "REVIEWED",
          "disclosure": "PARTIAL",
          "locators": [
            "Section 1, AMA evaluation pipeline link",
            "Appendices A, B, and C",
            "Table 2",
            "Figures 5 and 6"
          ],
          "finding": "The arXiv source package, appendix material, parameter table, and evolving dashboard expose part of the method and aggregate output. The frozen record does not identify an immutable code repository, versioned dataset export, complete configurations, raw run outputs, representative decision traces, exclusion ledger, or correction history needed for independent reconstruction.",
          "verdict": "INDEPENDENT_RECONSTRUCTION_INCOMPLETE",
          "reviewedAt": "2026-08-15"
        }
      ],
      "scopedFinding": "The frozen v2 paper supports describing a reported two-month, four-asset study of daily simulated directional signals across named agent frameworks and model families. The reviewed source does not fully establish realistic execution profitability, causal agent-versus-model attribution, repeated-run stability, independent reconstruction, or transfer beyond that window.",
      "verdict": "REPORTED_STUDY_WITH_MATERIAL_DISCLOSURE_LIMITS",
      "reviewedAt": "2026-08-15",
      "boundary": "DXRG reviewed source disclosure only. We did not rerun the benchmark, validate its inputs or outputs, reproduce its metrics, rank it against another benchmark, or assess an agent for deployment."
    },
    {
      "benchmarkId": "ktd-fin",
      "benchmarkTitle": "KTD-Fin",
      "sourceId": "ktd-fin",
      "auditState": "UNRUN",
      "reviewedSource": null,
      "intendedClaim": null,
      "unitResults": [],
      "scopedFinding": null,
      "verdict": null,
      "reviewedAt": null,
      "boundary": "No source audit has been completed; this row carries no finding."
    },
    {
      "benchmarkId": "clqt",
      "benchmarkTitle": "CLQT",
      "sourceId": "clqt",
      "auditState": "UNRUN",
      "reviewedSource": null,
      "intendedClaim": null,
      "unitResults": [],
      "scopedFinding": null,
      "verdict": null,
      "reviewedAt": null,
      "boundary": "No source audit has been completed; this row carries no finding."
    }
  ]
}
