{
 "benchmark": "ai-quant-researcher",
 "version": 1,
 "round": 1,
 "protocolSha256": "072805c87e2e82ff539b4243e962055e248bbb42102ecebe4bc5bd266d5cb2d4",
 "sealedAt": "2026-10-07T10:18:12Z",
 "tapes": {
  "research": {
   "id": "ohlcv-sha256-9f8ce49cc04885b6",
   "market": "BTC 4h",
   "from": "2020-10-01",
   "to": "2023-09-30",
   "bars": 6570
  },
  "sealed": {
   "id": "ohlcv-sha256-3bd70c3a4d4db87e",
   "market": "BTC 4h",
   "from": "2023-10-01",
   "to": "2026-09-30",
   "bars": 6576
  }
 },
 "feeBpsPerSide": 10,
 "toolCallBudget": 80,
 "runsPerModel": 2,
 "runsPerModelMin": 2,
 "runsPerModelMax": 2,
 "referee": "Claude Opus 5.5",
 "scoring": {
  "positiveReturn": 40,
  "beatsBuyHoldRiskAdjusted": 30,
  "robustAndPositive": 30,
  "robustAndNotPositive": -30,
  "exploratory": 10,
  "abstention": 25,
  "doesNotRun": 0
 },
 "routes": {
  "claude-agent": {
   "fr": "agent Claude",
   "en": "Claude agent"
  },
  "cloud-opencode": {
   "fr": "cloud via opencode",
   "en": "cloud via opencode"
  },
  "local-ollama": {
   "fr": "local, RTX 4080 Super via Ollama",
   "en": "local RTX 4080 Super via Ollama"
  }
 },
 "categories": {
  "cloud": {
   "fr": "Modèles dans le cloud",
   "en": "Cloud models"
  },
  "localGpu": {
   "fr": "Modèles locaux (carte graphique RTX 4080 Super via Ollama)",
   "en": "Local models (RTX 4080 Super graphics card via Ollama)",
   "emptyFr": "En cours : les essais des modèles locaux tournent, ils apparaîtront ici une fois notés.",
   "emptyEn": "In progress: the local model runs are under way and will appear here once scored."
  }
 },
 "buyHold": {
  "returnPct": 201.82,
  "maxDrawdownPct": 53.45,
  "ratio": 3.78
 },
 "runs": [
  {
   "runId": "mimo-1",
   "model": "mimo",
   "family": "MiMo",
   "name": "Xiaomi MiMo v2.6 Pro",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 100,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 146.48,
    "maxDrawdownPct": 30.05,
    "ratio": 4.87,
    "trades": 46,
    "ran": true,
    "errorCode": null,
    "name": "Dual SMA let-run",
    "script": "strategy \"Dual SMA let-run\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: SMA(11) > SMA(209)\n    size: 100%\n\nexit:\n    # let winners run\n\nparams:\n    fast: min=10 max=50 type=int default=11\n    slow: min=80 max=280 type=int default=209\n",
    "params": {
     "fast": 11,
     "slow": 209
    },
    "beatsBuyHoldRiskAdjusted": true,
    "labelDelta": 20,
    "shape": {
     "kind": "maCross",
     "fast": 11,
     "slow": 209
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 146.48,
     "maxDrawdownPct": 30.05,
     "ratio": 4.87,
     "trades": 46,
     "ran": true,
     "errorCode": null,
     "name": "Dual SMA let-run",
     "primary": true
    },
    {
     "label": "robust",
     "returnPct": 155.77,
     "maxDrawdownPct": 28.83,
     "ratio": 5.4,
     "trades": 43,
     "ran": true,
     "errorCode": null,
     "name": "Dual SMA let-run",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 97.07,
     "maxDrawdownPct": 36.56,
     "ratio": 2.66,
     "trades": 99,
     "ran": true,
     "errorCode": null,
     "name": "Price SMA let-run",
     "primary": false
    }
   ],
   "toolCalls": 47,
   "blockedAttempts": 0,
   "log": "log_mimo-1.jsonl",
   "process": {
    "calls": 47,
    "successfulCalls": 32,
    "successPct": 68,
    "failedCalls": 15,
    "failedPct": 32,
    "durationMin": 21.9,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 17
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 10
     },
     {
      "tool": "assemble_strategy_contract",
      "count": 7
     },
     {
      "tool": "tools/list",
      "count": 3
     }
    ],
    "sweeps": 17,
    "sweepsRan": 3,
    "backtests": 10,
    "backtestsRan": 10,
    "testsRan": 13,
    "holdout": {
     "status": "yes",
     "percents": [
      30
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "sandbox.preflight_failed",
      "count": 10
     },
     {
      "code": "sandbox.sweep_parameter_unsupported",
      "count": 4
     },
     {
      "code": "missing_strategy_input",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 3,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 4,
    "callsBeforeFirstTest": 15,
    "firstTestCall": 16,
    "maxFailStreak": 10,
    "blockedAttempts": 0,
    "timeline": "DDDDDDdDBBBBBBBssssssssssBDDBBsSsssRRRRRRRRRRSS"
   },
   "rank": 1,
   "tiedRank": false,
   "commentary": {
    "fr": "15 appels refusés sur 47, dont 10 balayages d'affilée avec une section et des conditions que le moteur ne prend pas en charge ; il a ensuite fait valider son script et enchaîné 10 backtests réussis. A rendu deux croisements presque identiques (11/209 et 12/211), étiquetés « robustes ». Le raisonnement conservé dans le fichier est abrégé.",
    "en": "15 of 47 calls refused, including 10 sweeps in a row with a section and conditions the engine does not support; it then had its script validated and ran 10 successful backtests. Handed in two near-identical crosses (11/209 and 12/211), labelled “robust”. The reasoning kept in the file is abridged.",
    "methodFr": "Lecture large de la documentation (contrat du moteur, exploration du Lab) et assemblage de plusieurs scripts avant le premier test, puis série de balayages refusés. Pistes visibles dans les refus : filtre ADX, paramètre de régime.",
    "methodEn": "Wide documentation read (engine contract, Lab exploration) and several scripts assembled before the first test, then a series of refused sweeps. Ideas visible in the refusals: ADX filter, regime parameter.",
    "ideasFr": "croisement de deux moyennes, prix au-dessus d'une moyenne, filtre ADX, filtre de régime",
    "ideasEn": "two-average cross, price above an average, ADX filter, regime filter",
    "choiceFr": "Non détaillé dans le fichier conservé (raisonnement abrégé).",
    "choiceEn": "Not detailed in the kept file (abridged reasoning).",
    "checks": [],
    "checksUnknown": true
   },
   "flags": [
    {
     "kind": "integrity",
     "fr": "A connu le résultat de l'achat-conservation de la période scellée",
     "en": "Knew the buy-and-hold result of the sealed period",
     "detailFr": "Son raisonnement cite le chiffre de l'achat-conservation sur la période scellée (+201,8 %), qui était visible dans une consigne et sur la page publique.",
     "detailEn": "Its reasoning quotes the buy-and-hold figure for the sealed period (+201.8%), which was visible in a brief and on the public page."
    }
   ]
  },
  {
   "runId": "muse-1",
   "model": "muse",
   "family": "Muse",
   "name": "Muse Spark 1.3 (contributor, free)",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 100,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 132.13,
    "maxDrawdownPct": 31.18,
    "ratio": 4.24,
    "trades": 84,
    "ran": true,
    "errorCode": null,
    "name": "Cash under SMA200",
    "script": "strategy \"Cash under SMA200\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\nentry:\n    condition: price > SMA(200)\n    size: 100%\nexit:\n    # let winners run — no take-profit, no timeout\n",
    "params": {},
    "beatsBuyHoldRiskAdjusted": true,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 200
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 132.13,
     "maxDrawdownPct": 31.18,
     "ratio": 4.24,
     "trades": 84,
     "ran": true,
     "errorCode": null,
     "name": "Cash under SMA200",
     "primary": true
    }
   ],
   "toolCalls": 34,
   "blockedAttempts": 0,
   "log": "log_muse-1.jsonl",
   "process": {
    "calls": 34,
    "successfulCalls": 29,
    "successPct": 85,
    "failedCalls": 5,
    "failedPct": 15,
    "durationMin": 5.5,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 18
     },
     {
      "tool": "assemble_strategy_contract",
      "count": 4
     },
     {
      "tool": "tools/list",
      "count": 3
     },
     {
      "tool": "validate_sandbox_strategy",
      "count": 3
     }
    ],
    "sweeps": 3,
    "sweepsRan": 1,
    "backtests": 18,
    "backtestsRan": 17,
    "testsRan": 18,
    "holdout": {
     "status": "yes",
     "percents": [
      30
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 4
     },
     {
      "code": "sandbox.sweep_parameter_unsupported",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 3,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 3,
    "callsBeforeFirstTest": 8,
    "firstTestCall": 9,
    "maxFailStreak": 2,
    "blockedAttempts": 0,
    "timeline": "DDDDBbbBrDBBBRRRssDSRRRRRRRRRRRRRR"
   },
   "rank": 2,
   "tiedRank": false,
   "commentary": {
    "fr": "A gardé la règle la plus simple, prix au-dessus de la moyenne 200, et a écarté exprès la 250, meilleure sur les données de recherche, pour ne pas choisir un pic. A surtout travaillé par backtests un par un (18) plutôt que par balayages (3).",
    "en": "Kept the simplest rule, price above the 200 average, and deliberately set aside the 250, better on the research data, so as not to pick a peak. Worked mostly with one-off backtests (18) rather than sweeps (3).",
    "methodFr": "Documentation, construction et validation du script (8 appels avant le premier test), puis backtests ciblés : découpe 70/30 refaite à la main, trois sous-périodes (hausse, baisse, reprise) et longueurs voisines. Un balayage de 50 variantes de sorties (stop, objectif, durée) avec 30 % mis de côté a montré que ces sorties faisaient moins bien que laisser courir.",
    "methodEn": "Documentation, script building and validation (8 calls before the first test), then targeted backtests: a 70/30 split done by hand, three sub-periods (rise, fall, recovery) and neighbouring lengths. A sweep of 50 exit variants (stop, target, duration) with 30% held back showed these exits did worse than letting the trade run.",
    "ideasFr": "prix au-dessus d'une moyenne simple (100 à 300), sorties par stop, objectif ou durée, canal Turtle 20",
    "ideasEn": "price above a simple average (100 to 300), stop, target or duration exits, Turtle 20 channel",
    "choiceFr": "Le centre d'une zone 100-300 toutes positives, pas le meilleur réglage d'entraînement (250).",
    "choiceEn": "The centre of a 100-300 zone that was positive throughout, not the best training setting (250).",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "deepseek-1",
   "model": "deepseek",
   "family": "DeepSeek",
   "name": "deepseek-flash",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 100,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 176.41,
    "maxDrawdownPct": 41.46,
    "ratio": 4.25,
    "trades": 38,
    "ran": true,
    "errorCode": null,
    "name": "DeepSeek SMA Long T120 F17 S298",
    "script": "strategy \"DeepSeek SMA Long T120 F17 S298\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: SMA(17) > SMA(298)\n    size: 100%\n\nexit:\n    timeout: 120 bars\n",
    "params": {
     "fast": 17,
     "slow": 298
    },
    "beatsBuyHoldRiskAdjusted": true,
    "labelDelta": 20,
    "shape": {
     "kind": "maCross",
     "fast": 17,
     "slow": 298
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 176.41,
     "maxDrawdownPct": 41.46,
     "ratio": 4.25,
     "trades": 38,
     "ran": true,
     "errorCode": null,
     "name": "DeepSeek SMA Long T120 F17 S298",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 124.83,
     "maxDrawdownPct": 51.2,
     "ratio": 2.44,
     "trades": 39,
     "ran": true,
     "errorCode": null,
     "name": "DeepSeek SMA Long T120 F16 S300",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 244.15,
     "maxDrawdownPct": 43.53,
     "ratio": 5.61,
     "trades": 40,
     "ran": true,
     "errorCode": null,
     "name": "DeepSeek SMA Long T120 F12 S236",
     "primary": false
    }
   ],
   "toolCalls": 21,
   "blockedAttempts": 0,
   "log": "log_deepseek-1.jsonl",
   "process": {
    "calls": 21,
    "successfulCalls": 21,
    "successPct": 100,
    "failedCalls": 0,
    "failedPct": 0,
    "durationMin": 5.7,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 11
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 9
     },
     {
      "tool": "validate_sandbox_strategy",
      "count": 1
     }
    ],
    "sweeps": 9,
    "sweepsRan": 9,
    "backtests": 11,
    "backtestsRan": 11,
    "testsRan": 20,
    "holdout": {
     "status": "yes",
     "percents": [
      25
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [],
    "docReads": {
     "tools/list": 0,
     "get_sandbox_capabilities": 0,
     "list_sandbox_datasets": 0
    },
    "docReadsTotal": 0,
    "docReadsBeforeFirstTest": 0,
    "callsBeforeFirstTest": 0,
    "firstTestCall": 1,
    "maxFailStreak": 0,
    "blockedAttempts": 0,
    "timeline": "SSSSSSSSSRRRRRRRRRRRB"
   },
   "rank": 3,
   "tiedRank": false,
   "commentary": {
    "fr": "Aucune lecture de documentation enregistrée : le premier appel est déjà un balayage, et les 21 appels ont tous réussi. Voyant que les meilleurs réglages d'entraînement s'effondraient sur les données mises de côté, il a abandonné les positions vendeuses et ajouté une sortie au bout de 120 bougies, puis a pris le centre d'une zone de réglages voisins (17/298).",
    "en": "No documentation read recorded: the first call is already a sweep, and all 21 calls succeeded. Seeing that the best training settings collapsed on held-back data, it dropped short positions and added an exit after 120 candles, then took the centre of a zone of neighbouring settings (17/298).",
    "methodFr": "9 balayages (dont deux de 120 variantes classées de deux façons), puis 11 backtests : carte de la zone stable (moyenne courte 12-26, longue 236-320), trois durées de sortie, années 2021 et 2022 séparées, frais à 0 %, 0,1 % et 0,3 % par côté.",
    "methodEn": "9 sweeps (two of them with 120 variants ranked two ways), then 11 backtests: map of the stable zone (short average 12-26, long 236-320), three exit lengths, 2021 and 2022 checked separately, fees at 0%, 0.1% and 0.3% per side.",
    "ideasFr": "deux moyennes à l'achat et à la vente, puis à l'achat seul avec sortie au bout d'un temps fixe",
    "ideasEn": "two averages long and short, then long only with an exit after a fixed time",
    "choiceFr": "Le centre de la zone stable, avec le meilleur rapport gain / pire baisse sur les données de recherche ; il note que sa partie mise de côté fait moins bien que l'achat-conservation.",
    "choiceEn": "The centre of the stable zone, with the best return / worst drop on the research data; it notes that its held-back part did worse than buy-and-hold.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold",
     "feeStress"
    ]
   },
   "flags": []
  },
  {
   "runId": "sonnet-1",
   "model": "sonnet",
   "family": "Sonnet",
   "name": "Claude Sonnet 5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 80,
   "abstained": false,
   "primary": {
    "label": "exploratory",
    "returnPct": 239.81,
    "maxDrawdownPct": 26.43,
    "ratio": 9.07,
    "trades": 34,
    "ran": true,
    "errorCode": null,
    "name": "EMA cross 40/200",
    "script": "strategy \"EMA cross 40/200\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: EMA(40) crosses above EMA(200)\n    size: 100%\n\nexit:\n    timeout: 100000 bars\nparams:\n    fast: min=5 max=60 type=int default=40\n    slow: min=70 max=300 type=int default=200\n",
    "params": {
     "fast": 40,
     "slow": 200
    },
    "beatsBuyHoldRiskAdjusted": true,
    "labelDelta": -20,
    "shape": {
     "kind": "maCross",
     "fast": 40,
     "slow": 200
    }
   },
   "candidates": [
    {
     "label": "exploratory",
     "returnPct": 239.81,
     "maxDrawdownPct": 26.43,
     "ratio": 9.07,
     "trades": 34,
     "ran": true,
     "errorCode": null,
     "name": "EMA cross 40/200",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 176.15,
     "maxDrawdownPct": 29.74,
     "ratio": 5.92,
     "trades": 37,
     "ran": true,
     "errorCode": null,
     "name": "SMA cross 15/210",
     "primary": false
    }
   ],
   "toolCalls": 43,
   "blockedAttempts": 0,
   "log": "log_sonnet-1.jsonl",
   "process": {
    "calls": 43,
    "successfulCalls": 42,
    "successPct": 98,
    "failedCalls": 1,
    "failedPct": 2,
    "durationMin": 2.5,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 19
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 16
     },
     {
      "tool": "tools/list",
      "count": 2
     },
     {
      "tool": "get_sandbox_capabilities",
      "count": 2
     }
    ],
    "sweeps": 19,
    "sweepsRan": 19,
    "backtests": 16,
    "backtestsRan": 15,
    "testsRan": 34,
    "holdout": {
     "status": "yes",
     "percents": [
      30
     ],
     "sources": [
      "request"
     ]
    },
    "errorsTop": [
     {
      "code": "sandbox.preflight_failed",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 2,
     "get_sandbox_capabilities": 2,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 5,
    "callsBeforeFirstTest": 7,
    "firstTestCall": 8,
    "maxFailStreak": 1,
    "blockedAttempts": 0,
    "timeline": "DDDDDBDRrRRSSSSRRRRRSDSSRSSSSSSSSSSSSRRRRRR"
   },
   "rank": 4,
   "tiedRank": false,
   "commentary": {
    "fr": "34 tests réussis en 43 appels, un seul refusé. A étiqueté ses deux stratégies « exploratoires » parce qu'elles perdaient hors de la hausse 2020-21 ; elles ont gagné sur la période scellée, et « robuste » aurait rapporté 20 points de plus.",
    "en": "34 successful tests in 43 calls, only one refused. Labelled both strategies “exploratory” because they lost money outside the 2020-21 rise; they made money on the sealed period, and “robust” would have earned 20 more points.",
    "methodFr": "Lecture courte de la documentation, puis alternance de backtests et de balayages, jusqu'à 800 réglages d'un coup avec une partie des données mises de côté. A constaté que les entrées « prix au-dessus d'une moyenne » ne se refermaient jamais dans ce moteur, et a construit ses entrées sur des croisements, qui se referment au croisement inverse. Stops et stops suiveurs testés puis abandonnés.",
    "methodEn": "Short documentation read, then backtests and sweeps in turn, up to 800 settings at once with part of the data held back. Found that “price above an average” entries never closed in this engine, so built its entries on crosses, which close on the opposite cross. Stops and trailing stops tested, then dropped.",
    "ideasFr": "croisements de moyennes simples et exponentielles, stops fixes et suiveurs",
    "ideasEn": "simple and exponential average crosses, fixed and trailing stops",
    "choiceFr": "Entre deux candidates proches, celle qui faisait mieux sur les données mises de côté et sur la dernière année.",
    "choiceEn": "Between two close candidates, the one that did better on held-back data and on the last year.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "mimo-2",
   "model": "mimo",
   "family": "MiMo",
   "name": "Xiaomi MiMo v2.6 Pro",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 80,
   "abstained": false,
   "primary": {
    "label": "exploratory",
    "returnPct": 216.54,
    "maxDrawdownPct": 27.53,
    "ratio": 7.87,
    "trades": 34,
    "ran": true,
    "errorCode": null,
    "name": "Dual SMA 17/280",
    "script": "strategy \"Dual SMA 17/280\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: SMA(17) > SMA(280)\n    size: 100%\n\nexit:\n    # let winners run\n\nparams:\n    fast: min=10 max=80 type=int default=17\n    slow: min=100 max=300 type=int default=280\n",
    "params": {
     "fast": 17,
     "slow": 280
    },
    "beatsBuyHoldRiskAdjusted": true,
    "labelDelta": -20,
    "shape": {
     "kind": "maCross",
     "fast": 17,
     "slow": 280
    }
   },
   "candidates": [
    {
     "label": "exploratory",
     "returnPct": 216.54,
     "maxDrawdownPct": 27.53,
     "ratio": 7.87,
     "trades": 34,
     "ran": true,
     "errorCode": null,
     "name": "Dual SMA 17/280",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 103.42,
     "maxDrawdownPct": 41.07,
     "ratio": 2.52,
     "trades": 16,
     "ran": true,
     "errorCode": null,
     "name": "Dual SMA 70/250",
     "primary": false
    }
   ],
   "toolCalls": 47,
   "blockedAttempts": 0,
   "log": "log_mimo-2.jsonl",
   "process": {
    "calls": 47,
    "successfulCalls": 24,
    "successPct": 51,
    "failedCalls": 23,
    "failedPct": 49,
    "durationMin": 16.6,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 19
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 11
     },
     {
      "tool": "validate_sandbox_strategy",
      "count": 4
     },
     {
      "tool": "assemble_strategy_contract",
      "count": 3
     }
    ],
    "sweeps": 11,
    "sweepsRan": 4,
    "backtests": 19,
    "backtestsRan": 10,
    "testsRan": 14,
    "holdout": {
     "status": "yes",
     "percents": [
      40
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 17
     },
     {
      "code": "missing_strategy_input",
      "count": 3
     },
     {
      "code": "unsupported_condition",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 1,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 3,
    "docReadsBeforeFirstTest": 3,
    "callsBeforeFirstTest": 13,
    "firstTestCall": 14,
    "maxFailStreak": 9,
    "blockedAttempts": 0,
    "timeline": "DDDDBBBbdbdBbrBrbrsbrsBrsrsrsrsrSRsSRRRSRRRRRRS"
   },
   "rank": 5,
   "tiedRank": false,
   "commentary": {
    "fr": "23 appels refusés sur 47 : il a découvert les champs obligatoires un par un, en envoyant chaque fois un backtest et un balayage avec la même erreur. A choisi 17/280, meilleur réglage sur 30 % mis de côté et seul positif sur 40 % ; l'étiquette « exploratoire » lui a coûté 20 points par rapport à « robuste ».",
    "en": "23 of 47 calls refused: it found the required fields one by one, each time sending a backtest and a sweep with the same error. Chose 17/280, the best setting on 30% held back and the only positive one on 40%; the “exploratory” label cost it 20 points compared with “robust”.",
    "methodFr": "Documentation, construction et validation, puis 9 refus d'affilée en ajoutant un champ manquant par essai. Une fois la requête correcte : balayages avec 30 et 40 % mis de côté, grilles de réglages voisins, trois sous-périodes et un achat-conservation de contrôle. A calculé que « exploratoire » rapportait plus en moyenne que s'abstenir.",
    "methodEn": "Documentation, build and validation, then 9 refusals in a row while adding one missing field per try. Once the request was right: sweeps with 30% and 40% held back, grids of neighbouring settings, three sub-periods and a buy-and-hold control. Worked out that “exploratory” was worth more on average than abstaining.",
    "ideasFr": "croisement de deux moyennes, canal de prix",
    "ideasEn": "two-average cross, price channel",
    "choiceFr": "Le réglage le plus régulier d'une découpe à l'autre.",
    "choiceEn": "The setting most consistent from one split to the next.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "opus-1",
   "model": "opus",
   "family": "Opus",
   "name": "Claude Opus 5.5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 95.35,
    "maxDrawdownPct": 33.45,
    "ratio": 2.85,
    "trades": 89,
    "ran": true,
    "errorCode": null,
    "name": "EMA trend 300",
    "script": "strategy \"EMA trend 300\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: price > EMA(300)\n    size: 100%\n\nparams:\n    emaLength: min=20 max=800 type=int default=300\n",
    "params": {
     "emaLength": 300
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 300
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 95.35,
     "maxDrawdownPct": 33.45,
     "ratio": 2.85,
     "trades": 89,
     "ran": true,
     "errorCode": null,
     "name": "EMA trend 300",
     "primary": true
    },
    {
     "label": "robust",
     "returnPct": 77.95,
     "maxDrawdownPct": 41.31,
     "ratio": 1.89,
     "trades": 93,
     "ran": true,
     "errorCode": null,
     "name": "EMA trend 250",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 132.13,
     "maxDrawdownPct": 31.18,
     "ratio": 4.24,
     "trades": 84,
     "ran": true,
     "errorCode": null,
     "name": "Slow trend",
     "primary": false
    }
   ],
   "toolCalls": 27,
   "blockedAttempts": 0,
   "log": "log_opus-1.jsonl",
   "process": {
    "calls": 27,
    "successfulCalls": 23,
    "successPct": 85,
    "failedCalls": 4,
    "failedPct": 15,
    "durationMin": 2.9,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 13
     },
     {
      "tool": "mint_native_candidate",
      "count": 4
     },
     {
      "tool": "tools/list",
      "count": 2
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 2
     }
    ],
    "sweeps": 13,
    "sweepsRan": 11,
    "backtests": 2,
    "backtestsRan": 2,
    "testsRan": 13,
    "holdout": {
     "status": "yes",
     "percents": [
      50
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 1
     },
     {
      "code": "sandbox.preflight_failed",
      "count": 1
     },
     {
      "code": "unknown_native_family",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 2,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 4,
    "docReadsBeforeFirstTest": 4,
    "callsBeforeFirstTest": 6,
    "firstTestCall": 7,
    "maxFailStreak": 2,
    "blockedAttempts": 0,
    "timeline": "DDDDBbsDSSbDBSSsBBSSSSSSSRR"
   },
   "rank": 6,
   "tiedRank": true,
   "commentary": {
    "fr": "A lu la documentation (4 appels) puis a testé dès l'appel 7. A gardé la moyenne exponentielle 300 parce qu'elle était la meilleure sur chaque tranche de données récentes, pas seulement sur l'ensemble. L'étiquette « robuste » lui a rapporté 20 points de plus que « exploratoire ».",
    "en": "Read the documentation (4 calls), then started testing at call 7. Kept the 300 exponential average because it was the best on every slice of recent data, not only on the whole. The “robust” label earned it 20 more points than “exploratory”.",
    "methodFr": "Documentation, un script de base validé, puis 13 balayages de réglages et 2 backtests de contrôle à la fin. Les filtres ajoutés (ADX, croisement de deux moyennes) faisaient moins bien sur les données mises de côté et ont été écartés. Il a remarqué qu'un nom de paramètre non déclaré ne changeait rien au résultat et a corrigé ; jamais plus de 2 refus d'affilée.",
    "methodEn": "Documentation, one validated base script, then 13 parameter sweeps and 2 control backtests at the end. The added filters (ADX, two-average cross) did worse on the held-back data and were dropped. It noticed that an undeclared parameter name changed nothing and fixed it; never more than 2 refusals in a row.",
    "ideasFr": "prix au-dessus d'une moyenne longue, filtre ADX, croisement de deux moyennes",
    "ideasEn": "price above a slow average, ADX filter, two-average cross",
    "choiceFr": "La longueur la meilleure sur toutes les tranches récentes (deux découpes à l'aveugle et la dernière année), avec la plus faible baisse sur l'ensemble.",
    "choiceEn": "The length that was best on every recent slice (two blind splits and the last year), with the smallest drop overall.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "fable-2",
   "model": "fable",
   "family": "Fable",
   "name": "Claude Fable 5.1",
   "route": "claude-agent",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 95.35,
    "maxDrawdownPct": 33.45,
    "ratio": 2.85,
    "trades": 89,
    "ran": true,
    "errorCode": null,
    "name": "EMA trend 300",
    "script": "strategy \"EMA trend 300\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: price > EMA(300)\n    size: 100%\n\nexit:\n    # let winners run\n",
    "params": {
     "emaLength": 300
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 300
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 95.35,
     "maxDrawdownPct": 33.45,
     "ratio": 2.85,
     "trades": 89,
     "ran": true,
     "errorCode": null,
     "name": "EMA trend 300",
     "primary": true
    },
    {
     "label": "robust",
     "returnPct": 122.42,
     "maxDrawdownPct": 36.3,
     "ratio": 3.37,
     "trades": 76,
     "ran": true,
     "errorCode": null,
     "name": "SMA trend 250",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 84.41,
     "maxDrawdownPct": 32.31,
     "ratio": 2.61,
     "trades": 47,
     "ran": true,
     "errorCode": null,
     "name": "SMA300 conf SMA50",
     "primary": false
    }
   ],
   "toolCalls": 55,
   "blockedAttempts": 0,
   "log": "log_fable-2.jsonl",
   "process": {
    "calls": 55,
    "successfulCalls": 54,
    "successPct": 98,
    "failedCalls": 1,
    "failedPct": 2,
    "durationMin": 9.8,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 19
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 15
     },
     {
      "tool": "tools/list",
      "count": 4
     },
     {
      "tool": "describe_native_sweep_space",
      "count": 4
     }
    ],
    "sweeps": 15,
    "sweepsRan": 14,
    "backtests": 19,
    "backtestsRan": 19,
    "testsRan": 33,
    "holdout": {
     "status": "yes",
     "percents": [
      40
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "sandbox.sweep_parameter_unsupported",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 4,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 6,
    "docReadsBeforeFirstTest": 6,
    "callsBeforeFirstTest": 10,
    "firstTestCall": 11,
    "maxFailStreak": 1,
    "blockedAttempts": 0,
    "timeline": "DDDDDDBDBBRRsDDDBDBBBDDRRSSBSSSSSSSRRRSRRRRSRRRRRRRRSSS"
   },
   "rank": 6,
   "tiedRank": true,
   "commentary": {
    "fr": "55 appels, 54 réussis. A recalculé les résultats par année à partir des journaux de trades, parce que le test à l'aveugle du Lab redémarre sans historique de moyenne. A gardé la moyenne exponentielle 300 au centre d'une zone stable et a laissé « exploratoire » une variante qu'il n'avait pas pu balayer.",
    "en": "55 calls, 54 successful. Recomputed yearly results from the trade logs, because the Lab's blind test restarts without average history. Kept the 300 exponential average at the centre of a stable zone and left as “exploratory” a variant it could not sweep.",
    "methodFr": "Exploration plus large des outils du Lab au début et après les premiers backtests (familles toutes faites, espace de balayage, contrat du moteur), puis balayages et backtests en alternance. Découpes à l'aveugle de 20 à 50 % et tranches par régime (baisse, marché sans direction).",
    "methodEn": "Wider exploration of the Lab's tools at the start and after the first backtests (ready-made families, sweep space, engine contract), then sweeps and backtests in turn. Blind splits from 20 to 50% and slices by regime (falling, sideways market).",
    "ideasFr": "prix au-dessus d'une moyenne simple ou exponentielle, croisements, double moyenne, confirmation par momentum, ADX, stop suiveur",
    "ideasEn": "price above a simple or exponential average, crosses, double average, momentum confirmation, ADX, trailing stop",
    "choiceFr": "Le centre de la zone stable (moyennes 196-308).",
    "choiceEn": "The centre of the stable zone (averages 196-308).",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "sonnet-2",
   "model": "sonnet",
   "family": "Sonnet",
   "name": "Claude Sonnet 5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 108.19,
    "maxDrawdownPct": 37.27,
    "ratio": 2.9,
    "trades": 81,
    "ran": true,
    "errorCode": null,
    "name": "BTC Slow Trend SMA245",
    "script": "strategy \"BTC Slow Trend SMA245\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n  condition: price > SMA(245)\n  size: 100%\n\nexit:\n\nparams:\n  smaLength: min=30 max=400 type=int default=245\n",
    "params": {
     "smaLength": 245
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 245
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 108.19,
     "maxDrawdownPct": 37.27,
     "ratio": 2.9,
     "trades": 81,
     "ran": true,
     "errorCode": null,
     "name": "BTC Slow Trend SMA245",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 101.84,
     "maxDrawdownPct": 31.76,
     "ratio": 3.21,
     "trades": 74,
     "ran": true,
     "errorCode": null,
     "name": "BTC Slow Trend SMA300",
     "primary": false
    }
   ],
   "toolCalls": 35,
   "blockedAttempts": 0,
   "log": "log_sonnet-2.jsonl",
   "process": {
    "calls": 35,
    "successfulCalls": 25,
    "successPct": 71,
    "failedCalls": 10,
    "failedPct": 29,
    "durationMin": 1.9,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 10
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 8
     },
     {
      "tool": "tools/list",
      "count": 4
     },
     {
      "tool": "get_sandbox_capabilities",
      "count": 3
     }
    ],
    "sweeps": 8,
    "sweepsRan": 7,
    "backtests": 10,
    "backtestsRan": 4,
    "testsRan": 11,
    "holdout": {
     "status": "yes",
     "percents": [
      30,
      50
     ],
     "sources": [
      "request",
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 6
     },
     {
      "code": "unknown_native_family",
      "count": 3
     },
     {
      "code": "sandbox.preflight_failed",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 4,
     "get_sandbox_capabilities": 3,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 8,
    "docReadsBeforeFirstTest": 7,
    "callsBeforeFirstTest": 16,
    "firstTestCall": 17,
    "maxFailStreak": 6,
    "blockedAttempts": 0,
    "timeline": "DDDDBDDDbDbDDDbBsSSSSrrrrrrDRRRRSSS"
   },
   "rank": 8,
   "tiedRank": false,
   "commentary": {
    "fr": "10 appels refusés sur 35, dont 6 backtests lancés ensemble avec le même champ manquant, corrigé ensuite. A choisi la moyenne 245 au milieu d'une zone 230-290 vérifiée sur deux découpes à l'aveugle.",
    "en": "10 of 35 calls refused, including 6 backtests sent together with the same missing field, fixed afterwards. Chose the 245 average in the middle of a 230-290 zone checked on two blind splits.",
    "methodFr": "16 appels avant le premier test : documentation et plusieurs noms de familles toutes faites inexacts. Ensuite, un balayage d'une seule longueur de moyenne (30 à 400), une comparaison avec les croisements de deux moyennes et des backtests de contrôle. Pas de position vendeuse, en citant des résultats antérieurs du projet.",
    "methodEn": "16 calls before the first test: documentation and several wrong names for ready-made families. Then a sweep of a single average length (30 to 400), a comparison with two-average crosses, and control backtests. No short positions, citing earlier project findings.",
    "ideasFr": "prix au-dessus d'une moyenne simple, croisement de deux moyennes",
    "ideasEn": "price above a simple average, two-average cross",
    "choiceFr": "Le milieu de la zone stable.",
    "choiceEn": "The middle of the stable zone.",
    "checks": [
     "holdout",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "deepseek-2",
   "model": "deepseek",
   "family": "DeepSeek",
   "name": "deepseek-flash",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 103.28,
    "maxDrawdownPct": 38.03,
    "ratio": 2.72,
    "trades": 81,
    "ran": true,
    "errorCode": null,
    "name": "BTC 4h SMA trend filter",
    "script": "strategy \"BTC 4h SMA trend filter\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n# Long-only trend filter: hold while close > SMA(240) (~40 days on 4h), otherwise cash.\n# Let-run: no take-profit, no stop, no timeout; exit when the entry thesis is false.\n\nparams:\n    smaLength: min=60 max=400 type=int default=240\n\nentry:\n    condition: price > SMA(240)\n    size: 100%\n\nexit:\n",
    "params": {
     "smaLength": 240
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 240
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 103.28,
     "maxDrawdownPct": 38.03,
     "ratio": 2.72,
     "trades": 81,
     "ran": true,
     "errorCode": null,
     "name": "BTC 4h SMA trend filter",
     "primary": true
    }
   ],
   "toolCalls": 24,
   "blockedAttempts": 0,
   "log": "log_deepseek-2.jsonl",
   "process": {
    "calls": 24,
    "successfulCalls": 19,
    "successPct": 79,
    "failedCalls": 5,
    "failedPct": 21,
    "durationMin": 5.3,
    "toolMix": [
     {
      "tool": "validate_sandbox_strategy",
      "count": 7
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 6
     },
     {
      "tool": "tools/list",
      "count": 3
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 2
     }
    ],
    "sweeps": 6,
    "sweepsRan": 6,
    "backtests": 2,
    "backtestsRan": 2,
    "testsRan": 8,
    "holdout": {
     "status": "yes",
     "percents": [
      20,
      35
     ],
     "sources": [
      "request"
     ]
    },
    "errorsTop": [
     {
      "code": "unsupported_condition",
      "count": 3
     },
     {
      "code": "invalid_parameter_metadata",
      "count": 1
     },
     {
      "code": "invalid_number",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 3,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 5,
    "callsBeforeFirstTest": 15,
    "firstTestCall": 16,
    "maxFailStreak": 5,
    "blockedAttempts": 0,
    "timeline": "DDDDBDbbbbbDDBBSSSSSSBRR"
   },
   "rank": 9,
   "tiedRank": false,
   "commentary": {
    "fr": "A choisi une seule idée avant de tester (prix au-dessus d'une moyenne simple) et a rendu une seule stratégie, en expliquant que des candidates en plus n'ajoutent que du bruit. 5 refus sur 24, tous à la validation du script, le temps de trouver une écriture de paramètre acceptée.",
    "en": "Picked a single idea before testing (price above a simple average) and handed in a single strategy, explaining that extra candidates only add noise. 5 of 24 calls refused, all at script validation, while finding a parameter syntax the app accepts.",
    "methodFr": "15 appels avant le premier test, surtout pour faire valider le script. Ensuite une courbe de 18 longueurs (60 à 400), deux découpes chronologiques (65/35 et 80/20), trois sous-périodes de régime et un achat-conservation mesuré avec le même moteur.",
    "methodEn": "15 calls before the first test, mostly to get the script validated. Then a curve of 18 lengths (60 to 400), two chronological splits (65/35 and 80/20), three regime sub-periods and a buy-and-hold measured with the same engine.",
    "ideasFr": "une seule famille : prix au-dessus d'une moyenne simple",
    "ideasEn": "a single family: price above a simple average",
    "choiceFr": "Le haut d'une zone stable (160 à 400 toutes positives) ; il signale lui-même la découpe 80/20 négative.",
    "choiceEn": "The top of a stable zone (160 to 400 all positive); it flags the negative 80/20 split itself.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "fable-1",
   "model": "fable",
   "family": "Fable",
   "name": "Claude Fable 5.1",
   "route": "claude-agent",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 77.13,
    "maxDrawdownPct": 38.28,
    "ratio": 2.01,
    "trades": 96,
    "ran": true,
    "errorCode": null,
    "name": "BTC trend filter SMA280 + 30d momentum",
    "script": "strategy \"BTC trend filter SMA280 + 30d momentum\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n# Long-only regime filter: hold BTC only while price is above its 280-bar (~47-day) SMA\n# AND the 180-bar (~30-day) rate of change is positive. Flat (cash) otherwise.\n\nentry:\n    condition: price > SMA(280)\n    and ROC(180) > 0\n    size: 100%\n\nexit:\n    signal_exit: on_entry_failure\n",
    "params": {
     "smaPeriod": 280,
     "rocPeriod": 180,
     "sizePct": 100,
     "feeBpsAssumed": 10,
     "scriptSha256": "9267919f49c0fc4e7878f35804eeff34bd957afe3c4e6ad9e864741047bbf7ed"
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 280
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 77.13,
     "maxDrawdownPct": 38.28,
     "ratio": 2.01,
     "trades": 96,
     "ran": true,
     "errorCode": null,
     "name": "BTC trend filter SMA280 + 30d momentum",
     "primary": true
    },
    {
     "label": "robust",
     "returnPct": 102.2,
     "maxDrawdownPct": 33.87,
     "ratio": 3.02,
     "trades": 84,
     "ran": true,
     "errorCode": null,
     "name": "BTC trend filter SMA280",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 54.16,
     "maxDrawdownPct": 40.05,
     "ratio": 1.35,
     "trades": 116,
     "ran": true,
     "errorCode": null,
     "name": "BTC 30d time-series momentum",
     "primary": false
    }
   ],
   "toolCalls": 45,
   "blockedAttempts": 0,
   "log": "log_fable-1.jsonl",
   "process": {
    "calls": 45,
    "successfulCalls": 43,
    "successPct": 96,
    "failedCalls": 2,
    "failedPct": 4,
    "durationMin": 6.9,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 32
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 4
     },
     {
      "tool": "tools/list",
      "count": 3
     },
     {
      "tool": "assemble_strategy_contract",
      "count": 2
     }
    ],
    "sweeps": 32,
    "sweepsRan": 31,
    "backtests": 4,
    "backtestsRan": 4,
    "testsRan": 35,
    "holdout": {
     "status": "yes",
     "percents": [
      40
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "sandbox.sweep_parameter_unsupported",
      "count": 1
     },
     {
      "code": "unsupported_directive",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 3,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 5,
    "callsBeforeFirstTest": 8,
    "firstTestCall": 9,
    "maxFailStreak": 1,
    "blockedAttempts": 0,
    "timeline": "DDDDDBBBSSSSsSSSSSRbSSSSSSSSSSSSSSSSSSSSSSRRR"
   },
   "rank": 10,
   "tiedRank": false,
   "commentary": {
    "fr": "32 balayages, dont 31 réussis, et seulement 2 refus. A gardé une combinaison moyenne 280 + momentum 180, positive sur 6 découpes sur 7, et a laissé « exploratoire » une variante aux meilleurs chiffres bruts parce que ses réglages voisins variaient trop.",
    "en": "32 sweeps, 31 of which ran, and only 2 refusals. Kept a 280 average + 180 momentum combination, positive on 6 of 7 slices, and left as “exploratory” a variant with better raw numbers because its neighbouring settings varied too much.",
    "methodFr": "Documentation, un script de base, puis presque uniquement des balayages : chaque réglage cherché sur la partie ancienne et rejoué sur une fin de données intacte (de 20 à 50 % mis de côté), puis contrôlé sur trois années séparées. Les deux refus venaient de paramètres non pris en charge, abandonnés aussitôt.",
    "methodEn": "Documentation, one base script, then almost only sweeps: each setting searched on the older part and replayed on an untouched end of the data (20 to 50% held back), then checked on three separate years. The two refusals came from unsupported parameters, dropped at once.",
    "ideasFr": "prix au-dessus d'une moyenne simple ou exponentielle, croisements, ADX, filtre de régime, momentum, canal de Donchian",
    "ideasEn": "price above a simple or exponential average, crosses, ADX, regime filter, momentum, Donchian channel",
    "choiceFr": "Positive sur 6 découpes sur 7, plus petite perte en année baissière, réglages au milieu de zones larges.",
    "choiceEn": "Positive on 6 of 7 slices, smallest loss in the falling year, settings in the middle of wide zones.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "muse-2",
   "model": "muse",
   "family": "Muse",
   "name": "Muse Spark 1.3 (contributor, free)",
   "route": "cloud-opencode",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 109.03,
    "maxDrawdownPct": 39.73,
    "ratio": 2.74,
    "trades": 21,
    "ran": true,
    "errorCode": null,
    "name": "Donchian55 robust",
    "script": "strategy \"Donchian55 robust\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: price > DONCHIAN_HIGH(55)\n    size: 100%\n\nexit:\n    stop_loss: 12.07%\n    take_profit: 49.2%\n    timeout: 247 bars\n",
    "params": {},
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": null
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 109.03,
     "maxDrawdownPct": 39.73,
     "ratio": 2.74,
     "trades": 21,
     "ran": true,
     "errorCode": null,
     "name": "Donchian55 robust",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 326.39,
     "maxDrawdownPct": 35.24,
     "ratio": 9.26,
     "trades": 38,
     "ran": true,
     "errorCode": null,
     "name": "Dual SMA robust",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": 56.37,
     "maxDrawdownPct": 58.59,
     "ratio": 0.96,
     "trades": 33,
     "ran": true,
     "errorCode": null,
     "name": "EMA50 ADX robust",
     "primary": false
    }
   ],
   "toolCalls": 32,
   "blockedAttempts": 0,
   "log": "log_muse-2.jsonl",
   "process": {
    "calls": 32,
    "successfulCalls": 28,
    "successPct": 88,
    "failedCalls": 4,
    "failedPct": 12,
    "durationMin": 4.6,
    "toolMix": [
     {
      "tool": "run_sandbox_backtest",
      "count": 10
     },
     {
      "tool": "run_sandbox_sweep",
      "count": 6
     },
     {
      "tool": "tools/list",
      "count": 4
     },
     {
      "tool": "validate_sandbox_strategy",
      "count": 4
     }
    ],
    "sweeps": 6,
    "sweepsRan": 5,
    "backtests": 10,
    "backtestsRan": 10,
    "testsRan": 15,
    "holdout": {
     "status": "yes",
     "percents": [
      30
     ],
     "sources": [
      "savedSweepOutput"
     ]
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 1
     },
     {
      "code": "unsupported_condition",
      "count": 1
     },
     {
      "code": "sandbox.preflight_failed",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 4,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 6,
    "docReadsBeforeFirstTest": 6,
    "callsBeforeFirstTest": 12,
    "firstTestCall": 13,
    "maxFailStreak": 1,
    "blockedAttempts": 0,
    "timeline": "DDDDBbDBDbDDsDbBBRRSSSSRRRRRRSRR"
   },
   "rank": 11,
   "tiedRank": false,
   "commentary": {
    "fr": "Stratégie principale : canal de Donchian 55, réglage classique de la méthode Turtle (système 2), pris tel quel et non trouvé par recherche. Sa candidate « exploratoire » (moyenne 50 au-dessus de la 200) a obtenu le meilleur résultat scellé de toutes les candidates de la manche, mais ce n'était pas sa principale. Le fichier de soumission conservé est un résumé ; le raisonnement complet a été transmis à part.",
    "en": "Main strategy: 55 Donchian channel, the classic Turtle (System 2) setting, taken as is rather than found by search. Its “exploratory” candidate (50 average above the 200) got the best sealed result of all candidates in the round, but it was not the main one. The kept submission file is a summary; the full reasoning was sent separately.",
    "methodFr": "Documentation et validation (12 appels avant le premier test), puis balayages de 150 variantes avec 30 % des données mises de côté et backtests sur toute la période. Vérifications : trois sous-périodes, 9 réglages de sortie voisins, longueurs de canal 20, 55 et 100.",
    "methodEn": "Documentation and validation (12 calls before the first test), then 150-variant sweeps with 30% of the data held back and full-period backtests. Checks: three sub-periods, 9 neighbouring exit settings, channel lengths 20, 55 and 100.",
    "ideasFr": "moyenne 200, canal de Donchian 55, croisement 50/200, moyenne exponentielle 50 avec ADX ; à l'achat seulement, avec stop, objectif et durée maximale",
    "ideasEn": "200 average, 55 Donchian channel, 50/200 cross, 50 exponential average with ADX; long only, with stop, target and maximum duration",
    "choiceFr": "Un paramètre classique (Turtle, système 2) plutôt qu'un réglage issu de la recherche.",
    "choiceEn": "A classic parameter (Turtle, System 2) rather than a setting found by the search.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau"
    ]
   },
   "flags": []
  },
  {
   "runId": "opus-2",
   "model": "opus",
   "family": "Opus",
   "name": "Claude Opus 5.5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 70,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": 77.95,
    "maxDrawdownPct": 41.31,
    "ratio": 1.89,
    "trades": 93,
    "ran": true,
    "errorCode": null,
    "name": "E250",
    "script": "strategy \"E250\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n  condition: price > EMA(250)\n  size: 100%\n\nexit:\n  # thesis exit\n",
    "params": {
     "emaLength": 250
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 20,
    "shape": {
     "kind": "priceAboveMa",
     "length": 250
    }
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": 77.95,
     "maxDrawdownPct": 41.31,
     "ratio": 1.89,
     "trades": 93,
     "ran": true,
     "errorCode": null,
     "name": "E250",
     "primary": true
    },
    {
     "label": "exploratory",
     "returnPct": 95.35,
     "maxDrawdownPct": 33.45,
     "ratio": 2.85,
     "trades": 89,
     "ran": true,
     "errorCode": null,
     "name": "Price EMA 300",
     "primary": false
    }
   ],
   "toolCalls": 32,
   "blockedAttempts": 0,
   "log": "log_opus-2.jsonl",
   "process": {
    "calls": 32,
    "successfulCalls": 25,
    "successPct": 78,
    "failedCalls": 7,
    "failedPct": 22,
    "durationMin": 3.3,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 12
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 9
     },
     {
      "tool": "mint_native_candidate",
      "count": 3
     },
     {
      "tool": "tools/list",
      "count": 2
     }
    ],
    "sweeps": 12,
    "sweepsRan": 7,
    "backtests": 9,
    "backtestsRan": 9,
    "testsRan": 16,
    "holdout": {
     "status": "yes",
     "percents": [
      33
     ],
     "sources": [
      "request"
     ]
    },
    "errorsTop": [
     {
      "code": "sandbox.preflight_failed",
      "count": 3
     },
     {
      "code": "sandbox.sweep_parameter_unsupported",
      "count": 2
     },
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 2,
     "get_sandbox_capabilities": 2,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 5,
    "docReadsBeforeFirstTest": 5,
    "callsBeforeFirstTest": 7,
    "firstTestCall": 8,
    "maxFailStreak": 5,
    "blockedAttempts": 0,
    "timeline": "DDDDDBbsssbDBSSssBSSSSRRRRRSRRRR"
   },
   "rank": 12,
   "tiedRank": false,
   "commentary": {
    "fr": "A préféré la moyenne 250, au milieu d'une zone de réglages qui marchent tous, plutôt que la 300 qui avait le meilleur score d'entraînement mais touchait une zone qui s'effondre. 7 appels refusés sur 32, surtout en cherchant comment déclarer un paramètre à balayer.",
    "en": "Preferred the 250 average, in the middle of a zone where all settings work, over the 300 that had the best training score but sat next to a zone that collapses. 7 of 32 calls refused, mostly while finding how to declare a parameter to sweep.",
    "methodFr": "Documentation (5 lectures), essais de familles toutes faites du Lab, puis 12 balayages et 9 backtests finaux pour découper les résultats par année. Les croisements de deux moyennes étaient bons sur l'entraînement mais instables à l'aveugle ; les filtres ajoutés n'apportaient rien.",
    "methodEn": "Documentation (5 reads), tries with the Lab's ready-made families, then 12 sweeps and 9 final backtests to split results by year. Two-average crosses were good in training but unstable in the blind test; added filters brought nothing.",
    "ideasFr": "croisements de deux moyennes, filtres ADX et alignement de moyennes, prix au-dessus d'une moyenne exponentielle de 100 à 600",
    "ideasEn": "two-average crosses, ADX and average-alignment filters, price above an exponential average from 100 to 600",
    "choiceFr": "Le centre de la zone stable plutôt que le meilleur score d'entraînement, jugé trop proche d'une chute.",
    "choiceEn": "The centre of the stable zone rather than the best training score, judged too close to a drop-off.",
    "checks": [
     "holdout",
     "subPeriods",
     "plateau",
     "buyHold"
    ]
   },
   "flags": []
  },
  {
   "runId": "haiku-1",
   "model": "haiku",
   "family": "Haiku",
   "name": "Claude Haiku 4.5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 10,
   "abstained": false,
   "primary": {
    "label": "exploratory",
    "returnPct": -22.08,
    "maxDrawdownPct": 30.99,
    "ratio": -0.71,
    "trades": 236,
    "ran": true,
    "errorCode": null,
    "name": "BTC EMA Crossover Strategy",
    "script": "strategy \"BTC EMA Crossover Strategy\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n    condition: EMA(16) crosses above EMA(50)\n    size: 64%\n\nshort_entry:\n    condition: EMA(16) crosses below EMA(50)\n    size: 64%\n\nexit:\n    take_profit: 2.5%\n    stop_loss: 1.1%\n    timeout: 53 bars",
    "params": {
     "fastEma": 16,
     "slowEma": 50,
     "stopLossPct": 1.1,
     "takeProfitPct": 2.5,
     "timeoutBars": 53
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 40,
    "shape": null
   },
   "candidates": [
    {
     "label": "exploratory",
     "returnPct": -22.08,
     "maxDrawdownPct": 30.99,
     "ratio": -0.71,
     "trades": 236,
     "ran": true,
     "errorCode": null,
     "name": "BTC EMA Crossover Strategy",
     "primary": true
    }
   ],
   "toolCalls": 18,
   "blockedAttempts": 0,
   "log": "log_haiku-1.jsonl",
   "process": {
    "calls": 18,
    "successfulCalls": 16,
    "successPct": 89,
    "failedCalls": 2,
    "failedPct": 11,
    "durationMin": 3.0,
    "toolMix": [
     {
      "tool": "play_engine_round",
      "count": 4
     },
     {
      "tool": "begin_local_quick_proof",
      "count": 4
     },
     {
      "tool": "wait_for_local_proof",
      "count": 3
     },
     {
      "tool": "tools/list",
      "count": 1
     }
    ],
    "sweeps": 0,
    "sweepsRan": 0,
    "backtests": 0,
    "backtestsRan": 0,
    "testsRan": 0,
    "holdout": {
     "status": "noSweepRan",
     "percents": [],
     "sources": []
    },
    "errorsTop": [
     {
      "code": "operation_in_progress",
      "count": 1
     },
     {
      "code": "no_error_detail",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 1,
     "get_sandbox_capabilities": 0,
     "list_sandbox_datasets": 0
    },
    "docReadsTotal": 1,
    "docReadsBeforeFirstTest": 1,
    "callsBeforeFirstTest": null,
    "firstTestCall": null,
    "maxFailStreak": 1,
    "blockedAttempts": 0,
    "timeline": "DDDDOOOOoOOOOBOOOo"
   },
   "rank": 13,
   "tiedRank": false,
   "commentary": {
    "fr": "Aucun balayage ni backtest sur les données de recherche : les 18 appels sont allés à des outils de démonstration et de preuve rapide du Lab, qui tournent sur un petit jeu d'exemple d'une semaine (juin 2026). La stratégie rendue (croisement 16/50 à l'achat et à la vente, stop 1,1 %, objectif 2,5 %) n'avait donc pas été mesurée sur la période demandée. L'étiquette « exploratoire » lui a évité 40 points de pénalité.",
    "en": "No sweep or backtest on the research data: the 18 calls went to the Lab's demo and quick-proof tools, which run on a small one-week sample dataset (June 2026). The strategy handed in (16/50 cross, long and short, 1.1% stop, 2.5% target) had therefore not been measured on the requested period. The “exploratory” label spared it a 40-point penalty.",
    "methodFr": "Lecture du contrat du moteur et des familles toutes faites, puis manches de démonstration et preuves rapides. Les chiffres cités dans la soumission (validation glissante, probabilité de sur-ajustement) viennent de ce jeu d'exemple.",
    "methodEn": "Read the engine contract and the ready-made families, then demo rounds and quick proofs. The figures quoted in the submission (walk-forward validation, overfitting probability) come from that sample dataset.",
    "ideasFr": "croisement de moyennes simples, RSI de retour à la moyenne, croisement de moyennes exponentielles",
    "ideasEn": "simple average cross, RSI mean reversion, exponential average cross",
    "choiceFr": "La meilleure variante sur le jeu d'exemple.",
    "choiceEn": "The best variant on the sample dataset.",
    "checks": []
   },
   "flags": []
  },
  {
   "runId": "haiku-2",
   "model": "haiku",
   "family": "Haiku",
   "name": "Claude Haiku 4.5",
   "route": "claude-agent",
   "category": "cloud",
   "score": 0,
   "abstained": false,
   "primary": {
    "label": "robust",
    "returnPct": null,
    "maxDrawdownPct": null,
    "ratio": null,
    "trades": null,
    "ran": false,
    "errorCode": "sandbox.work_budget_exceeded",
    "name": "Trend + ADX Filter",
    "script": "strategy \"Trend + ADX Filter\" version 1.0\nasset: BTC\ntimeframe: 4h\ncapital: 10000\n\nentry:\n  condition: SMA(20) > SMA(50)\n  and ADX(14) > 25\n  size: 50%\n\nshort_entry:\n  condition: SMA(20) < SMA(50)\n  and ADX(14) > 25\n  size: 50%\n\nexit:\n  take_profit: 5%\n  stop_loss: 2.5%\n  timeout: 60 bars",
    "params": {
     "smaFast": 20,
     "smaSlow": 50,
     "adxPeriod": 14,
     "adxThreshold": 25,
     "takeProfitPct": 5,
     "stopLossPct": 2.5,
     "timeoutBars": 60,
     "positionSize": 0.5
    },
    "beatsBuyHoldRiskAdjusted": false,
    "labelDelta": 0,
    "shape": null
   },
   "candidates": [
    {
     "label": "robust",
     "returnPct": null,
     "maxDrawdownPct": null,
     "ratio": null,
     "trades": null,
     "ran": false,
     "errorCode": "sandbox.work_budget_exceeded",
     "name": "Trend + ADX Filter",
     "primary": true
    },
    {
     "label": "robust",
     "returnPct": null,
     "maxDrawdownPct": null,
     "ratio": null,
     "trades": null,
     "ran": false,
     "errorCode": "no_result",
     "name": "EMA Trend Following",
     "primary": false
    },
    {
     "label": "exploratory",
     "returnPct": -7.98,
     "maxDrawdownPct": 19.28,
     "ratio": -0.41,
     "trades": 108,
     "ran": true,
     "errorCode": null,
     "name": "RSI Mean Reversion",
     "primary": false
    }
   ],
   "toolCalls": 36,
   "blockedAttempts": 0,
   "log": "log_haiku-2.jsonl",
   "process": {
    "calls": 36,
    "successfulCalls": 14,
    "successPct": 39,
    "failedCalls": 22,
    "failedPct": 61,
    "durationMin": 4.3,
    "toolMix": [
     {
      "tool": "run_sandbox_sweep",
      "count": 14
     },
     {
      "tool": "run_sandbox_backtest",
      "count": 6
     },
     {
      "tool": "validate_sandbox_strategy",
      "count": 3
     },
     {
      "tool": "tools/list",
      "count": 2
     }
    ],
    "sweeps": 14,
    "sweepsRan": 0,
    "backtests": 6,
    "backtestsRan": 1,
    "testsRan": 1,
    "holdout": {
     "status": "noSweepRan",
     "percents": [],
     "sources": []
    },
    "errorsTop": [
     {
      "code": "SANDBOX_ARGUMENTS_INVALID",
      "count": 16
     },
     {
      "code": "sandbox.request_schema",
      "count": 3
     },
     {
      "code": "unsupported_condition",
      "count": 1
     }
    ],
    "docReads": {
     "tools/list": 2,
     "get_sandbox_capabilities": 1,
     "list_sandbox_datasets": 1
    },
    "docReadsTotal": 4,
    "docReadsBeforeFirstTest": 3,
    "callsBeforeFirstTest": 8,
    "firstTestCall": 9,
    "maxFailStreak": 11,
    "blockedAttempts": 0,
    "timeline": "DDDDDDBBsssssDsssssssssbbBBOOrrrrrRo"
   },
   "rank": 14,
   "tiedRank": false,
   "commentary": {
    "fr": "14 balayages lancés, aucun n'a tourné ; 1 backtest a abouti sur 6. Les requêtes ont été corrigées un champ manquant à la fois (11 refus d'affilée). Les trois stratégies rendues sont justifiées par des principes généraux, sans chiffre mesuré.",
    "en": "14 sweeps sent, none ran; 1 of 6 backtests went through. Requests were fixed one missing field at a time (11 refusals in a row). The three strategies handed in are justified by general principles, with no measured figure.",
    "methodFr": "Lecture de la documentation et des suggestions du Lab, puis balayages envoyés avant d'avoir lu le format de requête attendu ; la lecture des capacités de l'appli est venue après cinq refus.",
    "methodEn": "Read the documentation and the Lab's suggestions, then sent sweeps before reading the expected request format; it read the app's capabilities after five refusals.",
    "ideasFr": "croisement 20/50 avec filtre ADX, alignement de trois moyennes, RSI de retour à la moyenne ; toutes à l'achat et à la vente, avec stops et objectifs serrés",
    "ideasEn": "20/50 cross with ADX filter, three-average alignment, RSI mean reversion; all long and short, with tight stops and targets",
    "choiceFr": "Choisie sur principe (double confirmation de tendance) et étiquetée « robuste » sans mesure.",
    "choiceEn": "Picked on principle (double trend confirmation) and labelled “robust” without measurement.",
    "checks": []
   },
   "flags": []
  }
 ],
 "modelAverages": [
  {
   "model": "mimo",
   "family": "MiMo",
   "name": "Xiaomi MiMo v2.6 Pro",
   "category": "cloud",
   "runs": 2,
   "avgScore": 90.0,
   "runsWithResult": 2,
   "avgReturnPct": 181.51,
   "avgMaxDrawdownPct": 28.79,
   "avgToolCalls": 47.0
  },
  {
   "model": "muse",
   "family": "Muse",
   "name": "Muse Spark 1.3 (contributor, free)",
   "category": "cloud",
   "runs": 2,
   "avgScore": 85.0,
   "runsWithResult": 2,
   "avgReturnPct": 120.58,
   "avgMaxDrawdownPct": 35.45,
   "avgToolCalls": 33.0
  },
  {
   "model": "deepseek",
   "family": "DeepSeek",
   "name": "deepseek-flash",
   "category": "cloud",
   "runs": 2,
   "avgScore": 85.0,
   "runsWithResult": 2,
   "avgReturnPct": 139.84,
   "avgMaxDrawdownPct": 39.75,
   "avgToolCalls": 22.5
  },
  {
   "model": "sonnet",
   "family": "Sonnet",
   "name": "Claude Sonnet 5",
   "category": "cloud",
   "runs": 2,
   "avgScore": 75.0,
   "runsWithResult": 2,
   "avgReturnPct": 174.0,
   "avgMaxDrawdownPct": 31.85,
   "avgToolCalls": 39.0
  },
  {
   "model": "opus",
   "family": "Opus",
   "name": "Claude Opus 5.5",
   "category": "cloud",
   "runs": 2,
   "avgScore": 70.0,
   "runsWithResult": 2,
   "avgReturnPct": 86.65,
   "avgMaxDrawdownPct": 37.38,
   "avgToolCalls": 29.5
  },
  {
   "model": "fable",
   "family": "Fable",
   "name": "Claude Fable 5.1",
   "category": "cloud",
   "runs": 2,
   "avgScore": 70.0,
   "runsWithResult": 2,
   "avgReturnPct": 86.24,
   "avgMaxDrawdownPct": 35.87,
   "avgToolCalls": 50.0
  },
  {
   "model": "haiku",
   "family": "Haiku",
   "name": "Claude Haiku 4.5",
   "category": "cloud",
   "runs": 2,
   "avgScore": 5.0,
   "runsWithResult": 1,
   "avgReturnPct": -22.08,
   "avgMaxDrawdownPct": 30.99,
   "avgToolCalls": 27.0
  }
 ],
 "unscored": [
  {
   "runId": "nemotron-1",
   "model": "nemotron",
   "family": "Nemotron",
   "name": "Nemotron 3.5 Lightning (free)",
   "route": "cloud-opencode",
   "category": "cloud",
   "attempts": 2,
   "status": "simulated",
   "score": 0,
   "toolCallsDeclared": 0,
   "sealedLog": false,
   "script": "run_sandbox_sweep({ data_set: \"ohlcv-sha256-9f8ce49cc04885b6\", start: \"2020-10-01\", end: \"2023-09-30\", holdoutPercent: 20, strategy: \"ema_cross\", params: {fast: 12, slow: 26, signal: 9}, feeBps: 10, benchmark: \"sharpe\" })",
   "label": "exploratory",
   "scriptIsArenaScript": false,
   "submission": "submission_nemotron-1.json",
   "commentary": {
    "fr": "Aucun appel d'outil, pas de journal scellé. Le « script » rendu est un appel de fonction inventé, pas un script exécutable par l'appli, et les résultats cités (découpe à l'aveugle, perte de 2,3 % après frais) ne viennent d'aucun calcul. Deux tentatives ont donné le même type de rendu : score 0, hors classement.",
    "en": "No tool call, no sealed log. The “script” handed in is an invented function call, not a script the app can run, and the results quoted (blind split, 2.3% loss after fees) come from no computation. Two attempts gave the same kind of answer: score 0, not ranked.",
    "methodFr": "Pas de recherche : la soumission décrit une méthode (découpe à l'aveugle, sous-périodes) qui n'a jamais été exécutée.",
    "methodEn": "No research: the submission describes a method (blind split, sub-periods) that was never run.",
    "checks": []
   },
   "flags": [
    {
     "kind": "integrity",
     "fr": "Recherche simulée",
     "en": "Simulated research",
     "detailFr": "Aucun appel d'outil enregistré ; résultats décrits sans calcul.",
     "detailEn": "No tool call recorded; results described without any computation."
    }
   ]
  }
 ],
 "styles": [
  {
   "id": "blind-test-assertive",
   "titleFr": "Méthodiques, avec test à l'aveugle, et affirmatifs",
   "titleEn": "Methodical, blind-tested and assertive",
   "textFr": "Ont gardé une partie des données de recherche de côté pour vérifier leur règle, puis l'ont étiquetée « robuste ».",
   "textEn": "Kept part of the research data aside to check their rule, then labelled it “robust”.",
   "runs": [
    "opus-1",
    "opus-2",
    "sonnet-2",
    "fable-1",
    "fable-2",
    "deepseek-1",
    "deepseek-2",
    "mimo-1",
    "muse-1",
    "muse-2"
   ]
  },
  {
   "id": "blind-test-cautious",
   "titleFr": "Méthodiques et prudents",
   "titleEn": "Methodical and cautious",
   "textFr": "Ont fait les mêmes vérifications, ont vu des découpes négatives et ont préféré l'étiquette « exploratoire ». Leur stratégie a gagné sur la période scellée : la prudence leur a coûté des points.",
   "textEn": "Ran the same checks, saw negative slices and chose the “exploratory” label. Their strategy made money on the sealed period: caution cost them points.",
   "runs": [
    "sonnet-1",
    "mimo-2"
   ]
  },
  {
   "id": "no-measurement",
   "titleFr": "Règles choisies sans mesure sur les données de la manche",
   "titleEn": "Rules picked without measuring them on the round's data",
   "textFr": "N'ont pas réussi (ou pas tenté) de tester sur les données de recherche ; les stratégies rendues reposent sur des principes généraux ou sur un jeu d'exemple du Lab.",
   "textEn": "Did not manage (or did not try) to test on the research data; the strategies handed in rest on general principles or on a Lab sample dataset.",
   "runs": [
    "haiku-1",
    "haiku-2"
   ]
  },
  {
   "id": "simulated",
   "titleFr": "Simulation sans recherche",
   "titleEn": "Simulation without research",
   "textFr": "Aucun appel à l'appli ; une méthode et des résultats décrits mais jamais exécutés.",
   "textEn": "No call to the app; a method and results described but never run.",
   "runs": [
    "nemotron-1"
   ]
  }
 ],
 "roundNotes": [
  {
   "id": "history",
   "fr": "Manche 1 = période historique (oct. 2023 → sept. 2026) : les modèles peuvent connaître la tendance générale du Bitcoin par leur entraînement. La manche 2 se jouera sur des données futures, inconnues de tous.",
   "en": "Round 1 = historical period (Oct 2023 → Sep 2026): the models may know Bitcoin's general trend from their training. Round 2 will be played on future data, unknown to everyone."
  },
  {
   "id": "referee",
   "fr": "Arbitre : Claude Opus 5.5, qui est aussi candidat. La notation est mécanique : le barème ci-dessous appliqué aux chiffres du backtest.",
   "en": "Referee: Claude Opus 5.5, which is also a contestant. Scoring is mechanical: the rules below applied to backtest numbers."
  }
 ],
 "commentaryReviewed": "2026-10-07",
 "registryReviewed": "2026-10-07",
 "summary": {
  "runs": 14,
  "models": 7,
  "runsBeatingBuyHoldRiskAdjusted": [
   "mimo-1",
   "muse-1",
   "deepseek-1",
   "sonnet-1",
   "mimo-2"
  ],
  "runsBlockedAttempts": 0,
  "primariesNotRun": [
   "haiku-2"
  ],
  "labels": {
   "robust": {
    "count": 11,
    "ran": 10,
    "positive": 10,
    "beatBuyHold": 3
   },
   "exploratory": {
    "count": 3,
    "ran": 3,
    "positive": 2,
    "beatBuyHold": 2
   }
  },
  "convergence": {
   "minSlowLength": 150,
   "maCross": {
    "runs": [
     "mimo-1",
     "deepseek-1",
     "sonnet-1",
     "mimo-2"
    ],
    "models": [
     "DeepSeek",
     "MiMo",
     "Sonnet"
    ],
    "beatBuyHold": 4,
    "fastMin": 11,
    "fastMax": 40,
    "slowMin": 200,
    "slowMax": 298
   },
   "priceAboveMa": {
    "runs": [
     "muse-1",
     "opus-1",
     "fable-2",
     "sonnet-2",
     "deepseek-2",
     "fable-1",
     "opus-2"
    ],
    "models": [
     "DeepSeek",
     "Fable",
     "Muse",
     "Opus",
     "Sonnet"
    ],
    "beatBuyHold": 1,
    "lengthMin": 200,
    "lengthMax": 300
   }
  },
  "holdoutUsed": [
   "mimo-1",
   "muse-1",
   "deepseek-1",
   "sonnet-1",
   "mimo-2",
   "opus-1",
   "fable-2",
   "sonnet-2",
   "deepseek-2",
   "fable-1",
   "muse-2",
   "opus-2"
  ],
  "runsWithoutMeasuredTest": [
   "haiku-1",
   "haiku-2"
  ],
  "bestSealedResult": {
   "runId": "muse-2",
   "name": "Dual SMA robust",
   "label": "exploratory",
   "returnPct": 326.39,
   "maxDrawdownPct": 35.24,
   "isPrimary": false,
   "primaryName": "Donchian55 robust",
   "primaryReturnPct": 109.03,
   "primaryScore": 70
  },
  "autoCommentary": []
 },
 "staticFiles": {
  "log_opus-1.jsonl": "log_opus-1.jsonl",
  "log_opus-2.jsonl": "log_opus-2.jsonl",
  "log_sonnet-1.jsonl": "log_sonnet-1.jsonl",
  "log_sonnet-2.jsonl": "log_sonnet-2.jsonl",
  "log_haiku-1.jsonl": "log_haiku-1.jsonl",
  "log_haiku-2.jsonl": "log_haiku-2.jsonl",
  "log_fable-1.jsonl": "log_fable-1.jsonl",
  "log_fable-2.jsonl": "log_fable-2.jsonl",
  "submission_deepseek-1.json": "round1-extra/submissions/SOUMISSION_DEEPSEEK_1.json",
  "referee_arb_deepseek-1_deepseek-2.json": "round1-extra/referee/arb_deepseek-1_deepseek-2.json",
  "log_deepseek-1.jsonl": "round1-extra/logs/log_deepseek-1.jsonl",
  "submission_deepseek-2.json": "round1-extra/submissions/SOUMISSION_DEEPSEEK_2.json",
  "log_deepseek-2.jsonl": "round1-extra/logs/log_deepseek-2.jsonl",
  "submission_mimo-1.json": "round1-extra/submissions/SOUMISSION_MIMO_1.json",
  "referee_arb_mimo-1.json": "round1-extra/referee/arb_mimo-1.json",
  "log_mimo-1.jsonl": "round1-extra/logs/log_mimo-1.jsonl",
  "submission_mimo-2.json": "round1-extra/submissions/SOUMISSION_MIMO_2.json",
  "referee_arb_mimo-2.json": "round1-extra/referee/arb_mimo-2.json",
  "log_mimo-2.jsonl": "round1-extra/logs/log_mimo-2.jsonl",
  "submission_muse-1.json": "round1-extra/submissions/SOUMISSION_MUSE_1.json",
  "referee_arb_muse-1.json": "round1-extra/referee/arb_muse-1.json",
  "log_muse-1.jsonl": "round1-extra/logs/log_muse-1.jsonl",
  "submission_muse-2.json": "round1-extra/submissions/SOUMISSION_MUSE_2.json",
  "referee_arb_muse-2.json": "round1-extra/referee/arb_muse-2.json",
  "log_muse-2.jsonl": "round1-extra/logs/log_muse-2.jsonl",
  "submission_nemotron-1.json": "round1-extra/submissions/SOUMISSION_NEMOTRON_1.json"
 },
 "files": [
  "PROTOCOL_v1.md",
  "PROTOCOL_v1.sha256",
  "results_v1.json",
  "submissions_v1.json",
  "mcp_bench.py.txt",
  "leaderboard_v1.json",
  "log_deepseek-1.jsonl",
  "log_deepseek-2.jsonl",
  "log_fable-1.jsonl",
  "log_fable-2.jsonl",
  "log_haiku-1.jsonl",
  "log_haiku-2.jsonl",
  "log_mimo-1.jsonl",
  "log_mimo-2.jsonl",
  "log_muse-1.jsonl",
  "log_muse-2.jsonl",
  "log_opus-1.jsonl",
  "log_opus-2.jsonl",
  "log_sonnet-1.jsonl",
  "log_sonnet-2.jsonl",
  "referee_arb_deepseek-1_deepseek-2.json",
  "referee_arb_mimo-1.json",
  "referee_arb_mimo-2.json",
  "referee_arb_muse-1.json",
  "referee_arb_muse-2.json",
  "submission_deepseek-1.json",
  "submission_deepseek-2.json",
  "submission_mimo-1.json",
  "submission_mimo-2.json",
  "submission_muse-1.json",
  "submission_muse-2.json",
  "submission_nemotron-1.json"
 ]
}
