diff --git a/app/src/components/Methodology.tsx b/app/src/components/Methodology.tsx index 730151b..859e11b 100644 --- a/app/src/components/Methodology.tsx +++ b/app/src/components/Methodology.tsx @@ -130,8 +130,11 @@ export default function Methodology({
+ Claude models skip extended thinking when the answer tool call is
+ forced, as it is in the identical request this board holds every
+ model to; other reasoning-by-default providers reason regardless.
+ Re-run with tool_choice: auto, Claude Fable 5 scores
+ 86.9 (would rank #2), Claude Opus 5 85.6 (#3), and Claude Sonnet 5
+ 80.2 (#8). The board below is unchanged — those runs sit beside it
+ as a{" "}
+
+ labeled sensitivity
+ {" "}
+ — and the{" "}
+
+ next board version
+ {" "}
+ moves every model to auto.
+