{"id":19710,"date":"2026-09-23T12:16:05","date_gmt":"2026-09-23T16:16:05","guid":{"rendered":"https:\/\/wp.glbgpt.com\/?p=19710"},"modified":"2026-09-23T12:16:06","modified_gmt":"2026-09-23T16:16:06","slug":"claude-opus-5-5-review","status":"publish","type":"post","link":"https:\/\/wp.glbgpt.com\/jp\/hub\/claude-opus-5-5-review","title":{"rendered":"Claude Opus 5.5 \u30ec\u30d3\u30e5\u30fc\uff1a\u4fa1\u683c\u3001\u30d9\u30f3\u30c1\u30de\u30fc\u30af\u3001\u304a\u3088\u3073\u5b9f\u969b\u306eAPI\u30c6\u30b9\u30c8"},"content":{"rendered":"<div class=\"wp-block-group opus-review is-layout-flow wp-block-group-is-layout-flow\">\n\n<style>\n\n    :root{--ink:#172033;--muted:#617087;--line:#dfe5ef;--paper:#fff;--page:#f2f5fa;--navy:#111a36;--violet:#6656d9;--violet-soft:#f5f2ff;--blue-soft:#eef5ff;--green:#137849;--green-soft:#edf9f1;--amber:#8a5a00;--amber-soft:#fff7df;--red:#a33838;--red-soft:#fff0f0;--shadow:0 18px 50px rgba(16,24,40,.11)}\n    *{box-sizing:border-box}html{scroll-behavior:smooth}body{margin:0;background:radial-gradient(circle at 5% 0,rgba(102,86,217,.14),transparent 27%),var(--page);color:var(--ink);font:17px\/1.72 Inter,ui-sans-serif,-apple-system,BlinkMacSystemFont,\"Segoe UI\",Arial,sans-serif}.opus-review{max-width:1080px;margin:32px auto 72px;background:var(--paper);border:1px solid #e7ebf2;border-radius:24px;box-shadow:var(--shadow);overflow:hidden}.opus-review>*:not(.article-hero){margin-left:54px;margin-right:54px}.article-hero{padding:54px 58px 46px;background:linear-gradient(135deg,#111a36,#2b2b70 58%,#5b3d91);color:#fff}.hero-kicker{display:inline-flex;padding:7px 11px;border:1px solid rgba(255,255,255,.28);border-radius:999px;color:#e4e7ff;font-size:12px;font-weight:800;letter-spacing:.08em;text-transform:uppercase}.article-hero h1{max-width:850px;margin:18px 0 16px;color:#fff;font-size:clamp(36px,5vw,58px);line-height:1.08;letter-spacing:-.045em}.article-hero p{max-width:850px;margin:0;color:#e6e9ff;font-size:19px}.lede{margin-top:38px;font-size:20px}.opus-review h2{margin-top:54px;padding-top:6px;border-left:5px solid var(--violet);padding-left:15px;font-size:31px;line-height:1.2;letter-spacing:-.025em}.opus-review h3{font-size:22px;line-height:1.28;margin:0 0 10px}.opus-review p{margin:0 0 18px}.opus-review a{color:#3d4fc6;text-underline-offset:3px}.opus-review code{padding:2px 6px;border-radius:5px;background:#eef1f7;color:#293657;font-size:.92em}.toc,.quick-answer,.method-note,.verdict-card{border:1px solid var(--line);border-radius:16px;padding:22px 26px;margin-top:28px}.toc{background:#f8f9fc}.toc p{margin:0 0 9px;font-weight:800}.toc ul{columns:2;margin:0;padding-left:22px}.toc li{margin:5px 0}.quick-answer{background:linear-gradient(135deg,#f3f0ff,#eef6ff);border-left:6px solid var(--violet)}.quick-answer h2{margin:0 0 12px;padding:0;border:0;font-size:24px}.quick-answer ul{margin:0;padding-left:21px}.quick-answer li{margin:7px 0}.evidence-strip{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:1px;margin:24px 0;background:var(--line);border:1px solid var(--line);border-radius:14px;overflow:hidden}.evidence-strip div{padding:17px;background:#fff}.evidence-strip strong{display:block;font-size:27px;line-height:1.1;color:var(--violet)}.evidence-strip span{display:block;color:var(--muted);font-size:13px;margin-top:5px}.official-proof{margin:26px 0;border:1px solid var(--line);border-radius:16px;background:#fff;overflow:hidden;box-shadow:0 8px 24px rgba(16,24,40,.06)}.official-proof a{display:block;line-height:0}.official-proof img{display:block;width:100%;height:auto}.official-proof figcaption{padding:12px 15px;color:#5a6677;background:#fafbfe;font-size:14px;line-height:1.55}.proof-grid{display:grid;grid-template-columns:repeat(2,minmax(0,1fr));gap:18px;margin:25px 0}.proof-grid .official-proof{margin:0}.benchmark-grid{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));gap:16px;margin:24px 0}.benchmark-card{border:1px solid var(--line);border-radius:16px;padding:21px;background:linear-gradient(180deg,#fff,#fafbff);box-shadow:0 10px 25px rgba(16,24,40,.05)}.benchmark-card .source-label{display:inline-block;margin-bottom:11px;color:#5d51c4;font-size:12px;font-weight:800;letter-spacing:.08em;text-transform:uppercase}.benchmark-card .metric{font-size:31px;font-weight:800;line-height:1.12;color:var(--navy);margin:6px 0 14px}.benchmark-card p{font-size:15px;color:#48556b}.benchmark-card .boundary{padding:11px 12px;border-left:4px solid #a9b4c8;background:#f4f6fa;font-size:14px;color:#536075}.price-grid{display:grid;grid-template-columns:repeat(2,minmax(0,1fr));gap:16px;margin:24px 0}.price-card{border:1px solid var(--line);border-radius:16px;padding:22px;background:#fff}.price-card.api{background:linear-gradient(135deg,#f2f0ff,#fff)}.price-card.consumer{background:linear-gradient(135deg,#fff8e7,#fff)}.price-card h3{margin:0 0 12px}.price-row{display:flex;justify-content:space-between;gap:14px;padding:11px 0;border-bottom:1px solid #e7ebf2}.price-row:last-child{border-bottom:0}.price-row strong{white-space:nowrap}.price-card p{font-size:14px;color:#536075;margin:12px 0 0}.small-note{font-size:14px;color:var(--muted)}.method-note{background:var(--blue-soft);border-color:#cdddf8}.method-note strong{color:#1d4487}.test-card{border:1px solid #dfe5ef;border-radius:18px;margin:26px 0;overflow:hidden;background:#fff;box-shadow:0 12px 30px rgba(16,24,40,.07)}.test-head{display:flex;justify-content:space-between;gap:14px;align-items:flex-start;padding:19px 23px;background:var(--navy);color:#fff}.test-kicker{display:block;color:#a9c7ff;font-size:11px;font-weight:800;letter-spacing:.08em;text-transform:uppercase}.test-head h3{margin:4px 0 0;color:#fff;font-size:23px}.status{display:inline-flex;align-items:center;padding:5px 10px;border-radius:999px;font-size:12px;font-weight:800;white-space:nowrap}.status.pass{background:#dff3e6;color:#176b3c}.status.limited{background:#fff0c8;color:#7a5000}.status.route{background:#ffe0e0;color:#8c3030}.test-body{padding:20px 23px}.test-meta{display:flex;flex-wrap:wrap;gap:8px;margin:0 0 15px}.chip{padding:5px 9px;border:1px solid #dfe5ef;border-radius:999px;background:#f8f9fc;color:#536075;font-size:12px;font-weight:700}.test-body h4{margin:16px 0 7px;font-size:14px;text-transform:uppercase;letter-spacing:.06em;color:#5c6b83}.test-body p{font-size:15px}.test-output{display:grid;grid-template-columns:1.2fr .8fr;gap:16px;margin-top:13px}.output-box{padding:15px;border-radius:11px;background:#f7f8fb;border:1px solid #e5e9f1}.output-box strong{display:block;font-size:12px;text-transform:uppercase;letter-spacing:.06em;color:#5d6c84;margin-bottom:6px}.output-box p{margin:0;font-size:14px;color:#39455a}.test-verdict{margin:17px 0 0;padding:12px 14px;border-left:4px solid var(--violet);background:var(--violet-soft);font-size:15px}.api-list{display:grid;grid-template-columns:repeat(2,minmax(0,1fr));gap:12px;margin:22px 0;padding:0;list-style:none}.api-list li{padding:15px 16px;border:1px solid var(--line);border-radius:12px;background:#fbfcff}.api-list strong{display:block;margin-bottom:5px}.verdict-card{background:var(--navy);color:#eff3ff;border:0}.verdict-card h2{margin:0 0 12px;padding:0;border:0;color:#fff;font-size:25px}.verdict-card p{color:#e5ebff}.faq h3{margin-top:25px;font-size:20px}.source-note{font-size:13px;color:var(--muted)}@media(max-width:760px){body{background:#fff}.opus-review{margin:0;border:0;border-radius:0;box-shadow:none}.opus-review>*:not(.article-hero){margin-left:20px;margin-right:20px}.article-hero{padding:38px 23px 34px}.article-hero h1{font-size:35px}.article-hero p{font-size:17px}.toc ul{columns:1}.evidence-strip{grid-template-columns:repeat(2,minmax(0,1fr))}.benchmark-grid,.price-grid,.proof-grid,.test-output,.api-list{grid-template-columns:1fr}.test-head{display:block}.test-head .status{margin-top:13px}}\n  \n<\/style>\n\n\n\n<header class=\"article-hero\">\n<span class=\"hero-kicker\">Model review \u00b7 controlled API test \u00b7 September 23, 2026<\/span>\n<h1>Claude Opus 5.5 \u30ec\u30d3\u30e5\u30fc\uff1a\u4fa1\u683c\u3001\u30d9\u30f3\u30c1\u30de\u30fc\u30af\u3001\u304a\u3088\u3073\u5b9f\u969b\u306eAPI\u30c6\u30b9\u30c8<\/h1>\n<p>Claude Opus 5.5 combines a 1M-token context window with a $4 \/ $20 per-million-token API price. This review separates Anthropic\u2019s published claims, independent benchmark snapshots, and five direct task runs through the Messages API route.<\/p>\n<\/header>\n\n\n\n<p class=\"lede wp-block-paragraph\"><strong>The measured result is specific.<\/strong> Three tasks returned clean, usable answers; the coding task returned a useful answer but hit the first 1,024-token ceiling; the Chinese-writing run produced garbled text on the tested route. The sample shows concrete behavior and route limits, not a universal success rate.<\/p>\n\n\n\n<div class=\"quick-answer\" id=\"quick-answer\">\n<h2>\u7c21\u5358\u306a\u56de\u7b54<\/h2>\n<ul>\n<li><strong>\u30d1\u30d5\u30a9\u30fc\u30de\u30f3\u30b9<\/strong> Anthropic reports Fable 5.1-level performance on most work, while Artificial Analysis recorded a 58 Intelligence Index score at max effort. METR\u2019s narrower judgment-skill evidence did not show a large improvement over Fable 5.1.<\/li>\n<li><strong>\u4fa1\u683c\uff1a<\/strong> the official API rate is <strong>$4 per million input tokens and $20 per million output tokens<\/strong>. Cache reads cost $0.20 per million; 5-minute cache writes cost $5 per million.<\/li>\n<li><strong>Hands-on result:<\/strong> long-text extraction, strict JSON, and arithmetic reasoning passed in the recorded runs. Coding was useful but truncated at the initial ceiling. The Chinese result is recorded as a route\/encoding issue, not a model-level language verdict.<\/li>\n<li><strong>API warning:<\/strong> thinking cannot be disabled, forced tool use returns an error, thinking blocks are tied to the model and conversation, and the older <code>computer_20251124<\/code> tool is not accepted on the Claude API and Google Cloud route.<\/li>\n<\/ul>\n<\/div>\n\n\n\n<div aria-label=\"Review evidence summary\" class=\"evidence-strip\">\n<div><strong>5<\/strong><span>API tasks run<\/span><\/div>\n<div><strong>3<\/strong><span>Clean usable answers<\/span><\/div>\n<div><strong>$4\/$20<\/strong><span>Input\/output per MTok<\/span><\/div>\n<div><strong>1M<\/strong><span>\u30b3\u30f3\u30c6\u30ad\u30b9\u30c8\u30a6\u30a3\u30f3\u30c9\u30a6<\/span><\/div>\n<\/div>\n\n\n\n<nav aria-label=\"\u76ee\u6b21\" class=\"toc\">\n<p>\u672c\u30ec\u30d3\u30e5\u30fc\u3067\u306f<\/p>\n<ul>\n<li><a href=\"#what-changed\">What the official docs say<\/a><\/li>\n<li><a href=\"#performance\">Performance evidence<\/a><\/li>\n<li><a href=\"#pricing\">Price and billing units<\/a><\/li>\n<li><a href=\"#hands-on-test\">Five API task cards<\/a><\/li>\n<li><a href=\"#api-caveats\">API and migration caveats<\/a><\/li>\n<li><a href=\"#verdict\">Evidence-based verdict<\/a><\/li>\n<li><a href=\"#faq\">\u3088\u304f\u3042\u308b\u3054\u8cea\u554f<\/a><\/li>\n<\/ul>\n<\/nav>\n\n\n\n<h2 class=\"wp-block-heading\" id=\"what-changed\">What the official docs say<\/h2>\n\n\n\n<p class=\"wp-block-paragraph\">Anthropic\u2019s release page dates Claude Opus 5.5 to September 22, 2026 and describes it as the first model in the Claude 5.5 family. The announcement claims performance at the level of Claude Fable 5.1 on most work, a 40% lower typical workload cost than Opus 5, and output more than 30% faster than Opus 5. Those are provider claims, so they are shown separately from the independent and hands-on evidence below.<\/p>\n\n\n\n<figure class=\"wp-block-image size-full official-proof\"><a href=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/01-anthropic-release-hero-tight.webp\"><img decoding=\"async\" src=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/01-anthropic-release-hero-tight.webp\" alt=\"Cropped official Anthropic Claude Opus 5.5 announcement showing the title and September 22, 2026 date\"\/><\/a><figcaption>Small crop from Anthropic\u2019s announcement: model name, September 22, 2026 release date, and the beginning of its introduction. The source page is <a href=\"https:\/\/www.anthropic.com\/claude-opus-5-5\">Anthropic\u2019s Claude Opus 5.5 announcement<\/a>; its claims are not independent test results.<\/figcaption><\/figure>\n\n\n\n<p class=\"wp-block-paragraph\">\u306b\u3064\u3044\u3066 <a href=\"https:\/\/platform.claude.com\/docs\/en\/models\/opus-5-5\/overview\">Claude Platform Docs<\/a> list the API model ID as <code>claude-opus-5-5<\/code>, a 1M-token context window, a 128K maximum output, and adaptive thinking that is always on. The docs also list the changes that affect existing Opus 5 integrations, covered in the migration section below.<\/p>\n\n\n\n<h2 class=\"wp-block-heading\" id=\"performance\">Performance evidence<\/h2>\n\n\n\n<p class=\"wp-block-paragraph\">There is no single \u201cperformance\u201d number here. The three public sources measure different things, so each card keeps its metric and evidence boundary visible.<\/p>\n\n\n\n<div class=\"wp-block-columns benchmark-grid is-layout-flex wp-container-core-columns-is-layout-7387b849 wp-block-columns-is-layout-flex\">\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<section class=\"benchmark-card\">\n<span class=\"source-label\">Provider release<\/span>\n<h3>Anthropic announcement<\/h3>\n<div class=\"metric\">Fable 5.1 level<\/div>\n<p><strong>Published signal:<\/strong> Anthropic says Opus 5.5 performs at the level of Claude Fable 5.1 on most work, costs 40% less than Opus 5 on typical workloads, and produces output more than 30% faster.<\/p>\n<p class=\"boundary\"><strong>\u5883\u754c\uff1a<\/strong> provider-published positioning and tester examples, not an independent benchmark.<\/p>\n<p class=\"source-note\"><a href=\"https:\/\/www.anthropic.com\/claude-opus-5-5\">Open the official release<\/a><\/p>\n<\/section>\n\n<\/div>\n\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<section class=\"benchmark-card\">\n<span class=\"source-label\">Independent snapshot \u00b7 Sep. 22, 2026<\/span>\n<h3>\u4eba\u5de5\u5206\u6790<\/h3>\n<div class=\"metric\">58 Intelligence Index<\/div>\n<p><strong>Published signal:<\/strong> the max-effort snapshot reports 59.6% on Terminal-Bench 4.0, 61.4% on Humanity\u2019s Last Exam, and 66.9% on SciCode.<\/p>\n<p class=\"boundary\"><strong>\u5883\u754c\uff1a<\/strong> scores depend on effort, harness, date, and metric definition. They do not predict every API prompt.<\/p>\n<p class=\"source-note\"><a href=\"https:\/\/artificialanalysis.ai\/articles\/claude-opus-5-5\">Open the Artificial Analysis report<\/a><\/p>\n<\/section>\n\n<\/div>\n\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<section class=\"benchmark-card\">\n<span class=\"source-label\">Independent qualification \u00b7 Sep. 22, 2026<\/span>\n<h3>METR predeployment evaluation<\/h3>\n<div class=\"metric\">No large judgment-skill gain<\/div>\n<p><strong>Published signal:<\/strong> METR says its evidence did not show a large improvement over Fable 5.1 on the judgment skills it examined, while still expecting meaningful acceleration for some R&amp;D work.<\/p>\n<p class=\"boundary\"><strong>\u5883\u754c\uff1a<\/strong> this is a narrower evaluation of judgment and task acceleration, not a complete quality ranking.<\/p>\n<p class=\"source-note\"><a href=\"https:\/\/metr.org\/blog\/2026-09-22-claude-opus-5-5\/\">Open the METR evaluation<\/a><\/p>\n<\/section>\n\n<\/div>\n\n<\/div>\n\n\n\n<p class=\"wp-block-paragraph\">Read together, the cards support a limited conclusion: Opus 5.5 is a leading model on several dated public evaluations, but the size of the advantage changes with the task, effort setting, and evaluation design. The cards do not establish a universal \u201cbest model\u201d result.<\/p>\n\n\n\n<h2 class=\"wp-block-heading\" id=\"pricing\">Claude Opus 5.5 price and billing units<\/h2>\n\n\n\n<p class=\"wp-block-paragraph\">The API price and the consumer subscription price are different products. The API figures below are the unit prices used for token budgeting; the consumer figures are the plan cards visible in the captured browser region.<\/p>\n\n\n\n<div class=\"wp-block-columns price-grid is-layout-flex wp-container-core-columns-is-layout-7387b849 wp-block-columns-is-layout-flex\">\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<section class=\"price-card api\">\n<h3>\u516c\u5f0fAPI\u4fa1\u683c<\/h3>\n<div class=\"price-row\"><span>\u30a4\u30f3\u30d7\u30c3\u30c8<\/span><strong>$4 \/ 1M tokens<\/strong><\/div>\n<div class=\"price-row\"><span>\u51fa\u529b<\/span><strong>$20 \/ 1M tokens<\/strong><\/div>\n<div class=\"price-row\"><span>\u30ad\u30e3\u30c3\u30b7\u30e5\u8aad\u307f\u53d6\u308a<\/span><strong>$0.20 \/ 1M<\/strong><\/div>\n<div class=\"price-row\"><span>5\u5206\u9593\u306e\u30ad\u30e3\u30c3\u30b7\u30e5\u66f8\u304d\u8fbc\u307f<\/span><strong>$5 \/ 1M<\/strong><\/div>\n<div class=\"price-row\"><span>\u30d0\u30c3\u30c1API<\/span><strong>50%\u5272\u5f15<\/strong><\/div>\n<p>Simple estimate: 10,000 input tokens plus 2,000 output tokens is about $0.08 at standard rates, before cache, batch, tax, or platform markup.<\/p>\n<\/section>\n\n<\/div>\n\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<section class=\"price-card consumer\">\n<h3>Consumer plan screenshot<\/h3>\n<div class=\"price-row\"><span>\u7121\u6599<\/span><strong>$0<\/strong><\/div>\n<div class=\"price-row\"><span>\u30d7\u30ed<\/span><strong>$18 annual \/ $22 monthly<\/strong><\/div>\n<div class=\"price-row\"><span>\u30de\u30c3\u30af\u30b9<\/span><strong>From $110 \/ month<\/strong><\/div>\n<p>The captured page also shows a 10% JCT note. These are region-sensitive displayed values, not a global price promise.<\/p>\n<\/section>\n\n<\/div>\n\n<\/div>\n\n\n\n<div class=\"wp-block-columns proof-grid is-layout-flex wp-container-core-columns-is-layout-7387b849 wp-block-columns-is-layout-flex\">\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<figure class=\"wp-block-image size-full official-proof\"><a href=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/02-platform-model-specs.webp\"><img decoding=\"async\" src=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/02-platform-model-specs.webp\" alt=\"Cropped Claude Platform Docs model card showing 1M context, 128K output and $4\/$20 pricing\"\/><\/a><figcaption>Docs crop showing the model ID, context\/output limits, and the top comparison row. <a href=\"https:\/\/platform.claude.com\/docs\/en\/models\/opus-5-5\/overview\">Source: Claude Platform Docs<\/a>.<\/figcaption><\/figure>\n\n<\/div>\n\n\n<div class=\"wp-block-column is-layout-flow wp-block-column-is-layout-flow\">\n\n<figure class=\"wp-block-image size-full official-proof\"><a href=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/03-claude-pricing-plans.webp\"><img decoding=\"async\" src=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/03-claude-pricing-plans.webp\" alt=\"Cropped official Claude pricing page showing Free, Pro and Max plan cards\"\/><\/a><figcaption>Pricing crop showing the Free, Pro, and Max cards visible in the captured region. <a href=\"https:\/\/claude.com\/pricing\">Source: Claude pricing<\/a>.<\/figcaption><\/figure>\n\n<\/div>\n\n<\/div>\n\n\n\n<p class=\"wp-block-paragraph\">For a separate explanation of token billing and context-window cost, see the <a href=\"https:\/\/www.glbgpt.com\/hub\/gpt-5-5-pricing\/\">GPT-5.5\u306e\u4fa1\u683c\u8a2d\u5b9a\u3068\u80cc\u666f\u306b\u95a2\u3059\u308b\u30ac\u30a4\u30c9<\/a>. The provider\u2019s API rate, a subscription price, and any multi-model platform credit balance should remain separate in your budget.<\/p>\n\n\n\n<h2 class=\"wp-block-heading\" id=\"hands-on-test\">Hands-on test: five task result cards<\/h2>\n\n\n\n<p class=\"wp-block-paragraph\">We sent five independent prompts through the same controlled Messages API route on September 23, 2026 with model ID <code>claude-opus-5-5<\/code>. The first pass used <code>max_tokens: 1024<\/code>. We recorded local elapsed time, input\/output tokens, stop reason, correctness, and literal format compliance. The Chinese-writing task was rerun with a 4,096-token ceiling after the first pass exposed the route\u2019s output issue. This is a five-task sample, not a reliability estimate.<\/p>\n\n\n\n<div class=\"method-note\"><strong>How to read the cards:<\/strong> \u201cPass\u201d means the returned answer was directly usable for the stated task. \u201cLimited\u201d marks a useful answer constrained by the configured output ceiling. \u201cRoute issue\u201d records unusable returned text without assigning the problem to the model\u2019s general language ability. Timings are local end-to-end elapsed time for this route.<\/div>\n\n\n\n<section class=\"test-card\" id=\"test-coding\">\n<div class=\"test-head\"><div><span class=\"test-kicker\">Task 01 \u00b7 Python debugging<\/span><h3>Can it repair a mixed-type deduplication function?<\/h3><\/div><span class=\"status limited\">Useful \u00b7 ceiling hit<\/span><\/div>\n<div class=\"test-body\">\n<div class=\"test-meta\"><span class=\"chip\">11.19s local<\/span><span class=\"chip\">145 input \/ 1,024 output tokens<\/span><span class=\"chip\">stop: max_tokens<\/span><\/div>\n<h4>\u30bf\u30b9\u30af\u306e\u8a2d\u5b9a<\/h4><p>Find the bug in a Python <code>dedupe<\/code> function, handle non-string and unhashable values, preserve the first spelling, and provide tests.<\/p>\n<div class=\"test-output\"><div class=\"output-box\"><strong>\u89b3\u6e2c\u3055\u308c\u305f\u51fa\u529b<\/strong><p>The response identified the <code>.lower()<\/code> failure on non-string values, switched strings to <code>casefold()<\/code>, added an unhashable fallback, and returned corrected code plus test cases.<\/p><\/div><div class=\"output-box\"><strong>\u5236\u9650<\/strong><p>The answer stopped at the 1,024-token ceiling. The fix was useful, but the configuration was too small for a full code review.<\/p><\/div><\/div>\n<p class=\"test-verdict\"><strong>Card verdict:<\/strong> technically useful answer returned; this run cannot support a latency or completeness claim because the token ceiling cut it off.<\/p>\n<\/div>\n<\/section>\n\n\n\n<section class=\"test-card\" id=\"test-extract\">\n<div class=\"test-head\"><div><span class=\"test-kicker\">Task 02 \u00b7 Long-text extraction<\/span><h3>Can it preserve facts and test limitations?<\/h3><\/div><span class=\"status pass\">\u30d1\u30b9<\/span><\/div>\n<div class=\"test-body\">\n<div class=\"test-meta\"><span class=\"chip\">4.47s local<\/span><span class=\"chip\">210 input \/ 381 output tokens<\/span><span class=\"chip\">stop: end_turn<\/span><\/div>\n<h4>\u30bf\u30b9\u30af\u306e\u8a2d\u5b9a<\/h4><p>Extract five numbered findings from a short evaluation memo, including the method and what the next round did not measure.<\/p>\n<div class=\"test-output\"><div class=\"output-box\"><strong>\u89b3\u6e2c\u3055\u308c\u305f\u51fa\u529b<\/strong><p>Returned exactly five numbered bullets. It preserved the assistant comparisons, the 20-prompt method, and the gaps around long-term retention and customer conversion.<\/p><\/div><div class=\"output-box\"><strong>\u5236\u9650<\/strong><p>This was a short source pack, so it does not test the 1M context window.<\/p><\/div><\/div>\n<p class=\"test-verdict\"><strong>Card verdict:<\/strong> clean extraction with the requested count and the source memo\u2019s evidence boundary preserved.<\/p>\n<\/div>\n<\/section>\n\n\n\n<section class=\"test-card\" id=\"test-json\">\n<div class=\"test-head\"><div><span class=\"test-kicker\">Task 03 \u00b7 Strict JSON<\/span><h3>Can it return machine-readable output without cleanup?<\/h3><\/div><span class=\"status pass\">\u30d1\u30b9<\/span><\/div>\n<div class=\"test-body\">\n<div class=\"test-meta\"><span class=\"chip\">4.92s local<\/span><span class=\"chip\">128 input \/ 271 output tokens<\/span><span class=\"chip\">stop: end_turn<\/span><\/div>\n<h4>\u30bf\u30b9\u30af\u306e\u8a2d\u5b9a<\/h4><p>Return a fixed JSON object with a title, audience, two pros, two cons, and a verdict. No Markdown fence was allowed.<\/p>\n<div class=\"test-output\"><div class=\"output-box\"><strong>\u89b3\u6e2c\u3055\u308c\u305f\u51fa\u529b<\/strong><p>Returned parseable JSON with the requested keys, two pros, two cons, and a concise recommendation. No Markdown wrapper was present.<\/p><\/div><div class=\"output-box\"><strong>\u5236\u9650<\/strong><p>One valid response does not prove schema reliability across larger or nested objects.<\/p><\/div><\/div>\n<p class=\"test-verdict\"><strong>Card verdict:<\/strong> literal JSON compliance passed in this run.<\/p>\n<\/div>\n<\/section>\n\n\n\n<section class=\"test-card\" id=\"test-math\">\n<div class=\"test-head\"><div><span class=\"test-kicker\">Task 04 \u00b7 Arithmetic reasoning<\/span><h3>Can it apply a repeated rounding rule correctly?<\/h3><\/div><span class=\"status pass\">\u30d1\u30b9<\/span><\/div>\n<div class=\"test-body\">\n<div class=\"test-meta\"><span class=\"chip\">4.39s local<\/span><span class=\"chip\">89 input \/ 264 output tokens<\/span><span class=\"chip\">stop: end_turn<\/span><\/div>\n<h4>\u30bf\u30b9\u30af\u306e\u8a2d\u5b9a<\/h4><p>Start with 12 items, multiply each next batch by 1.25, round down after each batch, and report the total across four batches.<\/p>\n<div class=\"test-output\"><div class=\"output-box\"><strong>\u89b3\u6e2c\u3055\u308c\u305f\u51fa\u529b<\/strong><p>Returned 12, 15, 18, and 22, then summed them to <strong>67<\/strong>. The intermediate arithmetic was visible.<\/p><\/div><div class=\"output-box\"><strong>\u5236\u9650<\/strong><p>The task is deterministic and small; it does not represent broad reasoning performance.<\/p><\/div><\/div>\n<p class=\"test-verdict\"><strong>Card verdict:<\/strong> correct result and transparent intermediate steps.<\/p>\n<\/div>\n<\/section>\n\n\n\n<section class=\"test-card\" id=\"test-chinese\">\n<div class=\"test-head\"><div><span class=\"test-kicker\">Task 05 \u00b7 Chinese SEO opening<\/span><h3>Did the tested route return usable Chinese text?<\/h3><\/div><span class=\"status route\">Route issue<\/span><\/div>\n<div class=\"test-body\">\n<div class=\"test-meta\"><span class=\"chip\">12.37s local<\/span><span class=\"chip\">132 input \/ 943 output tokens<\/span><span class=\"chip\">stop: end_turn<\/span><span class=\"chip\">rerun ceiling: 4,096<\/span><\/div>\n<h4>\u30bf\u30b9\u30af\u306e\u8a2d\u5b9a<\/h4><p>Draft a Chinese review opening that separates official claims, price, benchmark evidence, and hands-on testing.<\/p>\n<div class=\"test-output\"><div class=\"output-box\"><strong>\u89b3\u6e2c\u3055\u308c\u305f\u51fa\u529b<\/strong><p>The response structure was present, but Chinese characters were returned garbled through the tested route in the saved result.<\/p><\/div><div class=\"output-box\"><strong>\u5236\u9650<\/strong><p>The evidence does not distinguish model language quality from route or encoding behavior. No Chinese-quality score is assigned.<\/p><\/div><\/div>\n<p class=\"test-verdict\"><strong>Card verdict:<\/strong> unusable returned text on this route; rerun with a verified UTF-8 path before making a language-quality claim.<\/p>\n<\/div>\n<\/section>\n\n\n\n<p class=\"wp-block-paragraph\">Across the four language-neutral or English runs, the local mean was 6.24 seconds. That number describes five sequential requests on one route and is not provider-side latency. The strongest observed behaviors were concise extraction, literal JSON compliance, and correct arithmetic; the two unresolved issues were output-budget pressure on coding and garbled Chinese output on the tested route.<\/p>\n\n\n\n<h2 class=\"wp-block-heading\" id=\"api-caveats\">API and migration caveats<\/h2>\n\n\n\n<p class=\"wp-block-paragraph\">The quick answer includes these because they can change an integration even when the model\u2019s text quality looks good.<\/p>\n\n\n\n<ul class=\"wp-block-list\">\n<li><strong>Thinking is always on.<\/strong> Use the documented effort parameter to control depth, latency, and cost; there is no off switch.<\/li>\n<li><strong>Forced tool use returns an error.<\/strong> Existing code that requires forced tool selection needs a separate migration test.<\/li>\n<li><strong>Thinking blocks are bound to model and conversation.<\/strong> Do not assume a thinking block can be reused after changing models.<\/li>\n<li><strong><code>computer_20251124<\/code> is not accepted<\/strong> on the Claude API and Google Cloud route listed in the docs.<\/li>\n<li><strong>Output ceilings matter.<\/strong> The coding card shows a useful response can still be cut off by a small <code>max_tokens<\/code> value.<\/li>\n<li><strong>Batch and fast mode are separate billing paths.<\/strong> Keep their discounts or surcharges separate from standard token prices.<\/li>\n\n<\/ul>\n\n\n\n<figure class=\"wp-block-image size-full official-proof\"><a href=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/04-opus-5-5-system-card.webp\"><img decoding=\"async\" src=\"https:\/\/wp.glbgpt.com\/wp-content\/uploads\/2026\/09\/04-opus-5-5-system-card.webp\" alt=\"Official Anthropic System Card cover for Claude Opus 5.5 dated September 22, 2026\"\/><\/a><figcaption>System Card cover rendered from Anthropic\u2019s official PDF. It identifies the source and date; it is not evidence for the benchmark or API test results. <a href=\"https:\/\/www.anthropic.com\/claude-opus-5-5-system-card\">Source: Anthropic System Card<\/a>.<\/figcaption><\/figure>\n\n\n\n<div class=\"verdict-card\" id=\"verdict\">\n<h2 id=\"verdict-heading\">Evidence-based verdict<\/h2>\n<p>Claude Opus 5.5 has strong dated public benchmark evidence and a lower published standard API rate than the Opus 5 figures shown in Anthropic\u2019s announcement. In the five recorded task runs, it produced clean extraction, strict JSON, and arithmetic; the coding answer was useful but truncated by the initial ceiling; and the Chinese output was not usable through the tested route.<\/p>\n<p>The defensible conclusion is therefore narrow: the model demonstrated strong instruction following on several small tasks, while the route and configuration introduced visible limits. Any migration decision should repeat the same prompts with a suitable output budget, verify tool calls, and keep official token rates separate from subscription or platform credits.<\/p>\n<\/div>\n\n\n\n<p class=\"wp-block-paragraph\">If you want a second route for comparison, the <a href=\"https:\/\/www.glbgpt.com\/hub\/deepseek-v4-pro-vs-flash\/\">DeepSeek V4 Pro vs Flash test format<\/a> shows how task-level verdicts can stay separate from model-wide claims. The <a href=\"https:\/\/www.glbgpt.com\/hub\/gpt-5-5-use-cases\/\">GPT-5.5 \u30e6\u30fc\u30b9\u30b1\u30fc\u30b9\u30ac\u30a4\u30c9<\/a> adds a task-oriented reference point, while <a href=\"https:\/\/www.glbgpt.com\/hub\/gemini-3-6-flash-review\/\">Gemini Flash benchmark notes<\/a> \u305d\u3057\u3066 <a href=\"https:\/\/www.glbgpt.com\/hub\/best-ai-models\/\">AI model comparison guide<\/a> provide nearby model context. For multi-model access, see <a href=\"https:\/\/www.glbgpt.com\/hub\/all-in-one-ai-tools-subscription\/\">all-in-one AI subscriptions<\/a> \u305d\u3057\u3066 <a href=\"https:\/\/www.glbgpt.com\/hub\/best-all-in-one-ai-tools-and-platforms-11-options-compared\/\">the multi-model platform comparison<\/a>.<\/p>\n\n\n\n<h2 class=\"wp-block-heading\">\u3088\u304f\u3042\u308b\u8cea\u554f<\/h2>\n\n\n<h3 class=\"wp-block-heading\">How much does Claude Opus 5.5 cost?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">The official API rate is $4 per million input tokens and $20 per million output tokens. Cache reads are $0.20 per million tokens, and 5-minute cache writes are $5 per million tokens.<\/p>\n\n\n<h3 class=\"wp-block-heading\">What is the Claude Opus 5.5 context window?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">The Claude Platform Docs list a 1 million-token context window and a 128K-token maximum output. A separate documented beta path supports up to 300K output tokens for Message Batches.<\/p>\n\n\n<h3 class=\"wp-block-heading\">Is Claude Opus 5.5 better than Opus 5?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">Anthropic reports lower typical workload cost and faster output, while independent evaluations show strong results on several dated tasks. The size of any quality gain depends on the benchmark, effort setting, prompt, and route.<\/p>\n\n\n<h3 class=\"wp-block-heading\">Can thinking be turned off?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">No. The current model documentation says adaptive thinking is always on. Use the effort parameter to control thinking depth, latency, and cost.<\/p>\n\n\n<h3 class=\"wp-block-heading\">Did the five API tasks prove reliability?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">No. The cards show concrete behavior and configuration limits, but five tasks cannot estimate a general success rate. The coding ceiling issue and Chinese route issue are reported separately.<\/p>\n\n\n<h3 class=\"wp-block-heading\">Did this review prove Chinese quality?<\/h3>\n\n\n<p class=\"wp-block-paragraph\">No. The tested route returned garbled Chinese text, so the result is recorded as an encoding or route issue pending a verified UTF-8 rerun.<\/p>\n\n\n\n<script type=\"application\/ld+json\">{\n    \"@context\": \"https:\\\/\\\/schema.org\",\n    \"@type\": \"FAQPage\",\n    \"mainEntity\": [\n        {\n            \"@type\": \"Question\",\n            \"name\": \"How much does Claude Opus 5.5 cost?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"The official API rate is $4 per million input tokens and $20 per million output tokens. Cache reads are $0.20 per million tokens, and 5-minute cache writes are $5 per million tokens.\"\n            }\n        },\n        {\n            \"@type\": \"Question\",\n            \"name\": \"What is the Claude Opus 5.5 context window?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"The Claude Platform Docs list a 1 million-token context window and a 128K-token maximum output. A separate documented beta path supports up to 300K output tokens for Message Batches.\"\n            }\n        },\n        {\n            \"@type\": \"Question\",\n            \"name\": \"Is Claude Opus 5.5 better than Opus 5?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"Anthropic reports lower typical workload cost and faster output, while independent evaluations show strong results on several dated tasks. The size of any quality gain depends on the benchmark, effort setting, prompt, and route.\"\n            }\n        },\n        {\n            \"@type\": \"Question\",\n            \"name\": \"Can thinking be turned off?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"No. The current model documentation says adaptive thinking is always on. Use the effort parameter to control thinking depth, latency, and cost.\"\n            }\n        },\n        {\n            \"@type\": \"Question\",\n            \"name\": \"Did the five API tasks prove reliability?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"No. The cards show concrete behavior and configuration limits, but five tasks cannot estimate a general success rate. The coding ceiling issue and Chinese route issue are reported separately.\"\n            }\n        },\n        {\n            \"@type\": \"Question\",\n            \"name\": \"Did this review prove Chinese quality?\",\n            \"acceptedAnswer\": {\n                \"@type\": \"Answer\",\n                \"text\": \"No. The tested route returned garbled Chinese text, so the result is recorded as an encoding or route issue pending a verified UTF-8 rerun.\"\n            }\n        }\n    ]\n}<\/script>\n\n<\/div>","protected":false},"excerpt":{"rendered":"<p>Claude Opus 5.5 \u306f\u305d\u306e\u4fa1\u683c\u306b\u898b\u5408\u3046\u4fa1\u5024\u304c\u3042\u308b\u306e\u3067\u3057\u3087\u3046\u304b\uff1f\u516c\u5f0f\u4fa1\u683c\u3001\u30d9\u30f3\u30c1\u30de\u30fc\u30af\u306e\u80cc\u666f\u3001API\u306b\u95a2\u3059\u308b\u6ce8\u610f\u70b9\u3001\u304a\u3088\u3073\u5236\u5fa1\u3055\u308c\u305f\u30b3\u30fc\u30c7\u30a3\u30f3\u30b0\u3001JSON\u3001\u6570\u5b66\u3001\u9577\u6587\u306e\u30c6\u30b9\u30c8\u7d50\u679c\u3092\u3054\u89a7\u304f\u3060\u3055\u3044\u3002.<\/p>","protected":false},"author":13,"featured_media":19708,"comment_status":"closed","ping_status":"closed","sticky":false,"template":"","format":"standard","meta":{"_acf_changed":false,"_seopress_robots_primary_cat":"","_seopress_titles_title":"Claude Opus 5.5 Review: Price, Tests & API Limits","_seopress_titles_desc":"Is Claude Opus 5.5 worth the cost? See official prices, benchmark context, API caveats, and controlled coding, JSON, math, and long-text tests.","_seopress_robots_index":"","footnotes":""},"categories":[7],"tags":[],"class_list":["post-19710","post","type-post","status-publish","format-standard","has-post-thumbnail","hentry","category-ai-chat"],"acf":[],"_links":{"self":[{"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/posts\/19710","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/posts"}],"about":[{"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/types\/post"}],"author":[{"embeddable":true,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/users\/13"}],"replies":[{"embeddable":true,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/comments?post=19710"}],"version-history":[{"count":2,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/posts\/19710\/revisions"}],"predecessor-version":[{"id":19718,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/posts\/19710\/revisions\/19718"}],"wp:attachment":[{"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/media?parent=19710"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/categories?post=19710"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/wp.glbgpt.com\/jp\/wp-json\/wp\/v2\/tags?post=19710"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}